From acf4e1286d8fdee8343da9c8bcd491f7cf3659ff Mon Sep 17 00:00:00 2001 From: Xiaoyang Han Date: Wed, 7 Oct 2026 16:41:46 +0800 Subject: [PATCH 01/50] Complete native MST2 publication, retention and metadata integration Rebase the complete PR #54 code and test delta onto current main as one verified signed commit. Preserve the exact previously reviewed source tree, including fixed-content metadata optimizations and controlled native initialization. --- src/api/router/snapshot_content.rs | 291 +++- src/api/router/snapshot_content_tests.rs | 1239 +++++++++++++++++ src/api/router/snapshot_request.rs | 114 ++ src/api/router/snapshot_request_tests.rs | 353 +++++ src/api/router/snapshot_router.rs | 438 ++++-- src/callisto/mod.rs | 10 + src/callisto/mst2_metadata_payload.rs | 16 + src/callisto/mst2_metadata_prepare.rs | 32 + src/callisto/mst2_metadata_prepare_page.rs | 15 + src/callisto/mst2_native_head.rs | 20 + src/callisto/mst2_native_publication.rs | 27 + src/callisto/mst2_publication.rs | 8 +- src/callisto/mst2_queue_noop_receipt.rs | 34 + src/callisto/mst2_retention_edge.rs | 19 + src/callisto/mst2_retention_gc_op.rs | 26 + src/callisto/mst2_retention_node.rs | 25 + src/callisto/mst2_retention_root.rs | 20 + src/callisto/prelude.rs | 4 + src/callisto/push_queue.rs | 3 + src/ceres/api_service/mod.rs | 7 + src/ceres/api_service/mono_api_service.rs | 4 + src/ceres/snapshot/error.rs | 10 +- src/ceres/snapshot/metadata_install.rs | 463 ++++++ src/ceres/snapshot/mod.rs | 4 + src/ceres/snapshot/native_projection_tests.rs | 646 +++++++++ src/ceres/snapshot/pages.rs | 788 +++++++++-- src/ceres/snapshot/projection_observation.rs | 638 +++++++++ src/ceres/snapshot/projection_writer.rs | 506 +++++++ src/ceres/snapshot/projection_writer_tests.rs | 358 +++++ src/ceres/snapshot/retention.rs | 447 +++++- src/ceres/snapshot/retention_dag.rs | 554 ++++++++ src/ceres/snapshot/retention_dag_tests.rs | 345 +++++ src/ceres/snapshot/retention_prepare_tests.rs | 92 ++ src/ceres/snapshot/runtime.rs | 452 +++++- src/commands/config.rs | 111 ++ src/commands/mod.rs | 2 +- src/commands/service/mod.rs | 28 +- .../service/native_publication_init.rs | 198 +++ .../native_publication_init_preflight.rs | 162 +++ src/config/model.rs | 3 + src/config/reload.rs | 33 + src/config/validate.rs | 192 ++- src/context/mod.rs | 16 + ...100_add_mst2_publication_request_digest.rs | 48 + ...05_000100_add_mst2_retention_durability.rs | 119 ++ .../m20261005_000200_add_mst2_native_head.rs | 72 + ...1005_000200_harden_mst2_retention_graph.rs | 76 + ...261005_000300_add_mst2_metadata_install.rs | 76 + src/jupiter/migration/mod.rs | 19 +- .../service/native_publication_push_tests.rs | 384 +++++ src/jupiter/service/push_queue_service.rs | 740 +++++++++- src/jupiter/storage/mod.rs | 19 +- src/jupiter/storage/mono_storage.rs | 141 +- .../storage/mst2_publication_storage.rs | 514 +++++++ src/jupiter/storage/mst2_publication_tests.rs | 1091 +++++++++++++++ src/jupiter/storage/mst2_retention.rs | 633 +++++++++ src/jupiter/storage/mst2_retention_tests.rs | 629 +++++++++ .../storage/native_metadata_install.rs | 802 +++++++++++ .../storage/native_metadata_install_tests.rs | 1226 ++++++++++++++++ .../storage/native_publication_storage.rs | 656 +++++++++ .../storage/native_publication_tests.rs | 854 ++++++++++++ src/jupiter/storage/push_queue_storage.rs | 176 ++- src/jupiter/tests.rs | 8 +- tests/common/git_cli.rs | 82 +- tests/integration_git_cli.rs | 36 +- tests/integration_storage_events_runtime.rs | 13 +- 66 files changed, 16521 insertions(+), 646 deletions(-) create mode 100644 src/api/router/snapshot_content_tests.rs create mode 100644 src/api/router/snapshot_request.rs create mode 100644 src/api/router/snapshot_request_tests.rs create mode 100644 src/callisto/mst2_metadata_payload.rs create mode 100644 src/callisto/mst2_metadata_prepare.rs create mode 100644 src/callisto/mst2_metadata_prepare_page.rs create mode 100644 src/callisto/mst2_native_head.rs create mode 100644 src/callisto/mst2_native_publication.rs create mode 100644 src/callisto/mst2_queue_noop_receipt.rs create mode 100644 src/callisto/mst2_retention_edge.rs create mode 100644 src/callisto/mst2_retention_gc_op.rs create mode 100644 src/callisto/mst2_retention_node.rs create mode 100644 src/callisto/mst2_retention_root.rs create mode 100644 src/ceres/snapshot/metadata_install.rs create mode 100644 src/ceres/snapshot/native_projection_tests.rs create mode 100644 src/ceres/snapshot/projection_observation.rs create mode 100644 src/ceres/snapshot/projection_writer.rs create mode 100644 src/ceres/snapshot/projection_writer_tests.rs create mode 100644 src/ceres/snapshot/retention_dag.rs create mode 100644 src/ceres/snapshot/retention_dag_tests.rs create mode 100644 src/ceres/snapshot/retention_prepare_tests.rs create mode 100644 src/commands/service/native_publication_init.rs create mode 100644 src/commands/service/native_publication_init_preflight.rs create mode 100644 src/jupiter/migration/m20261005_000100_add_mst2_publication_request_digest.rs create mode 100644 src/jupiter/migration/m20261005_000100_add_mst2_retention_durability.rs create mode 100644 src/jupiter/migration/m20261005_000200_add_mst2_native_head.rs create mode 100644 src/jupiter/migration/m20261005_000200_harden_mst2_retention_graph.rs create mode 100644 src/jupiter/migration/m20261005_000300_add_mst2_metadata_install.rs create mode 100644 src/jupiter/service/native_publication_push_tests.rs create mode 100644 src/jupiter/storage/mst2_publication_storage.rs create mode 100644 src/jupiter/storage/mst2_publication_tests.rs create mode 100644 src/jupiter/storage/mst2_retention.rs create mode 100644 src/jupiter/storage/mst2_retention_tests.rs create mode 100644 src/jupiter/storage/native_metadata_install.rs create mode 100644 src/jupiter/storage/native_metadata_install_tests.rs create mode 100644 src/jupiter/storage/native_publication_storage.rs create mode 100644 src/jupiter/storage/native_publication_tests.rs diff --git a/src/api/router/snapshot_content.rs b/src/api/router/snapshot_content.rs index b52dbba6..08158723 100644 --- a/src/api/router/snapshot_content.rs +++ b/src/api/router/snapshot_content.rs @@ -9,16 +9,18 @@ use axum::{ http::HeaderMap, response::{IntoResponse, Response}, }; -use bytes::Bytes; use futures::stream::StreamExt; use serde::Deserialize; use serde_json::json; -use super::{abs_view_path, internal, mst2_error_response}; +use super::{abs_view_path, internal, mst2_error_response, request::Mst2Bytes, treeframe_response}; use crate::ceres::snapshot::{ chunks::{ChunkProjection, get_or_project}, error::{SnapshotError, SnapshotErrorCode}, - pages::{WalkOutcome, base64_of, hex_of, resolve_abs}, + pages::{ + MetadataWalkOutcome, WalkOutcome, base64_of, fetch_raw_blob, hex_of, resolve_abs, + resolve_abs_metadata, + }, resolver::FsKind, runtime::runtime, view::validate_scope_relative_path, @@ -26,12 +28,130 @@ use crate::ceres::snapshot::{ /// One file resolved at a fixed path with verified content. struct ResolvedFile { - fs_kind: FsKind, digest: [u8; 32], size: u64, raw: Vec, } +struct ResolvedFileMetadata { + fs_kind: FsKind, + oid: String, + digest: [u8; 32], + size: u64, +} + +#[allow(clippy::result_large_err)] +async fn fixed_root_tree( + handler: &T, + oid: &str, +) -> Result { + let tree = handler.get_tree_by_hash(oid).await.map_err(internal)?; + if tree.id.to_string() != oid { + return Err(mst2_error_response(SnapshotError::new( + SnapshotErrorCode::IntegrityError, + "fetched fixed root tree identity mismatch", + ))); + } + Ok(tree) +} + +#[allow(clippy::result_large_err)] +async fn resolve_file_metadata( + handler: &T, + root_tree: &git_internal::internal::object::tree::Tree, + scope: &str, + path: &str, + expected_digest: Option<&str>, +) -> Result { + validate_scope_relative_path(path).map_err(mst2_error_response)?; + let (fs_kind, oid) = match resolve_abs_metadata(handler, root_tree, &abs_view_path(scope, path)) + .await + .map_err(mst2_error_response)? + { + MetadataWalkOutcome::FoundFile { fs_kind, oid } => (fs_kind, oid), + MetadataWalkOutcome::FoundDir => { + return Err(mst2_error_response(SnapshotError::new( + SnapshotErrorCode::NotDirectory, + format!("{path} is a directory"), + ))); + } + MetadataWalkOutcome::Absent => { + return Err(mst2_error_response(SnapshotError::new( + SnapshotErrorCode::PathNotFound, + format!("{path} absent in the fixed view"), + ))); + } + MetadataWalkOutcome::NotDirectory { symlink } => { + return Err(mst2_error_response(SnapshotError::new( + if symlink { + SnapshotErrorCode::SymlinkTraversal + } else { + SnapshotErrorCode::NotDirectory + }, + format!("{path}: intermediate component is not a directory"), + ))); + } + }; + let storage = handler.get_context(); + let mut verified = storage + .mono_storage() + .get_verified_blobs(vec![oid.clone()]) + .await + .map_err(|error| { + let code = if matches!( + error, + crate::common::errors::MegaError::ObjStorageInconsistent(_) + ) { + SnapshotErrorCode::IntegrityError + } else { + SnapshotErrorCode::Internal + }; + tracing::warn!(error = %error, "fixed content metadata lookup failed"); + mst2_error_response(SnapshotError::new( + code, + "fixed content metadata lookup failed", + )) + })?; + let fact = verified.remove(&oid).ok_or_else(|| { + mst2_error_response(SnapshotError::new( + SnapshotErrorCode::MetadataNotReady, + "fixed object has no current verified size and digest fact", + )) + })?; + let digest: [u8; 32] = fact.raw_sha256.as_slice().try_into().map_err(|_| { + mst2_error_response(SnapshotError::new( + SnapshotErrorCode::IntegrityError, + "invalid verified content digest", + )) + })?; + let size = u64::try_from(fact.size).map_err(|_| { + mst2_error_response(SnapshotError::new( + SnapshotErrorCode::IntegrityError, + "invalid verified content size", + )) + })?; + if fs_kind == FsKind::Symlink && !(1..=4095).contains(&size) { + return Err(mst2_error_response(SnapshotError::new( + SnapshotErrorCode::IntegrityError, + "verified symlink size is outside the fixed filesystem profile", + ))); + } + if let Some(expected) = expected_digest + && expected != format!("sha256:{}", hex_of(&digest)) + { + return Err(mst2_error_response(SnapshotError::new( + SnapshotErrorCode::DigestMismatch, + format!("{path}: content does not match expected_digest"), + ))); + } + Ok(ResolvedFileMetadata { + fs_kind, + oid, + digest, + size, + }) +} + pub(super) fn fs_kind_str(k: FsKind) -> &'static str { match k { FsKind::Regular => "regular", @@ -58,11 +178,7 @@ async fn resolve_file( .map_err(mst2_error_response)? { WalkOutcome::FoundFile { - fs_kind, - raw, - size, - digest, - .. + raw, size, digest, .. } => { if let Some(expected) = expected_digest && expected != format!("sha256:{}", hex_of(&digest)) @@ -72,12 +188,7 @@ async fn resolve_file( format!("{path}: content does not match expected_digest"), ))); } - Ok(ResolvedFile { - fs_kind, - digest, - size, - raw, - }) + Ok(ResolvedFile { digest, size, raw }) } WalkOutcome::FoundDir => Err(mst2_error_response(SnapshotError::new( SnapshotErrorCode::NotDirectory, @@ -129,11 +240,8 @@ pub(super) async fn blob_head( .api_handler(std::path::Path::new("/")) .await .map_err(internal)?; - let root_tree = handler - .get_tree_by_hash(&ctx.root_tree_oid) - .await - .map_err(internal)?; - let f = resolve_file( + let root_tree = fixed_root_tree(handler.as_ref(), &ctx.root_tree_oid).await?; + let f = resolve_file_metadata( handler.as_ref(), &root_tree, &ctx.built.descriptor.scope, @@ -145,6 +253,7 @@ pub(super) async fn blob_head( .header("content-length", f.size.to_string()) .header("etag", format!("\"sha256:{}\"", hex_of(&f.digest))) .header("cache-control", "private, no-cache, no-transform") + .header("vary", "Authorization, Accept") .header("x-mega-fs-kind", fs_kind_str(f.fs_kind)) .header("x-mega-content-size", f.size.to_string()) .body(axum::body::Body::empty()) @@ -174,7 +283,7 @@ const OBJECT_TOTAL_MAX: usize = 8 * 1024 * 1024; pub(super) async fn objects( state: State, AxumPath(snapshot_id): AxumPath, - body: Bytes, + Mst2Bytes(body): Mst2Bytes, ) -> Result { ensure(&state)?; let ctx = runtime() @@ -202,10 +311,7 @@ pub(super) async fn objects( .api_handler(std::path::Path::new("/")) .await .map_err(internal)?; - let root_tree = handler - .get_tree_by_hash(&ctx.root_tree_oid) - .await - .map_err(internal)?; + let root_tree = fixed_root_tree(handler.as_ref(), &ctx.root_tree_oid).await?; let scope = ctx.built.descriptor.scope.clone(); // Copied references: each per-item future borrows the shared walk state // without moving it (the stream is an FnMut over owned items). @@ -290,11 +396,7 @@ pub(super) async fn objects( ); out.extend_from_slice(&end); - axum::response::Response::builder() - .header("content-type", "application/octet-stream") - .header("cache-control", "private, no-cache, no-transform") - .body(axum::body::Body::from(Bytes::from(out))) - .map_err(|e| mst2_error_response(internal(format!("body build failed: {e}")))) + treeframe_response(&snapshot_id, &body, out).map_err(mst2_error_response) } #[derive(Deserialize, Debug)] @@ -305,6 +407,12 @@ pub(super) struct ChunkMapQuery { pub(super) expected_digest: Option, #[serde(default)] pub(super) page: Option, + /// Canonical v3 page requests bind the page to the already verified map + /// instead of repeating the file digest. + #[serde(default)] + pub(super) map_id: Option, + #[serde(default)] + pub(super) page_index: Option, } /// Project a fixed-path file into its range-readable representation. @@ -316,23 +424,40 @@ async fn project_for( path: &str, expected_digest: Option<&str>, ) -> Result, Response> { - let f = resolve_file(handler, root_tree, scope, path, expected_digest).await?; + let f = resolve_file_metadata(handler, root_tree, scope, path, expected_digest).await?; // The first request for a digest builds the projection from the fixed // Git object; later requests slice the cached representation. A miss // rebuilds, never errors with "missing chunk". let digest = f.digest; - get_or_project(digest, || async move { - let raw = f.raw; - if raw.len() as u64 != f.size { + let size = f.size; + let projection = get_or_project(digest, || async move { + let raw = fetch_raw_blob(handler, &f.oid).await?; + if raw.len() as u64 != size { return Err(SnapshotError::new( - SnapshotErrorCode::Internal, - "resolved blob size disagrees with its length", + SnapshotErrorCode::IntegrityError, + "fixed blob length disagrees with its verified size fact", )); } Ok(raw) }) .await - .map_err(mst2_error_response) + .map_err(|error| { + mst2_error_response(if error.code == SnapshotErrorCode::DigestMismatch { + SnapshotError::new( + SnapshotErrorCode::IntegrityError, + "fixed blob digest disagrees with its verified fact", + ) + } else { + error + }) + })?; + if projection.map.file_size != size || projection.map.file_content_id != digest { + return Err(mst2_error_response(SnapshotError::new( + SnapshotErrorCode::IntegrityError, + "cached projection disagrees with the fixed verified fact", + ))); + } + Ok(projection) } #[allow(clippy::result_large_err)] @@ -350,10 +475,7 @@ pub(super) async fn chunk_map( .api_handler(std::path::Path::new("/")) .await .map_err(internal)?; - let root_tree = handler - .get_tree_by_hash(&ctx.root_tree_oid) - .await - .map_err(internal)?; + let root_tree = fixed_root_tree(handler.as_ref(), &ctx.root_tree_oid).await?; let scope = ctx.built.descriptor.scope.clone(); let proj = project_for( handler.as_ref(), @@ -363,17 +485,23 @@ pub(super) async fn chunk_map( q.expected_digest.as_deref(), ) .await?; + // Canonical v3 uses a closed top-level envelope and a nested map + // descriptor. Keeping the map under `map` is part of profile selection; + // the client rejects the legacy flat shape once canonical capabilities + // have been advertised. let body = json!({ "snapshot_id": snapshot_id, "path": q.path, - "schema_version": 2, - "file_content_id": format!("sha256:{}", hex_of(&proj.map.file_content_id)), - "file_size": proj.map.file_size.to_string(), - "chunk_size": mst2_codec::chunkmap::CHUNK_SIZE, - "chunk_count": proj.map.chunk_count.to_string(), - "page_count": proj.map.page_count.to_string(), - "pages_root": format!("sha256:{}", hex_of(&proj.map.pages_root)), - "map_id": format!("sha256:{}", hex_of(&proj.map_id)), + "map": { + "schema_version": 2, + "file_content_id": format!("sha256:{}", hex_of(&proj.map.file_content_id)), + "file_size": proj.map.file_size.to_string(), + "chunk_size": mst2_codec::chunkmap::CHUNK_SIZE, + "chunk_count": proj.map.chunk_count.to_string(), + "page_count": proj.map.page_count.to_string(), + "pages_root": format!("sha256:{}", hex_of(&proj.map.pages_root)), + "map_id": format!("sha256:{}", hex_of(&proj.map_id)), + }, }); Ok(Json(body).into_response()) } @@ -389,18 +517,30 @@ pub(super) async fn chunk_map_pages( .context(&snapshot_id) .map_err(mst2_error_response)?; validate_scope_relative_path(&q.path).map_err(mst2_error_response)?; - let page_index: u64 = match q.page.as_deref() { - Some(s) => parse_decimal_count(s, "page").map_err(mst2_error_response)?, - None => 0, + // Canonical v3 uses `map_id` + `page_index`; retain parsing of the old + // names only while the legacy client is still present in this checkout. + let page_index: u64 = match (q.page_index.as_deref(), q.page.as_deref()) { + (Some(_), Some(_)) => { + return Err(mst2_error_response(SnapshotError::new( + SnapshotErrorCode::ScopeInvalid, + "page_index and page are mutually exclusive", + ))); + } + (Some(s), None) => parse_decimal_count(s, "page_index").map_err(mst2_error_response)?, + (None, Some(s)) => parse_decimal_count(s, "page").map_err(mst2_error_response)?, + (None, None) => 0, }; + if q.page_index.is_some() != q.map_id.is_some() { + return Err(mst2_error_response(SnapshotError::new( + SnapshotErrorCode::ScopeInvalid, + "canonical page requests require map_id and page_index together", + ))); + } let handler = state .api_handler(std::path::Path::new("/")) .await .map_err(internal)?; - let root_tree = handler - .get_tree_by_hash(&ctx.root_tree_oid) - .await - .map_err(internal)?; + let root_tree = fixed_root_tree(handler.as_ref(), &ctx.root_tree_oid).await?; let scope = ctx.built.descriptor.scope.clone(); let proj = project_for( handler.as_ref(), @@ -410,6 +550,15 @@ pub(super) async fn chunk_map_pages( q.expected_digest.as_deref(), ) .await?; + if let Some(expected_map) = q.map_id.as_deref() { + let actual_map = format!("sha256:{}", hex_of(&proj.map_id)); + if expected_map != actual_map { + return Err(mst2_error_response(SnapshotError::new( + SnapshotErrorCode::DigestMismatch, + "chunk-map page map_id does not bind to the fixed file", + ))); + } + } let (leaf, proof) = proj .leaf_and_proof(page_index) .map_err(mst2_error_response)?; @@ -429,16 +578,13 @@ pub(super) async fn chunk_map_pages( }) }) .collect(); + // Canonical v3 page responses are closed and carry the encoded leaf + // directly. The map descriptor already authenticated page_count and the + // client verifies this page_index against that fixed descriptor. let body = json!({ - "snapshot_id": snapshot_id, - "path": q.path, "map_id": format!("sha256:{}", hex_of(&proj.map_id)), - "page_count": proj.map.page_count.to_string(), - "leaf": { - "page_index": leaf.page_index.to_string(), - "count": leaf.chunk_sha256.len().to_string(), - "data_base64": base64_of(&leaf_bytes), - }, + "page_index": leaf.page_index.to_string(), + "leaf_base64": base64_of(&leaf_bytes), "proof": proof_json, }); Ok(Json(body).into_response()) @@ -473,7 +619,7 @@ struct Planned { pub(super) async fn chunks( state: State, AxumPath(snapshot_id): AxumPath, - body: Bytes, + Mst2Bytes(body): Mst2Bytes, ) -> Result { ensure(&state)?; let ctx = runtime() @@ -500,10 +646,7 @@ pub(super) async fn chunks( .api_handler(std::path::Path::new("/")) .await .map_err(internal)?; - let root_tree = handler - .get_tree_by_hash(&ctx.root_tree_oid) - .await - .map_err(internal)?; + let root_tree = fixed_root_tree(handler.as_ref(), &ctx.root_tree_oid).await?; let scope = ctx.built.descriptor.scope.clone(); let mut planned: Vec = Vec::new(); let mut units: Vec<(String, u64)> = Vec::new(); @@ -589,11 +732,7 @@ pub(super) async fn chunks( ); out.extend_from_slice(&end); - axum::response::Response::builder() - .header("content-type", "application/octet-stream") - .header("cache-control", "private, no-cache, no-transform") - .body(axum::body::Body::from(Bytes::from(out))) - .map_err(|e| mst2_error_response(internal(format!("body build failed: {e}")))) + treeframe_response(&snapshot_id, &body, out).map_err(mst2_error_response) } /// Strict decimal-string parse for unsigned counts (spec 04 §1: no leading @@ -635,3 +774,7 @@ fn ensure(state: &crate::api::MonoApiServiceState) -> Result<(), Response> { ))) } } + +#[cfg(test)] +#[path = "snapshot_content_tests.rs"] +mod tests; diff --git a/src/api/router/snapshot_content_tests.rs b/src/api/router/snapshot_content_tests.rs new file mode 100644 index 00000000..72590edb --- /dev/null +++ b/src/api/router/snapshot_content_tests.rs @@ -0,0 +1,1239 @@ +use std::{ + sync::{ + Arc, + atomic::{AtomicUsize, Ordering}, + }, + time::Duration, +}; + +use axum::{ + Router, + body::{Body, to_bytes}, + http::{Method, Request}, + response::Response, +}; +use base64::{Engine, engine::general_purpose::STANDARD}; +use bytes::Bytes; +use futures::StreamExt; +use git_internal::{ + hash::{HashKind, ObjectHash}, + internal::object::{ + commit::Commit, + tree::{Tree, TreeItem, TreeItemMode}, + }, +}; +use mst2_codec::{ + chunkmap::{CHUNK_SIZE, ChunkLeaf, verify_leaf}, + treeframe::{Frame, parse_stream}, +}; +use sea_orm::{ + ActiveModelTrait, ColumnTrait, ConnectionTrait, EntityTrait, IntoActiveModel, QueryFilter, Set, +}; +use serde_json::{Value, json}; +use sha2::{Digest, Sha256}; +use tower::ServiceExt; + +use crate::{ + api::{ + MonoApiServiceState, oauth::api_store::BrowserSessionStore, + router::snapshot_router::routers, + }, + callisto::{mega_refs, mst2_verified_object, sea_orm_active_enums::PushQueueKindEnum}, + ceres::{ + api_service::{ApiHandler, cache::GitObjectCache, mono_api_service::MonoApiService}, + snapshot::{ + error::SnapshotErrorCode, + pages::{MetadataWalkOutcome, hex_of, resolve_abs_metadata}, + resolver::FsKind, + }, + }, + common::utils::MEGA_BRANCH_NAME, + config::{PushPolicy, testing::isolated_config}, + jupiter::{ + service::{ + git_service::GitService, + push_queue_service::{ + EnqueueRequest, ExecuteOutcome, ExecuteRequest, PushExecContext, PushPayload, + push_operation_id, + }, + }, + storage::{ + base_storage::StorageConnector, + mono_storage::MST2_VERIFICATION_VERSION, + object_storage::{MegaObjectStorageWrapper, build_object_storage}, + push_queue_storage::{ClaimOutcome, EnqueueOutcome}, + }, + tests::{test_redis_manager, test_storage_with_config, with_test_vault}, + }, + orbit_api::{ + error::OrbitResult, + log_storage::{LogManifest, LogStorage}, + object_storage::{ + MegaObjectStorage, ObjectByteStream, ObjectKey, ObjectMeta, ObjectNamespace, + }, + }, +}; + +const TOKEN: &str = "mst2-fixed-content-test"; + +#[derive(Default)] +struct ReadCounts { + whole: AtomicUsize, + range: AtomicUsize, + bytes: AtomicUsize, +} + +impl ReadCounts { + fn reset(&self) { + self.whole.store(0, Ordering::SeqCst); + self.range.store(0, Ordering::SeqCst); + self.bytes.store(0, Ordering::SeqCst); + } + + fn assert(&self, whole: usize, bytes: usize) { + assert_eq!(self.whole.load(Ordering::SeqCst), whole); + assert_eq!(self.range.load(Ordering::SeqCst), 0); + assert_eq!(self.bytes.load(Ordering::SeqCst), bytes); + } +} + +struct CountingStorage { + inner: MegaObjectStorageWrapper, + counts: Arc, +} + +#[async_trait::async_trait] +impl MegaObjectStorage for CountingStorage { + async fn put_stream( + &self, + key: &ObjectKey, + data: ObjectByteStream, + meta: ObjectMeta, + ) -> OrbitResult<()> { + self.inner.inner.put_stream(key, data, meta).await + } + + async fn get_stream(&self, key: &ObjectKey) -> OrbitResult<(ObjectByteStream, ObjectMeta)> { + self.counts.whole.fetch_add(1, Ordering::SeqCst); + let (stream, meta) = self.inner.inner.get_stream(key).await?; + let counts = self.counts.clone(); + let stream = stream.map(move |part| { + if let Ok(bytes) = &part { + counts.bytes.fetch_add(bytes.len(), Ordering::SeqCst); + } + part + }); + Ok((Box::pin(stream), meta)) + } + + async fn get_range_stream( + &self, + key: &ObjectKey, + start: u64, + end: Option, + ) -> OrbitResult<(ObjectByteStream, ObjectMeta)> { + self.counts.range.fetch_add(1, Ordering::SeqCst); + self.inner.inner.get_range_stream(key, start, end).await + } + + async fn exists(&self, key: &ObjectKey) -> OrbitResult { + self.inner.inner.exists(key).await + } + + async fn signed_url( + &self, + key: &ObjectKey, + method: Method, + expires_in: Duration, + ) -> OrbitResult> { + self.inner.inner.signed_url(key, method, expires_in).await + } + + async fn delete(&self, key: &ObjectKey) -> OrbitResult<()> { + self.inner.inner.delete(key).await + } +} + +#[async_trait::async_trait] +impl LogStorage for CountingStorage { + async fn append( + &self, + key: &ObjectKey, + data: ObjectByteStream, + meta: ObjectMeta, + ) -> OrbitResult<()> { + self.inner.inner.append(key, data, meta).await + } + + async fn read_range( + &self, + key: &ObjectKey, + offset: u64, + length: u64, + ) -> OrbitResult { + self.inner.inner.read_range(key, offset, length).await + } + + async fn read_lines_range( + &self, + key: &ObjectKey, + start_line: u64, + end_line: u64, + ) -> OrbitResult { + self.inner + .inner + .read_lines_range(key, start_line, end_line) + .await + } + + async fn append_concurrently( + &self, + key: &ObjectKey, + data: ObjectByteStream, + meta: ObjectMeta, + ) -> OrbitResult<()> { + self.inner.inner.append_concurrently(key, data, meta).await + } + + async fn load_manifest(&self, key: &ObjectKey) -> OrbitResult { + self.inner.inner.load_manifest(key).await + } + + async fn log_exists(&self, key: &ObjectKey) -> OrbitResult { + self.inner.inner.log_exists(key).await + } +} + +struct Fixture { + app: Router, + state: MonoApiServiceState, + snapshot: String, + lease: String, + oid: String, + raw: Vec, + digest: [u8; 32], + counts: Arc, + _temp: tempfile::TempDir, +} + +fn tree(items: Vec) -> Tree { + crate::ceres::view::tree_source::build_tree(HashKind::Sha1, items).unwrap() +} + +fn item(mode: TreeItemMode, oid: ObjectHash, name: &str) -> TreeItem { + TreeItem::new(mode, oid, name.to_string()) +} + +fn digest(bytes: &[u8]) -> [u8; 32] { + Sha256::digest(bytes).into() +} + +async fn publish_native_push( + storage: &crate::jupiter::storage::Storage, + path: &str, + old: ObjectHash, + new: &Commit, +) { + let old_id = old.to_string(); + let new_id = new.id.to_string(); + let payload = PushPayload { + commits: vec![new_id.clone()], + fork_base: Some(old_id.clone()), + n: 1, + }; + let outcome = storage + .push_queue_service + .enqueue(EnqueueRequest { + kind: PushQueueKindEnum::Push, + operation_id: push_operation_id(&old_id, &new_id), + path: path.to_string(), + old_id, + new_id, + requester: None, + payload: payload.to_json(), + ref_name: Some(MEGA_BRANCH_NAME.to_string()), + is_delete: false, + }) + .await + .unwrap(); + let EnqueueOutcome::Inserted { id } = outcome else { + panic!("fresh fixture push must insert: {outcome:?}"); + }; + assert_eq!( + storage + .push_queue_service + .storage() + .claim_for_execution(id) + .await + .unwrap(), + ClaimOutcome::Claimed + ); + let context = PushExecContext { + storage: storage.clone(), + git_object_cache: Arc::new(GitObjectCache { + connection: test_redis_manager().await, + prefix: String::new(), + }), + pre_apply_enter_barrier: None, + pre_apply_release_barrier: None, + }; + let outcome = storage + .push_queue_service + .execute_b3( + ExecuteRequest { + id, + ..Default::default() + }, + None, + None, + Some(&context), + ) + .await + .unwrap(); + assert!(matches!( + outcome, + ExecuteOutcome::Done { + root_cas_writes: 1, + .. + } + )); +} + +impl Fixture { + async fn new() -> Self { + let temp = tempfile::tempdir().unwrap(); + let mut config = isolated_config(temp.path().join("config")); + config.monorepo.push_policy = PushPolicy::Trunk; + config.mst2.enabled = true; + config.mst2.publication_enabled = true; + config.mst2.instance_uuid = Some(uuid::Uuid::new_v4().to_string()); + config.mst2.auth_token = Some(TOKEN.to_string()); + let backend = build_object_storage(&config.object_storage).await.unwrap(); + let counts = Arc::new(ReadCounts::default()); + let mut storage = test_storage_with_config(temp.path(), config).await; + storage.git_service = GitService { + obj_storage: MegaObjectStorageWrapper::new(Arc::new(CountingStorage { + inner: backend, + counts: counts.clone(), + })), + }; + let storage = with_test_vault(storage, temp.path()).await; + // Unique bytes prevent another test's process-wide digest cache from + // turning a required cold read into a hit. These bytes also look like + // a Git object header, and contain NUL and non-UTF-8 content. + let prefix = format!("blob 3\0abc\0{}", uuid::Uuid::new_v4()); + let mut raw = vec![0xff; CHUNK_SIZE as usize + 113]; + raw[..prefix.len()].copy_from_slice(prefix.as_bytes()); + let oid = storage + .git_service + .save_object_from_raw(Bytes::copy_from_slice(&raw)) + .await + .unwrap(); + let blob_oid = ObjectHash::from_hex_for_kind(HashKind::Sha1, &oid).unwrap(); + let link_oid = storage + .git_service + .save_object_from_raw(Bytes::from_static(b"file")) + .await + .unwrap(); + let link_oid = ObjectHash::from_hex_for_kind(HashKind::Sha1, &link_oid).unwrap(); + let empty_oid = storage + .git_service + .save_object_from_raw(Bytes::new()) + .await + .unwrap(); + let empty_oid = ObjectHash::from_hex_for_kind(HashKind::Sha1, &empty_oid).unwrap(); + let empty_dir = tree(vec![]); + let nested = tree(vec![item(TreeItemMode::Blob, blob_oid, "file")]); + let project = tree(vec![ + item(TreeItemMode::Blob, blob_oid, "alias"), + item(TreeItemMode::Tree, empty_dir.id, "directory"), + item(TreeItemMode::Blob, empty_oid, "empty"), + item(TreeItemMode::BlobExecutable, blob_oid, "executable"), + item(TreeItemMode::Blob, blob_oid, "file"), + item(TreeItemMode::Link, link_oid, "link"), + item(TreeItemMode::Tree, nested.id, "nested"), + ]); + let old_tip = Commit::from_tree_id_with_kind( + HashKind::Sha1, + project.id, + vec![], + "fixed content initial path tip", + ) + .unwrap(); + let new_tip = Commit::from_tree_id_with_kind( + HashKind::Sha1, + project.id, + vec![old_tip.id], + "fixed content published path tip", + ) + .unwrap(); + let root = tree(vec![ + item(TreeItemMode::Blob, blob_oid, "outside"), + item(TreeItemMode::Tree, project.id, "project"), + ]); + let commit = Commit::from_tree_id_with_kind( + HashKind::Sha1, + root.id, + vec![], + "fixed content HTTP test", + ) + .unwrap(); + let mono = storage.mono_storage(); + mono.save_mega_trees( + vec![empty_dir, nested, project, root.clone()], + commit.id, + None, + ) + .await + .unwrap(); + mono.save_mega_commits(vec![commit.clone(), old_tip.clone(), new_tip.clone()], None) + .await + .unwrap(); + mono.save_refs( + mega_refs::Model::new( + "/", + MEGA_BRANCH_NAME.to_string(), + commit.id.to_string(), + root.id.to_string(), + false, + ), + None, + ) + .await + .unwrap(); + mono.save_refs( + mega_refs::Model::new( + "/project", + MEGA_BRANCH_NAME.to_string(), + old_tip.id.to_string(), + old_tip.tree_id.to_string(), + false, + ), + None, + ) + .await + .unwrap(); + mono.initialize_native_publication(storage.config().mst2.instance_uuid.as_deref().unwrap()) + .await + .unwrap(); + publish_native_push(&storage, "/project", old_tip.id, &new_tip).await; + let head = mono + .read_native_publication_head(storage.config().mst2.instance_uuid.as_deref().unwrap()) + .await + .unwrap(); + assert_eq!(head.token.sequence, 1); + assert!(head.token.certificate.is_some()); + assert_eq!(head.root.commit, commit.id.to_string()); + assert_eq!(head.root.tree, root.id.to_string()); + let state = MonoApiServiceState { + entity_store: storage.entity_store.clone(), + storage, + session_store: BrowserSessionStore::Anonymous, + git_object_cache: Arc::new(GitObjectCache { + connection: test_redis_manager().await, + prefix: String::new(), + }), + listen_addr: "127.0.0.1:0".to_string(), + }; + let app = Router::new().nest("/api/v2", routers(state.clone()).with_state(state.clone())); + let response = app + .clone() + .oneshot( + Request::builder() + .method("POST") + .uri("/api/v2/snapshots/resolve") + .header("authorization", format!("Bearer {TOKEN}")) + .header("content-type", "application/json") + .body(Body::from( + json!({"target":{"kind":"latest"},"scope":"/project"}).to_string(), + )) + .unwrap(), + ) + .await + .unwrap(); + let resolved = success_json(response).await; + assert!(counts.whole.load(Ordering::SeqCst) > 0); + assert!( + !mono + .get_verified_blobs(vec![oid.clone()]) + .await + .unwrap() + .is_empty() + ); + counts.reset(); + Self { + app, + state, + snapshot: resolved["descriptor"]["snapshot_id"] + .as_str() + .unwrap() + .to_string(), + lease: resolved["lease_id"].as_str().unwrap().to_string(), + oid, + digest: Sha256::digest(&raw).into(), + raw, + counts, + _temp: temp, + } + } + + fn request(&self, method: &str, suffix: &str, body: Body) -> Request { + Request::builder() + .method(method) + .uri(format!("/api/v2/snapshots/{}/{suffix}", self.snapshot)) + .header("authorization", format!("Bearer {TOKEN}")) + .header("x-mega-snapshot-lease", &self.lease) + .header("content-type", "application/json") + .body(body) + .unwrap() + } + + async fn send(&self, method: &str, suffix: &str, body: Body) -> Response { + self.app + .clone() + .oneshot(self.request(method, suffix, body)) + .await + .unwrap() + } + + fn digest_string(&self) -> String { + format!("sha256:{}", hex_of(&self.digest)) + } + + async fn map(&self, path: &str) -> Value { + success_json( + self.send("GET", &format!("chunk-map?path={path}"), Body::empty()) + .await, + ) + .await + } + + async fn fact(&self) -> mst2_verified_object::Model { + self.state + .storage + .mono_storage() + .get_verified_blobs(vec![self.oid.clone()]) + .await + .unwrap() + .remove(&self.oid) + .unwrap() + } + + async fn delete_fact(&self) { + mst2_verified_object::Entity::delete_many() + .filter(mst2_verified_object::Column::GitOid.eq(&self.oid)) + .exec(self.state.storage.mono_storage().get_connection()) + .await + .unwrap(); + } + + async fn replace_fact(&self, fact: mst2_verified_object::Model) { + self.delete_fact().await; + mst2_verified_object::Entity::insert(fact.into_active_model()) + .exec(self.state.storage.mono_storage().get_connection()) + .await + .unwrap(); + } + + async fn write_raw(&self, bytes: Vec) { + self.state + .storage + .git_service + .save_object_from_model(bytes, &self.oid) + .await + .unwrap(); + } + + fn chunk_body(&self, path: &str, map_id: &str, index: &str) -> Value { + json!({"items":[{ + "path":path,"expected_digest":self.digest_string(), + "map_id":map_id,"chunk_index":index + }],"encoding":"identity"}) + } +} + +async fn success_json(response: Response) -> Value { + let status = response.status(); + let bytes = to_bytes(response.into_body(), 2 * 1024 * 1024) + .await + .unwrap(); + assert_eq!(status.as_u16(), 200, "{}", String::from_utf8_lossy(&bytes)); + serde_json::from_slice(&bytes).unwrap() +} + +async fn error(response: Response, status: u16, code: &str, retryable: bool) { + assert_eq!(response.status().as_u16(), status); + assert_eq!(response.headers()["content-type"], "application/json"); + let request_id = response.headers()["x-request-id"] + .to_str() + .unwrap() + .to_string(); + assert!(!request_id.is_empty()); + let bytes = to_bytes(response.into_body(), 16 * 1024).await.unwrap(); + let value: Value = serde_json::from_slice(&bytes).unwrap(); + assert_eq!(value["error"]["code"], code); + assert_eq!(value["error"]["retryable"], retryable); + assert_eq!(value["error"]["request_id"], request_id); +} + +#[tokio::test] +async fn mst2_fixed_head_uses_verified_facts_without_body_reads_and_preserves_raw_bytes() { + let fixture = Fixture::new().await; + for (path, kind, size, digest) in [ + ( + "/file", + "regular", + fixture.raw.len(), + fixture.digest_string(), + ), + ( + "/alias", + "regular", + fixture.raw.len(), + fixture.digest_string(), + ), + ( + "/executable", + "executable", + fixture.raw.len(), + fixture.digest_string(), + ), + ( + "/empty", + "regular", + 0, + format!("sha256:{}", hex_of(&digest(&[]))), + ), + ( + "/link", + "symlink", + 4, + format!("sha256:{}", hex_of(&digest(b"file"))), + ), + ] { + let response = fixture + .send("HEAD", &format!("blob?path={path}"), Body::empty()) + .await; + assert_eq!(response.status(), 200); + assert_eq!(response.headers()["content-length"], size.to_string()); + assert_eq!(response.headers()["x-mega-content-size"], size.to_string()); + assert_eq!(response.headers()["x-mega-fs-kind"], kind); + assert_eq!(response.headers()["etag"], format!("\"{digest}\"")); + assert_eq!(response.headers()["vary"], "Authorization, Accept"); + assert_eq!( + response.headers()["cache-control"], + "private, no-cache, no-transform" + ); + assert!( + to_bytes(response.into_body(), 1024) + .await + .unwrap() + .is_empty() + ); + fixture.counts.assert(0, 0); + } + let response = fixture.send("GET", "blob?path=/file", Body::empty()).await; + assert_eq!(response.status(), 200); + let bytes = to_bytes(response.into_body(), 2 * 1024 * 1024) + .await + .unwrap(); + assert_eq!(bytes.as_ref(), fixture.raw); + fixture.counts.assert(1, fixture.raw.len()); +} + +#[tokio::test] +async fn mst2_fixed_warm_map_leaf_and_chunk_aliases_skip_body_reads() { + let fixture = Fixture::new().await; + let initial = fixture.map("/file").await; + fixture.counts.assert(1, fixture.raw.len()); + assert_eq!(initial["map"]["file_content_id"], fixture.digest_string()); + assert_eq!(initial["map"]["file_size"], fixture.raw.len().to_string()); + assert_eq!(initial["map"]["chunk_count"], "2"); + let map_id = initial["map"]["map_id"].as_str().unwrap(); + fixture.counts.reset(); + fixture + .state + .storage + .git_service + .obj_storage + .inner + .delete(&ObjectKey { + namespace: ObjectNamespace::Git, + key: fixture.oid.clone(), + }) + .await + .unwrap(); + for path in ["/file", "/alias", "/executable", "/nested/file"] { + let cached = fixture.map(path).await; + assert_eq!(cached["path"], path); + assert_eq!(cached["map"], initial["map"]); + let leaf_response = fixture + .send( + "GET", + &format!("chunk-map/pages?path={path}&map_id={map_id}&page_index=0"), + Body::empty(), + ) + .await; + let value = success_json(leaf_response).await; + let leaf = ChunkLeaf::decode( + &STANDARD + .decode(value["leaf_base64"].as_str().unwrap()) + .unwrap(), + ) + .unwrap(); + assert_eq!(leaf.page_index, 0); + let hashes: Vec<[u8; 32]> = fixture + .raw + .chunks(CHUNK_SIZE as usize) + .map(|bytes| Sha256::digest(bytes).into()) + .collect(); + assert_eq!(leaf.chunk_sha256, hashes); + assert_eq!(value["proof"], json!([])); + assert_eq!( + initial["map"]["pages_root"], + format!("sha256:{}", hex_of(&leaf.leaf_hash().unwrap())) + ); + verify_leaf( + 1, + 0, + leaf.leaf_hash().unwrap(), + &[], + leaf.leaf_hash().unwrap(), + ) + .unwrap(); + for index in 0..2 { + let body = fixture + .chunk_body(path, map_id, &index.to_string()) + .to_string(); + let response = fixture + .send("POST", "chunks", Body::from(body.clone())) + .await; + assert_eq!(response.status(), 200); + let wire = to_bytes(response.into_body(), 2 * 1024 * 1024) + .await + .unwrap(); + let frames = parse_stream(&wire).unwrap(); + let [Frame::Chunk(chunk), Frame::End(end)] = frames.as_slice() else { + panic!("expected one CHUNK and authenticated END"); + }; + let begin = index as usize * CHUNK_SIZE as usize; + let want = &fixture.raw[begin..fixture.raw.len().min(begin + CHUNK_SIZE as usize)]; + assert_eq!(chunk.chunk_index, index); + assert_eq!(chunk.file_content_id, fixture.digest); + assert_eq!(format!("sha256:{}", hex_of(&chunk.map_id)), map_id); + assert_eq!(chunk.chunk_bytes, want); + assert_eq!(end.request_item_count, 1); + assert_eq!(end.unique_unit_count, 1); + assert_eq!(end.logical_bytes, want.len() as u64); + assert_eq!( + end.request_body_sha256, + <[u8; 32]>::from(Sha256::digest(body.as_bytes())) + ); + } + fixture.counts.assert(0, 0); + } +} + +#[tokio::test] +async fn mst2_fixed_missing_and_legacy_facts_never_backfill_or_use_warm_cache() { + let fixture = Fixture::new().await; + fixture.map("/file").await; + let original = fixture.fact().await; + fixture.counts.reset(); + fixture.delete_fact().await; + error( + fixture + .send("GET", "chunk-map?path=/file", Body::empty()) + .await, + 503, + "METADATA_NOT_READY", + true, + ) + .await; + let response = fixture.send("HEAD", "blob?path=/file", Body::empty()).await; + assert_eq!(response.status(), 503); + assert!( + to_bytes(response.into_body(), 1024) + .await + .unwrap() + .is_empty() + ); + fixture.counts.assert(0, 0); + for (domain, kind, generation) in [ + ("git", "blob", 1), + ("other", "blob", MST2_VERIFICATION_VERSION), + ("git", "tree", MST2_VERIFICATION_VERSION), + ] { + let mut fact = original.clone(); + fact.storage_domain = domain.to_string(); + fact.object_kind = kind.to_string(); + fact.verification_version = generation; + fixture.replace_fact(fact).await; + error( + fixture + .send("GET", "chunk-map?path=/alias", Body::empty()) + .await, + 503, + "METADATA_NOT_READY", + true, + ) + .await; + fixture.counts.assert(0, 0); + } + fixture.replace_fact(original).await; + fixture.map("/file").await; + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn mst2_fixed_invalid_facts_and_cached_size_conflict_fail_before_body_read() { + let fixture = Fixture::new().await; + fixture.map("/file").await; + let original = fixture.fact().await; + fixture.counts.reset(); + for case in 0..6 { + let mut fact = original.clone(); + match case { + 0 => fact.state = "PENDING".to_string(), + 1 => fact.verification_version = MST2_VERIFICATION_VERSION + 1, + 2 => fact.size = -1, + 3 => fact.size = 8_796_093_022_209, + 4 => { + fact.raw_sha256.pop().unwrap(); + } + 5 => { + fact.size += 1; + } + _ => unreachable!(), + }; + fixture.replace_fact(fact).await; + error( + fixture + .send("GET", "chunk-map?path=/file", Body::empty()) + .await, + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + fixture.counts.assert(0, 0); + } + fixture.replace_fact(original).await; + fixture.map("/file").await; + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn mst2_fixed_verified_fact_db_error_is_not_missing_metadata_or_empty_content() { + let fixture = Fixture::new().await; + let mono = fixture.state.storage.mono_storage(); + mono.get_connection() + .execute_unprepared( + "ALTER TABLE mst2_verified_object RENAME TO mst2_verified_object_unavailable", + ) + .await + .unwrap(); + error( + fixture + .send("GET", "chunk-map?path=/file", Body::empty()) + .await, + 500, + "INTERNAL", + true, + ) + .await; + fixture.counts.assert(0, 0); + mono.get_connection() + .execute_unprepared( + "ALTER TABLE mst2_verified_object_unavailable RENAME TO mst2_verified_object", + ) + .await + .unwrap(); + fixture.map("/file").await; + fixture.counts.assert(1, fixture.raw.len()); +} + +#[tokio::test] +async fn mst2_fixed_symlink_facts_outside_profile_fail_without_body_reads() { + let fixture = Fixture::new().await; + let mono = fixture.state.storage.mono_storage(); + let main = mono.get_main_ref("/").await.unwrap().unwrap(); + let handler = MonoApiService::from(&fixture.state); + let root = handler.get_tree_by_hash(&main.ref_tree_hash).await.unwrap(); + let MetadataWalkOutcome::FoundFile { oid, .. } = + resolve_abs_metadata(&handler, &root, "/project/link") + .await + .unwrap() + else { + panic!("fixture link must be a fixed file"); + }; + let original = mono + .get_verified_blobs(vec![oid.clone()]) + .await + .unwrap() + .remove(&oid) + .unwrap(); + for size in [0, 4096] { + let mut fact = original.clone().into_active_model(); + fact.size = Set(size); + fact.update(mono.get_connection()).await.unwrap(); + error( + fixture + .send("GET", "chunk-map?path=/link", Body::empty()) + .await, + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + fixture.counts.assert(0, 0); + } + let mut restored = original.into_active_model(); + restored.size = Set(4); + restored.update(mono.get_connection()).await.unwrap(); + let response = fixture.send("HEAD", "blob?path=/link", Body::empty()).await; + assert_eq!(response.status(), 200); + assert_eq!(response.headers()["x-mega-content-size"], "4"); + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn mst2_fixed_digest_and_path_semantics_are_checked_before_projection_cache() { + let fixture = Fixture::new().await; + let wrong = format!("sha256:{}", "0".repeat(64)); + error( + fixture + .send( + "GET", + &format!("chunk-map?path=/file&expected_digest={wrong}"), + Body::empty(), + ) + .await, + 409, + "EXPECTED_DIGEST_MISMATCH", + false, + ) + .await; + fixture.counts.assert(0, 0); + fixture.map("/file").await; + fixture.counts.reset(); + for (path, status, code) in [ + ("/absent", 404, "PATH_NOT_FOUND"), + ("/outside", 404, "PATH_NOT_FOUND"), + ("/directory", 409, "NOT_DIRECTORY"), + ("/file/child", 409, "NOT_DIRECTORY"), + ("/link/child", 409, "SYMLINK_TRAVERSAL"), + ("/../outside", 400, "SCOPE_INVALID"), + ] { + error( + fixture + .send( + "GET", + &format!( + "chunk-map?path={path}&expected_digest={}", + fixture.digest_string() + ), + Body::empty(), + ) + .await, + status, + code, + false, + ) + .await; + fixture.counts.assert(0, 0); + } + error( + fixture + .send( + "GET", + &format!("chunk-map?path=/alias&expected_digest={wrong}"), + Body::empty(), + ) + .await, + 409, + "EXPECTED_DIGEST_MISMATCH", + false, + ) + .await; + let mut request = fixture.request("HEAD", "blob?path=/file", Body::empty()); + request + .headers_mut() + .insert("range", "bytes=0-9".parse().unwrap()); + assert_eq!( + fixture.app.clone().oneshot(request).await.unwrap().status(), + 400 + ); + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn mst2_fixed_cold_corrupt_or_missing_bodies_publish_no_projection_and_retry() { + for mode in 0..3 { + let fixture = Fixture::new().await; + match mode { + 0 => { + let mut corrupt = fixture.raw.clone(); + corrupt[0] ^= 1; + fixture.write_raw(corrupt).await; + } + 1 => { + fixture + .write_raw(fixture.raw[..fixture.raw.len() - 1].to_vec()) + .await + } + 2 => fixture + .state + .storage + .git_service + .obj_storage + .inner + .delete(&ObjectKey { + namespace: ObjectNamespace::Git, + key: fixture.oid.clone(), + }) + .await + .unwrap(), + _ => unreachable!(), + } + let (status, code, bytes) = match mode { + 0 => (502, "INTEGRITY_ERROR", fixture.raw.len()), + 1 => (502, "INTEGRITY_ERROR", fixture.raw.len() - 1), + 2 => (503, "OBJECT_UNAVAILABLE", 0), + _ => unreachable!(), + }; + error( + fixture + .send("GET", "chunk-map?path=/file", Body::empty()) + .await, + status, + code, + false, + ) + .await; + fixture.counts.assert(1, bytes); + fixture.write_raw(fixture.raw.clone()).await; + fixture.counts.reset(); + fixture.map("/file").await; + fixture.counts.assert(1, fixture.raw.len()); + fixture.counts.reset(); + fixture.map("/alias").await; + fixture.counts.assert(0, 0); + } +} + +#[tokio::test] +async fn mst2_fixed_warm_map_still_checks_map_indices_batches_and_http_lease() { + let fixture = Fixture::new().await; + let map = fixture.map("/file").await; + let map_id = map["map"]["map_id"].as_str().unwrap(); + let wrong_map = format!("sha256:{}", "1".repeat(64)); + fixture.counts.reset(); + error( + fixture + .send( + "GET", + &format!("chunk-map/pages?path=/file&map_id={wrong_map}&page_index=0"), + Body::empty(), + ) + .await, + 409, + "EXPECTED_DIGEST_MISMATCH", + false, + ) + .await; + error( + fixture + .send( + "GET", + &format!("chunk-map/pages?path=/file&map_id={map_id}&page_index=1"), + Body::empty(), + ) + .await, + 404, + "PATH_NOT_FOUND", + false, + ) + .await; + for (id, index, status, code) in [ + (wrong_map.as_str(), "0", 409, "EXPECTED_DIGEST_MISMATCH"), + (map_id, "2", 400, "SCOPE_INVALID"), + ] { + error( + fixture + .send( + "POST", + "chunks", + Body::from(fixture.chunk_body("/file", id, index).to_string()), + ) + .await, + status, + code, + false, + ) + .await; + } + let mut duplicate = fixture.chunk_body("/file", map_id, "0"); + let entry = duplicate["items"][0].clone(); + duplicate["items"].as_array_mut().unwrap().push(entry); + error( + fixture + .send("POST", "chunks", Body::from(duplicate.to_string())) + .await, + 400, + "SCOPE_INVALID", + false, + ) + .await; + let mut request = fixture.request("GET", "chunk-map?path=/file", Body::empty()); + request.headers_mut().remove("authorization"); + error( + fixture.app.clone().oneshot(request).await.unwrap(), + 401, + "UNAUTHENTICATED", + false, + ) + .await; + let mut request = fixture.request("GET", "chunk-map?path=/file", Body::empty()); + request.headers_mut().remove("x-mega-snapshot-lease"); + error( + fixture.app.clone().oneshot(request).await.unwrap(), + 401, + "UNAUTHENTICATED", + false, + ) + .await; + let response = fixture + .app + .clone() + .oneshot( + Request::builder() + .method("DELETE") + .uri(format!("/api/v2/snapshots/leases/{}", fixture.lease)) + .header("authorization", format!("Bearer {TOKEN}")) + .body(Body::empty()) + .unwrap(), + ) + .await + .unwrap(); + assert_eq!(response.status(), 200); + error( + fixture + .send("GET", "chunk-map?path=/file", Body::empty()) + .await, + 410, + "LEASE_EXPIRED", + false, + ) + .await; + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn mst2_fixed_old_snapshot_reads_its_original_oid_after_real_ref_advances() { + let fixture = Fixture::new().await; + let mono = fixture.state.storage.mono_storage(); + let main = mono.get_main_ref("/").await.unwrap().unwrap(); + let project = mono.get_main_ref("/project").await.unwrap().unwrap(); + let old_commit = + ObjectHash::from_hex_for_kind(HashKind::Sha1, &project.ref_commit_hash).unwrap(); + let blob_oid = ObjectHash::from_hex_for_kind(HashKind::Sha1, &fixture.oid).unwrap(); + let new_tree = tree(vec![item(TreeItemMode::Blob, blob_oid, "new-only")]); + let new_commit = Commit::from_tree_id_with_kind( + HashKind::Sha1, + new_tree.id, + vec![old_commit], + "advanced project without old paths", + ) + .unwrap(); + mono.save_mega_trees(vec![new_tree.clone()], new_commit.id, None) + .await + .unwrap(); + mono.save_mega_commits(vec![new_commit.clone()], None) + .await + .unwrap(); + publish_native_push(&fixture.state.storage, "/project", old_commit, &new_commit).await; + let head = mono + .read_native_publication_head( + fixture + .state + .storage + .config() + .mst2 + .instance_uuid + .as_deref() + .unwrap(), + ) + .await + .unwrap(); + assert_eq!(head.token.sequence, 2); + assert!(head.token.certificate.is_some()); + assert_ne!(head.root.commit, main.ref_commit_hash); + assert_ne!(head.root.tree, main.ref_tree_hash); + let published = mono.get_main_ref("/").await.unwrap().unwrap(); + assert_eq!(head.root.commit, published.ref_commit_hash); + assert_eq!(head.root.tree, published.ref_tree_hash); + let published_project = mono.get_main_ref("/project").await.unwrap().unwrap(); + assert_eq!(published_project.ref_commit_hash, new_commit.id.to_string()); + assert_eq!(published_project.ref_tree_hash, new_tree.id.to_string()); + let response = fixture.send("HEAD", "blob?path=/file", Body::empty()).await; + assert_eq!(response.status(), 200); + assert_eq!( + response.headers()["etag"], + format!("\"{}\"", fixture.digest_string()) + ); + fixture.counts.assert(0, 0); + let map = fixture.map("/file").await; + assert_eq!(map["map"]["file_content_id"], fixture.digest_string()); + fixture.counts.assert(1, fixture.raw.len()); +} + +#[tokio::test] +async fn mst2_fixed_metadata_walker_rejects_real_tree_gitlinks_without_body_reads() { + let fixture = Fixture::new().await; + let handler = MonoApiService::from(&fixture.state); + let commit_oid = ObjectHash::from_hex_for_kind( + HashKind::Sha1, + &fixture + .state + .storage + .mono_storage() + .get_main_ref("/") + .await + .unwrap() + .unwrap() + .ref_commit_hash, + ) + .unwrap(); + let root = tree(vec![item(TreeItemMode::Commit, commit_oid, "gitlink")]); + fixture + .state + .storage + .mono_storage() + .save_mega_trees(vec![root.clone()], commit_oid, None) + .await + .unwrap(); + let error = resolve_abs_metadata(&handler, &root, "/gitlink") + .await + .unwrap_err(); + assert_eq!(error.code, SnapshotErrorCode::UnsupportedEntry); + let file_oid = ObjectHash::from_hex_for_kind(HashKind::Sha1, &fixture.oid).unwrap(); + let file_root = tree(vec![item(TreeItemMode::Blob, file_oid, "file")]); + match resolve_abs_metadata(&handler, &file_root, "/file") + .await + .unwrap() + { + MetadataWalkOutcome::FoundFile { fs_kind, oid } => { + assert_eq!(fs_kind, FsKind::Regular); + assert_eq!(oid, fixture.oid); + } + other => panic!("unexpected fixed outcome: {other:?}"), + } + assert!(matches!( + resolve_abs_metadata(&handler, &file_root, "/missing/deep") + .await + .unwrap(), + MetadataWalkOutcome::Absent + )); + fixture.counts.assert(0, 0); +} diff --git a/src/api/router/snapshot_request.rs b/src/api/router/snapshot_request.rs new file mode 100644 index 00000000..ec43af70 --- /dev/null +++ b/src/api/router/snapshot_request.rs @@ -0,0 +1,114 @@ +//! Bounded raw JSON input for the MST/2 POST surface (spec 14). + +use std::{collections::HashSet, fmt, time::Duration}; + +use axum::{ + extract::{FromRequest, Request}, + http::StatusCode, +}; +use bytes::Bytes; +use serde::{ + Deserialize, Deserializer, + de::{self, MapAccess, SeqAccess, Visitor}, +}; + +use crate::ceres::snapshot::error::{SnapshotError, SnapshotErrorCode}; + +/// One overall read deadline, including a body which keeps trickling bytes. +/// This bounds input collection only, not handler work or response streams. +pub(super) const JSON_REQUEST_TIMEOUT: Duration = Duration::from_secs(10); + +/// Preserve the original bytes for TreeFrame request-body digests. The router +/// supplies DefaultBodyLimit; its rejection is converted to the MST envelope. +pub(super) struct Mst2Bytes(pub(super) Bytes); + +/// Decode keys before comparing them, including escaped spellings. This +/// separate pass also catches duplicate optional fields whose first value is +/// null, which a derived DTO can otherwise treat as an absent field. +pub(super) fn validate_json_keys(body: &[u8]) -> Result<(), serde_json::Error> { + serde_json::from_slice::(body).map(|_| ()) +} + +struct UniqueKeys; + +impl<'de> Deserialize<'de> for UniqueKeys { + fn deserialize>(deserializer: D) -> Result { + deserializer.deserialize_any(UniqueKeyVisitor) + } +} + +struct UniqueKeyVisitor; + +impl<'de> Visitor<'de> for UniqueKeyVisitor { + type Value = UniqueKeys; + + fn expecting(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result { + formatter.write_str("JSON without duplicate object keys") + } + + fn visit_bool(self, _: bool) -> Result { + Ok(UniqueKeys) + } + fn visit_i64(self, _: i64) -> Result { + Ok(UniqueKeys) + } + fn visit_u64(self, _: u64) -> Result { + Ok(UniqueKeys) + } + fn visit_f64(self, _: f64) -> Result { + Ok(UniqueKeys) + } + fn visit_str(self, _: &str) -> Result { + Ok(UniqueKeys) + } + fn visit_unit(self) -> Result { + Ok(UniqueKeys) + } + + fn visit_seq>(self, mut sequence: A) -> Result { + while sequence.next_element::()?.is_some() {} + Ok(UniqueKeys) + } + + fn visit_map>(self, mut object: A) -> Result { + let mut keys = HashSet::new(); + while let Some(key) = object.next_key::()? { + if !keys.insert(key) { + return Err(de::Error::custom("duplicate JSON object key")); + } + object.next_value::()?; + } + Ok(UniqueKeys) + } +} + +impl FromRequest for Mst2Bytes +where + S: Send + Sync, +{ + type Rejection = SnapshotError; + + async fn from_request(req: Request, state: &S) -> Result { + match tokio::time::timeout(JSON_REQUEST_TIMEOUT, Bytes::from_request(req, state)).await { + Ok(Ok(body)) => Ok(Self(body)), + Ok(Err(error)) => { + let (code, message) = if error.status() == StatusCode::PAYLOAD_TOO_LARGE { + ( + SnapshotErrorCode::LimitExceeded, + "request body over the spec 14 limit", + ) + } else { + ( + SnapshotErrorCode::InvalidRequest, + "could not read request body", + ) + }; + Err(SnapshotError::new(code, message)) + } + Err(_) => Err(SnapshotError::new( + SnapshotErrorCode::TemporaryUnavailable, + "request body read deadline exceeded", + )), + } + } +} diff --git a/src/api/router/snapshot_request_tests.rs b/src/api/router/snapshot_request_tests.rs new file mode 100644 index 00000000..fc577c85 --- /dev/null +++ b/src/api/router/snapshot_request_tests.rs @@ -0,0 +1,353 @@ +use std::{ + convert::Infallible, + pin::Pin, + sync::{ + Arc, + atomic::{AtomicBool, AtomicUsize, Ordering}, + }, + task::{Context, Poll}, + time::Duration, +}; + +use axum::{ + Router, + body::{Body, to_bytes}, + http::Request, + middleware, + response::Response, + routing::get, +}; +use bytes::Bytes; +use futures::{Stream, stream}; +use serde_json::Value; +use tower::ServiceExt; + +use super::{JSON_REQUEST_LIMIT, request::JSON_REQUEST_TIMEOUT, routers}; +use crate::{ + api::{MonoApiServiceState, oauth::api_store::BrowserSessionStore}, + ceres::{ + api_service::cache::GitObjectCache, + snapshot::{descriptor::build, runtime::runtime, view::SnapshotView}, + }, + config::testing::isolated_config, + jupiter::tests::{test_redis_manager, test_storage_with_config}, + server::trace_context::{TraceContext, inject_trace_context}, +}; + +struct Fixture { + app: Router, + snapshot_id: String, + lease_id: String, + expires: u64, + _temp: tempfile::TempDir, +} + +impl Fixture { + async fn new() -> Self { + let temp = tempfile::TempDir::new().unwrap(); + let mut config = isolated_config(temp.path().join("config")); + config.mst2.enabled = true; + config.mst2.instance_uuid = Some(uuid::Uuid::new_v4().to_string()); + config.mst2.auth_token = Some("mst2-input-test".to_string()); + let storage = test_storage_with_config(temp.path(), config).await; + let state = MonoApiServiceState { + entity_store: storage.entity_store.clone(), + storage, + session_store: BrowserSessionStore::Anonymous, + git_object_cache: Arc::new(GitObjectCache { + connection: test_redis_manager().await, + prefix: String::new(), + }), + listen_addr: "127.0.0.1:0".to_string(), + }; + let view = SnapshotView::from_commit(&"1".repeat(40), &"2".repeat(40)); + let built = build(&state.storage.config().mst2, &view, "/", [3; 32]).unwrap(); + let ctx = runtime() + .insert_context(built, &view.commit_oid, &view.root_tree_oid, 60) + .expect("fixture lease registration must succeed"); + // Reproduce the server's nest-after-layer order: MST must establish + // its own request context even when the earlier layer does not run. + let app = Router::new() + .route("/outside", get(|| async { "ok" })) + .layer(middleware::from_fn(inject_trace_context)) + .nest("/api/v2", routers(state.clone()).with_state(state)); + Self { + app, + snapshot_id: ctx.built.snapshot_id, + lease_id: ctx.lease_id, + expires: ctx.lease_expires_at_unix, + _temp: temp, + } + } + + fn request(&self, suffix: &str, body: Body) -> Request { + Request::builder() + .method("POST") + .uri(format!("/api/v2/snapshots/{suffix}")) + .header("authorization", "Bearer mst2-input-test") + .header("x-mega-snapshot-lease", &self.lease_id) + .header("content-type", "application/json") + .body(body) + .unwrap() + } + + fn renew_path(&self) -> String { + format!("leases/{}/renew", self.lease_id) + } + + fn assert_lease_unchanged(&self) { + let ctx = runtime().context(&self.snapshot_id).unwrap(); + assert_eq!(ctx.lease_id, self.lease_id); + assert_eq!(ctx.lease_expires_at_unix, self.expires); + } +} + +async fn assert_error(response: Response, status: u16, code: &str, retryable: bool) -> Value { + assert_eq!(response.status().as_u16(), status); + assert_eq!(response.headers()["content-type"], "application/json"); + let request_id = response.headers()["x-request-id"] + .to_str() + .unwrap() + .to_string(); + assert!(!request_id.is_empty()); + let body = to_bytes(response.into_body(), 16 * 1024).await.unwrap(); + let value: Value = serde_json::from_slice(&body).unwrap(); + assert_eq!(value["error"]["code"], code); + assert_eq!(value["error"]["request_id"], request_id); + assert_eq!(value["error"]["retryable"], retryable); + assert!(!value["error"]["message"].as_str().unwrap().is_empty()); + value +} + +fn chunked(bytes: Vec) -> Body { + let chunks: Vec> = bytes + .chunks(4096) + .map(|chunk| Ok(Bytes::copy_from_slice(chunk))) + .collect(); + Body::from_stream(stream::iter(chunks)) +} + +#[tokio::test] +async fn mst2_json_actual_chunked_limit_has_typed_errors_on_every_post() { + let fixture = Fixture::new().await; + let paths = [ + "resolve".to_string(), + fixture.renew_path(), + format!("{}/lookup", fixture.snapshot_id), + format!("{}/metadata/pages", fixture.snapshot_id), + format!("{}/objects", fixture.snapshot_id), + format!("{}/chunks", fixture.snapshot_id), + ]; + for path in paths { + for declared in [None, Some("2")] { + let mut request = fixture.request(&path, chunked(vec![b' '; JSON_REQUEST_LIMIT + 1])); + if let Some(length) = declared { + request + .headers_mut() + .insert("content-length", length.parse().unwrap()); + } + let response = fixture.app.clone().oneshot(request).await.unwrap(); + assert_error(response, 413, "LIMIT_EXCEEDED", false).await; + fixture.assert_lease_unchanged(); + } + } + + let mut body = br#"{"lease_seconds":120}"#.to_vec(); + body.resize(JSON_REQUEST_LIMIT, b' '); + let response = fixture + .app + .clone() + .oneshot(fixture.request(&fixture.renew_path(), chunked(body))) + .await + .unwrap(); + assert_eq!(response.status(), 200); + assert!(!response.headers()["x-request-id"].is_empty()); + let renewed = runtime().context(&fixture.snapshot_id).unwrap(); + assert!(renewed.lease_expires_at_unix > fixture.expires); +} + +#[tokio::test] +async fn mst2_json_declared_oversize_is_rejected_without_reading_body() { + let fixture = Fixture::new().await; + let polls = Arc::new(AtomicUsize::new(0)); + let observed = polls.clone(); + let body = Body::from_stream(stream::poll_fn(move |_| { + observed.fetch_add(1, Ordering::SeqCst); + Poll::>>::Pending + })); + let mut request = fixture.request(&fixture.renew_path(), body); + request.headers_mut().insert( + "content-length", + (JSON_REQUEST_LIMIT + 1).to_string().parse().unwrap(), + ); + let response = fixture.app.clone().oneshot(request).await.unwrap(); + assert_error(response, 413, "LIMIT_EXCEEDED", false).await; + assert_eq!(polls.load(Ordering::SeqCst), 0); + fixture.assert_lease_unchanged(); +} + +#[tokio::test] +async fn mst2_json_closed_dtos_duplicates_and_malformed_input_do_not_renew() { + let fixture = Fixture::new().await; + let renew = fixture.renew_path(); + let lookup = format!("{}/lookup", fixture.snapshot_id); + let metadata = format!("{}/metadata/pages", fixture.snapshot_id); + let objects = format!("{}/objects", fixture.snapshot_id); + let chunks = format!("{}/chunks", fixture.snapshot_id); + let cases: Vec<(&str, &[u8])> = vec![ + ("resolve", br#"{"target":{"kind":"latest"},"scope":"/","scope":"/"}"#), + ("resolve", br#"{"target":{"kind":"latest","kind":"latest"}}"#), + ("resolve", br#"{"target":{"kind":"latest","extra":true}}"#), + ("resolve", br#"{"target":{"kind":"latest"},"extra":true}"#), + (&renew, br#"{"lease_seconds":3600,"lease_seconds":120}"#), + (&renew, br#"{"lease_seconds":3600,"\u006cease_seconds":120}"#), + (&renew, br#"{"lease_seconds":null,"lease_seconds":120}"#), + (&renew, br#"{"lease_seconds":null,"\u006cease_seconds":120}"#), + (&renew, br#"{"lease_seconds":null,"lease_seconds":null}"#), + (&renew, br#"{"lease_seconds":3600,"extra":true}"#), + (&renew, br#"{"lease_seconds":"120"}"#), + (&renew, br#"{"lease_seconds":1.5}"#), + (&renew, br#"{"lease_seconds":-1}"#), + (&renew, br#"{"lease_seconds":NaN}"#), + (&renew, br#"{"lease_seconds":Infinity}"#), + (&renew, br#"{"lease_seconds":18446744073709551616}"#), + (&renew, br#"{"lease_seconds":"#), + (&renew, b"{\"lease_seconds\":\xff}"), + (&renew, b"{\"lease_seconds\":3600} {}"), + (&renew, b"{\"lease_seconds\":3600} x"), + (&lookup, br#"{"paths":[],"paths":[]}"#), + (&lookup, br#"{"paths":[],"extra":true}"#), + (&metadata, br#"{"items":[],"items":[]}"#), + (&metadata, br#"{"items":[],"extra":true}"#), + (&metadata, br#"{"items":[{"directory_path":"/","route":[],"route":[]}]}"#), + (&metadata, br#"{"items":[{"directory_path":"/","extra":true}]}"#), + (&objects, br#"{"items":[],"items":[]}"#), + (&objects, br#"{"items":[],"extra":true}"#), + (&objects, br#"{"items":[{"path":"/a","path":"/b","expected_digest":"x"}]}"#), + (&objects, br#"{"items":[{"path":"/a","expected_digest":"x","extra":true}]}"#), + (&chunks, br#"{"items":[],"items":[]}"#), + (&chunks, br#"{"items":[],"extra":true}"#), + (&chunks, br#"{"items":[{"path":"/a","expected_digest":"x","map_id":"x","chunk_index":"0","chunk_index":"1"}]}"#), + (&chunks, br#"{"items":[{"path":"/a","expected_digest":"x","map_id":"x","chunk_index":"0","extra":true}]}"#), + ]; + for (path, body) in cases { + let response = fixture + .app + .clone() + .oneshot(fixture.request(path, Body::from(body.to_vec()))) + .await + .unwrap(); + assert_error(response, 400, "INVALID_REQUEST", false).await; + fixture.assert_lease_unchanged(); + } + + for body in [Body::empty(), Body::from("{} \r\n\t")] { + let response = fixture + .app + .clone() + .oneshot(fixture.request(&renew, body)) + .await + .unwrap(); + assert_eq!(response.status(), 200); + assert!(!response.headers()["x-request-id"].is_empty()); + } +} + +struct SlowBody { + interval: Option, + chunks: Arc, + dropped: Arc, +} + +impl Stream for SlowBody { + type Item = Result; + + fn poll_next(self: Pin<&mut Self>, cx: &mut Context<'_>) -> Poll> { + let this = self.get_mut(); + match &mut this.interval { + Some(interval) => match interval.poll_tick(cx) { + Poll::Ready(_) => { + this.chunks.fetch_add(1, Ordering::SeqCst); + Poll::Ready(Some(Ok(Bytes::from_static(b" ")))) + } + Poll::Pending => Poll::Pending, + }, + None => Poll::Pending, + } + } +} + +impl Drop for SlowBody { + fn drop(&mut self) { + self.dropped.store(true, Ordering::SeqCst); + } +} + +#[tokio::test] +async fn mst2_json_real_post_pending_and_trickle_bodies_hit_overall_deadline() { + let fixture = Fixture::new().await; + let exercise = |trickle: bool| { + let app = fixture.app.clone(); + let dropped = Arc::new(AtomicBool::new(false)); + let chunks = Arc::new(AtomicUsize::new(0)); + let body = Body::from_stream(SlowBody { + interval: trickle.then(|| tokio::time::interval(Duration::from_millis(50))), + chunks: chunks.clone(), + dropped: dropped.clone(), + }); + let request = fixture.request(&fixture.renew_path(), body); + async move { + let response = tokio::time::timeout( + JSON_REQUEST_TIMEOUT + Duration::from_secs(5), + app.oneshot(request), + ) + .await + .expect("MST input deadline must terminate the request") + .unwrap(); + assert_error(response, 503, "TEMPORARY_UNAVAILABLE", true).await; + assert!(dropped.load(Ordering::SeqCst)); + if trickle { + assert!(chunks.load(Ordering::SeqCst) > 1); + } else { + assert_eq!(chunks.load(Ordering::SeqCst), 0); + } + } + }; + tokio::join!(exercise(false), exercise(true)); + fixture.assert_lease_unchanged(); + let response = fixture + .app + .clone() + .oneshot(fixture.request(&fixture.renew_path(), Body::from("{}"))) + .await + .unwrap(); + assert_eq!(response.status(), 200); +} + +#[tokio::test] +async fn mst2_json_body_read_failure_is_typed_and_preserves_request_context() { + let fixture = Fixture::new().await; + let body = Body::from_stream(stream::once(async { + Err::(std::io::Error::other("private backend key must not leak")) + })); + let mut request = fixture.request(&fixture.renew_path(), body); + request + .headers_mut() + .insert("x-request-id", "different-inbound-id".parse().unwrap()); + request.extensions_mut().insert(TraceContext { + trace_id: Arc::from("existing-trace-id"), + }); + let response = fixture.app.clone().oneshot(request).await.unwrap(); + let value = assert_error(response, 400, "INVALID_REQUEST", false).await; + assert_eq!(value["error"]["request_id"], "existing-trace-id"); + assert_eq!(value["error"]["message"], "could not read request body"); + fixture.assert_lease_unchanged(); + + let mut request = fixture.request(&fixture.renew_path(), Body::from("{")); + request + .headers_mut() + .insert("x-request-id", "accepted-inbound-id".parse().unwrap()); + let response = fixture.app.clone().oneshot(request).await.unwrap(); + let value = assert_error(response, 400, "INVALID_REQUEST", false).await; + assert_eq!(value["error"]["request_id"], "accepted-inbound-id"); +} diff --git a/src/api/router/snapshot_router.rs b/src/api/router/snapshot_router.rs index 556897e3..3cfa4963 100644 --- a/src/api/router/snapshot_router.rs +++ b/src/api/router/snapshot_router.rs @@ -15,16 +15,23 @@ use axum::{ }; use base64::Engine; use bytes::Bytes; +use git_internal::hash::{ObjectHash, get_hash_kind}; use mst2_codec::descriptor; +use request::Mst2Bytes; use serde::Deserialize; use serde_json::json; +use sha2::{Digest, Sha256}; use crate::{ api::MonoApiServiceState, ceres::snapshot::{ descriptor::build as build_descriptor, error::{SnapshotError, SnapshotErrorCode}, - pages::{WalkOutcome, base64_of, build_directory_page, hex_of, proof_pages, resolve_abs}, + pages::{ + WalkOutcome, base64_of, build_directory_page, build_directory_page_with_work, hex_of, + proof_pages, resolve_abs, + }, + projection_observation::{NativeResolveSource, ResolvedProjection}, runtime::{now_unix, runtime}, view::{SnapshotView, validate_scope_relative_path}, }, @@ -72,9 +79,43 @@ pub fn routers(api_state: MonoApiServiceState) -> Router { #[path = "snapshot_content.rs"] mod content; +#[path = "snapshot_request.rs"] +mod request; + +#[cfg(test)] +#[path = "snapshot_request_tests.rs"] +mod request_tests; + /// Spec 14 §4: JSON request bytes hard limit. pub(crate) const JSON_REQUEST_LIMIT: usize = 131_072; +/// The media type is part of the MST/2 TreeFrame wire contract (spec 06 +/// §1). Keep it in one place so every frame-producing endpoint has the same +/// response representation. +pub(crate) const TREEFRAME_MEDIA_TYPE: &str = "application/vnd.mega.treeframe;version=2"; + +/// Build a TreeFrame response with the protocol identity headers. The request +/// digest covers the exact bytes that were parsed, including JSON whitespace +/// and key ordering, so callers must pass the original body. +pub(crate) fn treeframe_response( + snapshot_id: &str, + request_body: &[u8], + body: Vec, +) -> Result { + let request_digest: [u8; 32] = Sha256::digest(request_body).into(); + Response::builder() + .header("content-type", TREEFRAME_MEDIA_TYPE) + .header("x-mega-snapshot-id", snapshot_id) + .header( + "x-mega-request-digest", + format!("sha256:{}", hex_of(&request_digest)), + ) + .header("cache-control", "private, no-cache, no-transform") + .header("vary", "Authorization, Accept") + .body(axum::body::Body::from(Bytes::from(body))) + .map_err(|e| internal(format!("TreeFrame response build failed: {e}"))) +} + tokio::task_local! { /// Per-request id for the error envelope (spec 14 §5). Sourced from the /// global `TraceContext` so the envelope, the `X-Request-Id` response @@ -82,6 +123,30 @@ tokio::task_local! { static REQUEST_ID: String; } +#[cfg(test)] +tokio::task_local! { + static NATIVE_RESOLVE_BARRIERS: (std::sync::Arc, std::sync::Arc); + static REJECT_NATIVE_OBSERVATION_SOURCE: bool; +} + +#[cfg(test)] +pub(crate) async fn with_rejected_native_observation_source( + future: F, +) -> F::Output { + REJECT_NATIVE_OBSERVATION_SOURCE.scope(true, future).await +} + +#[cfg(test)] +pub(crate) async fn with_native_resolve_barriers( + captured: std::sync::Arc, + release: std::sync::Arc, + future: F, +) -> F::Output { + NATIVE_RESOLVE_BARRIERS + .scope((captured, release), future) + .await +} + /// The request id of the in-flight request, for error envelopes. pub(crate) fn current_request_id() -> String { REQUEST_ID.try_with(|id| id.clone()).unwrap_or_default() @@ -97,28 +162,36 @@ const LEASE_HEADER: &str = "x-mega-snapshot-lease"; /// knowing the snapshot id alone is not a capability. async fn snapshot_auth_middleware( State(state): State, - req: axum::extract::Request, + mut req: axum::extract::Request, next: axum::middleware::Next, ) -> Response { - // Same id the global trace layer echoes on responses and logs. + // Nested routes may be added after the server's trace layer. Reuse an + // existing context or establish one here, including rejection responses. let id = req .extensions() .get::() - .map(|c| c.trace_id.to_string()) - .unwrap_or_default(); - REQUEST_ID - .scope(id, async { + .map(|c| c.trace_id.clone()) + .unwrap_or_else(|| crate::server::trace_context::resolve_trace_id(req.headers())); + req.extensions_mut() + .insert(crate::server::trace_context::TraceContext { + trace_id: id.clone(), + }); + let mut response = REQUEST_ID + .scope(id.to_string(), async { if let Some(res) = auth_error(&state, req.headers(), req.uri().path()) { return res; } next.run(req).await }) - .await + .await; + if let Ok(value) = HeaderValue::from_str(&id) { + response.headers_mut().insert("x-request-id", value); + } + response } -/// Spec 14 §4 enforcement with the MST/2 error envelope (DefaultBodyLimit's -/// own rejection is plain-text). Content-Length is checked here; a lying -/// chunked body still trips DefaultBodyLimit inside the extractor. +/// Reject an oversized declared length before consuming any body. Actual +/// bytes and the overall read deadline are checked by `Mst2Bytes`. async fn reject_oversize_body( req: axum::extract::Request, next: axum::middleware::Next, @@ -142,12 +215,14 @@ async fn reject_oversize_body( /// (spec 14 §5 INVALID_REQUEST). Size is enforced by the router layers. #[allow(clippy::result_large_err)] pub(crate) fn parse_json_body(body: &Bytes) -> Result { - serde_json::from_slice(body).map_err(|e| { - mst2_error_response(SnapshotError::new( - SnapshotErrorCode::InvalidRequest, - format!("malformed request body: {e}"), - )) - }) + request::validate_json_keys(body) + .and_then(|()| serde_json::from_slice(body)) + .map_err(|e| { + mst2_error_response(SnapshotError::new( + SnapshotErrorCode::InvalidRequest, + format!("malformed request body: {e}"), + )) + }) } /// Bearer-token check shared by the auth middleware; the parsing rule is the @@ -216,7 +291,10 @@ fn mst2_error_response(err: SnapshotError) -> Response { "request_id": current_request_id(), "retryable": matches!( err.code, - SnapshotErrorCode::SnapshotNotReady | SnapshotErrorCode::Internal + SnapshotErrorCode::SnapshotNotReady + | SnapshotErrorCode::MetadataNotReady + | SnapshotErrorCode::TemporaryUnavailable + | SnapshotErrorCode::Internal ), } })), @@ -230,6 +308,12 @@ impl From for Response { } } +impl IntoResponse for SnapshotError { + fn into_response(self) -> Response { + mst2_error_response(self) + } +} + /// MegaError → snapshot error: storage/lease failures must surface as errors, /// never as absence (spec 00 SYS-04). fn internal(e: E) -> SnapshotError { @@ -237,32 +321,45 @@ fn internal(e: E) -> SnapshotError { } async fn capabilities() -> Json { - // Honest capability set for this build (spec 04 §3): only what this slice - // serves is true; everything else stays false until accepted. + // Keep this document on the canonical discovery contract consumed by the + // v3 reader. The full delivery path serves the complete metadata/object + // closure, so advertising `full_hydration` as false would make a typed + // resolve reject before it reaches this router. Json(json!({ "protocol_versions": [2], "metadata_codecs": [1], - "frame_encodings": ["identity", "zstd"], + "frame_encodings": ["identity"], "features": { - "resolve": true, + "strict_publication": true, "directory": true, - "leases": true, "lookup": true, "metadata_pages": true, "raw_blob": true, - "objects": true, + "small_objects": true, "chunk_reads": true, - "full_hydration": false, + "full_hydration": true, + "region_hints": false, "offline_export": false, - "bindings": false, - "immutable_release": false, }, "limits": { + "max_file_bytes": "8796093022208", + "max_path_bytes": 4096, + "max_path_components": 256, "metadata_page_bytes": 16384, "metadata_leaf_entries": 128, - "max_directory_page_limit": 256, + "max_json_request_bytes": 131072, + "max_json_response_bytes": 1048576, + "max_directory_entries": 256, + "max_request_items": 128, + "max_metadata_items": 64, + "small_object_bytes": 262144, + "small_batch_bytes": 8388608, + "object_frame_raw_bytes": 1048576, + "chunk_frame_raw_bytes": 1048652, + "frame_wire_bytes": 2097152, + "zstd_window_bytes": 8388608, "chunk_size": 1048576, - "small_object_bytes": 262144 + "chunk_batch_bytes": 134217728 } })) } @@ -319,7 +416,10 @@ fn ensure_enabled(state: &MonoApiServiceState) -> Result<(), SnapshotError> { // `lfs_router::enforce_lfs_access` (where the error is the rare arm and is // boxed), there is nothing to gain here, so the lint is allowed outright. #[allow(clippy::result_large_err)] -async fn resolve(state: State, body: Bytes) -> Result { +async fn resolve( + state: State, + Mst2Bytes(body): Mst2Bytes, +) -> Result { ensure_enabled(&state).map_err(mst2_error_response)?; let req: ResolveRequest = parse_json_body(&body)?; // Unknown target kinds are client errors, never a silent fallback to @@ -349,21 +449,103 @@ async fn resolve(state: State, body: Bytes) -> Result Some(source), + Err(_) => { + if let Some(sink) = &state.storage.projection_observation_sink { + sink.reject_binding(); + } + tracing::warn!("native resolve observation source rejected"); + None + } + }; + ( + head.root.commit, + head.root.tree, + head.token.sequence.to_string(), + head.token.epoch.to_string(), + ) + } else { + let main = state + .storage + .mono_storage() + .get_main_ref("/") + .await + .map_err(internal)? + .ok_or_else(|| { + mst2_error_response(SnapshotError::new( + SnapshotErrorCode::SnapshotNotReady, + "monorepo main ref missing", + )) + })?; + let sequence = runtime() + .publication_sequence(&main.ref_commit_hash) + .to_string(); + ( + main.ref_commit_hash, + main.ref_tree_hash, + sequence, + "1".to_owned(), + ) + }; + + #[cfg(test)] + if let Ok((captured, release)) = NATIVE_RESOLVE_BARRIERS.try_with(|value| value.clone()) { + captured.wait().await; + release.wait().await; + } let view = SnapshotView::from_commit(&commit_oid, &tree_oid); if let Some(want) = &req.target.view_id { let kind_matches = req.target.kind == "view"; @@ -385,47 +567,54 @@ async fn resolve(state: State, body: Bytes) -> Result { + if let Some(sink) = &state.storage.projection_observation_sink { + let _ = sink.enqueue(&observation); + } + observation.emit(); + } + Err(_) => { + if let Some(sink) = &state.storage.projection_observation_sink { + sink.reject_binding(); + } + tracing::warn!("native resolve observation context rejected"); + } + } + } let body = json!({ "descriptor": descriptor_json(&built), - "publication_sequence": seq.to_string(), - "writer_epoch": "1", + "publication_sequence": sequence, + "writer_epoch": writer_epoch, "lease_id": ctx.lease_id, "lease_expires_at": crate::ceres::snapshot::runtime::rfc3339(ctx.lease_expires_at_unix), "authorization_epoch": "1", @@ -483,18 +672,13 @@ struct RenewRequest { async fn lease_renew( state: State, AxumPath(lease_id): AxumPath, - body: Bytes, + Mst2Bytes(body): Mst2Bytes, ) -> Result { ensure_enabled(&state).map_err(mst2_error_response)?; let req: RenewRequest = if body.is_empty() { RenewRequest::default() } else { - serde_json::from_slice(&body).map_err(|e| { - mst2_error_response(SnapshotError::new( - SnapshotErrorCode::ScopeInvalid, - format!("malformed renew body: {e}"), - )) - })? + parse_json_body(&body)? }; let renewed = runtime() .renew_lease(&lease_id, req.lease_seconds.unwrap_or(600)) @@ -854,7 +1038,7 @@ async fn lookup( state: State, AxumPath(snapshot_id): AxumPath, _headers: HeaderMap, - body: Bytes, + Mst2Bytes(body): Mst2Bytes, ) -> Result { ensure_enabled(&state).map_err(mst2_error_response)?; let req: LookupRequest = parse_json_body(&body)?; @@ -1011,7 +1195,7 @@ async fn metadata_pages( state: State, AxumPath(snapshot_id): AxumPath, _headers: HeaderMap, - body: Bytes, + Mst2Bytes(body): Mst2Bytes, ) -> Result { ensure_enabled(&state).map_err(mst2_error_response)?; let req: MetadataPagesRequest = parse_json_body(&body)?; @@ -1136,9 +1320,75 @@ async fn metadata_pages( ); out.extend_from_slice(&end); - Response::builder() - .header("content-type", "application/octet-stream") - .header("cache-control", "private, no-cache, no-transform") - .body(axum::body::Body::from(Bytes::from(out))) - .map_err(|e| mst2_error_response(internal(format!("body build failed: {e}")))) + treeframe_response(&snapshot_id, &body, out).map_err(mst2_error_response) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[tokio::test] + async fn capabilities_advertise_the_canonical_full_delivery_contract() { + let Json(value) = capabilities().await; + assert_eq!( + value, + json!({ + "protocol_versions": [2], + "metadata_codecs": [1], + "frame_encodings": ["identity"], + "features": { + "strict_publication": true, + "directory": true, + "lookup": true, + "metadata_pages": true, + "raw_blob": true, + "small_objects": true, + "chunk_reads": true, + "full_hydration": true, + "region_hints": false, + "offline_export": false, + }, + "limits": { + "max_file_bytes": "8796093022208", + "max_path_bytes": 4096, + "max_path_components": 256, + "metadata_page_bytes": 16384, + "metadata_leaf_entries": 128, + "max_json_request_bytes": 131072, + "max_json_response_bytes": 1048576, + "max_directory_entries": 256, + "max_request_items": 128, + "max_metadata_items": 64, + "small_object_bytes": 262144, + "small_batch_bytes": 8388608, + "object_frame_raw_bytes": 1048576, + "chunk_frame_raw_bytes": 1048652, + "frame_wire_bytes": 2097152, + "zstd_window_bytes": 8388608, + "chunk_size": 1048576, + "chunk_batch_bytes": 134217728 + } + }) + ); + } + + #[test] + fn treeframe_response_emits_protocol_identity_headers() { + let snapshot_id = "sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa"; + let request_body = br#"{"items":[], "encoding":"identity"}"#; + let response = treeframe_response(snapshot_id, request_body, vec![1, 2, 3]).unwrap(); + + assert_eq!(response.headers()["content-type"], TREEFRAME_MEDIA_TYPE); + assert_eq!(response.headers()["x-mega-snapshot-id"], snapshot_id); + let digest: [u8; 32] = Sha256::digest(request_body).into(); + assert_eq!( + response.headers()["x-mega-request-digest"], + format!("sha256:{}", hex_of(&digest)) + ); + assert_eq!( + response.headers()["cache-control"], + "private, no-cache, no-transform" + ); + assert_eq!(response.headers()["vary"], "Authorization, Accept"); + } } diff --git a/src/callisto/mod.rs b/src/callisto/mod.rs index dab5b7c5..ca18ec38 100644 --- a/src/callisto/mod.rs +++ b/src/callisto/mod.rs @@ -66,8 +66,18 @@ pub mod mega_view_root_chain_scan; pub mod mega_webhook; pub mod mega_webhook_delivery; pub mod mega_webhook_event_type; +pub mod mst2_metadata_payload; +pub mod mst2_metadata_prepare; +pub mod mst2_metadata_prepare_page; +pub mod mst2_native_head; +pub mod mst2_native_publication; pub mod mst2_publication; pub mod mst2_publication_outbox; +pub mod mst2_queue_noop_receipt; +pub mod mst2_retention_edge; +pub mod mst2_retention_gc_op; +pub mod mst2_retention_node; +pub mod mst2_retention_root; pub mod mst2_verified_object; pub mod notification_event_types; pub mod oci_blob_ref; diff --git a/src/callisto/mst2_metadata_payload.rs b/src/callisto/mst2_metadata_payload.rs new file mode 100644 index 00000000..798aab27 --- /dev/null +++ b/src/callisto/mst2_metadata_payload.rs @@ -0,0 +1,16 @@ +use sea_orm::entity::prelude::*; + +#[derive(Clone, Debug, PartialEq, DeriveEntityModel, Eq)] +#[sea_orm(table_name = "mst2_metadata_payload")] +pub struct Model { + #[sea_orm(primary_key, auto_increment = false)] + pub page_id: Vec, + pub metadata_codec: i16, + pub byte_size: i32, + pub payload: Vec, + pub created_at: DateTimeWithTimeZone, +} + +#[derive(Copy, Clone, Debug, EnumIter, DeriveRelation)] +pub enum Relation {} +impl ActiveModelBehavior for ActiveModel {} diff --git a/src/callisto/mst2_metadata_prepare.rs b/src/callisto/mst2_metadata_prepare.rs new file mode 100644 index 00000000..9401282b --- /dev/null +++ b/src/callisto/mst2_metadata_prepare.rs @@ -0,0 +1,32 @@ +use sea_orm::entity::prelude::*; + +#[derive(Clone, Debug, PartialEq, DeriveEntityModel, Eq)] +#[sea_orm(table_name = "mst2_metadata_prepare")] +pub struct Model { + #[sea_orm(primary_key, auto_increment = false)] + pub prepare_id: String, + pub operation_id: String, + pub manifest_digest: Vec, + pub canonical_plan: Vec, + pub source_domain: String, + pub tagged_root_tree_oid: String, + pub scope: String, + pub schema_version: i16, + pub metadata_codec: i16, + pub materialization_policy: i16, + pub fs_semantics: i16, + pub access_projection: i16, + pub verification_revision: i32, + pub projection_revision: i16, + pub metadata_root: Vec, + pub node_count: i32, + pub edge_count: i32, + pub total_bytes: i64, + pub state: String, + pub created_at: DateTimeWithTimeZone, + pub committed_at: Option, +} + +#[derive(Copy, Clone, Debug, EnumIter, DeriveRelation)] +pub enum Relation {} +impl ActiveModelBehavior for ActiveModel {} diff --git a/src/callisto/mst2_metadata_prepare_page.rs b/src/callisto/mst2_metadata_prepare_page.rs new file mode 100644 index 00000000..61299f7e --- /dev/null +++ b/src/callisto/mst2_metadata_prepare_page.rs @@ -0,0 +1,15 @@ +use sea_orm::entity::prelude::*; + +#[derive(Clone, Debug, PartialEq, DeriveEntityModel, Eq)] +#[sea_orm(table_name = "mst2_metadata_prepare_page")] +pub struct Model { + #[sea_orm(primary_key, auto_increment = false)] + pub prepare_id: String, + #[sea_orm(primary_key, auto_increment = false)] + pub page_id: Vec, + pub expected_size: i32, +} + +#[derive(Copy, Clone, Debug, EnumIter, DeriveRelation)] +pub enum Relation {} +impl ActiveModelBehavior for ActiveModel {} diff --git a/src/callisto/mst2_native_head.rs b/src/callisto/mst2_native_head.rs new file mode 100644 index 00000000..68a26e00 --- /dev/null +++ b/src/callisto/mst2_native_head.rs @@ -0,0 +1,20 @@ +use sea_orm::entity::prelude::*; + +#[derive(Clone, Debug, PartialEq, DeriveEntityModel, Eq)] +#[sea_orm(table_name = "mst2_native_head")] +pub struct Model { + #[sea_orm(primary_key, auto_increment = false)] + pub namespace: String, + pub instance_id: String, + pub sequence: i64, + pub writer_epoch: i64, + pub root_commit: String, + pub root_tree: String, + pub state: String, + pub certificate_receipt_id: Option, +} + +#[derive(Copy, Clone, Debug, EnumIter, DeriveRelation)] +pub enum Relation {} + +impl ActiveModelBehavior for ActiveModel {} diff --git a/src/callisto/mst2_native_publication.rs b/src/callisto/mst2_native_publication.rs new file mode 100644 index 00000000..449be1db --- /dev/null +++ b/src/callisto/mst2_native_publication.rs @@ -0,0 +1,27 @@ +use sea_orm::entity::prelude::*; + +#[derive(Clone, Debug, PartialEq, DeriveEntityModel, Eq)] +#[sea_orm(table_name = "mst2_native_publication")] +pub struct Model { + #[sea_orm(primary_key, auto_increment = false)] + pub receipt_id: i64, + pub namespace: String, + pub instance_id: String, + pub sequence: i64, + pub writer_epoch: i64, + pub old_root_commit: String, + pub old_root_tree: String, + pub root_commit: String, + pub root_tree: String, + pub origin_path: String, + pub origin_ref: String, + pub old_path_commit: Option, + pub old_path_tree: Option, + pub path_commit: String, + pub path_tree: String, +} + +#[derive(Copy, Clone, Debug, EnumIter, DeriveRelation)] +pub enum Relation {} + +impl ActiveModelBehavior for ActiveModel {} diff --git a/src/callisto/mst2_publication.rs b/src/callisto/mst2_publication.rs index ff69518e..603b6509 100644 --- a/src/callisto/mst2_publication.rs +++ b/src/callisto/mst2_publication.rs @@ -16,7 +16,7 @@ use serde::{Deserialize, Serialize}; pub struct Model { #[sea_orm(primary_key, auto_increment = true)] pub id: i64, - /// Deterministic writer-side operation id (e.g. push old→new). + /// Server-assigned writer identity (e.g. `mst2:trunk-queue:`). pub operation_id: String, /// Namespace path this publication advances (e.g. "/"). pub namespace: String, @@ -29,6 +29,12 @@ pub struct Model { pub writer_epoch: i64, /// `trunk_push` | `web_edit` | `import` | … (spec 09 §5 writer matrix). pub writer_kind: String, + /// Missing on legacy receipts; never reconstructed from current refs. + #[sea_orm(column_type = "Text", nullable)] + pub request_digest: Option, + pub request_digest_version: Option, + /// NULL preserves historical path-only receipts; version 1 requires its certificate. + pub native_certificate_version: Option, pub created_at: DateTimeWithTimeZone, } diff --git a/src/callisto/mst2_queue_noop_receipt.rs b/src/callisto/mst2_queue_noop_receipt.rs new file mode 100644 index 00000000..15c3b137 --- /dev/null +++ b/src/callisto/mst2_queue_noop_receipt.rs @@ -0,0 +1,34 @@ +//! Committed trunk-queue no-op results, separate from visible publications. + +use sea_orm::entity::prelude::*; +use serde::{Deserialize, Serialize}; + +#[derive(Clone, Debug, PartialEq, DeriveEntityModel, Eq, Serialize, Deserialize)] +#[sea_orm(table_name = "mst2_queue_noop_receipt")] +pub struct Model { + #[sea_orm(primary_key, auto_increment = true)] + pub id: i64, + #[sea_orm(column_type = "Text")] + pub operation_id: String, + #[sea_orm(column_type = "Text")] + pub namespace: String, + #[sea_orm(column_type = "Text")] + pub request_digest: String, + pub request_digest_version: i32, + pub writer_epoch: i64, + #[sea_orm(column_type = "Text")] + pub writer_kind: String, + pub observed_sequence: i64, + #[sea_orm(column_type = "Text")] + pub observed_root_commit: String, + #[sea_orm(column_type = "Text")] + pub observed_root_tree: String, + #[sea_orm(column_type = "Text")] + pub landed_commit_id: String, + pub created_at: DateTimeWithTimeZone, +} + +#[derive(Copy, Clone, Debug, EnumIter, DeriveRelation)] +pub enum Relation {} + +impl ActiveModelBehavior for ActiveModel {} diff --git a/src/callisto/mst2_retention_edge.rs b/src/callisto/mst2_retention_edge.rs new file mode 100644 index 00000000..2c861bdf --- /dev/null +++ b/src/callisto/mst2_retention_edge.rs @@ -0,0 +1,19 @@ +//! MST/2 retention graph edges (spec 10 §6). + +use sea_orm::entity::prelude::*; +use serde::{Deserialize, Serialize}; + +#[derive(Clone, Debug, PartialEq, DeriveEntityModel, Eq, Serialize, Deserialize)] +#[sea_orm(table_name = "mst2_retention_edge")] +pub struct Model { + #[sea_orm(primary_key, auto_increment = true)] + pub id: i64, + pub parent_id: String, + pub child_id: String, + pub created_at: Option, +} + +#[derive(Copy, Clone, Debug, EnumIter, DeriveRelation)] +pub enum Relation {} + +impl ActiveModelBehavior for ActiveModel {} diff --git a/src/callisto/mst2_retention_gc_op.rs b/src/callisto/mst2_retention_gc_op.rs new file mode 100644 index 00000000..833c949d --- /dev/null +++ b/src/callisto/mst2_retention_gc_op.rs @@ -0,0 +1,26 @@ +//! Durable MST/2 GC operation log (spec 10 §6 crash replay). + +use sea_orm::entity::prelude::*; +use serde::{Deserialize, Serialize}; + +#[derive(Clone, Debug, PartialEq, DeriveEntityModel, Eq, Serialize, Deserialize)] +#[sea_orm(table_name = "mst2_retention_gc_op")] +pub struct Model { + /// Caller supplied idempotency key. A retry of one physical GC step must + /// address the same row, so replay cannot decrement an edge twice. + #[sea_orm(primary_key, auto_increment = false)] + pub operation_id: String, + pub node_id: String, + /// `MARK_DELETING` or `REMOVE`. + pub operation: String, + /// `PENDING`, `APPLIED`, or `FAILED`; workers may replay `PENDING`. + pub state: String, + pub attempts: i32, + pub created_at: DateTimeWithTimeZone, + pub completed_at: Option, +} + +#[derive(Copy, Clone, Debug, EnumIter, DeriveRelation)] +pub enum Relation {} + +impl ActiveModelBehavior for ActiveModel {} diff --git a/src/callisto/mst2_retention_node.rs b/src/callisto/mst2_retention_node.rs new file mode 100644 index 00000000..40ce2999 --- /dev/null +++ b/src/callisto/mst2_retention_node.rs @@ -0,0 +1,25 @@ +//! MST/2 retention graph nodes (spec 10 §6). + +use sea_orm::entity::prelude::*; +use serde::{Deserialize, Serialize}; + +#[derive(Clone, Debug, PartialEq, DeriveEntityModel, Eq, Serialize, Deserialize)] +#[sea_orm(table_name = "mst2_retention_node")] +pub struct Model { + #[sea_orm(primary_key, auto_increment = false)] + pub node_id: String, + pub kind: String, + pub state: String, + #[sea_orm(column_type = "BigInteger")] + pub bytes: i64, + /// Number of unique incoming retention edges. Root coverage is kept in + /// `mst2_retention_root` and checked in the same transaction. + #[sea_orm(column_type = "BigInteger")] + pub incoming_refs: i64, + pub created_at: Option, +} + +#[derive(Copy, Clone, Debug, EnumIter, DeriveRelation)] +pub enum Relation {} + +impl ActiveModelBehavior for ActiveModel {} diff --git a/src/callisto/mst2_retention_root.rs b/src/callisto/mst2_retention_root.rs new file mode 100644 index 00000000..9b60cb5a --- /dev/null +++ b/src/callisto/mst2_retention_root.rs @@ -0,0 +1,20 @@ +//! MST/2 retention roots (spec 10 §5/§6). + +use sea_orm::entity::prelude::*; +use serde::{Deserialize, Serialize}; + +#[derive(Clone, Debug, PartialEq, DeriveEntityModel, Eq, Serialize, Deserialize)] +#[sea_orm(table_name = "mst2_retention_root")] +pub struct Model { + #[sea_orm(primary_key, auto_increment = true)] + pub id: i64, + pub node_id: String, + pub root_key: String, + pub root_kind: String, + pub created_at: Option, +} + +#[derive(Copy, Clone, Debug, EnumIter, DeriveRelation)] +pub enum Relation {} + +impl ActiveModelBehavior for ActiveModel {} diff --git a/src/callisto/prelude.rs b/src/callisto/prelude.rs index f85bd9cb..de2fd09c 100644 --- a/src/callisto/prelude.rs +++ b/src/callisto/prelude.rs @@ -25,6 +25,10 @@ pub use super::{ mega_view_root_chain_scan::Entity as MegaViewRootChainScan, mega_webhook::Entity as MegaWebhook, mega_webhook_delivery::Entity as MegaWebhookDelivery, mega_webhook_event_type::Entity as MegaWebhookEventType, + mst2_retention_edge::Entity as Mst2RetentionEdge, + mst2_retention_gc_op::Entity as Mst2RetentionGcOp, + mst2_retention_node::Entity as Mst2RetentionNode, + mst2_retention_root::Entity as Mst2RetentionRoot, notification_event_types::Entity as NotificationEventTypes, oci_blob_ref::Entity as OciBlobRef, oci_manifest::Entity as OciManifest, oci_tag::Entity as OciTag, oci_upload::Entity as OciUpload, path_check_configs::Entity as PathCheckConfigs, diff --git a/src/callisto/push_queue.rs b/src/callisto/push_queue.rs index 01638727..a8dea3d6 100644 --- a/src/callisto/push_queue.rs +++ b/src/callisto/push_queue.rs @@ -42,6 +42,9 @@ pub struct Model { pub expected_commit_hash: Option, #[sea_orm(column_type = "Text", nullable)] pub expected_tree_hash: Option, + pub expected_native_sequence: Option, + pub expected_native_epoch: Option, + pub expected_native_certificate: Option, pub pending_action: Option, pub enqueued_at: DateTimeWithTimeZone, pub started_at: Option, diff --git a/src/ceres/api_service/mod.rs b/src/ceres/api_service/mod.rs index 856087b2..e5e0f4be 100644 --- a/src/ceres/api_service/mod.rs +++ b/src/ceres/api_service/mod.rs @@ -60,6 +60,13 @@ mod un25_freeze; pub trait ApiHandler: Send + Sync { fn get_context(&self) -> Storage; + /// Only native monorepo trees use the native snapshot memoization domain. + /// Import and custom handlers remain uncached unless they implement their + /// own source identity and verification contract. + fn native_snapshot_projection(&self) -> bool { + false + } + fn object_cache(&self) -> &GitObjectCache; fn strip_relative(&self, path: &Path) -> Result; diff --git a/src/ceres/api_service/mono_api_service.rs b/src/ceres/api_service/mono_api_service.rs index f499b159..0a204c16 100644 --- a/src/ceres/api_service/mono_api_service.rs +++ b/src/ceres/api_service/mono_api_service.rs @@ -1177,6 +1177,10 @@ impl ApiHandler for MonoApiService { self.storage.clone() } + fn native_snapshot_projection(&self) -> bool { + true + } + fn object_cache(&self) -> &GitObjectCache { &self.git_object_cache } diff --git a/src/ceres/snapshot/error.rs b/src/ceres/snapshot/error.rs index 4fe54274..4c427126 100644 --- a/src/ceres/snapshot/error.rs +++ b/src/ceres/snapshot/error.rs @@ -18,6 +18,10 @@ pub enum SnapshotErrorCode { ScopeForbidden, ViewNotFound, SnapshotNotReady, + /// Verified fixed-object size/digest facts are not yet available. + MetadataNotReady, + /// A bounded operation could not complete in time (spec 14 §5). + TemporaryUnavailable, /// The fixed view no longer exists: spec 14 §5 SNAPSHOT_GONE (410). SnapshotGone, PathNotFound, @@ -49,6 +53,8 @@ impl SnapshotErrorCode { SnapshotErrorCode::ScopeForbidden => "SCOPE_FORBIDDEN", SnapshotErrorCode::ViewNotFound => "VIEW_NOT_FOUND", SnapshotErrorCode::SnapshotNotReady => "SNAPSHOT_NOT_READY", + SnapshotErrorCode::MetadataNotReady => "METADATA_NOT_READY", + SnapshotErrorCode::TemporaryUnavailable => "TEMPORARY_UNAVAILABLE", SnapshotErrorCode::SnapshotGone => "SNAPSHOT_GONE", SnapshotErrorCode::PathNotFound => "PATH_NOT_FOUND", SnapshotErrorCode::NotDirectory => "NOT_DIRECTORY", @@ -82,7 +88,9 @@ impl SnapshotErrorCode { SnapshotErrorCode::ViewNotFound | SnapshotErrorCode::PathNotFound | SnapshotErrorCode::LeaseUnknown => 404, - SnapshotErrorCode::SnapshotNotReady => 503, + SnapshotErrorCode::SnapshotNotReady + | SnapshotErrorCode::MetadataNotReady + | SnapshotErrorCode::TemporaryUnavailable => 503, SnapshotErrorCode::ObjectUnavailable => 503, SnapshotErrorCode::NotDirectory | SnapshotErrorCode::Conflict diff --git a/src/ceres/snapshot/metadata_install.rs b/src/ceres/snapshot/metadata_install.rs new file mode 100644 index 00000000..ab946290 --- /dev/null +++ b/src/ceres/snapshot/metadata_install.rs @@ -0,0 +1,463 @@ +//! Storage-owned identity for a durable native metadata installation. +//! This plan grants no publication, lease or content-retention authority. + +use std::collections::{BTreeMap, BTreeSet}; + +use mst2_codec::metapage::{HEADER_LEN, PAGE_MAX_BYTES}; +use sha2::{Digest, Sha256}; + +use super::{ + error::{SnapshotError, SnapshotErrorCode}, + retention_dag::{MetadataDagLimits, MetadataPageId, ValidatedMetadataDag}, +}; + +const DOMAIN: &[u8] = b"mega.mst2.metadata-install.v1\0"; +pub(crate) const MAX_PLAN_BYTES: usize = 2 * 1024 * 1024; + +#[derive(Debug, Clone, PartialEq, Eq)] +pub(crate) struct MetadataInstallIdentity { + pub source_domain: String, + pub tagged_root_tree_oid: String, + pub scope: String, + pub schema_version: u16, + pub metadata_codec: u16, + pub materialization_policy: u16, + pub fs_semantics: u16, + pub access_projection: u16, + pub verification_revision: i32, + pub projection_revision: u16, +} + +#[derive(Debug, Clone, PartialEq, Eq)] +pub(crate) struct MetadataInstallPlan { + pub identity: MetadataInstallIdentity, + pub root: MetadataPageId, + pub pages: BTreeMap, + pub edges: BTreeSet<(MetadataPageId, MetadataPageId)>, + pub total_bytes: u64, +} + +impl MetadataInstallPlan { + pub(super) fn from_validated( + identity: MetadataInstallIdentity, + dag: &ValidatedMetadataDag, + ) -> Result { + dag.check_limits(MetadataDagLimits::default())?; + let edges = dag + .edges() + .iter() + .map(|edge| Ok((parse_node_id(&edge.parent)?, parse_node_id(&edge.child)?))) + .collect::>()?; + let plan = Self { + identity, + root: dag.root(), + pages: dag + .payloads() + .iter() + .map(|page| (page.id, page.size)) + .collect(), + edges, + total_bytes: dag.payload_bytes(), + }; + plan.validate()?; + Ok(plan) + } + + pub fn digest(&self) -> Result<[u8; 32], SnapshotError> { + Ok(Sha256::digest(self.encode()?).into()) + } + + pub fn encode(&self) -> Result, SnapshotError> { + self.validate()?; + let mut bytes = DOMAIN.to_vec(); + bytes.extend_from_slice(&1u16.to_be_bytes()); + for value in [ + &self.identity.source_domain, + &self.identity.tagged_root_tree_oid, + &self.identity.scope, + ] { + bytes.extend_from_slice(&(value.len() as u32).to_be_bytes()); + bytes.extend_from_slice(value.as_bytes()); + } + for value in [ + self.identity.schema_version, + self.identity.metadata_codec, + self.identity.materialization_policy, + self.identity.fs_semantics, + self.identity.access_projection, + ] { + bytes.extend_from_slice(&value.to_be_bytes()); + } + bytes.extend_from_slice(&self.identity.verification_revision.to_be_bytes()); + bytes.extend_from_slice(&self.identity.projection_revision.to_be_bytes()); + bytes.extend_from_slice(&self.root); + bytes.extend_from_slice(&(self.pages.len() as u32).to_be_bytes()); + for (id, size) in &self.pages { + bytes.extend_from_slice(id); + bytes.extend_from_slice(&size.to_be_bytes()); + } + bytes.extend_from_slice(&(self.edges.len() as u32).to_be_bytes()); + for (parent, child) in &self.edges { + bytes.extend_from_slice(parent); + bytes.extend_from_slice(child); + } + if bytes.len() > MAX_PLAN_BYTES { + return Err(limit("metadata installation plan exceeds its byte budget")); + } + Ok(bytes) + } + + pub fn decode(bytes: &[u8], expected_digest: &[u8; 32]) -> Result { + if bytes.len() > MAX_PLAN_BYTES { + return Err(limit( + "stored metadata installation plan exceeds its byte budget", + )); + } + if Sha256::digest(bytes).as_slice() != expected_digest { + return Err(integrity( + "stored metadata installation plan digest mismatch", + )); + } + let mut reader = PlanReader(bytes); + if reader.take(DOMAIN.len())? != DOMAIN || reader.u16()? != 1 { + return Err(integrity("unsupported metadata installation plan encoding")); + } + let identity = MetadataInstallIdentity { + source_domain: reader.string(64)?, + tagged_root_tree_oid: reader.string(128)?, + scope: reader.string(4096)?, + schema_version: reader.u16()?, + metadata_codec: reader.u16()?, + materialization_policy: reader.u16()?, + fs_semantics: reader.u16()?, + access_projection: reader.u16()?, + verification_revision: i32::from_be_bytes(reader.array()?), + projection_revision: reader.u16()?, + }; + let root = reader.array()?; + let count = reader.count(MetadataDagLimits::default().nodes)?; + let mut pages = BTreeMap::new(); + let mut total_bytes = 0u64; + let mut last = None; + for _ in 0..count { + let id = reader.array()?; + let size = reader.u64()?; + if last.is_some_and(|previous| previous >= id) { + return Err(integrity( + "metadata installation pages are not uniquely ordered", + )); + } + last = Some(id); + total_bytes = total_bytes + .checked_add(size) + .ok_or_else(|| limit("metadata installation byte count overflow"))?; + pages.insert(id, size); + } + let count = reader.count(MetadataDagLimits::default().edges)?; + let mut edges = BTreeSet::new(); + let mut last = None; + for _ in 0..count { + let edge = (reader.array()?, reader.array()?); + if last.is_some_and(|previous| previous >= edge) { + return Err(integrity( + "metadata installation edges are not uniquely ordered", + )); + } + last = Some(edge); + edges.insert(edge); + } + if !reader.0.is_empty() { + return Err(integrity("metadata installation plan has trailing bytes")); + } + let plan = Self { + identity, + root, + pages, + edges, + total_bytes, + }; + plan.validate()?; + if plan.encode()?.as_slice() != bytes { + return Err(integrity("metadata installation plan is not canonical")); + } + Ok(plan) + } + + fn validate(&self) -> Result<(), SnapshotError> { + let identity = &self.identity; + if identity.source_domain != "native-git" + || identity.schema_version != mst2_codec::descriptor::SCHEMA_VERSION + || identity.metadata_codec != mst2_codec::descriptor::METADATA_CODEC + || identity.materialization_policy + != mst2_codec::descriptor::MATERIALIZATION_POLICY_GIT_RAW_V1 + || identity.fs_semantics != mst2_codec::descriptor::FS_SEMANTICS_LINUX_CODE_V1 + || identity.access_projection != mst2_codec::descriptor::ACCESS_PROJECTION_EXACT_FULL + || identity.verification_revision + != crate::jupiter::storage::mono_storage::MST2_VERIFICATION_VERSION + || identity.projection_revision != 1 + { + return Err(integrity( + "unsupported stored native metadata installation profile", + )); + } + let tagged = identity.tagged_root_tree_oid.as_str(); + let (kind, hex) = tagged + .split_once(':') + .ok_or_else(|| integrity("untagged fixed root tree"))?; + let length = match kind { + "sha1" => 40, + "sha256" | "blake3" => 64, + _ => return Err(integrity("unsupported fixed root tree hash kind")), + }; + if hex.len() != length + || !hex + .bytes() + .all(|byte| byte.is_ascii_digit() || (b'a'..=b'f').contains(&byte)) + { + return Err(integrity("noncanonical fixed root tree identity")); + } + super::view::validate_scope_relative_path(&identity.scope) + .map_err(|_| integrity("noncanonical metadata installation scope"))?; + let limits = MetadataDagLimits::default(); + if self.pages.is_empty() + || self.pages.len() > limits.nodes + || self.edges.len() > limits.edges + || self.total_bytes > limits.payload_bytes + { + return Err(limit("metadata installation group exceeds its budget")); + } + if !self.pages.contains_key(&self.root) + || self + .pages + .values() + .any(|size| *size < HEADER_LEN as u64 || *size > PAGE_MAX_BYTES as u64) + || self + .pages + .values() + .try_fold(0u64, |total, size| total.checked_add(*size)) + != Some(self.total_bytes) + { + return Err(integrity( + "metadata installation root or page size is invalid", + )); + } + let mut incoming: BTreeMap<_, usize> = self.pages.keys().map(|id| (*id, 0)).collect(); + let mut children: BTreeMap<_, Vec<_>> = BTreeMap::new(); + for (parent, child) in &self.edges { + if parent == child || !self.pages.contains_key(parent) { + return Err(integrity("invalid metadata installation edge")); + } + *incoming + .get_mut(child) + .ok_or_else(|| integrity("metadata installation child is missing"))? += 1; + children.entry(*parent).or_default().push(*child); + } + let mut ready: Vec<_> = incoming + .iter() + .filter_map(|(id, count)| (*count == 0).then_some(*id)) + .collect(); + let mut visited = 0; + while let Some(parent) = ready.pop() { + visited += 1; + for child in children.get(&parent).into_iter().flatten() { + let count = incoming + .get_mut(child) + .ok_or_else(|| integrity("missing installation child"))?; + *count -= 1; + if *count == 0 { + ready.push(*child); + } + } + } + if visited != self.pages.len() { + return Err(integrity("cyclic metadata installation plan")); + } + let mut reachable = BTreeSet::new(); + let mut pending = vec![self.root]; + while let Some(id) = pending.pop() { + if reachable.insert(id) { + pending.extend(children.get(&id).into_iter().flatten()); + } + } + if reachable.len() != self.pages.len() { + return Err(integrity("unreachable metadata installation pages")); + } + Ok(()) + } +} + +struct PlanReader<'a>(&'a [u8]); + +impl<'a> PlanReader<'a> { + fn take(&mut self, count: usize) -> Result<&'a [u8], SnapshotError> { + if count > self.0.len() { + return Err(integrity("truncated metadata installation plan")); + } + let (bytes, rest) = self.0.split_at(count); + self.0 = rest; + Ok(bytes) + } + fn array(&mut self) -> Result<[u8; N], SnapshotError> { + self.take(N)? + .try_into() + .map_err(|_| integrity("invalid installation field")) + } + fn u16(&mut self) -> Result { + Ok(u16::from_be_bytes(self.array()?)) + } + fn u32(&mut self) -> Result { + Ok(u32::from_be_bytes(self.array()?)) + } + fn u64(&mut self) -> Result { + Ok(u64::from_be_bytes(self.array()?)) + } + fn count(&mut self, maximum: usize) -> Result { + let count = self.u32()? as usize; + if count > maximum { + return Err(limit("metadata installation count exceeds its budget")); + } + Ok(count) + } + fn string(&mut self, maximum: usize) -> Result { + let count = self.count(maximum)?; + String::from_utf8(self.take(count)?.to_vec()) + .map_err(|_| integrity("non-UTF8 installation identity")) + } +} + +fn parse_node_id(value: &str) -> Result { + let value = value + .strip_prefix("page:sha256:") + .ok_or_else(|| integrity("non-page installation node"))?; + hex::decode(value) + .map_err(|_| integrity("invalid installation page ID"))? + .try_into() + .map_err(|_| integrity("invalid installation page ID length")) +} + +fn integrity(message: &str) -> SnapshotError { + SnapshotError::new(SnapshotErrorCode::IntegrityError, message) +} +fn limit(message: &str) -> SnapshotError { + SnapshotError::new(SnapshotErrorCode::LimitExceeded, message) +} + +#[cfg(test)] +mod tests { + use std::sync::Arc; + + use mst2_codec::metapage::{Page, page_id}; + + use super::*; + use crate::ceres::snapshot::{ + pages::PreparedNativeMetadataRetention, retention_dag::MetadataDagBuilder, + }; + + fn plan() -> MetadataInstallPlan { + let bytes = Page::build(&[]).unwrap(); + let mut builder = MetadataDagBuilder::new(MetadataDagLimits::default()); + builder.add_directory(&bytes, &[]).unwrap(); + PreparedNativeMetadataRetention::test_installation( + Arc::new(builder.finish(page_id(&bytes)).unwrap()), + "/", + ) + .install_plan() + .unwrap() + } + + #[test] + fn native_installation_plan_round_trip_binds_every_source_and_profile_field() { + let plan = plan(); + let bytes = plan.encode().unwrap(); + let digest = plan.digest().unwrap(); + assert_eq!(MetadataInstallPlan::decode(&bytes, &digest).unwrap(), plan); + let mut changed = plan.clone(); + changed.identity.tagged_root_tree_oid = format!("sha1:{}", "b".repeat(40)); + assert_ne!(changed.digest().unwrap(), digest); + changed = plan.clone(); + changed.identity.scope = "/other".into(); + assert_ne!(changed.digest().unwrap(), digest); + changed = plan; + changed.identity.projection_revision += 1; + assert_eq!( + changed.encode().unwrap_err().code, + SnapshotErrorCode::IntegrityError + ); + } + + #[test] + fn native_installation_stored_plan_rejects_bad_digest_truncation_trailing_bytes_and_profile() { + let plan = plan(); + let bytes = plan.encode().unwrap(); + let digest = plan.digest().unwrap(); + let mut bad_digest = digest; + bad_digest[0] ^= 1; + assert!(MetadataInstallPlan::decode(&bytes, &bad_digest).is_err()); + for length in [0, DOMAIN.len(), bytes.len() - 1] { + let truncated = &bytes[..length]; + let digest = Sha256::digest(truncated).into(); + assert!(MetadataInstallPlan::decode(truncated, &digest).is_err()); + } + let mut trailing = bytes.clone(); + trailing.push(0); + assert!(MetadataInstallPlan::decode(&trailing, &Sha256::digest(&trailing).into()).is_err()); + let mut wrong_domain = bytes; + wrong_domain[0] ^= 1; + assert!( + MetadataInstallPlan::decode(&wrong_domain, &Sha256::digest(&wrong_domain).into()) + .is_err() + ); + } + + #[test] + fn native_installation_plan_rejects_oversized_stored_bytes_before_allocating_fields() { + let bytes = vec![0; MAX_PLAN_BYTES + 1]; + assert_eq!( + MetadataInstallPlan::decode(&bytes, &[0; 32]) + .unwrap_err() + .code, + SnapshotErrorCode::LimitExceeded + ); + } + + #[test] + fn native_installation_posix_scopes_preserve_utf8_backslash_control_and_path_boundaries() { + let original = plan(); + let at_byte_limit = format!("/{}", vec!["a".repeat(255); 16].join("/")); + assert_eq!(at_byte_limit.len(), 4096); + let at_component_limit = format!("/{}", vec!["a"; 256].join("/")); + for scope in [ + "/a\\b".to_owned(), + "/目录/é".to_owned(), + "/line\ncontrol\t".to_owned(), + format!("/{}", "a".repeat(255)), + at_byte_limit.clone(), + at_component_limit, + ] { + let mut value = original.clone(); + value.identity.scope = scope.clone(); + let encoded = value.encode().unwrap(); + assert_eq!( + MetadataInstallPlan::decode(&encoded, &value.digest().unwrap()) + .unwrap() + .identity + .scope, + scope + ); + } + for scope in [ + format!("/{}", "a".repeat(256)), + format!("/{}", vec!["a"; 257].join("/")), + format!("{at_byte_limit}/a"), + "/a\0b".to_owned(), + "/a/..".to_owned(), + ] { + let mut value = original.clone(); + value.identity.scope = scope; + assert_eq!( + value.encode().unwrap_err().code, + SnapshotErrorCode::IntegrityError + ); + } + } +} diff --git a/src/ceres/snapshot/mod.rs b/src/ceres/snapshot/mod.rs index a7c44a61..97a1a50b 100644 --- a/src/ceres/snapshot/mod.rs +++ b/src/ceres/snapshot/mod.rs @@ -7,10 +7,14 @@ pub mod chunks; pub mod descriptor; pub mod error; pub mod frame_stream; +pub(crate) mod metadata_install; pub mod namespace; pub mod pages; +pub(crate) mod projection_observation; +pub(crate) mod projection_writer; pub mod publication; pub mod resolver; pub mod retention; +pub mod retention_dag; pub mod runtime; pub mod view; diff --git a/src/ceres/snapshot/native_projection_tests.rs b/src/ceres/snapshot/native_projection_tests.rs new file mode 100644 index 00000000..0cdddbd5 --- /dev/null +++ b/src/ceres/snapshot/native_projection_tests.rs @@ -0,0 +1,646 @@ +//! Real persisted commits and object storage, with an independent raw-content +//! oracle. This exercises on-demand memoization, not durable publication. + +use std::{collections::BTreeMap, sync::Arc}; + +use git_internal::{ + hash::{HashKind, ObjectHash}, + internal::object::{ + blob::Blob, + commit::Commit, + tree::{Tree, TreeItem, TreeItemMode}, + types::ObjectType, + }, +}; +use sha2::{Digest, Sha256}; + +use super::{FsKind, ProjectionWork, build_directory_page, build_directory_page_with_work}; +use crate::{ + ceres::api_service::{ApiHandler, cache::GitObjectCache, mono_api_service::MonoApiService}, + config::testing::isolated_config, + jupiter::{ + service::git_service::GitService, + storage::object_storage::build_object_storage, + tests::{test_redis_manager, test_storage_with_config}, + }, +}; + +type Files = BTreeMap>; + +#[tokio::test] +async fn mst2_native_retention_prepares_full_shared_radix_closure_once() { + use crate::ceres::snapshot::retention_dag::MetadataDagLimits; + + let temp = tempfile::tempdir().unwrap(); + let handler = handler(temp.path()).await; + let mut files = Files::new(); + for prefix in ["/project/alpha", "/project/beta"] { + for index in 0..129 { + files.insert( + format!("{prefix}/f{index:03}"), + format!("raw-{index}").into_bytes(), + ); + } + files.insert(format!("{prefix}/nested/file"), b"shared nested".to_vec()); + } + let (_, root) = persist_commit(&handler, &files, vec![], "retention closure").await; + let scope = build_directory_page(&handler, &root, "/project") + .await + .unwrap(); + let prepared = super::prepare_native_metadata_retention( + &handler, + &root, + "/project", + MetadataDagLimits::default(), + ) + .await + .unwrap(); + assert_eq!(prepared.fixed_root_tree_oid(), root.id.to_tagged_string()); + assert_eq!(prepared.scope(), "/project"); + assert_eq!(prepared.metadata_codec(), 1); + assert_eq!(prepared.schema_version(), 2); + assert_eq!(prepared.dag().root(), scope.page_id); + assert!(prepared.dag().payloads().iter().any(|payload| { + matches!( + mst2_codec::metapage::Page::decode(&payload.bytes) + .unwrap() + .0, + mst2_codec::metapage::Page::Branch { .. } + ) + })); + assert!(prepared.dag().payloads().len() > 3); + let root_node = format!("page:sha256:{}", super::hex(&scope.page_id)); + assert_eq!( + prepared + .dag() + .edges() + .iter() + .filter(|edge| edge.parent == root_node) + .count(), + 1 + ); + + // Remove all directory memo entries: the prepared hit must return its Arc + // without rebuilding or walking even this fixed view's child Git trees. + { + let mut state = handler + .storage + .native_projection_cache + .state + .lock() + .unwrap(); + state.pages.clear(); + state.retained_payload_bytes = 0; + } + let again = super::prepare_native_metadata_retention( + &handler, + &root, + "/project", + MetadataDagLimits::default(), + ) + .await + .unwrap(); + assert!(Arc::ptr_eq(prepared.dag(), again.dag())); + { + let state = handler + .storage + .native_projection_cache + .state + .lock() + .unwrap(); + assert!( + state.pages.is_empty(), + "prepare hit must not rebuild directory projections" + ); + } + let tightened = super::prepare_native_metadata_retention( + &handler, + &root, + "/project", + MetadataDagLimits { + nodes: 1, + ..MetadataDagLimits::default() + }, + ) + .await + .unwrap_err(); + assert_eq!( + tightened.code, + crate::ceres::snapshot::error::SnapshotErrorCode::LimitExceeded + ); +} + +#[tokio::test] +async fn mst2_native_retention_checks_source_budgets_before_unknown_child_fetch() { + use crate::ceres::snapshot::{error::SnapshotErrorCode, retention_dag::MetadataDagLimits}; + + let temp = tempfile::tempdir().unwrap(); + let handler = handler(temp.path()).await; + let unknown = ObjectHash::from_hex_for_kind(HashKind::Sha1, &"a".repeat(40)).unwrap(); + let root = Tree::from_tree_items_with_kind( + HashKind::Sha1, + vec![TreeItem::new( + TreeItemMode::Tree, + unknown, + "unknown-child".into(), + )], + ) + .unwrap(); + for limits in [ + MetadataDagLimits { + nodes: 1, + ..MetadataDagLimits::default() + }, + MetadataDagLimits { + edges: 0, + ..MetadataDagLimits::default() + }, + MetadataDagLimits { + entries: 0, + ..MetadataDagLimits::default() + }, + MetadataDagLimits { + payload_bytes: 19, + ..MetadataDagLimits::default() + }, + ] { + let error = super::prepare_native_metadata_retention(&handler, &root, "/", limits) + .await + .unwrap_err(); + assert_eq!( + error.code, + SnapshotErrorCode::LimitExceeded, + "budget must reject before querying the nonexistent tree: {error}" + ); + } + let state = handler + .storage + .native_projection_cache + .state + .lock() + .unwrap(); + assert!(state.retention_dags.is_empty()); +} + +async fn handler(temp: &std::path::Path) -> MonoApiService { + let config = isolated_config(temp.join("config")); + let object_storage = build_object_storage(&config.object_storage).await.unwrap(); + let mut storage = test_storage_with_config(temp, config).await; + storage.git_service = GitService { + obj_storage: object_storage, + }; + MonoApiService { + storage, + git_object_cache: Arc::new(GitObjectCache { + connection: test_redis_manager().await, + prefix: String::new(), + }), + } +} + +#[derive(Default)] +struct Directory { + children: BTreeMap, + files: BTreeMap, +} + +fn build_trees(directory: Directory, trees: &mut Vec) -> Tree { + let mut items: BTreeMap = directory + .files + .into_iter() + .map(|(name, oid)| { + let item = TreeItem::new(TreeItemMode::Blob, oid, name.clone()); + (name, item) + }) + .collect(); + for (name, child) in directory.children { + let child = build_trees(child, trees); + items.insert( + name.clone(), + TreeItem::new(TreeItemMode::Tree, child.id, name), + ); + } + let tree = + Tree::from_tree_items_with_kind(HashKind::Sha1, items.into_values().collect()).unwrap(); + trees.push(tree.clone()); + tree +} + +async fn persist_commit( + handler: &MonoApiService, + files: &Files, + parents: Vec, + message: &str, +) -> (Commit, Tree) { + let mut directory = Directory::default(); + for (path, raw) in files { + let blob = Blob::from_content_bytes_with_kind(HashKind::Sha1, raw.clone()).unwrap(); + handler + .storage + .git_service + .save_object_from_model(raw.clone(), &blob.id.to_string()) + .await + .unwrap(); + let components: Vec<_> = path.trim_start_matches('/').split('/').collect(); + let mut parent = &mut directory; + for component in &components[..components.len() - 1] { + parent = parent.children.entry((*component).to_string()).or_default(); + } + parent + .files + .insert(components.last().unwrap().to_string(), blob.id); + } + let mut trees = Vec::new(); + let root = build_trees(directory, &mut trees); + persist_tree_commit(handler, root, trees, parents, message).await +} + +async fn persist_tree_commit( + handler: &MonoApiService, + root: Tree, + trees: Vec, + parents: Vec, + message: &str, +) -> (Commit, Tree) { + let commit = Commit::from_tree_id_with_kind(HashKind::Sha1, root.id, parents, message).unwrap(); + let mono = handler.storage.mono_storage(); + mono.save_mega_trees(trees, commit.id, None).await.unwrap(); + mono.save_mega_commits(vec![commit.clone()], None) + .await + .unwrap(); + // Read back the real persisted commit and its fixed tree; no fake blob OID + // is used as a commit identity and no live ref is required by the builder. + let persisted = handler + .get_commit_by_hash(&commit.id.to_string()) + .await + .unwrap(); + assert_eq!(persisted.tree_id, root.id); + let root = handler + .get_tree_by_hash(&persisted.tree_id.to_string()) + .await + .unwrap(); + (persisted, root) +} + +fn fixture(modules: usize, buckets: usize, files_per_bucket: usize, bytes: usize) -> Files { + let mut files = Files::new(); + for module in 0..modules { + for bucket in 0..buckets { + for file in 0..files_per_bucket { + let path = format!("/project/m{module:03}/d{bucket:02}/f{file:03}"); + let mut raw = vec![b'x'; bytes.max(path.len())]; + raw[..path.len()].copy_from_slice(path.as_bytes()); + files.insert(path, raw); + } + } + } + files +} + +async fn assert_raw_oracle(handler: &MonoApiService, root: &Tree, scope: &str, files: &Files) { + let mut paths = vec![scope.to_string()]; + let mut projected = BTreeMap::new(); + while let Some(path) = paths.pop() { + let directory = build_directory_page(handler, root, &path).await.unwrap(); + for entry in &directory.entries { + let child = format!("{}/{name}", path.trim_end_matches('/'), name = entry.name); + if entry.fs_kind == FsKind::Directory { + paths.push(child); + } else { + assert_eq!(entry.fs_kind, FsKind::Regular); + projected.insert(child, (entry.size.unwrap(), entry.content_digest.unwrap())); + } + } + } + let expected: BTreeMap<_, _> = files + .iter() + .map(|(path, raw)| { + ( + path.clone(), + (raw.len() as u64, <[u8; 32]>::from(Sha256::digest(raw))), + ) + }) + .collect(); + assert_eq!(projected, expected); +} + +async fn leaf_update_case( + modules: usize, + buckets: usize, + files_per_bucket: usize, + bytes: usize, +) -> ProjectionWork { + let temp = tempfile::tempdir().unwrap(); + let handler = handler(temp.path()).await; + let mut files = fixture(modules, buckets, files_per_bucket, bytes); + let original = files.clone(); + let (v1, root1) = persist_commit(&handler, &files, vec![], "native projection V1").await; + let (page1, cold) = build_directory_page_with_work(&handler, &root1, "/project") + .await + .unwrap(); + assert_eq!( + cold.directories_rebuilt, + (1 + modules + modules * buckets) as u64 + ); + assert_eq!(cold.reused_subtree_roots, 0); + assert_eq!(cold.directory_root_pages_built, cold.directories_rebuilt); + assert_eq!(cold.verified_blob_misses, files.len() as u64); + + files.get_mut("/project/m000/d00/f000").unwrap().push(b'2'); + let (v2, root2) = persist_commit(&handler, &files, vec![v1.id], "native projection V2").await; + let (page2, update) = build_directory_page_with_work(&handler, &root2, "/project") + .await + .unwrap(); + assert_ne!(page1.page_id, page2.page_id); + assert_eq!(update.directories_rebuilt, 3); + assert_eq!(update.directory_root_pages_built, 3); + assert_eq!( + update.tree_fetches, 3, + "scope + two changed child trees only" + ); + assert_eq!( + update.reused_subtree_roots, + (modules - 1 + buckets - 1) as u64 + ); + assert_eq!( + update.directory_entries_scanned, + (modules + buckets + files_per_bucket) as u64 + ); + assert_eq!(update.verified_blob_hits, (files_per_bucket - 1) as u64); + assert_eq!(update.verified_blob_misses, 1); + assert_eq!( + update.raw_bytes_fetched, + files["/project/m000/d00/f000"].len() as u64 + ); + assert_eq!(update.raw_bytes_hashed, update.raw_bytes_fetched); + let scope_tree = + super::fetch_tree_with_work(&handler, &root2, "/project", &mut ProjectionWork::default()) + .await + .unwrap(); + let mut full_work = ProjectionWork::default(); + let full = super::build_subtree( + &handler, + &handler.storage, + &scope_tree, + "/project", + false, + &mut full_work, + ) + .await + .unwrap(); + assert_eq!( + page2.page_id, full.page_id, + "cache reuse equals a fresh full projection" + ); + assert_eq!(full_work.directories_rebuilt, cold.directories_rebuilt); + assert_eq!(full_work.reused_subtree_roots, 0); + assert_raw_oracle(&handler, &root2, "/project", &files).await; + + // A third real commit repeats the operation; this cannot be a special + // first-update shortcut. Reading older fixed roots remains correct. + let second_files = files.clone(); + files.get_mut("/project/m000/d00/f000").unwrap().push(b'3'); + let (_, root3) = persist_commit(&handler, &files, vec![v2.id], "native projection V3").await; + let (_, third) = build_directory_page_with_work(&handler, &root3, "/project") + .await + .unwrap(); + assert_eq!(third.directories_rebuilt, update.directories_rebuilt); + assert_eq!( + third.directory_entries_scanned, + update.directory_entries_scanned + ); + assert_eq!(third.reused_subtree_roots, update.reused_subtree_roots); + assert_eq!(third.verified_blob_misses, 1); + assert_raw_oracle(&handler, &root3, "/project", &files).await; + assert_raw_oracle(&handler, &root2, "/project", &second_files).await; + + let (old, old_work) = build_directory_page_with_work(&handler, &root1, "/project") + .await + .unwrap(); + assert_eq!(old.page_id, page1.page_id); + assert_eq!(old_work.directories_rebuilt, 0); + assert_eq!(old_work.directory_entries_scanned, 0); + assert_eq!(old_work.reused_subtree_roots, 1); + assert_raw_oracle(&handler, &root1, "/project", &original).await; + let old_file = super::resolve_abs(&handler, &root1, "/project/m000/d00/f000") + .await + .unwrap(); + assert!( + matches!(old_file, super::WalkOutcome::FoundFile { raw, .. } if raw == original["/project/m000/d00/f000"]) + ); + update +} + +#[tokio::test] +async fn mst2_native_projection_real_commit_leaf_update_only_rebuilds_ancestors() { + leaf_update_case(4, 3, 4, 64).await; +} + +#[tokio::test] +#[ignore = "opt-in 16,384-file/256MiB real object-storage projection workload"] +async fn mst2_native_projection_medium_real_commit_leaf_update() { + let update = leaf_update_case(64, 8, 32, 16 * 1024).await; + assert_eq!(update.directory_entries_scanned, 104); + assert_eq!(update.reused_subtree_roots, 70); + eprintln!("native projection memoization work: {update:?}"); +} + +#[tokio::test] +async fn mst2_native_projection_same_oid_isolated_storage_does_not_share_cache() { + let temp_a = tempfile::tempdir().unwrap(); + let temp_b = tempfile::tempdir().unwrap(); + let a = handler(temp_a.path()).await; + let b = handler(temp_b.path()).await; + let files = fixture(2, 2, 2, 64); + let (_, root) = persist_commit(&a, &files, vec![], "independent storage A").await; + let (cached, _) = build_directory_page_with_work(&a, &root, "/") + .await + .unwrap(); + assert!( + build_directory_page_with_work(&b, &root, "/") + .await + .is_err(), + "B must fetch its own missing child tree" + ); + let (_, own_root) = persist_commit(&b, &files, vec![], "independent storage B").await; + assert_eq!(root.id, own_root.id); + let (rebuilt, work) = build_directory_page_with_work(&b, &own_root, "/") + .await + .unwrap(); + assert_eq!(cached.page_id, rebuilt.page_id); + assert_eq!(work.directories_rebuilt, 8); + assert_eq!(work.reused_subtree_roots, 0); + let clone = a.clone(); + let (_, clone_work) = build_directory_page_with_work(&clone, &root, "/") + .await + .unwrap(); + assert_eq!(clone_work.directories_rebuilt, 0); + assert_eq!(clone_work.tree_fetches, 0); +} + +#[tokio::test] +async fn mst2_native_projection_real_commit_directory_move_reuses_identical_tree() { + let temp = tempfile::tempdir().unwrap(); + let handler = handler(temp.path()).await; + let files = Files::from([("/project/old/nested/file".into(), b"move me".to_vec())]); + let (v1, root1) = persist_commit(&handler, &files, vec![], "before move").await; + let (before, _) = build_directory_page_with_work(&handler, &root1, "/project") + .await + .unwrap(); + let moved = Files::from([("/project/new/nested/file".into(), b"move me".to_vec())]); + let (_, root2) = persist_commit(&handler, &moved, vec![v1.id], "after move").await; + let (after, work) = build_directory_page_with_work(&handler, &root2, "/project") + .await + .unwrap(); + assert_eq!( + before.entries[0].directory_root, + after.entries[0].directory_root + ); + assert_eq!(work.directories_rebuilt, 1); + assert_eq!(work.reused_subtree_roots, 1); + assert_eq!( + work.tree_fetches, 1, + "only scope; moved root is reused by OID" + ); + assert_eq!(work.verified_blob_hits + work.verified_blob_misses, 0); + assert_raw_oracle(&handler, &root2, "/project", &moved).await; +} + +fn byte_prefix(bytes: usize) -> String { + let mut path = "/project".to_string(); + while path.len() < bytes { + let length = (bytes - path.len() - 1).min(255); + assert_ne!(length, 0); + path.push('/'); + path.extend(std::iter::repeat_n('a', length)); + } + path +} + +fn wrap_tree_at(prefix: &str, mut child: Tree, trees: &mut Vec) -> Tree { + for name in prefix[1..].split('/').rev() { + child = Tree::from_tree_items_with_kind( + HashKind::Sha1, + vec![TreeItem::new( + TreeItemMode::Tree, + child.id, + name.to_string(), + )], + ) + .unwrap(); + trees.push(child.clone()); + } + child +} + +#[tokio::test] +async fn mst2_native_projection_cached_empty_child_move_rechecks_path_budgets() { + let temp = tempfile::tempdir().unwrap(); + let handler = handler(temp.path()).await; + let empty = Tree { + id: ObjectHash::from_type_and_data_for_kind(HashKind::Sha1, ObjectType::Tree, &[]).unwrap(), + tree_items: vec![], + }; + let source = Tree::from_tree_items_with_kind( + HashKind::Sha1, + vec![TreeItem::new( + TreeItemMode::Tree, + empty.id, + "empty".to_string(), + )], + ) + .unwrap(); + let mut trees = vec![empty.clone(), source.clone()]; + let root = wrap_tree_at("/project/source", source.clone(), &mut trees); + let (mut commit, root) = + persist_tree_commit(&handler, root, trees, vec![], "empty source").await; + build_directory_page_with_work(&handler, &root, "/project/source") + .await + .unwrap(); + for (index, (prefix, allowed)) in [ + (byte_prefix(4090), true), + (byte_prefix(4091), false), + (format!("/project/{}", vec!["a"; 254].join("/")), true), + (format!("/project/{}", vec!["a"; 255].join("/")), false), + ] + .into_iter() + .enumerate() + { + let mut trees = vec![empty.clone(), source.clone()]; + let root = wrap_tree_at(&prefix, source.clone(), &mut trees); + let (next, root) = persist_tree_commit( + &handler, + root, + trees, + vec![commit.id], + &format!("empty move {index}"), + ) + .await; + commit = next; + let result = build_directory_page_with_work(&handler, &root, &prefix).await; + if allowed { + let (directory, work) = result.unwrap(); + assert_eq!(directory.entries[0].name, "empty"); + assert_eq!(work.directories_rebuilt, 0); + assert_eq!(work.reused_subtree_roots, 1); + } else { + assert!( + matches!(result, Err(error) if error.code == super::SnapshotErrorCode::ScopeInvalid) + ); + } + } +} + +#[test] +fn mst2_native_projection_cache_separates_equal_hex_different_hash_kinds() { + let cache = super::NativeProjectionCache::default(); + let sha = ObjectHash::from_bytes_for_kind(HashKind::Sha256, &[7; 32]).unwrap(); + let blake = ObjectHash::from_bytes_for_kind(HashKind::Blake3, &[7; 32]).unwrap(); + assert_eq!(sha.to_string(), blake.to_string()); + let page_bytes = mst2_codec::metapage::Page::build(&[]).unwrap(); + let directory = Arc::new(super::BuiltDirectory { + page_id: mst2_codec::metapage::page_id(&page_bytes), + page_bytes, + entries: vec![], + codec_entries: vec![], + path_budget: super::DescendantPathBudget::default(), + }); + cache.insert(sha, Arc::clone(&directory)); + assert!(cache.get(blake).is_none()); + assert!(Arc::ptr_eq(&cache.get(sha).unwrap(), &directory)); +} + +#[tokio::test] +async fn mst2_native_projection_cached_move_rechecks_full_scope_path_budgets() { + let temp = tempfile::tempdir().unwrap(); + let handler = handler(temp.path()).await; + let raw = b"multibyte basename".to_vec(); + let files = Files::from([("/project/source/é".into(), raw.clone())]); + let (mut commit, root) = persist_commit(&handler, &files, vec![], "warm source").await; + build_directory_page_with_work(&handler, &root, "/project/source") + .await + .unwrap(); + let prefixes = [ + (byte_prefix(4093), true), + (byte_prefix(4094), false), + (format!("/project/{}", vec!["a"; 254].join("/")), true), + (format!("/project/{}", vec!["a"; 255].join("/")), false), + ]; + for (index, (prefix, allowed)) in prefixes.into_iter().enumerate() { + let files = Files::from([(format!("{prefix}/é"), raw.clone())]); + let (next, root) = + persist_commit(&handler, &files, vec![commit.id], &format!("move {index}")).await; + commit = next; + let result = build_directory_page_with_work(&handler, &root, &prefix).await; + if allowed { + let (_, work) = result.unwrap(); + assert_eq!(work.directories_rebuilt, 0); + assert_eq!(work.reused_subtree_roots, 1); + assert_eq!(work.directory_entries_scanned, 0); + } else { + assert!( + matches!(result, Err(error) if error.code == super::SnapshotErrorCode::ScopeInvalid) + ); + } + } +} diff --git a/src/ceres/snapshot/pages.rs b/src/ceres/snapshot/pages.rs index 555ed913..05e4e67f 100644 --- a/src/ceres/snapshot/pages.rs +++ b/src/ceres/snapshot/pages.rs @@ -6,25 +6,36 @@ //! yields identical page_ids across requests. use std::{ - collections::HashMap, - sync::{Arc, Mutex, OnceLock}, + borrow::Cow, + collections::{HashMap, HashSet}, + sync::{Arc, Mutex}, }; use base64::Engine; -use mst2_codec::metapage::{Entry, EntryKind, Page, page_id}; +use git_internal::{hash::ObjectHash, internal::object::tree::Tree}; +use mst2_codec::{ + descriptor::{ + ACCESS_PROJECTION_EXACT_FULL, FS_SEMANTICS_LINUX_CODE_V1, + MATERIALIZATION_POLICY_GIT_RAW_V1, METADATA_CODEC, SCHEMA_VERSION, + }, + metapage::{Entry, EntryKind, Page, page_id}, +}; use sea_orm::ActiveValue::Set; use sha2::{Digest, Sha256}; +pub use crate::ceres::snapshot::projection_observation::ProjectionWork; use crate::{ ceres::{ api_service::ApiHandler, snapshot::{ error::{SnapshotError, SnapshotErrorCode}, + projection_observation::NATIVE_PROJECTION_REVISION, resolver::FsKind, + retention_dag::{MetadataDagBuilder, MetadataDagLimits, ValidatedMetadataDag}, view::hex, }, }, - jupiter::storage::mono_storage::MST2_VERIFICATION_VERSION, + jupiter::storage::{Storage, mono_storage::MST2_VERIFICATION_VERSION}, }; /// A directory page plus everything the JSON layer needs. @@ -38,6 +49,189 @@ pub struct BuiltDirectory { /// The codec entries the page was built from, so callers can walk a route /// through the same canonical tree (`Page::pages_along_route`). pub codec_entries: Vec, + /// Bounds proven while validating this subtree; rechecked at every new + /// prefix so an identical tree cannot bypass full-path limits after a move. + path_budget: DescendantPathBudget, +} + +#[derive(Debug, Clone, Copy, Default)] +struct DescendantPathBudget { + /// Includes the slash before each descendant name. + suffix_bytes: usize, + components: usize, +} + +impl DescendantPathBudget { + fn include(&mut self, name: &str, child: Self) { + self.suffix_bytes = self.suffix_bytes.max(1 + name.len() + child.suffix_bytes); + self.components = self.components.max(1 + child.components); + } + + fn validate_at(self, prefix: &str) -> Result<(), SnapshotError> { + crate::ceres::snapshot::view::validate_scope_relative_path(prefix)?; + let (bytes, components) = if prefix == "/" { + (0, 0) + } else { + (prefix.len(), prefix[1..].split('/').count()) + }; + if bytes + self.suffix_bytes > 4096 || components + self.components > 256 { + return Err(SnapshotError::new( + SnapshotErrorCode::ScopeInvalid, + "subtree exceeds full-path budget at this prefix", + )); + } + Ok(()) + } +} + +#[derive(Debug, Clone, PartialEq, Eq, Hash)] +struct NativeProjectionKey { + source_domain: &'static str, + /// Tagged OID preserves SHA-256 vs BLAKE3 even though both are 64 hex. + tree_oid: String, + schema_version: u16, + metadata_codec: u16, + materialization_policy: u16, + fs_semantics: u16, + access_projection: u16, + verification_revision: i32, + projection_revision: u16, +} + +impl NativeProjectionKey { + fn new(oid: ObjectHash) -> Self { + Self { + source_domain: "native-git", + tree_oid: oid.to_tagged_string(), + schema_version: SCHEMA_VERSION, + metadata_codec: METADATA_CODEC, + materialization_policy: MATERIALIZATION_POLICY_GIT_RAW_V1, + fs_semantics: FS_SEMANTICS_LINUX_CODE_V1, + access_projection: ACCESS_PROJECTION_EXACT_FULL, + verification_revision: MST2_VERIFICATION_VERSION, + projection_revision: NATIVE_PROJECTION_REVISION, + } + } +} + +/// Storage-owned, disposable derived facts. This cache grants no authorization +/// and provides no durable retention evidence. Clearing it only costs a rebuild. +#[derive(Default)] +pub(crate) struct NativeProjectionCache { + state: Mutex, +} + +#[derive(Default)] +struct NativeProjectionCacheState { + pages: HashMap>, + retained_payload_bytes: usize, + retention_dags: HashMap>, + retained_dag_bytes: usize, +} + +#[derive(Debug, Clone, PartialEq, Eq, Hash)] +struct NativeRetentionKey { + projection: NativeProjectionKey, + scope: String, +} + +impl NativeProjectionCache { + fn retention_dag(&self, key: &NativeRetentionKey) -> Option> { + self.state.lock().ok()?.retention_dags.get(key).cloned() + } + + fn insert_retention_dag(&self, key: NativeRetentionKey, dag: Arc) { + // The memo has one aggregate residency bound, independent of the page + // cache. Caller Arcs may outlive eviction; this is no request admission. + const MAX_DAG_RESIDENT_BYTES: usize = 64 * 1024 * 1024; + const MAX_DAG_ENTRIES: usize = 16; + self.insert_retention_dag_with_budget(key, dag, MAX_DAG_RESIDENT_BYTES, MAX_DAG_ENTRIES); + } + + fn insert_retention_dag_with_budget( + &self, + key: NativeRetentionKey, + dag: Arc, + maximum_bytes: usize, + maximum_entries: usize, + ) { + let Some(bytes) = dag + .residency_bytes() + .and_then(|bytes| { + bytes.checked_add( + std::mem::size_of::() + + std::mem::size_of::>(), + ) + }) + .and_then(|bytes| bytes.checked_add(key.scope.capacity())) + .and_then(|bytes| bytes.checked_add(key.projection.tree_oid.capacity())) + else { + return; + }; + if bytes > maximum_bytes || maximum_entries == 0 { + return; + } + if let Ok(mut state) = self.state.lock() { + if state.retention_dags.contains_key(&key) { + return; + } + if state.retention_dags.len() >= maximum_entries + || state.retained_dag_bytes > maximum_bytes - bytes + { + state.retention_dags.clear(); + state.retained_dag_bytes = 0; + } + state.retained_dag_bytes += bytes; + state.retention_dags.insert(key, dag); + } + } + + fn get(&self, oid: ObjectHash) -> Option> { + // A poisoned optimization cache is a miss, not a serving failure. + self.state + .lock() + .ok()? + .pages + .get(&NativeProjectionKey::new(oid)) + .cloned() + } + + fn insert(&self, oid: ObjectHash, directory: Arc) { + // Bound retained payload as well as directory count: a wide directory + // retains all entries and codec names even though its root is <=16 KiB. + const MAX_PAYLOAD_BYTES: usize = 64 * 1024 * 1024; + let payload_bytes = std::mem::size_of::() + + directory.page_bytes.capacity() + + directory.entries.capacity() * std::mem::size_of::() + + directory.codec_entries.capacity() * std::mem::size_of::() + + directory + .entries + .iter() + .map(|entry| entry.name.capacity() + entry.oid.capacity()) + .sum::() + + directory + .codec_entries + .iter() + .map(|entry| entry.name.capacity()) + .sum::(); + if payload_bytes > MAX_PAYLOAD_BYTES { + return; + } + if let Ok(mut state) = self.state.lock() { + let key = NativeProjectionKey::new(oid); + if state.pages.contains_key(&key) { + return; + } + if state.pages.len() >= 8192 + || state.retained_payload_bytes + payload_bytes > MAX_PAYLOAD_BYTES + { + state.pages.clear(); + state.retained_payload_bytes = 0; + } + state.retained_payload_bytes += payload_bytes; + state.pages.insert(key, directory); + } + } } #[derive(Debug, Clone)] @@ -52,54 +246,375 @@ pub struct DirEntry { pub content_digest: Option<[u8; 32]>, /// Child directory page id (directories only). pub directory_root: Option<[u8; 32]>, + /// Preserve the source hash kind when preparing a cached child closure. + directory_tree_oid: Option, +} + +/// Fixed source/profile identity for a prepared metadata closure. This grants +/// no authorization, durable retention or coverage of file/chunk payloads. +#[derive(Debug)] +pub struct PreparedNativeMetadataRetention { + key: NativeRetentionKey, + dag: Arc, +} + +impl PreparedNativeMetadataRetention { + #[cfg(test)] + pub(crate) fn test_installation(dag: Arc, scope: &str) -> Self { + let tree_oid = + ObjectHash::from_hex_for_kind(git_internal::hash::HashKind::Sha1, &"a".repeat(40)) + .unwrap(); + Self { + key: NativeRetentionKey { + projection: NativeProjectionKey::new(tree_oid), + scope: scope.to_owned(), + }, + dag, + } + } + pub(crate) fn install_plan( + &self, + ) -> Result { + use super::metadata_install::{MetadataInstallIdentity, MetadataInstallPlan}; + let key = &self.key.projection; + MetadataInstallPlan::from_validated( + MetadataInstallIdentity { + source_domain: key.source_domain.to_owned(), + tagged_root_tree_oid: key.tree_oid.clone(), + scope: self.key.scope.clone(), + schema_version: key.schema_version, + metadata_codec: key.metadata_codec, + materialization_policy: key.materialization_policy, + fs_semantics: key.fs_semantics, + access_projection: key.access_projection, + verification_revision: key.verification_revision, + projection_revision: key.projection_revision, + }, + &self.dag, + ) + } + pub fn fixed_root_tree_oid(&self) -> &str { + &self.key.projection.tree_oid + } + pub fn scope(&self) -> &str { + &self.key.scope + } + pub fn metadata_codec(&self) -> u16 { + self.key.projection.metadata_codec + } + pub fn schema_version(&self) -> u16 { + self.key.projection.schema_version + } + pub fn dag(&self) -> &Arc { + &self.dag + } +} + +/// Explicit preparation, separate from resolve/lease success. A hit returns +/// the immutable Arc before traversing any child Git tree. A miss collects +/// from existing native codec entries, rebuilding an evicted child as needed. +pub async fn prepare_native_metadata_retention( + handler: &T, + root_tree: &Tree, + scope: &str, + limits: MetadataDagLimits, +) -> Result { + crate::ceres::snapshot::view::validate_scope_relative_path(scope)?; + if !handler.native_snapshot_projection() { + return Err(SnapshotError::new( + SnapshotErrorCode::ScopeInvalid, + "retention preparation requires native Git projection", + )); + } + let storage = handler.get_context(); + let limits = limits.effective(); + let key = NativeRetentionKey { + projection: NativeProjectionKey::new(root_tree.id), + scope: scope.to_owned(), + }; + if let Some(dag) = storage.native_projection_cache.retention_dag(&key) { + dag.check_limits(limits)?; + return Ok(PreparedNativeMetadataRetention { key, dag }); + } + let mut projection_budget = MetadataProjectionBudget::new(limits); + let mut work = ProjectionWork::default(); + let root = if scope == "/" { + build_subtree_inner( + handler, + &storage, + root_tree, + scope, + true, + &mut work, + Some(&mut projection_budget), + ) + .await? + } else { + let tree = fetch_tree_with_work(handler, root_tree, scope, &mut work).await?; + build_subtree_inner( + handler, + &storage, + &tree, + scope, + true, + &mut work, + Some(&mut projection_budget), + ) + .await? + }; + let root_id = root.page_id; + let mut builder = MetadataDagBuilder::new(limits); + let mut scheduled = HashSet::from([root_id]); + let mut pending = vec![(root, scope.to_owned())]; + while let Some((directory, path)) = pending.pop() { + builder.add_directory(&directory.page_bytes, &directory.codec_entries)?; + for entry in directory + .entries + .iter() + .filter(|entry| entry.fs_kind == FsKind::Directory) + { + let child_id = entry.directory_root.ok_or_else(|| { + SnapshotError::new( + SnapshotErrorCode::IntegrityError, + "directory lacks its metadata root", + ) + })?; + if !scheduled.insert(child_id) { + continue; + } + let child_oid = entry.directory_tree_oid.ok_or_else(|| { + SnapshotError::new( + SnapshotErrorCode::IntegrityError, + "directory lacks its tagged source identity", + ) + })?; + let child_path = if path == "/" { + format!("/{}", entry.name) + } else { + format!("{path}/{}", entry.name) + }; + let child = if let Some(hit) = + cached_subtree(&storage, child_oid, &child_path, true, &mut work)? + { + hit + } else { + let tree = handler + .get_tree_by_hash(&entry.oid) + .await + .map_err(|error| { + SnapshotError::new(SnapshotErrorCode::Internal, error.to_string()) + })?; + if tree.id != child_oid { + return Err(SnapshotError::new( + SnapshotErrorCode::IntegrityError, + "retention child tree identity mismatch", + )); + } + build_subtree_inner( + handler, + &storage, + &tree, + &child_path, + true, + &mut work, + Some(&mut projection_budget), + ) + .await? + }; + if child.page_id != child_id { + return Err(SnapshotError::new( + SnapshotErrorCode::DigestMismatch, + "retention child page disagrees with fixed parent", + )); + } + pending.push((child, child_path)); + } + } + let dag = Arc::new(builder.finish(root_id)?); + storage + .native_projection_cache + .insert_retention_dag(key.clone(), Arc::clone(&dag)); + Ok(PreparedNativeMetadataRetention { key, dag }) } /// Build the MTP2 page for one directory of the fixed view. `rel_path` is -/// scope-relative ("/" = the root directory); `root_tree` is the fixed view's -/// root tree — never a current-ref read. Every entry is represented; +/// an absolute path within the fixed global view, including the descriptor +/// scope prefix ("/" = the global root); `root_tree` is that view's global +/// root tree — never a rebased scope tree or current-ref read. Every entry is represented; /// unsupported entries reject the whole projection (spec: no silent drops). /// -/// Results are memoized per (root tree, path): a page is a pure function of -/// the pinned root tree, and the recursive build otherwise re-walks the same -/// subtree once per ancestor and once per request (a sync over N directories -/// would rebuild the tree N times). Blob sizes/digests are persisted in -/// `mst2_verified_object`, so a cache miss after eviction is a bounded, -/// correctness-identical recomputation. -/// Page memoization table: (root tree id, scope-relative path) → built page. -type PageCache = Mutex>>; - +/// Native results are memoized by storage assembly, tagged subtree tree OID +/// and verification/profile revision. A new global root does not invalidate +/// unchanged subtrees. The fixed root is resolved to `rel_path` once; recursive +/// descent then follows direct child OIDs rather than re-walking from the root. pub async fn build_directory_page( handler: &T, root_tree: &git_internal::internal::object::tree::Tree, rel_path: &str, ) -> Result, SnapshotError> { - static PAGE_CACHE: OnceLock = OnceLock::new(); - // Pages are small (≤16 KiB + entries); 200k pages is far beyond any real - // view. On overflow the cache clears wholesale — a miss only costs a - // rebuild, never correctness. - const PAGE_CACHE_MAX: usize = 200_000; - - let key = (root_tree.id.to_string(), rel_path.to_string()); - let cache = PAGE_CACHE.get_or_init(|| Mutex::new(HashMap::new())); - if let Some(hit) = cache.lock().unwrap().get(&key) { - return Ok(Arc::clone(hit)); - } - let built = Arc::new(build_directory_page_uncached(handler, root_tree, rel_path).await?); - let mut cache = cache.lock().unwrap(); - if cache.len() >= PAGE_CACHE_MAX { - cache.clear(); - } - cache.insert(key, Arc::clone(&built)); - Ok(built) + let (directory, work) = build_directory_page_with_work(handler, root_tree, rel_path).await?; + tracing::debug!( + ?work, + path = rel_path, + "native snapshot projection memoization work" + ); + Ok(directory) } -async fn build_directory_page_uncached( +/// Build one directory, returning operation-local counters. Authorization and +/// the fixed-view source selection remain the caller's responsibility. +pub async fn build_directory_page_with_work( handler: &T, - root_tree: &git_internal::internal::object::tree::Tree, + root_tree: &Tree, rel_path: &str, -) -> Result { - let tree = fetch_tree(handler, root_tree, rel_path).await?; - let dirents = crate::ceres::snapshot::resolver::direct_entries(&tree)?; +) -> Result<(Arc, ProjectionWork), SnapshotError> { + let storage = handler.get_context(); + let mut work = ProjectionWork::default(); + let native = handler.native_snapshot_projection(); + if rel_path == "/" { + // Even cloning a wide fixed root would copy all its entries on a hit. + let directory = + build_subtree(handler, &storage, root_tree, rel_path, native, &mut work).await?; + return Ok((directory, work)); + } + let tree = fetch_tree_with_work(handler, root_tree, rel_path, &mut work).await?; + let directory = build_subtree(handler, &storage, &tree, rel_path, native, &mut work).await?; + Ok((directory, work)) +} + +fn cached_subtree( + storage: &Storage, + oid: ObjectHash, + rel_path: &str, + native: bool, + work: &mut ProjectionWork, +) -> Result>, SnapshotError> { + if native && let Some(hit) = storage.native_projection_cache.get(oid) { + hit.path_budget.validate_at(rel_path)?; + work.reused_subtree_roots += 1; + work.directory_root_pages_reused += 1; + work.directory_root_page_bytes_reused += hit.page_bytes.len() as u64; + return Ok(Some(hit)); + } + Ok(None) +} + +async fn build_subtree( + handler: &T, + storage: &Storage, + tree: &Tree, + rel_path: &str, + native: bool, + work: &mut ProjectionWork, +) -> Result, SnapshotError> { + build_subtree_inner(handler, storage, tree, rel_path, native, work, None).await +} + +struct MetadataProjectionBudget { + limits: MetadataDagLimits, + seen: HashSet, + required: HashSet, + edges: HashSet<(ObjectHash, ObjectHash)>, + entries: usize, + encoded_entry_bytes: u64, +} + +impl MetadataProjectionBudget { + fn new(limits: MetadataDagLimits) -> Self { + Self { + limits, + seen: HashSet::new(), + required: HashSet::new(), + edges: HashSet::new(), + entries: 0, + encoded_entry_bytes: 0, + } + } + + fn tree(&mut self, tree: &Tree) -> Result<(), SnapshotError> { + let failed = || { + SnapshotError::new( + SnapshotErrorCode::LimitExceeded, + "native metadata source projection budget exceeded", + ) + }; + if self.seen.contains(&tree.id) { + return Ok(()); + } + self.require_node(tree.id)?; + self.entries = self + .entries + .checked_add(tree.tree_items.len()) + .filter(|count| *count <= self.limits.entries) + .ok_or_else(failed)?; + self.encoded_entry_bytes = self + .encoded_entry_bytes + .checked_add(mst2_codec::metapage::HEADER_LEN as u64) + .filter(|bytes| *bytes <= self.limits.payload_bytes) + .ok_or_else(failed)?; + for item in &tree.tree_items { + let kind = FsKind::from_git_mode(item.mode).ok_or_else(|| { + SnapshotError::new( + SnapshotErrorCode::UnsupportedEntry, + "unsupported entry in native retention projection", + ) + })?; + let value_bytes = if kind == FsKind::Directory { 32 } else { 40 }; + self.encoded_entry_bytes = self + .encoded_entry_bytes + .checked_add((3 + item.name.len() + value_bytes) as u64) + .filter(|bytes| *bytes <= self.limits.payload_bytes) + .ok_or_else(failed)?; + if kind == FsKind::Directory { + self.require_node(item.id)?; + if !self.edges.contains(&(tree.id, item.id)) { + if self.edges.len() >= self.limits.edges { + return Err(failed()); + } + self.edges.insert((tree.id, item.id)); + } + } + } + if self.required.len() > self.limits.nodes { + return Err(failed()); + } + self.seen.insert(tree.id); + Ok(()) + } + + fn require_node(&mut self, id: ObjectHash) -> Result<(), SnapshotError> { + if !self.required.contains(&id) { + if self.required.len() >= self.limits.nodes { + return Err(SnapshotError::new( + SnapshotErrorCode::LimitExceeded, + "native metadata source node budget exceeded", + )); + } + self.required.insert(id); + } + Ok(()) + } +} + +async fn build_subtree_inner( + handler: &T, + storage: &Storage, + tree: &Tree, + rel_path: &str, + native: bool, + work: &mut ProjectionWork, + mut budget: Option<&mut MetadataProjectionBudget>, +) -> Result, SnapshotError> { + crate::ceres::snapshot::view::validate_scope_relative_path(rel_path)?; + if let Some(budget) = budget.as_deref_mut() { + budget.tree(tree)?; + } + if let Some(hit) = cached_subtree(storage, tree.id, rel_path, native, work)? { + return Ok(hit); + } + let dirents = crate::ceres::snapshot::resolver::direct_entries(tree)?; + work.directories_rebuilt += 1; + work.directory_entries_scanned += dirents.len() as u64; // T03 write-through verification: consult verified records first, then // fetch + hash the misses and persist them. Invalid records and lookup @@ -109,8 +624,7 @@ async fn build_directory_page_uncached( .filter(|(_, k, _)| *k != FsKind::Directory) .map(|(_, _, oid)| oid.clone()) .collect(); - let verified = handler - .get_context() + let verified = storage .mono_storage() .get_verified_blobs(blob_oids) .await @@ -128,6 +642,7 @@ async fn build_directory_page_uncached( })?; let mut new_verified = HashMap::new(); + let mut path_budget = DescendantPathBudget::default(); let mut entries = Vec::with_capacity(dirents.len()); let mut codec_entries = Vec::with_capacity(dirents.len()); for (name, fs_kind, oid) in dirents { @@ -139,8 +654,37 @@ async fn build_directory_page_uncached( crate::ceres::snapshot::view::validate_scope_relative_path(&child_rel)?; match fs_kind { FsKind::Directory => { - let child_page = - Box::pin(build_directory_page(handler, root_tree, &child_rel)).await?; + let child_oid = + ObjectHash::from_hex_for_kind(tree.id.kind(), &oid).map_err(|e| { + SnapshotError::new(SnapshotErrorCode::IntegrityError, e.to_string()) + })?; + let child_page = if let Some(hit) = + cached_subtree(storage, child_oid, &child_rel, native, work)? + { + hit + } else { + work.tree_fetches += 1; + let child_tree = handler.get_tree_by_hash(&oid).await.map_err(|e| { + SnapshotError::new(SnapshotErrorCode::Internal, e.to_string()) + })?; + if child_tree.id != child_oid { + return Err(SnapshotError::new( + SnapshotErrorCode::IntegrityError, + "fetched child tree identity mismatch", + )); + } + Box::pin(build_subtree_inner( + handler, + storage, + &child_tree, + &child_rel, + native, + work, + budget.as_deref_mut(), + )) + .await? + }; + path_budget.include(&name, child_page.path_budget); codec_entries.push(Entry::dir(name.as_bytes(), child_page.page_id)); entries.push(DirEntry { name, @@ -149,10 +693,13 @@ async fn build_directory_page_uncached( size: None, content_digest: None, directory_root: Some(child_page.page_id), + directory_tree_oid: Some(child_oid), }); } FsKind::Regular | FsKind::Executable | FsKind::Symlink => { + path_budget.include(&name, DescendantPathBudget::default()); let (size, digest) = if let Some(v) = verified.get(&oid) { + work.verified_blob_hits += 1; // Verified record: 64-bit size + raw digest, no content read. let d: [u8; 32] = v.raw_sha256.as_slice().try_into().map_err(|_| { SnapshotError::new( @@ -168,9 +715,12 @@ async fn build_directory_page_uncached( })?; (size, d) } else { + work.verified_blob_misses += 1; let raw = fetch_raw_blob(handler, &oid).await?; + work.raw_bytes_fetched += raw.len() as u64; let mut h = Sha256::new(); h.update(&raw); + work.raw_bytes_hashed += raw.len() as u64; let digest: [u8; 32] = h.finalize().into(); new_verified.entry(oid.clone()).or_insert( crate::callisto::mst2_verified_object::ActiveModel { @@ -201,6 +751,7 @@ async fn build_directory_page_uncached( size: Some(size), content_digest: Some(digest), directory_root: None, + directory_tree_oid: None, }); } } @@ -208,8 +759,7 @@ async fn build_directory_page_uncached( if !new_verified.is_empty() { // Records only ever describe already-fetched content; a persist // failure loses an optimization, never correctness. - if let Err(e) = handler - .get_context() + if let Err(e) = storage .mono_storage() .insert_verified_blobs(new_verified.into_values().collect()) .await @@ -224,13 +774,22 @@ async fn build_directory_page_uncached( format!("MTP2 build failed for {rel_path}: {e}"), ) })?; + work.directory_root_pages_built += 1; + work.directory_root_page_bytes_built += page_bytes.len() as u64; let pid = page_id(&page_bytes); - Ok(BuiltDirectory { + let built = Arc::new(BuiltDirectory { page_bytes, page_id: pid, entries, codec_entries, - }) + path_budget, + }); + if native { + storage + .native_projection_cache + .insert(tree.id, Arc::clone(&built)); + } + Ok(built) } /// Proof pages from the scope root down to (and including) `rel_path`, @@ -259,23 +818,24 @@ pub fn base64_of(data: &[u8]) -> String { base64::engine::general_purpose::STANDARD.encode(data) } -async fn fetch_tree( +async fn fetch_tree_with_work( handler: &T, root_tree: &git_internal::internal::object::tree::Tree, rel_path: &str, + work: &mut ProjectionWork, ) -> Result { crate::ceres::snapshot::view::validate_scope_relative_path(rel_path)?; if rel_path == "/" { return Ok(root_tree.clone()); } let comps: Vec<&str> = rel_path[1..].split('/').collect(); - let mut current = root_tree.clone(); - for (i, comp) in comps.iter().enumerate() { - let last = i == comps.len() - 1; - let item = current - .tree_items - .iter() - .find(|x| x.name == *comp) + let mut current = Cow::Borrowed(root_tree); + for comp in comps { + let position = current.tree_items.iter().position(|x| x.name == comp); + work.scope_path_entries_examined += + position.map_or(current.tree_items.len(), |index| index + 1) as u64; + let item = position + .map(|index| ¤t.tree_items[index]) .ok_or_else(|| { SnapshotError::new( SnapshotErrorCode::PathNotFound, @@ -288,30 +848,15 @@ async fn fetch_tree( "gitlink entries are not supported in this profile", ) })?; - if last { - if kind != FsKind::Directory { - return Err(SnapshotError::new( - SnapshotErrorCode::NotDirectory, - format!("{rel_path} is not a directory"), - )); - } - return handler - .get_tree_by_hash(&item.id.to_string()) - .await - .map_err(|e| { - SnapshotError::new( - SnapshotErrorCode::Internal, - format!("tree fetch failed for {rel_path}: {e}"), - ) - }); - } if kind != FsKind::Directory { return Err(SnapshotError::new( SnapshotErrorCode::NotDirectory, - "intermediate component is not a directory", + format!("{rel_path} is not a directory"), )); } - current = handler + let expected_oid = item.id; + work.tree_fetches += 1; + let fetched = handler .get_tree_by_hash(&item.id.to_string()) .await .map_err(|e| { @@ -320,8 +865,15 @@ async fn fetch_tree( format!("tree fetch failed for component '{comp}': {e}"), ) })?; + if fetched.id != expected_oid { + return Err(SnapshotError::new( + SnapshotErrorCode::IntegrityError, + "fetched scope tree identity mismatch", + )); + } + current = Cow::Owned(fetched); } - unreachable!("loop returns on the last component") + Ok(current.into_owned()) } /// Outcome of resolving one absolute view path (spec 04 §7 statuses). @@ -352,16 +904,51 @@ pub async fn resolve_abs( root_tree: &git_internal::internal::object::tree::Tree, abs_path: &str, ) -> Result { + Ok( + match resolve_abs_metadata(handler, root_tree, abs_path).await? { + MetadataWalkOutcome::FoundDir => WalkOutcome::FoundDir, + MetadataWalkOutcome::Absent => WalkOutcome::Absent, + MetadataWalkOutcome::NotDirectory { symlink } => WalkOutcome::NotDirectory { symlink }, + MetadataWalkOutcome::FoundFile { fs_kind, oid } => { + let raw = fetch_raw_blob(handler, &oid).await?; + let mut h = Sha256::new(); + h.update(&raw); + WalkOutcome::FoundFile { + fs_kind, + oid, + size: raw.len() as u64, + digest: h.finalize().into(), + raw, + } + } + }, + ) +} + +/// Fixed Git path metadata, without reading or hashing file content. +#[derive(Debug)] +pub(crate) enum MetadataWalkOutcome { + FoundDir, + FoundFile { fs_kind: FsKind, oid: String }, + Absent, + NotDirectory { symlink: bool }, +} + +pub(crate) async fn resolve_abs_metadata( + handler: &T, + root_tree: &git_internal::internal::object::tree::Tree, + abs_path: &str, +) -> Result { crate::ceres::snapshot::view::validate_scope_relative_path(abs_path)?; if abs_path == "/" { - return Ok(WalkOutcome::FoundDir); + return Ok(MetadataWalkOutcome::FoundDir); } let comps: Vec<&str> = abs_path[1..].split('/').collect(); - let mut current = root_tree.clone(); + let mut current = Cow::Borrowed(root_tree); for (i, comp) in comps.iter().enumerate() { let last = i == comps.len() - 1; let Some(item) = current.tree_items.iter().find(|x| x.name == *comp) else { - return Ok(WalkOutcome::Absent); + return Ok(MetadataWalkOutcome::Absent); }; let Some(kind) = FsKind::from_git_mode(item.mode) else { return Err(SnapshotError::new( @@ -372,33 +959,31 @@ pub async fn resolve_abs( let oid = item.id.to_string(); if last { return match kind { - FsKind::Directory => Ok(WalkOutcome::FoundDir), + FsKind::Directory => Ok(MetadataWalkOutcome::FoundDir), FsKind::Regular | FsKind::Executable | FsKind::Symlink => { - let raw = fetch_raw_blob(handler, &oid).await?; - let mut h = Sha256::new(); - h.update(&raw); - let digest: [u8; 32] = h.finalize().into(); - Ok(WalkOutcome::FoundFile { - fs_kind: kind, - oid, - size: raw.len() as u64, - digest, - raw, - }) + Ok(MetadataWalkOutcome::FoundFile { fs_kind: kind, oid }) } }; } if kind != FsKind::Directory { - return Ok(WalkOutcome::NotDirectory { + return Ok(MetadataWalkOutcome::NotDirectory { symlink: kind == FsKind::Symlink, }); } - current = handler.get_tree_by_hash(&oid).await.map_err(|e| { + let expected_oid = item.id; + let fetched = handler.get_tree_by_hash(&oid).await.map_err(|e| { SnapshotError::new( SnapshotErrorCode::Internal, format!("tree fetch failed for component '{comp}': {e}"), ) })?; + if fetched.id != expected_oid { + return Err(SnapshotError::new( + SnapshotErrorCode::IntegrityError, + "fetched fixed tree identity mismatch", + )); + } + current = Cow::Owned(fetched); } unreachable!("loop returns on the last component") } @@ -446,6 +1031,14 @@ pub fn hex_of(id: &[u8; 32]) -> String { hex(id) } +#[cfg(test)] +#[path = "native_projection_tests.rs"] +mod native_projection_tests; + +#[cfg(test)] +#[path = "retention_prepare_tests.rs"] +mod retention_prepare_tests; + #[cfg(test)] mod tests { use std::sync::Arc; @@ -829,9 +1422,16 @@ mod tests { ]) .unwrap(); let handler = crate::ceres::api_service::mono_api_service::MonoApiService::from(&state); - let built = build_directory_page_uncached(&handler, &tree, "/") - .await - .unwrap(); + let built = build_subtree( + &handler, + &state.storage, + &tree, + "/", + false, + &mut ProjectionWork::default(), + ) + .await + .unwrap(); assert_eq!(built.entries.len(), 2); assert!(built.entries.iter().all(|entry| entry.size == Some(10) && entry.content_digest == Some(Sha256::digest(raw).into()))); diff --git a/src/ceres/snapshot/projection_observation.rs b/src/ceres/snapshot/projection_observation.rs new file mode 100644 index 00000000..819e0e86 --- /dev/null +++ b/src/ceres/snapshot/projection_observation.rs @@ -0,0 +1,638 @@ +//! Operation-local resolve diagnostics, not publication or retention receipts. + +use std::time::Duration; + +use git_internal::hash::ObjectHash; +use mst2_codec::descriptor::{ + ACCESS_PROJECTION_EXACT_FULL, FS_SEMANTICS_LINUX_CODE_V1, MATERIALIZATION_POLICY_GIT_RAW_V1, + METADATA_CODEC, SCHEMA_VERSION, ServingDescriptor, +}; +use serde::Serialize; +use sha2::{Digest, Sha256}; +use uuid::Uuid; + +use crate::{ + ceres::snapshot::error::{SnapshotError, SnapshotErrorCode}, + jupiter::storage::mono_storage::MST2_VERIFICATION_VERSION, +}; + +pub(crate) const NATIVE_PROJECTION_REVISION: u16 = 1; + +/// Work in one directory projection. Page counters cover returned directory +/// roots, excluding codec-internal radix encoding and later route traversal. +#[derive(Debug, Clone, Default, PartialEq, Eq, Serialize)] +pub struct ProjectionWork { + pub directories_rebuilt: u64, + /// Cache-hit boundary visits, not all descendants or unique source OIDs. + pub reused_subtree_roots: u64, + pub directory_root_pages_built: u64, + pub directory_root_page_bytes_built: u64, + pub directory_root_pages_reused: u64, + pub directory_root_page_bytes_reused: u64, + pub directory_entries_scanned: u64, + pub scope_path_entries_examined: u64, + /// Backend tree requests, excluding the caller-supplied fixed root tree. + pub tree_fetches: u64, + /// Verification-record outcomes per file entry, not unique body OIDs. + pub verified_blob_hits: u64, + pub verified_blob_misses: u64, + /// Returned raw-body lengths, not physical disk/network traffic. + pub raw_bytes_fetched: u64, + /// Bytes passed to this builder's explicit raw-file SHA-256. + pub raw_bytes_hashed: u64, +} + +/// Captured only from the validated native head, before resolve projects it. +pub(crate) struct NativeResolveSource { + instance: Uuid, + commit: ObjectHash, + tree: ObjectHash, + certificate_receipt_id: u64, + writer_epoch: u64, + publication_sequence: u64, +} + +pub(crate) struct ResolvedProjection<'a> { + pub descriptor: &'a ServingDescriptor, + pub snapshot_id: &'a str, + pub metadata_root: &'a str, + pub context_commit: &'a str, + pub context_root_tree: &'a str, + pub fixed_root_tree: ObjectHash, + pub requested_scope: &'a str, + pub request_id: &'a str, +} + +/// Immutable successful resolve observation. This grants no authorization, +/// retention, durable publication or proof of codec-internal construction work. +pub(crate) struct NativeProjectionObservation { + source: NativeResolveSource, + scope: String, + namespace_view_id: String, + snapshot_id: String, + metadata_root: String, + request_id: String, + projection_elapsed_micros: u64, + work: ProjectionWork, +} + +/// Closed wire fields, borrowed only from the already validated observation. +#[derive(Serialize)] +pub(crate) struct ProjectionWireRecord<'a> { + observation_revision: u16, + phase: &'static str, + source_domain: &'static str, + request_id: &'a str, + instance_id: String, + #[serde(serialize_with = "tagged_oid")] + root_commit_oid: ObjectHash, + #[serde(serialize_with = "tagged_oid")] + root_tree_oid: ObjectHash, + native_certificate_receipt_id: u64, + native_writer_epoch: u64, + native_publication_sequence: u64, + scope: &'a str, + schema_version: u16, + metadata_codec: u16, + materialization_policy: u16, + fs_semantics: u16, + access_projection: u16, + verification_revision: i32, + projection_revision: u16, + namespace_view_id: &'a str, + snapshot_id: &'a str, + metadata_root: &'a str, + projection_elapsed_micros: u64, + page_counter_scope: &'static str, + codec_radix_work: &'static str, + #[serde(flatten)] + work: &'a ProjectionWork, + message: &'static str, +} + +fn tagged_oid(oid: &ObjectHash, serializer: S) -> Result { + serializer.serialize_str(&oid.to_tagged_string()) +} + +fn invalid() -> SnapshotError { + SnapshotError::new( + SnapshotErrorCode::IntegrityError, + "native resolve observation does not match its fixed source and context", + ) +} + +impl NativeResolveSource { + pub(crate) fn capture( + instance_id: &str, + commit: ObjectHash, + tree: ObjectHash, + certificate_receipt_id: Option, + writer_epoch: i64, + publication_sequence: i64, + ) -> Result { + let instance = Uuid::parse_str(instance_id).map_err(|_| invalid())?; + let positive = |value: i64| { + u64::try_from(value) + .ok() + .filter(|value| *value > 0) + .ok_or_else(invalid) + }; + if instance.to_string() != instance_id || commit.kind() != tree.kind() { + return Err(invalid()); + } + Ok(Self { + instance, + commit, + tree, + certificate_receipt_id: positive(certificate_receipt_id.ok_or_else(invalid)?)?, + writer_epoch: positive(writer_epoch)?, + publication_sequence: positive(publication_sequence)?, + }) + } + + pub(crate) fn observe( + self, + resolved: ResolvedProjection<'_>, + work: ProjectionWork, + projection_elapsed: Duration, + ) -> Result { + let mut view_hash = Sha256::new(); + view_hash.update(b"mega.mst2.namespaceview\0"); + view_hash.update(self.commit.to_string().as_bytes()); + let expected_view: [u8; 32] = view_hash.finalize().into(); + let descriptor = resolved.descriptor; + let digest_text = |bytes: &[u8]| format!("sha256:{}", hex::encode(bytes)); + let snapshot = descriptor.snapshot_id().map_err(|_| invalid())?; + let bounded_request_id = !resolved.request_id.is_empty() + && resolved.request_id.len() <= 128 + && resolved.request_id.bytes().all(|byte| { + matches!(byte, b'a'..=b'z' | b'A'..=b'Z' | b'0'..=b'9' | b'-' | b'_' | b'.' | b':' | b'/') + }); + if resolved.context_commit != self.commit.to_string() + || resolved.context_root_tree != self.tree.to_string() + || resolved.fixed_root_tree != self.tree + || descriptor.instance_uuid != *self.instance.as_bytes() + || descriptor.namespace_view_id != expected_view + || descriptor.scope != resolved.requested_scope + || resolved.snapshot_id != digest_text(&snapshot) + || resolved.metadata_root != digest_text(&descriptor.metadata_root) + || !bounded_request_id + { + return Err(invalid()); + } + Ok(NativeProjectionObservation { + source: self, + scope: descriptor.scope.clone(), + namespace_view_id: digest_text(&descriptor.namespace_view_id), + snapshot_id: resolved.snapshot_id.to_owned(), + metadata_root: resolved.metadata_root.to_owned(), + request_id: resolved.request_id.to_owned(), + projection_elapsed_micros: u64::try_from(projection_elapsed.as_micros()) + .map_err(|_| invalid())?, + work, + }) + } +} + +impl NativeProjectionObservation { + pub(crate) fn wire_record(&self) -> ProjectionWireRecord<'_> { + ProjectionWireRecord { + observation_revision: 1, + phase: "resolve_directory_projection", + source_domain: "native-git", + request_id: &self.request_id, + instance_id: self.source.instance.to_string(), + root_commit_oid: self.source.commit, + root_tree_oid: self.source.tree, + native_certificate_receipt_id: self.source.certificate_receipt_id, + native_writer_epoch: self.source.writer_epoch, + native_publication_sequence: self.source.publication_sequence, + scope: &self.scope, + schema_version: SCHEMA_VERSION, + metadata_codec: METADATA_CODEC, + materialization_policy: MATERIALIZATION_POLICY_GIT_RAW_V1, + fs_semantics: FS_SEMANTICS_LINUX_CODE_V1, + access_projection: ACCESS_PROJECTION_EXACT_FULL, + verification_revision: MST2_VERIFICATION_VERSION, + projection_revision: NATIVE_PROJECTION_REVISION, + namespace_view_id: &self.namespace_view_id, + snapshot_id: &self.snapshot_id, + metadata_root: &self.metadata_root, + projection_elapsed_micros: self.projection_elapsed_micros, + page_counter_scope: "returned-directory-root-pages", + codec_radix_work: "NOT_EXPOSED", + work: &self.work, + message: "native resolve directory projection succeeded", + } + } + + pub(crate) fn emit(self) { + tracing::debug!( + target: "mst2::native_projection_observation", + observation_revision = 1u16, + phase = "resolve_directory_projection", + source_domain = "native-git", + request_id = %self.request_id, + instance_id = %self.source.instance, + root_commit_oid = %self.source.commit.to_tagged_string(), + root_tree_oid = %self.source.tree.to_tagged_string(), + native_certificate_receipt_id = self.source.certificate_receipt_id, + native_writer_epoch = self.source.writer_epoch, + native_publication_sequence = self.source.publication_sequence, + scope = ?self.scope, + schema_version = SCHEMA_VERSION, + metadata_codec = METADATA_CODEC, + materialization_policy = MATERIALIZATION_POLICY_GIT_RAW_V1, + fs_semantics = FS_SEMANTICS_LINUX_CODE_V1, + access_projection = ACCESS_PROJECTION_EXACT_FULL, + verification_revision = MST2_VERIFICATION_VERSION, + projection_revision = NATIVE_PROJECTION_REVISION, + namespace_view_id = %self.namespace_view_id, + snapshot_id = %self.snapshot_id, + metadata_root = %self.metadata_root, + projection_elapsed_micros = self.projection_elapsed_micros, + page_counter_scope = "returned-directory-root-pages", + codec_radix_work = "NOT_EXPOSED", + directories_rebuilt = self.work.directories_rebuilt, + reused_subtree_roots = self.work.reused_subtree_roots, + directory_root_pages_built = self.work.directory_root_pages_built, + directory_root_page_bytes_built = self.work.directory_root_page_bytes_built, + directory_root_pages_reused = self.work.directory_root_pages_reused, + directory_root_page_bytes_reused = self.work.directory_root_page_bytes_reused, + directory_entries_scanned = self.work.directory_entries_scanned, + scope_path_entries_examined = self.work.scope_path_entries_examined, + tree_fetches = self.work.tree_fetches, + verified_blob_hits = self.work.verified_blob_hits, + verified_blob_misses = self.work.verified_blob_misses, + raw_bytes_fetched = self.work.raw_bytes_fetched, + raw_bytes_hashed = self.work.raw_bytes_hashed, + "native resolve directory projection succeeded" + ); + #[cfg(test)] + let _ = OBSERVATIONS.try_with(|observations| observations.lock().unwrap().push(self)); + } +} + +#[cfg(test)] +tokio::task_local! { + static OBSERVATIONS: std::sync::Arc>>; +} + +#[cfg(test)] +pub(crate) async fn with_observations( + future: F, +) -> (F::Output, Vec) { + let observations = std::sync::Arc::new(std::sync::Mutex::new(Vec::new())); + let result = OBSERVATIONS.scope(observations.clone(), future).await; + let records = std::mem::take(&mut *observations.lock().unwrap()); + (result, records) +} + +#[cfg(test)] +impl NativeProjectionObservation { + pub(crate) fn test_identity(&self) -> serde_json::Value { + serde_json::json!({ + "instance_id": self.source.instance.to_string(), + "root_commit_oid": self.source.commit.to_tagged_string(), + "root_tree_oid": self.source.tree.to_tagged_string(), + "native_certificate_receipt_id": self.source.certificate_receipt_id, + "native_writer_epoch": self.source.writer_epoch.to_string(), + "native_publication_sequence": self.source.publication_sequence.to_string(), + "scope": self.scope, + "namespace_view_id": self.namespace_view_id, + "snapshot_id": self.snapshot_id, + "metadata_root": self.metadata_root, + "request_id": self.request_id, + }) + } +} + +#[cfg(test)] +mod tests { + use std::{ + collections::BTreeMap, + sync::{Arc, Mutex}, + }; + + use git_internal::hash::HashKind; + use tracing::{ + Event, Metadata, Subscriber, + field::{Field, Visit}, + span::{Attributes, Id, Record}, + }; + + use super::*; + + const INSTANCE: &str = "11111111-2222-4333-8444-555555555559"; + + struct Fixture { + commit: ObjectHash, + tree: ObjectHash, + descriptor: ServingDescriptor, + snapshot_id: String, + metadata_root: String, + } + + impl Fixture { + fn new() -> Self { + let commit = ObjectHash::from_hex_for_kind(HashKind::Sha256, &"a".repeat(64)).unwrap(); + let tree = ObjectHash::from_hex_for_kind(HashKind::Sha256, &"b".repeat(64)).unwrap(); + let mut view_hash = Sha256::new(); + view_hash.update(b"mega.mst2.namespaceview\0"); + view_hash.update(commit.to_string().as_bytes()); + let descriptor = ServingDescriptor { + instance_uuid: *Uuid::parse_str(INSTANCE).unwrap().as_bytes(), + namespace_view_id: view_hash.finalize().into(), + scope: "/project".to_owned(), + metadata_root: [0xc; 32], + }; + Self { + commit, + tree, + snapshot_id: format!("sha256:{}", hex::encode(descriptor.snapshot_id().unwrap())), + metadata_root: format!("sha256:{}", hex::encode(descriptor.metadata_root)), + descriptor, + } + } + + fn source(&self) -> NativeResolveSource { + NativeResolveSource::capture(INSTANCE, self.commit, self.tree, Some(97), 3, 7).unwrap() + } + + fn resolved<'a>(&'a self, commit: &'a str, tree: &'a str) -> ResolvedProjection<'a> { + ResolvedProjection { + descriptor: &self.descriptor, + snapshot_id: &self.snapshot_id, + metadata_root: &self.metadata_root, + context_commit: commit, + context_root_tree: tree, + fixed_root_tree: self.tree, + requested_scope: "/project", + request_id: "resolve:request-1", + } + } + } + + #[test] + fn observation_binds_fixed_source_context_descriptor_and_operation_work() { + let fixture = Fixture::new(); + let commit = fixture.commit.to_string(); + let tree = fixture.tree.to_string(); + let work = ProjectionWork { + directories_rebuilt: 3, + reused_subtree_roots: 70, + directory_entries_scanned: 104, + verified_blob_misses: 1, + ..ProjectionWork::default() + }; + let observation = fixture + .source() + .observe( + fixture.resolved(&commit, &tree), + work.clone(), + Duration::from_micros(123), + ) + .unwrap(); + assert_eq!(observation.source.certificate_receipt_id, 97); + assert_eq!(observation.source.publication_sequence, 7); + assert_eq!(observation.source.writer_epoch, 3); + assert_eq!( + observation.source.tree.to_tagged_string(), + format!("sha256:{tree}") + ); + assert_eq!(observation.projection_elapsed_micros, 123); + assert_eq!(observation.work, work); + assert_eq!(observation.scope, "/project"); + assert_eq!(observation.snapshot_id, fixture.snapshot_id); + } + + #[test] + fn observation_rejects_each_changed_binding_or_unbounded_trace_token() { + let fixture = Fixture::new(); + let commit = fixture.commit.to_string(); + let tree = fixture.tree.to_string(); + for mode in 0..8 { + let mut descriptor = fixture.descriptor.clone(); + let mut resolved = fixture.resolved(&commit, &tree); + match mode { + 0 => resolved.context_commit = &tree, + 1 => resolved.context_root_tree = &commit, + 2 => resolved.fixed_root_tree = fixture.commit, + 3 => descriptor.instance_uuid[0] ^= 1, + 4 => descriptor.namespace_view_id[0] ^= 1, + 5 => resolved.requested_scope = "/different", + 6 => resolved.snapshot_id = &fixture.metadata_root, + 7 => resolved.metadata_root = &fixture.snapshot_id, + _ => unreachable!(), + } + resolved.descriptor = &descriptor; + assert!( + fixture + .source() + .observe(resolved, ProjectionWork::default(), Duration::ZERO) + .is_err(), + "mode {mode}" + ); + } + for request_id in [ + "".to_owned(), + "x".repeat(129), + "bad\ntrace".to_owned(), + "actor secret".to_owned(), + ] { + let mut resolved = fixture.resolved(&commit, &tree); + resolved.request_id = &request_id; + assert!( + fixture + .source() + .observe(resolved, ProjectionWork::default(), Duration::ZERO) + .is_err() + ); + } + } + + #[test] + fn capture_requires_positive_native_certificate_and_preserves_hash_kind() { + let fixture = Fixture::new(); + for (certificate, epoch, sequence) in [ + (None, 1, 1), + (Some(0), 1, 1), + (Some(-1), 1, 1), + (Some(1), 0, 1), + (Some(1), 1, 0), + ] { + assert!( + NativeResolveSource::capture( + INSTANCE, + fixture.commit, + fixture.tree, + certificate, + epoch, + sequence + ) + .is_err() + ); + } + let blake_tree = + ObjectHash::from_hex_for_kind(HashKind::Blake3, &fixture.tree.to_string()).unwrap(); + assert!( + NativeResolveSource::capture(INSTANCE, fixture.commit, blake_tree, Some(1), 1, 1) + .is_err() + ); + } + + struct TraceCapture(Arc>>>); + + impl Subscriber for TraceCapture { + fn enabled(&self, _: &Metadata<'_>) -> bool { + true + } + fn new_span(&self, _: &Attributes<'_>) -> Id { + Id::from_u64(1) + } + fn record(&self, _: &Id, _: &Record<'_>) {} + fn record_follows_from(&self, _: &Id, _: &Id) {} + fn enter(&self, _: &Id) {} + fn exit(&self, _: &Id) {} + fn event(&self, event: &Event<'_>) { + if event.metadata().target() != "mst2::native_projection_observation" { + return; + } + struct Fields(BTreeMap); + impl Visit for Fields { + fn record_debug(&mut self, field: &Field, value: &dyn std::fmt::Debug) { + self.0.insert(field.name().to_owned(), format!("{value:?}")); + } + fn record_u64(&mut self, field: &Field, value: u64) { + self.0.insert(field.name().to_owned(), value.to_string()); + } + fn record_str(&mut self, field: &Field, value: &str) { + self.0.insert(field.name().to_owned(), value.to_owned()); + } + } + let mut fields = Fields(BTreeMap::new()); + event.record(&mut fields); + self.0.lock().unwrap().push(fields.0); + } + } + + #[test] + fn actual_trace_has_closed_identity_profile_counter_fields_and_no_body_or_credentials() { + let fixture = Fixture::new(); + let commit = fixture.commit.to_string(); + let tree = fixture.tree.to_string(); + let observation = fixture + .source() + .observe( + fixture.resolved(&commit, &tree), + ProjectionWork::default(), + Duration::from_micros(4), + ) + .unwrap(); + let wire = serde_json::to_value(observation.wire_record()).unwrap(); + let events = Arc::new(Mutex::new(Vec::new())); + tracing::subscriber::with_default(TraceCapture(events.clone()), || observation.emit()); + let events = events.lock().unwrap(); + assert_eq!(events.len(), 1); + let fields = &events[0]; + let expected = [ + "observation_revision", + "phase", + "source_domain", + "request_id", + "instance_id", + "root_commit_oid", + "root_tree_oid", + "native_certificate_receipt_id", + "native_writer_epoch", + "native_publication_sequence", + "scope", + "schema_version", + "metadata_codec", + "materialization_policy", + "fs_semantics", + "access_projection", + "verification_revision", + "projection_revision", + "namespace_view_id", + "snapshot_id", + "metadata_root", + "projection_elapsed_micros", + "page_counter_scope", + "codec_radix_work", + "directories_rebuilt", + "reused_subtree_roots", + "directory_root_pages_built", + "directory_root_page_bytes_built", + "directory_root_pages_reused", + "directory_root_page_bytes_reused", + "directory_entries_scanned", + "scope_path_entries_examined", + "tree_fetches", + "verified_blob_hits", + "verified_blob_misses", + "raw_bytes_fetched", + "raw_bytes_hashed", + "message", + ]; + assert_eq!(fields.len(), expected.len()); + assert!(expected.iter().all(|field| fields.contains_key(*field))); + let wire = wire.as_object().unwrap(); + assert_eq!(wire.len(), expected.len()); + for key in expected { + let typed = &wire[key]; + let expected = if key == "scope" { + typed.to_string() + } else if let Some(text) = typed.as_str() { + text.into() + } else { + typed.to_string() + }; + assert_eq!(fields[key], expected, "typed observation diverged at {key}"); + } + assert_eq!(fields["phase"], "resolve_directory_projection"); + assert_eq!( + fields["page_counter_scope"], + "returned-directory-root-pages" + ); + assert_eq!(fields["codec_radix_work"], "NOT_EXPOSED"); + assert_eq!(fields["projection_elapsed_micros"], "4"); + assert_eq!(fields["native_certificate_receipt_id"], "97"); + assert_eq!(fields["scope"], "\"/project\""); + assert_eq!(fields["root_tree_oid"], fixture.tree.to_tagged_string()); + } + + #[tokio::test] + async fn emitted_observations_are_operation_local_with_no_publication_aggregation() { + let collect = |request: &'static str, sequence: i64| async move { + let fixture = Fixture::new(); + let commit = fixture.commit.to_string(); + let tree = fixture.tree.to_string(); + let mut resolved = fixture.resolved(&commit, &tree); + resolved.request_id = request; + let source = NativeResolveSource::capture( + INSTANCE, + fixture.commit, + fixture.tree, + Some(97), + 3, + sequence, + ) + .unwrap(); + source + .observe(resolved, ProjectionWork::default(), Duration::ZERO) + .unwrap() + .emit(); + }; + let (left, right) = tokio::join!( + with_observations(collect("request-left", 7)), + with_observations(collect("request-right", 8)) + ); + assert_eq!(left.1.len(), 1); + assert_eq!(right.1.len(), 1); + assert_eq!(left.1[0].request_id, "request-left"); + assert_eq!(right.1[0].request_id, "request-right"); + assert_eq!(left.1[0].source.publication_sequence, 7); + assert_eq!(right.1[0].source.publication_sequence, 8); + } +} diff --git a/src/ceres/snapshot/projection_writer.rs b/src/ceres/snapshot/projection_writer.rs new file mode 100644 index 00000000..a3d51852 --- /dev/null +++ b/src/ceres/snapshot/projection_writer.rs @@ -0,0 +1,506 @@ +//! Default-off typed projection observations. Buffer capacity is reserved +//! before serialization; logger filters and HTTP response bodies are unchanged. + +use std::{ + fs::{self, File, OpenOptions}, + io::{self, Write}, + path::{Path, PathBuf}, + sync::{ + Arc, Mutex, + atomic::{AtomicU8, AtomicU64, Ordering}, + mpsc::{self, Receiver, SyncSender, TrySendError}, + }, + thread::{self, JoinHandle}, + time::{Duration, Instant}, +}; + +use serde::Serialize; +use sha2::{Digest, Sha256}; +use uuid::Uuid; + +use super::projection_observation::NativeProjectionObservation; + +pub(crate) const RECORD_BYTES: usize = 32 * 1024; +pub(crate) const RECORD_LIMIT: usize = 64; +const STATUS_BYTES: usize = 4096; + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +#[repr(u8)] +pub(crate) enum WriterFailure { + QueueSaturated = 1, + WriterUnavailable, + RecordTooLarge, + RecordLimitExceeded, + ObservationBindingRejected, + IoFailure, + WorkerPanic, + DrainTimeout, +} + +impl WriterFailure { + fn label(self) -> &'static str { + match self { + Self::QueueSaturated => "QUEUE_SATURATED", + Self::WriterUnavailable => "WRITER_UNAVAILABLE", + Self::RecordTooLarge => "RECORD_TOO_LARGE", + Self::RecordLimitExceeded => "RECORD_LIMIT_EXCEEDED", + Self::ObservationBindingRejected => "OBSERVATION_BINDING_REJECTED", + Self::IoFailure => "IO_FAILURE", + Self::WorkerPanic => "WORKER_PANIC", + Self::DrainTimeout => "DRAIN_TIMEOUT", + } + } +} + +struct Health { + first_error: AtomicU8, + accepted: AtomicU64, +} + +impl Health { + fn fail(&self, error: WriterFailure) { + if self + .first_error + .compare_exchange(0, error as u8, Ordering::SeqCst, Ordering::SeqCst) + .is_ok() + { + tracing::warn!(target: "mst2::projection_writer", failure_code = error.label(), "projection observation delivery failed"); + } + } +} + +// Exactly 64 preallocated payload slots exist, including the slot currently +// being serialized and the worker's slot. No serialize-to-Vec-then-check path. +struct RecordBuffer { + bytes: Box<[u8; CAPACITY]>, + len: usize, +} + +impl Write for RecordBuffer { + fn write(&mut self, bytes: &[u8]) -> io::Result { + let end = self + .len + .checked_add(bytes.len()) + .filter(|end| *end <= CAPACITY) + .ok_or_else(|| io::Error::other("projection record capacity exhausted"))?; + self.bytes[self.len..end].copy_from_slice(bytes); + self.len = end; + Ok(bytes.len()) + } + fn flush(&mut self) -> io::Result<()> { + Ok(()) + } +} + +struct QueuedRecord { + sequence: u64, + buffer: RecordBuffer, +} +type Pool = Arc>>; +struct Producer { + sender: Option>, +} + +pub(crate) struct ProjectionObservationSink { + id: Uuid, + producer: Mutex, + pool: Pool, + health: Arc, + worker: Mutex>>, +} + +impl ProjectionObservationSink { + pub(crate) fn start(cache: &Path) -> io::Result> { + let id = Uuid::new_v4(); + let root = prepare_directory(cache, id)?; + let records = create_private(&root.join("records.jsonl"))?; + let health = Arc::new(Health { + first_error: AtomicU8::new(0), + accepted: AtomicU64::new(0), + }); + let pool = Arc::new(Mutex::new( + (0..RECORD_LIMIT) + .map(|_| RecordBuffer { + bytes: Box::new([0; RECORD_BYTES]), + len: 0, + }) + .collect::>(), + )); + let mut writer = FileWriter::new(root, records, id, health.clone()); + writer.status(false)?; + let (sender, receiver) = mpsc::sync_channel(RECORD_LIMIT); + let worker_pool = pool.clone(); + let worker_health = health.clone(); + let worker = thread::Builder::new() + .name("mst2-projection-writer".into()) + .spawn(move || { + let outcome = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| { + writer.run(receiver, worker_pool); + })); + if outcome.is_err() { + worker_health.fail(WriterFailure::WorkerPanic); + let _ = writer.status(true); + } + })?; + tracing::info!(target: "mst2::projection_writer", writer_revision = 1u16, sink_instance = %id, + payload_slots = RECORD_LIMIT, payload_slot_bytes = RECORD_BYTES, + "typed projection writer started"); + Ok(Arc::new(Self { + id, + producer: Mutex::new(Producer { + sender: Some(sender), + }), + pool, + health, + worker: Mutex::new(Some(worker)), + })) + } + + pub(crate) fn reject_binding(&self) { + self.health.fail(WriterFailure::ObservationBindingRejected); + } + + pub(crate) fn enqueue( + &self, + observation: &NativeProjectionObservation, + ) -> Result<(), WriterFailure> { + let result = self.enqueue_inner(observation); + if let Err(error) = result { + self.health.fail(error); + } + result + } + + fn enqueue_inner( + &self, + observation: &NativeProjectionObservation, + ) -> Result<(), WriterFailure> { + if self.health.first_error.load(Ordering::SeqCst) != 0 { + return Err(WriterFailure::WriterUnavailable); + } + let producer = self + .producer + .try_lock() + .map_err(|_| WriterFailure::QueueSaturated)?; + let sender = producer + .sender + .as_ref() + .ok_or(WriterFailure::WriterUnavailable)?; + let sequence = self.health.accepted.load(Ordering::SeqCst) + 1; + if sequence > RECORD_LIMIT as u64 { + return Err(WriterFailure::RecordLimitExceeded); + } + let mut buffer = self + .pool + .try_lock() + .map_err(|_| WriterFailure::QueueSaturated)? + .pop() + .ok_or(WriterFailure::QueueSaturated)?; + buffer.len = 0; + let serialized = (|| -> io::Result<()> { + write!( + buffer, + "{{\"writer_revision\":1,\"sink_instance\":\"{}\",\"record_sequence\":{},\"payload\":", + self.id, sequence + )?; + let begin = buffer.len; + serde_json::to_writer(&mut buffer, &observation.wire_record()) + .map_err(io::Error::other)?; + let digest = Sha256::digest(&buffer.bytes[begin..buffer.len]); + writeln!( + buffer, + ",\"payload_sha256\":\"sha256:{}\"}}", + hex::encode(digest) + ) + })(); + if serialized.is_err() { + self.return_slot(buffer); + return Err(WriterFailure::RecordTooLarge); + } + self.health.accepted.store(sequence, Ordering::SeqCst); + match sender.try_send(QueuedRecord { sequence, buffer }) { + Ok(()) => Ok(()), + Err(error) => { + self.health.accepted.store(sequence - 1, Ordering::SeqCst); + let (error, record) = match error { + TrySendError::Full(record) => (WriterFailure::QueueSaturated, record), + TrySendError::Disconnected(record) => { + (WriterFailure::WriterUnavailable, record) + } + }; + self.return_slot(record.buffer); + Err(error) + } + } + } + + fn return_slot(&self, mut buffer: RecordBuffer) { + buffer.len = 0; + if let Ok(mut pool) = self.pool.lock() { + pool.push(buffer); + } + } + + pub(crate) async fn shutdown(&self, deadline: Instant) -> Result<(), WriterFailure> { + self.producer + .lock() + .map_err(|_| WriterFailure::WriterUnavailable)? + .sender + .take(); + loop { + let finished = self + .worker + .lock() + .map_err(|_| WriterFailure::WriterUnavailable)? + .as_ref() + .is_none_or(JoinHandle::is_finished); + if finished { + break; + } + if Instant::now() >= deadline { + self.health.fail(WriterFailure::DrainTimeout); + return Err(WriterFailure::DrainTimeout); + } + tokio::time::sleep( + Duration::from_millis(10).min(deadline.saturating_duration_since(Instant::now())), + ) + .await; + } + if let Some(worker) = self + .worker + .lock() + .map_err(|_| WriterFailure::WriterUnavailable)? + .take() + && worker.join().is_err() + { + self.health.fail(WriterFailure::WorkerPanic); + } + if self.health.first_error.load(Ordering::SeqCst) != 0 { + return Err(WriterFailure::WriterUnavailable); + } + if Instant::now() >= deadline { + self.health.fail(WriterFailure::DrainTimeout); + return Err(WriterFailure::DrainTimeout); + } + Ok(()) + } + + #[cfg(test)] + pub(crate) fn test_directory(&self, cache: &Path) -> PathBuf { + cache + .join("logs/mst2-native-projection") + .join(self.id.to_string()) + } +} + +#[derive(Serialize)] +struct WriterStatus { + writer_revision: u16, + sink_instance: String, + accepted_records: u64, + written_sequence: u64, + written_records: u64, + written_bytes: u64, + rolling_sha256: String, + first_error_code: u8, + closed: bool, +} + +struct FileWriter { + root: PathBuf, + records: File, + id: Uuid, + health: Arc, + sequence: u64, + bytes: u64, + digest: Sha256, +} +impl FileWriter { + fn new(root: PathBuf, records: File, id: Uuid, health: Arc) -> Self { + Self { + root, + records, + id, + health, + sequence: 0, + bytes: 0, + digest: Sha256::new(), + } + } + fn status(&self, closed: bool) -> io::Result<()> { + let status = WriterStatus { + writer_revision: 1, + sink_instance: self.id.to_string(), + accepted_records: self.health.accepted.load(Ordering::SeqCst), + written_sequence: self.sequence, + written_records: self.sequence, + written_bytes: self.bytes, + rolling_sha256: format!("sha256:{}", hex::encode(self.digest.clone().finalize())), + first_error_code: self.health.first_error.load(Ordering::SeqCst), + closed, + }; + let mut buffer = RecordBuffer:: { + bytes: Box::new([0; STATUS_BYTES]), + len: 0, + }; + serde_json::to_writer(&mut buffer, &status).map_err(io::Error::other)?; + let temporary = self.root.join("status.tmp"); + let mut file = create_private(&temporary)?; + file.write_all(&buffer.bytes[..buffer.len])?; + file.sync_all()?; + fs::rename(temporary, self.root.join("status.json"))?; + sync_directory(&self.root) + } + fn run(&mut self, receiver: Receiver, pool: Pool) { + let mut reported_error = 0; + loop { + match receiver.recv_timeout(Duration::from_millis(100)) { + Ok(mut record) => { + let result = (|| -> io::Result<()> { + if record.sequence != self.sequence + 1 + || record.sequence > RECORD_LIMIT as u64 + { + return Err(io::Error::other("projection writer sequence mismatch")); + } + self.records + .write_all(&record.buffer.bytes[..record.buffer.len])?; + self.records.flush()?; + self.records.sync_all()?; + self.digest + .update(&record.buffer.bytes[..record.buffer.len]); + self.bytes += record.buffer.len as u64; + self.sequence = record.sequence; + self.status(false) + })(); + record.buffer.len = 0; + if let Ok(mut slots) = pool.lock() { + slots.push(record.buffer); + } + if result.is_err() { + self.health.fail(WriterFailure::IoFailure); + break; + } + } + Err(mpsc::RecvTimeoutError::Timeout) => {} + Err(mpsc::RecvTimeoutError::Disconnected) => break, + } + let error = self.health.first_error.load(Ordering::SeqCst); + if error != 0 && error != reported_error { + if self.status(false).is_err() { + self.health.fail(WriterFailure::IoFailure); + break; + } + reported_error = error; + } + } + if self.status(true).is_err() { + self.health.fail(WriterFailure::IoFailure); + } + } +} + +fn create_private(path: &Path) -> io::Result { + let mut options = OpenOptions::new(); + options.write(true).create_new(true); + #[cfg(unix)] + { + use std::os::unix::fs::OpenOptionsExt; + options.mode(0o600); + } + options.open(path) +} + +fn prepare_directory(cache: &Path, id: Uuid) -> io::Result { + prepare_directory_with(cache, id, sync_directory) +} + +fn real_directory(path: &Path) -> io::Result<()> { + if !fs::symlink_metadata(path)?.is_dir() { + return Err(io::Error::other( + "projection storage must be a real directory", + )); + } + Ok(()) +} + +fn create_directory(path: &Path, private: bool) -> io::Result<()> { + #[cfg(unix)] + let mut directory = fs::DirBuilder::new(); + #[cfg(not(unix))] + let directory = fs::DirBuilder::new(); + #[cfg(unix)] + { + use std::os::unix::fs::DirBuilderExt; + directory.mode(if private { 0o700 } else { 0o755 }); + } + #[cfg(not(unix))] + let _ = private; + directory.create(path) +} + +fn private_directory(path: &Path) -> io::Result<()> { + real_directory(path)?; + #[cfg(unix)] + { + use std::os::unix::fs::{MetadataExt, PermissionsExt}; + let metadata = fs::symlink_metadata(path)?; + // SAFETY: geteuid reads the process identity and has no pointer arguments. + let owner = unsafe { libc::geteuid() }; + if metadata.uid() != owner || metadata.permissions().mode() & 0o7777 != 0o700 { + return Err(io::Error::other( + "projection directory must be privately owned", + )); + } + } + Ok(()) +} + +fn prepare_directory_with( + cache: &Path, + id: Uuid, + sync: impl Fn(&Path) -> io::Result<()>, +) -> io::Result { + // The actual configured cache is an existing anchor. No arbitrary output + // path or recursively created ancestor is accepted by this writer. + real_directory(cache)?; + let logs = cache.join("logs"); + match create_directory(&logs, false) { + Ok(()) => {} + Err(error) if error.kind() == io::ErrorKind::AlreadyExists => {} + Err(error) => return Err(error), + } + real_directory(&logs)?; + sync(cache)?; + let parent = logs.join("mst2-native-projection"); + match create_directory(&parent, true) { + Ok(()) => {} + Err(error) if error.kind() == io::ErrorKind::AlreadyExists => {} + Err(error) => return Err(error), + } + private_directory(&parent)?; + sync(&logs)?; + let root = parent.join(id.to_string()); + create_directory(&root, true)?; + private_directory(&root)?; + sync(&parent)?; + Ok(root) +} + +fn sync_directory(path: &Path) -> io::Result<()> { + #[cfg(unix)] + { + File::open(path)?.sync_all() + } + #[cfg(not(unix))] + { + let _ = path; + Err(io::Error::new( + io::ErrorKind::Unsupported, + "durable projection writer requires directory fsync", + )) + } +} + +#[cfg(test)] +#[path = "projection_writer_tests.rs"] +mod tests; diff --git a/src/ceres/snapshot/projection_writer_tests.rs b/src/ceres/snapshot/projection_writer_tests.rs new file mode 100644 index 00000000..2de737a2 --- /dev/null +++ b/src/ceres/snapshot/projection_writer_tests.rs @@ -0,0 +1,358 @@ +use git_internal::hash::{HashKind, ObjectHash}; +use mst2_codec::descriptor::ServingDescriptor; + +use super::*; +use crate::ceres::snapshot::projection_observation::{ + NativeResolveSource, ProjectionWork, ResolvedProjection, +}; + +fn observation(scope: &str) -> NativeProjectionObservation { + let commit = ObjectHash::from_hex_for_kind(HashKind::Sha256, &"a".repeat(64)).unwrap(); + let tree = ObjectHash::from_hex_for_kind(HashKind::Sha256, &"b".repeat(64)).unwrap(); + let instance = Uuid::new_v4(); + let mut namespace = Sha256::new(); + namespace.update(b"mega.mst2.namespaceview\0"); + namespace.update(commit.to_string().as_bytes()); + let descriptor = ServingDescriptor { + instance_uuid: *instance.as_bytes(), + namespace_view_id: namespace.finalize().into(), + scope: scope.into(), + metadata_root: [3; 32], + }; + let snapshot = format!("sha256:{}", hex::encode(descriptor.snapshot_id().unwrap())); + let root = format!("sha256:{}", hex::encode(descriptor.metadata_root)); + NativeResolveSource::capture(&instance.to_string(), commit, tree, Some(1), 2, 3) + .unwrap() + .observe( + ResolvedProjection { + descriptor: &descriptor, + snapshot_id: &snapshot, + metadata_root: &root, + context_commit: &commit.to_string(), + context_root_tree: &tree.to_string(), + fixed_root_tree: tree, + requested_scope: scope, + request_id: "writer:test:a1", + }, + ProjectionWork::default(), + Duration::from_micros(4), + ) + .unwrap() +} + +fn idle_sink() -> (ProjectionObservationSink, Receiver) { + let (sender, receiver) = mpsc::sync_channel(RECORD_LIMIT); + let sink = ProjectionObservationSink { + id: Uuid::new_v4(), + producer: Mutex::new(Producer { + sender: Some(sender), + }), + pool: Arc::new(Mutex::new( + (0..RECORD_LIMIT) + .map(|_| RecordBuffer { + bytes: Box::new([0; RECORD_BYTES]), + len: 0, + }) + .collect(), + )), + health: Arc::new(Health { + first_error: AtomicU8::new(0), + accepted: AtomicU64::new(0), + }), + worker: Mutex::new(None), + }; + (sink, receiver) +} + +#[test] +fn bounded_slots_precede_serialization_and_saturation_is_sticky_with_no_extra_allocation() { + let (sink, receiver) = idle_sink(); + let observation = observation("/project"); + for _ in 0..RECORD_LIMIT { + sink.enqueue(&observation).unwrap(); + } + assert_eq!(sink.pool.lock().unwrap().len(), 0); + assert_eq!( + sink.health.accepted.load(Ordering::SeqCst), + RECORD_LIMIT as u64 + ); + assert_eq!( + sink.enqueue(&observation), + Err(WriterFailure::RecordLimitExceeded) + ); + assert_eq!( + sink.health.first_error.load(Ordering::SeqCst), + WriterFailure::RecordLimitExceeded as u8 + ); + let record = receiver.try_recv().unwrap(); + sink.return_slot(record.buffer); + assert_eq!( + sink.enqueue(&observation), + Err(WriterFailure::WriterUnavailable) + ); + assert_eq!(sink.pool.lock().unwrap().len(), 1); +} + +#[test] +fn pool_saturation_and_disconnection_have_typed_failure_and_return_the_actual_slot() { + let (sink, receiver) = idle_sink(); + let observation = observation("/project"); + let held = sink.pool.lock().unwrap().drain(..).collect::>(); + assert_eq!( + sink.enqueue(&observation), + Err(WriterFailure::QueueSaturated) + ); + assert_eq!(held.len(), RECORD_LIMIT); + assert_eq!(sink.health.accepted.load(Ordering::SeqCst), 0); + let (sink, receiver2) = idle_sink(); + drop(receiver2); + assert_eq!( + sink.enqueue(&observation), + Err(WriterFailure::WriterUnavailable) + ); + assert_eq!(sink.pool.lock().unwrap().len(), RECORD_LIMIT); + assert_eq!(sink.health.accepted.load(Ordering::SeqCst), 0); + drop(receiver); +} + +#[test] +fn typed_wire_contract_has_exact_38_fields_and_rejects_oversize_without_growing_buffer() { + let (sink, receiver) = idle_sink(); + let captured = observation("/project\"quoted"); + sink.enqueue(&captured).unwrap(); + let record = receiver.try_recv().unwrap(); + assert_eq!(record.buffer.bytes.len(), RECORD_BYTES); + let envelope: serde_json::Value = + serde_json::from_slice(&record.buffer.bytes[..record.buffer.len]).unwrap(); + let payload = &envelope["payload"]; + assert_eq!(payload.as_object().unwrap().len(), 38); + assert_eq!(payload["scope"], "/project\"quoted"); + assert_eq!(payload["request_id"], "writer:test:a1"); + assert_eq!(payload["projection_elapsed_micros"], 4); + assert_eq!(payload["codec_radix_work"], "NOT_EXPOSED"); + let mut small = RecordBuffer::<64> { + bytes: Box::new([0; 64]), + len: 0, + }; + // Serialize a genuine validated observation into an insufficient reserved + // slot, rather than bypassing the descriptor's component length checks. + assert!(serde_json::to_writer(&mut small, &captured.wire_record()).is_err()); + assert!(small.len <= 64); + assert_eq!(small.bytes.len(), 64); + let len = small.len; + assert!(small.write_all(&[0; 65]).is_err()); + assert_eq!(small.len, len); + let wire = &record.buffer.bytes[..record.buffer.len]; + let prefix = b"\"payload\":"; + let begin = wire + .windows(prefix.len()) + .position(|part| part == prefix) + .unwrap() + + prefix.len(); + let suffix = b",\"payload_sha256\":"; + let end = wire + .windows(suffix.len()) + .position(|part| part == suffix) + .unwrap(); + assert_eq!( + envelope["payload_sha256"], + format!("sha256:{}", hex::encode(Sha256::digest(&wire[begin..end]))) + ); +} + +#[cfg(unix)] +#[tokio::test] +async fn actual_writer_fsync_ack_tracks_exact_bytes_and_closed_drain() { + let temp = tempfile::tempdir().unwrap(); + let sink = ProjectionObservationSink::start(temp.path()).unwrap(); + sink.enqueue(&observation("/project")).unwrap(); + sink.shutdown(Instant::now() + Duration::from_secs(5)) + .await + .unwrap(); + let root = temp + .path() + .join("logs/mst2-native-projection") + .join(sink.id.to_string()); + let wire = fs::read(root.join("records.jsonl")).unwrap(); + let status: serde_json::Value = + serde_json::from_slice(&fs::read(root.join("status.json")).unwrap()).unwrap(); + assert_eq!(status["written_bytes"], wire.len()); + assert_eq!(status["written_sequence"], 1); + assert_eq!(status["accepted_records"], 1); + assert_eq!(status["first_error_code"], 0); + assert_eq!(status["closed"], true); + assert_eq!( + status["rolling_sha256"], + format!("sha256:{}", hex::encode(Sha256::digest(&wire))) + ); + let record: serde_json::Value = serde_json::from_slice(&wire).unwrap(); + assert_eq!(record["payload"].as_object().unwrap().len(), 38); + assert_eq!(sink.pool.lock().unwrap().len(), RECORD_LIMIT); + use std::os::unix::fs::PermissionsExt; + assert_eq!( + fs::metadata(&root).unwrap().permissions().mode() & 0o777, + 0o700 + ); + assert_eq!( + fs::metadata(root.parent().unwrap()) + .unwrap() + .permissions() + .mode() + & 0o7777, + 0o700 + ); + for path in [root.join("records.jsonl"), root.join("status.json")] { + assert_eq!( + fs::metadata(path).unwrap().permissions().mode() & 0o777, + 0o600 + ); + } +} + +#[tokio::test] +async fn expired_drain_deadline_cannot_report_a_successful_capture() { + let (sink, _) = idle_sink(); + assert_eq!( + sink.shutdown(Instant::now()).await, + Err(WriterFailure::DrainTimeout) + ); + assert_eq!( + sink.health.first_error.load(Ordering::SeqCst), + WriterFailure::DrainTimeout as u8 + ); + assert_eq!( + sink.enqueue(&observation("/project")), + Err(WriterFailure::WriterUnavailable) + ); +} + +#[cfg(unix)] +#[tokio::test] +async fn actual_status_write_failure_and_binding_rejection_cannot_drain_successfully() { + let temp = tempfile::tempdir().unwrap(); + let sink = ProjectionObservationSink::start(temp.path()).unwrap(); + let root = temp + .path() + .join("logs/mst2-native-projection") + .join(sink.id.to_string()); + fs::write(root.join("status.tmp"), b"blocked exclusive temporary").unwrap(); + sink.enqueue(&observation("/project")).unwrap(); + assert!( + sink.shutdown(Instant::now() + Duration::from_secs(5)) + .await + .is_err() + ); + let status: serde_json::Value = + serde_json::from_slice(&fs::read(root.join("status.json")).unwrap()).unwrap(); + assert_eq!( + status["written_sequence"], 0, + "no durable sidecar acknowledgement for this record" + ); + assert_eq!( + sink.health.first_error.load(Ordering::SeqCst), + WriterFailure::IoFailure as u8 + ); + let temp = tempfile::tempdir().unwrap(); + let sink = ProjectionObservationSink::start(temp.path()).unwrap(); + sink.reject_binding(); + assert!( + sink.shutdown(Instant::now() + Duration::from_secs(5)) + .await + .is_err() + ); + let root = temp + .path() + .join("logs/mst2-native-projection") + .join(sink.id.to_string()); + let status: serde_json::Value = + serde_json::from_slice(&fs::read(root.join("status.json")).unwrap()).unwrap(); + assert_eq!( + status["first_error_code"], + WriterFailure::ObservationBindingRejected as u8 + ); +} + +#[cfg(unix)] +#[test] +fn actual_writer_rejects_a_symlinked_output_directory_before_creating_records() { + use std::os::unix::fs::symlink; + let temp = tempfile::tempdir().unwrap(); + let target = tempfile::tempdir().unwrap(); + fs::create_dir(temp.path().join("logs")).unwrap(); + symlink( + target.path(), + temp.path().join("logs/mst2-native-projection"), + ) + .unwrap(); + assert!(ProjectionObservationSink::start(temp.path()).is_err()); + assert_eq!(target.path().read_dir().unwrap().count(), 0); + let temp = tempfile::tempdir().unwrap(); + symlink(target.path(), temp.path().join("logs")).unwrap(); + assert!(ProjectionObservationSink::start(temp.path()).is_err()); + assert_eq!(target.path().read_dir().unwrap().count(), 0); +} + +#[cfg(unix)] +#[test] +fn directory_chain_is_synced_from_existing_anchor_before_any_record_can_be_created() { + use std::cell::RefCell; + let temp = tempfile::tempdir().unwrap(); + let calls = RefCell::new(Vec::new()); + let root = prepare_directory_with(temp.path(), Uuid::new_v4(), |path| { + sync_directory(path)?; + calls.borrow_mut().push(path.to_path_buf()); + Ok(()) + }) + .unwrap(); + assert_eq!( + *calls.borrow(), + [ + temp.path().to_path_buf(), + temp.path().join("logs"), + temp.path().join("logs/mst2-native-projection") + ] + ); + assert_eq!(root.read_dir().unwrap().count(), 0); + for fail_at in 0..3 { + let temp = tempfile::tempdir().unwrap(); + let id = Uuid::new_v4(); + let calls = RefCell::new(0); + assert!( + prepare_directory_with(temp.path(), id, |path| { + let current = *calls.borrow(); + *calls.borrow_mut() += 1; + if current == fail_at { + Err(io::Error::other("injected directory durability failure")) + } else { + sync_directory(path) + } + }) + .is_err() + ); + assert_eq!(*calls.borrow(), fail_at + 1); + let root = temp + .path() + .join("logs/mst2-native-projection") + .join(id.to_string()); + assert!(!root.join("records.jsonl").exists()); + assert!(!root.join("status.json").exists()); + } +} + +#[cfg(unix)] +#[test] +fn missing_or_symlink_anchor_and_nonprivate_existing_parent_fail_before_records() { + use std::os::unix::fs::{PermissionsExt, symlink}; + let temp = tempfile::tempdir().unwrap(); + assert!(ProjectionObservationSink::start(&temp.path().join("missing")).is_err()); + assert!(!temp.path().join("missing").exists()); + let alias = temp.path().join("alias"); + symlink(temp.path(), &alias).unwrap(); + assert!(ProjectionObservationSink::start(&alias).is_err()); + assert!(!temp.path().join("logs").exists()); + let parent = temp.path().join("logs/mst2-native-projection"); + fs::create_dir_all(&parent).unwrap(); + fs::set_permissions(&parent, fs::Permissions::from_mode(0o755)).unwrap(); + assert!(ProjectionObservationSink::start(temp.path()).is_err()); + assert_eq!(parent.read_dir().unwrap().count(), 0); +} diff --git a/src/ceres/snapshot/retention.rs b/src/ceres/snapshot/retention.rs index 54bba4da..fa9a12f6 100644 --- a/src/ceres/snapshot/retention.rs +++ b/src/ceres/snapshot/retention.rs @@ -105,7 +105,7 @@ pub enum ReapDecision { #[derive(Debug, Clone, PartialEq, Eq, Default)] pub struct CollectionReport { - /// Node ids that became unreachable this run. + /// Node ids atomically marked DELETING this run. pub unreachable: Vec, /// Bytes that would be / were reclaimed. pub reclaimed_bytes: u64, @@ -140,16 +140,30 @@ impl Reaper for NoopReaper { pub trait RetentionStore { fn node(&self, id: &str) -> Option; fn root_covers(&self, node_id: &str) -> bool; + /// Count incoming edges from parents not yet removed, including + /// DELETING parents whose physical reclaim has not succeeded. fn live_incoming(&self, node_id: &str) -> usize; - /// Idempotent: create the node if absent, upsert edges and roots in - /// one atomic step (spec §6 "no empty window"). + /// Idempotent: retain one LIVE node with its edges and root coverage. fn retain( &self, node: RetentionNode, edges: &[RetentionEdge], roots: &[RetentionRoot], + ) -> Result<(), SnapshotError> { + self.retain_group(std::slice::from_ref(&node), edges, roots) + } + /// Atomically retain the entire group, covering every supplied node + /// with `roots`. All nodes and edge endpoints must be LIVE; acquiring + /// a DELETING node is forbidden. Any error leaves the graph unchanged. + fn retain_group( + &self, + nodes: &[RetentionNode], + edges: &[RetentionEdge], + roots: &[RetentionRoot], ) -> Result<(), SnapshotError>; - /// Atomically CAS a node LIVE→DELETING; false if it is no longer LIVE. + /// Atomically check zero roots/incoming references and CAS LIVE→DELETING. + /// A parent still protects its children until the parent is removed, + /// including while the parent is DELETING and physical reclaim is pending. fn mark_deleting(&self, id: &str) -> bool; /// Remove a DELETING node after the reaper succeeds. fn remove(&self, id: &str); @@ -182,14 +196,6 @@ impl RetentionCoordinator { parents: &[String], roots: &[RetentionRoot], ) -> Result<(), SnapshotError> { - for p in parents { - if self.store.node(p).is_none() { - return Err(SnapshotError::new( - SnapshotErrorCode::Internal, - format!("retention edge to unknown parent {p}"), - )); - } - } let edges: Vec = parents .iter() .map(|p| RetentionEdge { @@ -200,38 +206,24 @@ impl RetentionCoordinator { self.store.retain(node, &edges, roots) } - /// Add coverage from a root to a node, atomically re-lifting it out of - /// a DELETING state (spec §10). The root is also materialized as a live - /// anchor node so reachability can start from it. The store's `retain` - /// is the upsert, so both the fresh and the already-known cases take - /// the same path. + /// Acquire root coverage for the complete group in one store operation. + /// The root is also materialized as a LIVE anchor. If any covered node + /// is DELETING, neither the anchor nor any partial coverage is retained. pub fn pin_root( &self, root: &RetentionRoot, covered: &[RetentionNode], ) -> Result<(), SnapshotError> { - let roots = std::slice::from_ref(root); - self.store.retain( - RetentionNode { - id: root.key(), - kind: RetainedKind::Frame, - state: NodeState::Live, - bytes: 0, - }, - &[], - roots, - )?; - for n in covered { - self.store.retain( - RetentionNode { - state: NodeState::Live, - ..n.clone() - }, - &[], - roots, - )?; - } - Ok(()) + let mut nodes = Vec::with_capacity(covered.len() + 1); + nodes.push(RetentionNode { + id: root.key(), + kind: RetainedKind::Frame, + state: NodeState::Live, + bytes: 0, + }); + nodes.extend_from_slice(covered); + self.store + .retain_group(&nodes, &[], std::slice::from_ref(root)) } pub fn release(&self, root: &RetentionRoot) { @@ -239,11 +231,10 @@ impl RetentionCoordinator { } /// One collection pass (spec 10 §6 steps 1–7): - /// mark unreachable LIVE nodes DELETING, then reap. Nodes marked - /// DELETING are skipped entirely on later passes if a new root/edge - /// raced in only after the CAS — here reachability is recomputed under - /// the store lock, so a retain concurrent with a pass either observes - /// the node (keeping it LIVE) or lands before the next pass. + /// mark unreachable LIVE nodes DELETING, then reap. The reachability + /// scan selects candidates; the store rechecks roots and incoming + /// references atomically with the CAS. Acquisition either wins before + /// that CAS or fails because the node is already DELETING. pub fn collect(&self, reaper: &R) -> Result { let live = self.store.all_live(); let reachable = self.reachable_set(); @@ -254,18 +245,11 @@ impl RetentionCoordinator { continue; } if !self.store.mark_deleting(&node.id) { - // Lost the CAS: a new reference made it LIVE again. + // Another collector or a newly acquired reference won. continue; } report.unreachable.push(node.id.clone()); report.reclaimed_bytes += node.bytes; - // Re-verify after the CAS: the store serializes retain against - // mark, so a zero incoming count here is stable for this pass. - if self.store.root_covers(&node.id) || self.store.live_incoming(&node.id) > 0 { - // A concurrent retain resurrected it; the CAS semantics of - // the store keep it LIVE, so do not reap. - continue; - } if reaper.physical() { reaper.reap(&node)?; self.store.remove(&node.id); @@ -311,6 +295,8 @@ pub mod mem { edges: HashSet, /// node id -> set of root keys covering it. roots: HashMap>, + #[cfg(test)] + fail_next_retain: bool, } /// Single-Mutex store; the lock is the serialization point that makes @@ -327,6 +313,14 @@ pub mod mem { } } + #[cfg(test)] + impl InMemoryRetentionStore { + /// Inject a backend failure after validation and before committing. + pub(crate) fn fail_next_retain_for_test(&self) { + self.inner.lock().unwrap().fail_next_retain = true; + } + } + impl RetentionStore for InMemoryRetentionStore { fn node(&self, id: &str) -> Option { self.inner.lock().unwrap().nodes.get(id).cloned() @@ -350,41 +344,82 @@ pub mod mem { .count() } - fn retain( + fn retain_group( &self, - node: RetentionNode, + nodes: &[RetentionNode], edges: &[RetentionEdge], roots: &[RetentionRoot], ) -> Result<(), SnapshotError> { let mut g = self.inner.lock().unwrap(); - // Atomic re-lift: never downgrade a LIVE node to DELETING. - g.nodes - .entry(node.id.clone()) - .and_modify(|existing| { - existing.bytes = node.bytes; - existing.kind = node.kind; - existing.state = NodeState::Live; - }) - .or_insert_with(|| node.clone()); - // Defensive: do not insert an edge to a missing child node. - for e in edges { - if !g.nodes.contains_key(&e.child) || !g.nodes.contains_key(&e.parent) { + // Validate the entire transaction before mutating any graph data. + let mut staged = HashMap::new(); + for node in nodes { + if node.state != NodeState::Live + || g.nodes + .get(&node.id) + .is_some_and(|existing| existing.state != NodeState::Live) + { + return Err(SnapshotError::new( + SnapshotErrorCode::ObjectUnavailable, + format!("retention acquire requires LIVE node {}", node.id), + )); + } + if staged + .insert(node.id.as_str(), node) + .is_some_and(|previous| previous != node) + { return Err(SnapshotError::new( SnapshotErrorCode::Internal, - "retention edge references missing node", + "retention group has conflicting node definitions", )); } - g.edges.insert(e.clone()); } - let entry = g.roots.entry(node.id.clone()).or_default(); - for r in roots { - entry.insert(r.key()); + for e in edges { + for id in [&e.parent, &e.child] { + match staged.get(id.as_str()).copied().or_else(|| g.nodes.get(id)) { + Some(node) if node.state == NodeState::Live => {} + Some(_) => { + return Err(SnapshotError::new( + SnapshotErrorCode::ObjectUnavailable, + format!("retention edge requires LIVE endpoint {id}"), + )); + } + None => { + return Err(SnapshotError::new( + SnapshotErrorCode::Internal, + format!("retention edge references missing node {id}"), + )); + } + } + } + } + #[cfg(test)] + if std::mem::take(&mut g.fail_next_retain) { + return Err(SnapshotError::new( + SnapshotErrorCode::Internal, + "injected retention commit failure", + )); + } + for node in nodes { + g.nodes.insert(node.id.clone(), node.clone()); + let entry = g.roots.entry(node.id.clone()).or_default(); + for root in roots { + entry.insert(root.key()); + } } + g.edges.extend(edges.iter().cloned()); Ok(()) } fn mark_deleting(&self, id: &str) -> bool { let mut g = self.inner.lock().unwrap(); + if g.roots.get(id).is_some_and(|roots| !roots.is_empty()) + || g.edges + .iter() + .any(|edge| edge.child == id && g.nodes.contains_key(&edge.parent)) + { + return false; + } match g.nodes.get_mut(id) { Some(n) if n.state == NodeState::Live => { n.state = NodeState::Deleting; @@ -396,6 +431,13 @@ pub mod mem { fn remove(&self, id: &str) { let mut g = self.inner.lock().unwrap(); + if !g + .nodes + .get(id) + .is_some_and(|node| node.state == NodeState::Deleting) + { + return; + } g.nodes.remove(id); g.edges.retain(|e| e.parent != id && e.child != id); g.roots.remove(id); @@ -435,6 +477,15 @@ pub mod mem { #[cfg(test)] mod tests { + use std::{ + sync::{ + Arc, + mpsc::{Receiver, SyncSender, sync_channel}, + }, + thread, + time::Duration, + }; + use super::{mem::InMemoryRetentionStore, *}; fn node(id: &str, bytes: u64) -> RetentionNode { @@ -450,6 +501,243 @@ mod tests { RetentionCoordinator::new(InMemoryRetentionStore::default()) } + /// Pause a collection after its scan, immediately before the store CAS. + /// Timeouts make both sides of the forced interleaving bounded. + struct PausedMarkStore { + inner: InMemoryRetentionStore, + ready: SyncSender<()>, + resume: Mutex>, + } + + impl RetentionStore for PausedMarkStore { + fn node(&self, id: &str) -> Option { + self.inner.node(id) + } + + fn root_covers(&self, node_id: &str) -> bool { + self.inner.root_covers(node_id) + } + + fn live_incoming(&self, node_id: &str) -> usize { + self.inner.live_incoming(node_id) + } + + fn retain_group( + &self, + nodes: &[RetentionNode], + edges: &[RetentionEdge], + roots: &[RetentionRoot], + ) -> Result<(), SnapshotError> { + self.inner.retain_group(nodes, edges, roots) + } + + fn mark_deleting(&self, id: &str) -> bool { + self.ready.send(()).unwrap(); + self.resume + .lock() + .unwrap() + .recv_timeout(Duration::from_secs(5)) + .expect("test did not resume deletion"); + self.inner.mark_deleting(id) + } + + fn remove(&self, id: &str) { + self.inner.remove(id); + } + + fn release_root(&self, root: &RetentionRoot) { + self.inner.release_root(root); + } + + fn all_live(&self) -> Vec { + self.inner.all_live() + } + + fn children(&self, parent: &str) -> Vec { + self.inner.children(parent) + } + } + + #[test] + fn failed_pin_group_leaves_no_anchor_or_partial_coverage() { + let c = coord(); + let original = node("existing", 10); + c.store().retain(original.clone(), &[], &[]).unwrap(); + c.store().retain(node("deleting", 7), &[], &[]).unwrap(); + assert!(c.store().mark_deleting("deleting")); + let root = RetentionRoot::Lease("new".into()); + + // The last node fails after earlier entries would have been written + // by the old per-node pin loop, including an update to existing. + let err = c + .pin_root( + &root, + &[node("fresh", 1), node("existing", 99), node("deleting", 7)], + ) + .unwrap_err(); + assert_eq!(err.code, SnapshotErrorCode::ObjectUnavailable); + assert!(c.store().node(&root.key()).is_none()); + assert!(c.store().node("fresh").is_none()); + assert_eq!(c.store().node("existing"), Some(original)); + assert!(!c.store().root_covers("existing")); + assert!(!c.store().root_covers("deleting")); + assert_eq!( + c.store().node("deleting").unwrap().state, + NodeState::Deleting + ); + } + + #[test] + fn supplied_deleting_node_cannot_be_created_or_pinned() { + let c = coord(); + let root = RetentionRoot::Pin("invalid".into()); + let mut deleting = node("absent", 1); + deleting.state = NodeState::Deleting; + assert_eq!( + c.pin_root(&root, &[deleting]).unwrap_err().code, + SnapshotErrorCode::ObjectUnavailable + ); + assert!(c.store().all_live().is_empty()); + assert!(c.store().node("absent").is_none()); + assert!(c.store().node(&root.key()).is_none()); + } + + #[test] + fn late_invalid_edge_rolls_back_nodes_edges_and_roots() { + let c = coord(); + c.store().retain(node("parent", 0), &[], &[]).unwrap(); + let root = RetentionRoot::Lease("new".into()); + let edges = [ + RetentionEdge { + parent: "parent".into(), + child: "fresh".into(), + }, + RetentionEdge { + parent: "missing".into(), + child: "fresh".into(), + }, + ]; + let err = c + .store() + .retain(node("fresh", 2), &edges, &[root]) + .unwrap_err(); + assert_eq!(err.code, SnapshotErrorCode::Internal); + assert!(c.store().node("fresh").is_none()); + assert!(c.store().children("parent").is_empty()); + assert!(!c.store().root_covers("fresh")); + assert_eq!(c.store().live_incoming("fresh"), 0); + } + + #[test] + fn injected_commit_failure_preserves_existing_root_and_can_retry() { + let c = coord(); + let old = RetentionRoot::Pin("old".into()); + let new = RetentionRoot::Lease("new".into()); + c.store() + .retain(node("existing", 1), &[], std::slice::from_ref(&old)) + .unwrap(); + c.store().fail_next_retain_for_test(); + assert_eq!( + c.pin_root(&new, &[node("existing", 1), node("fresh", 2)]) + .unwrap_err() + .code, + SnapshotErrorCode::Internal + ); + assert!(c.store().node(&new.key()).is_none()); + assert!(c.store().node("fresh").is_none()); + assert!(c.store().root_covers("existing")); + c.release(&old); + assert!( + !c.store().root_covers("existing"), + "failed pin added no new root" + ); + + c.pin_root(&new, &[node("existing", 1), node("fresh", 2)]) + .unwrap(); + assert!(c.is_retained("existing")); + assert!(c.is_retained("fresh")); + } + + #[test] + fn collection_rechecks_roots_and_edges_acquired_after_its_scan() { + for via_edge in [false, true] { + let (ready_tx, ready_rx) = sync_channel(1); + let (resume_tx, resume_rx) = sync_channel(1); + let c = Arc::new(RetentionCoordinator::new(PausedMarkStore { + inner: InMemoryRetentionStore::default(), + ready: ready_tx, + resume: Mutex::new(resume_rx), + })); + c.store().retain(node("candidate", 1), &[], &[]).unwrap(); + let root = RetentionRoot::Lease("late".into()); + if via_edge { + c.store() + .retain(node("parent", 0), &[], std::slice::from_ref(&root)) + .unwrap(); + } + let collector = { + let c = Arc::clone(&c); + thread::spawn(move || c.collect(&NoopReaper)) + }; + ready_rx + .recv_timeout(Duration::from_secs(5)) + .expect("collector did not reach its deletion CAS"); + // Both references land after candidate selection, before CAS. + if via_edge { + c.retain(node("candidate", 1), &["parent".into()], &[]) + .unwrap(); + } else { + c.pin_root(&root, &[node("candidate", 1)]).unwrap(); + } + resume_tx.send(()).unwrap(); + let report = collector.join().unwrap().unwrap(); + assert!(report.unreachable.is_empty()); + assert_eq!(c.store().node("candidate").unwrap().state, NodeState::Live); + assert!(c.is_retained("candidate")); + } + } + + #[test] + fn deleting_parent_rejects_new_edges_and_protects_existing_children_until_removed() { + let c = coord(); + c.store().retain(node("parent", 0), &[], &[]).unwrap(); + c.retain(node("child", 1), &["parent".into()], &[]).unwrap(); + assert!(c.store().mark_deleting("parent")); + assert!(!c.store().mark_deleting("child")); + assert_eq!(c.store().live_incoming("child"), 1); + assert_eq!( + c.retain(node("new-child", 2), &["parent".into()], &[]) + .unwrap_err() + .code, + SnapshotErrorCode::ObjectUnavailable + ); + assert!(c.store().node("new-child").is_none()); + c.store().remove("parent"); + assert_eq!(c.store().live_incoming("child"), 0); + assert!(c.store().mark_deleting("child")); + } + + #[test] + fn retain_cannot_add_an_edge_to_a_deleting_child() { + let c = coord(); + c.store().retain(node("child", 1), &[], &[]).unwrap(); + assert!(c.store().mark_deleting("child")); + let edge = RetentionEdge { + parent: "new-parent".into(), + child: "child".into(), + }; + assert_eq!( + c.store() + .retain(node("new-parent", 0), &[edge], &[]) + .unwrap_err() + .code, + SnapshotErrorCode::ObjectUnavailable + ); + assert!(c.store().node("new-parent").is_none()); + assert_eq!(c.store().node("child").unwrap().state, NodeState::Deleting); + assert_eq!(c.store().live_incoming("child"), 0); + } + #[test] fn leaf_dies_when_root_released_but_other_root_keeps_it() { // root A and B both cover a shared leaf; releasing A must not @@ -486,8 +774,8 @@ mod tests { let r = c.collect(&NoopReaper).unwrap(); assert_eq!(r.unreachable, vec!["orphan".to_string()]); assert_eq!(r.reclaimed_bytes, 9); - // Releasing the root cascades unreachability to a and b only after - // the anchor itself is collected (the anchor node is root-covered). + // Releasing the root makes the chain unreachable, but children stay + // LIVE while their parents await physical removal under NoopReaper. c.release(&root); let _ = c.collect(&NoopReaper).unwrap(); assert!(!c.is_retained("b")); @@ -521,7 +809,7 @@ mod tests { ) .unwrap(); assert_eq!(c.store().live_incoming("shared"), 2); - // Remove p1's coverage and p1 itself; shared still has p2. + // Drop p1's coverage; its pending deletion must not affect p2's view. c.release(&RetentionRoot::Pin("r1".into())); c.collect(&NoopReaper).unwrap(); assert!(c.store().node("shared").is_some()); @@ -562,12 +850,12 @@ mod tests { let c = coord(); c.store().retain(node("x", 3), &[], &[]).unwrap(); assert!(c.collect(&Fail).is_err()); - // Node remains (DELETING) so a retry can replay; it is not lost. - assert!(c.store().node("x").is_some()); + // Node remains DELETING; physical retry/recovery is a later slice. + assert_eq!(c.store().node("x").unwrap().state, NodeState::Deleting); } #[test] - fn new_root_revives_a_node_before_it_is_collected() { + fn new_root_protects_a_live_node_before_it_is_collected() { let c = coord(); c.store().retain(node("x", 4), &[], &[]).unwrap(); // Before any collect, a lease covers it. @@ -644,7 +932,12 @@ mod tests { c.retain(node("ch", 2), &["p".to_string()], &[]).unwrap(); c.release(&RetentionRoot::Lease("l".into())); c.collect(&Delete).unwrap(); + // A child selected before its parent was removed waits for the + // next pass; iteration order must not affect the final result. + c.collect(&Delete).unwrap(); // Both nodes and their edge are gone; no stale incoming edge. + assert!(c.store().node("p").is_none()); + assert!(c.store().node("ch").is_none()); assert_eq!(c.store().live_incoming("ch"), 0); } } diff --git a/src/ceres/snapshot/retention_dag.rs b/src/ceres/snapshot/retention_dag.rs new file mode 100644 index 00000000..b9f3cb4d --- /dev/null +++ b/src/ceres/snapshot/retention_dag.rs @@ -0,0 +1,554 @@ +//! Bounded canonical native metadata closure, ready for future durable storage. +//! This module retains no leases, persists no bytes and starts no collector. + +use std::collections::{BTreeMap, BTreeSet, VecDeque}; + +use mst2_codec::{ + descriptor::METADATA_CODEC, + metapage::{Entry, Page, page_id}, +}; + +use crate::ceres::snapshot::{ + error::{SnapshotError, SnapshotErrorCode}, + retention::{NodeState, RetainedKind, RetentionEdge, RetentionNode}, +}; + +pub type MetadataPageId = [u8; 32]; + +#[derive(Debug, Clone, Copy)] +pub struct MetadataDagLimits { + pub nodes: usize, + pub edges: usize, + pub payload_bytes: u64, + pub entries: usize, + pub prepare_entry_visits: usize, +} + +impl Default for MetadataDagLimits { + fn default() -> Self { + Self { + nodes: 4096, + edges: 16_384, + payload_bytes: 64 * 1024 * 1024, + entries: 131_072, + prepare_entry_visits: 64 * 1024 * 1024, + } + } +} + +impl MetadataDagLimits { + /// Caller budgets may tighten, never widen, the absolute group ceilings. + pub(crate) fn effective(self) -> Self { + let hard = Self::default(); + Self { + nodes: self.nodes.min(hard.nodes), + edges: self.edges.min(hard.edges), + payload_bytes: self.payload_bytes.min(hard.payload_bytes), + entries: self.entries.min(hard.entries), + prepare_entry_visits: self.prepare_entry_visits.min(hard.prepare_entry_visits), + } + } +} + +#[derive(Debug, Clone)] +pub struct MetadataPagePayload { + pub id: MetadataPageId, + pub size: u64, + pub bytes: Vec, +} + +#[derive(Debug, Clone)] +pub struct MetadataDagCandidate { + pub metadata_codec: u16, + pub root: MetadataPageId, + pub pages: Vec, + pub edges: Vec<(MetadataPageId, MetadataPageId)>, +} + +/// Immutable validated output. Supply these slices to PostgreSQL retain_group +/// only after the payloads have been durably stored. +#[derive(Debug)] +pub struct ValidatedMetadataDag { + root: MetadataPageId, + pages: Vec, + nodes: Vec, + edges: Vec, + payload_bytes: u64, + entry_count: usize, +} + +impl ValidatedMetadataDag { + pub fn validate( + candidate: MetadataDagCandidate, + limits: MetadataDagLimits, + ) -> Result { + let limits = limits.effective(); + if candidate.metadata_codec != METADATA_CODEC { + return Err(SnapshotError::new( + SnapshotErrorCode::ScopeInvalid, + "native metadata DAG requires metadata codec 1", + )); + } + check_count(candidate.pages.len(), limits.nodes, "metadata nodes")?; + check_count(candidate.edges.len(), limits.edges, "metadata edges")?; + let mut payload_bytes = 0u64; + let mut entry_count = 0usize; + let mut pages = BTreeMap::new(); + let mut decoded = BTreeMap::new(); + let mut expected_edges = BTreeSet::new(); + let mut directory_roots = BTreeSet::from([candidate.root]); + for payload in candidate.pages { + payload_bytes = payload_bytes + .checked_add(payload.bytes.len() as u64) + .filter(|total| *total <= limits.payload_bytes) + .ok_or_else(|| limit("metadata payload budget exceeded"))?; + if payload.size != payload.bytes.len() as u64 || page_id(&payload.bytes) != payload.id { + return Err(SnapshotError::new( + SnapshotErrorCode::DigestMismatch, + "metadata page digest or advertised size mismatch", + )); + } + if pages.contains_key(&payload.id) { + return Err(integrity("duplicate metadata page identity")); + } + let (page, subtree_entries) = Page::decode(&payload.bytes).map_err(codec_error)?; + let entries: &[Entry] = match &page { + Page::Leaf { entries } => entries, + Page::Branch { + terminal, children, .. + } => { + for child in children { + expected_edges.insert((payload.id, child.child_page_id)); + } + terminal.as_slice() + } + }; + entry_count = entry_count + .checked_add(entries.len()) + .filter(|count| *count <= limits.entries) + .ok_or_else(|| limit("metadata entry budget exceeded"))?; + for entry in entries.iter().filter(|entry| entry.is_dir()) { + expected_edges.insert((payload.id, entry.child_root)); + directory_roots.insert(entry.child_root); + } + check_count(expected_edges.len(), limits.edges, "metadata edges")?; + decoded.insert(payload.id, (page, subtree_entries)); + pages.insert(payload.id, payload); + } + let supplied_edges: BTreeSet<_> = candidate.edges.iter().copied().collect(); + if supplied_edges.len() != candidate.edges.len() { + return Err(integrity("duplicate metadata edge")); + } + validate_graph(candidate.root, &pages, &supplied_edges)?; + if supplied_edges != expected_edges { + return Err(integrity( + "metadata edges disagree with canonical page references", + )); + } + for (page, _) in decoded.values() { + if let Page::Branch { children, .. } = page { + for child in children { + if decoded.get(&child.child_page_id).map(|(_, count)| *count) + != Some(child.subtree_entries) + { + return Err(integrity("metadata branch subtree count mismatch")); + } + } + } + } + let mut validation_visits = 0usize; + for root in directory_roots { + let advertised_entries = decoded + .get(&root) + .map(|(_, count)| *count) + .ok_or_else(|| unavailable("metadata child is missing"))?; + if advertised_entries > limits.entries as u64 { + return Err(limit("metadata directory entry budget exceeded")); + } + let mut entries = Vec::new(); + let mut pending = vec![root]; + while let Some(id) = pending.pop() { + let (page, _) = decoded + .get(&id) + .ok_or_else(|| unavailable("metadata child payload is missing"))?; + let direct_entries = match page { + Page::Leaf { entries } => entries.as_slice(), + Page::Branch { + terminal, children, .. + } => { + pending.extend(children.iter().map(|child| child.child_page_id)); + terminal.as_slice() + } + }; + validation_visits = validation_visits + .checked_add(direct_entries.len().saturating_add(1)) + .filter(|count| *count <= limits.prepare_entry_visits) + .ok_or_else(|| limit("metadata canonical validation work budget exceeded"))?; + entries.extend_from_slice(direct_entries); + } + entries.sort_by(|left, right| left.name.cmp(&right.name)); + let canonical = Page::build(&entries).map_err(codec_error)?; + if pages.get(&root).map(|payload| payload.bytes.as_slice()) + != Some(canonical.as_slice()) + { + return Err(integrity( + "metadata directory is not its canonical Build(entries)", + )); + } + } + let nodes = pages + .values() + .map(|payload| RetentionNode { + id: node_id(&payload.id), + kind: RetainedKind::Page, + state: NodeState::Live, + bytes: payload.size, + }) + .collect(); + let edges = supplied_edges + .into_iter() + .map(|(parent, child)| RetentionEdge { + parent: node_id(&parent), + child: node_id(&child), + }) + .collect(); + Ok(Self { + root: candidate.root, + pages: pages.into_values().collect(), + nodes, + edges, + payload_bytes, + entry_count, + }) + } + + pub fn root(&self) -> MetadataPageId { + self.root + } + pub fn payloads(&self) -> &[MetadataPagePayload] { + &self.pages + } + pub fn nodes(&self) -> &[RetentionNode] { + &self.nodes + } + pub fn edges(&self) -> &[RetentionEdge] { + &self.edges + } + pub fn payload_bytes(&self) -> u64 { + self.payload_bytes + } + + pub fn check_limits(&self, limits: MetadataDagLimits) -> Result<(), SnapshotError> { + let limits = limits.effective(); + check_count(self.nodes.len(), limits.nodes, "metadata nodes")?; + check_count(self.edges.len(), limits.edges, "metadata edges")?; + check_count(self.entry_count, limits.entries, "metadata entries")?; + if self.payload_bytes > limits.payload_bytes { + return Err(limit("metadata payload budget exceeded")); + } + Ok(()) + } + + /// Cache residency only; outstanding caller Arcs are outside that bound. + pub(crate) fn residency_bytes(&self) -> Option { + let mut bytes = std::mem::size_of::() + .checked_add( + self.pages + .capacity() + .checked_mul(std::mem::size_of::())?, + )? + .checked_add( + self.nodes + .capacity() + .checked_mul(std::mem::size_of::())?, + )? + .checked_add( + self.edges + .capacity() + .checked_mul(std::mem::size_of::())?, + )?; + for payload in &self.pages { + bytes = bytes.checked_add(payload.bytes.capacity())?; + } + for node in &self.nodes { + bytes = bytes.checked_add(node.id.capacity())?; + } + for edge in &self.edges { + bytes = bytes + .checked_add(edge.parent.capacity())? + .checked_add(edge.child.capacity())?; + } + Some(bytes) + } +} + +pub(crate) struct MetadataDagBuilder { + limits: MetadataDagLimits, + pages: BTreeMap, + edges: BTreeSet<(MetadataPageId, MetadataPageId)>, + required_nodes: BTreeSet, + directories: BTreeSet, + payload_bytes: u64, + entries: usize, + entry_visits: usize, + failed: bool, +} + +impl MetadataDagBuilder { + pub(crate) fn new(limits: MetadataDagLimits) -> Self { + Self { + limits: limits.effective(), + pages: BTreeMap::new(), + edges: BTreeSet::new(), + required_nodes: BTreeSet::new(), + directories: BTreeSet::new(), + payload_bytes: 0, + entries: 0, + entry_visits: 0, + failed: false, + } + } + + pub(crate) fn contains_directory(&self, root: MetadataPageId) -> bool { + self.directories.contains(&root) + } + + pub(crate) fn add_directory( + &mut self, + root_bytes: &[u8], + entries: &[Entry], + ) -> Result<(), SnapshotError> { + if self.failed { + return Err(integrity("metadata preparation already failed")); + } + let result = self.add_directory_inner(root_bytes, entries); + if result.is_err() { + self.failed = true; + } + result + } + + fn add_directory_inner( + &mut self, + root_bytes: &[u8], + entries: &[Entry], + ) -> Result<(), SnapshotError> { + let root = page_id(root_bytes); + if self.contains_directory(root) { + return Ok(()); + } + self.entries = self + .entries + .checked_add(entries.len()) + .filter(|count| *count <= self.limits.entries) + .ok_or_else(|| limit("metadata entry budget exceeded"))?; + self.require_node(root)?; + if !self.pages.contains_key(&root) + && root_bytes.len() as u64 + > self.limits.payload_bytes.saturating_sub(self.payload_bytes) + { + return Err(limit("metadata payload budget exceeded")); + } + self.charge_work(entries.len())?; + let canonical = Page::build(entries).map_err(codec_error)?; + if canonical != root_bytes { + return Err(integrity("directory root disagrees with canonical entries")); + } + let mut routes = VecDeque::from([(Vec::::new(), root)]); + while let Some((route, expected_id)) = routes.pop_front() { + if self.pages.contains_key(&expected_id) { + continue; + } + self.charge_work(entries.len())?; + let pages = Page::pages_along_route(entries, &route).map_err(codec_error)?; + let bytes = pages + .into_iter() + .last() + .ok_or_else(|| integrity("canonical route returned no page"))?; + let id = page_id(&bytes); + if id != expected_id { + return Err(integrity("canonical route child identity mismatch")); + } + self.payload_bytes = self + .payload_bytes + .checked_add(bytes.len() as u64) + .filter(|total| *total <= self.limits.payload_bytes) + .ok_or_else(|| limit("metadata payload budget exceeded"))?; + let (page, _) = Page::decode(&bytes).map_err(codec_error)?; + match page { + Page::Leaf { entries } => { + for entry in entries.iter().filter(|entry| entry.is_dir()) { + self.add_edge(id, entry.child_root)?; + } + } + Page::Branch { + terminal, children, .. + } => { + if let Some(entry) = terminal.filter(Entry::is_dir) { + self.add_edge(id, entry.child_root)?; + } + for child in children { + self.add_edge(id, child.child_page_id)?; + let mut child_route = route.clone(); + child_route.push(child.label); + routes.push_back((child_route, child.child_page_id)); + } + } + } + self.pages.insert( + id, + MetadataPagePayload { + id, + size: bytes.len() as u64, + bytes, + }, + ); + } + self.directories.insert(root); + Ok(()) + } + + fn require_node(&mut self, id: MetadataPageId) -> Result<(), SnapshotError> { + if !self.required_nodes.contains(&id) { + check_count( + self.required_nodes.len().saturating_add(1), + self.limits.nodes, + "metadata nodes", + )?; + self.required_nodes.insert(id); + } + Ok(()) + } + + fn charge_work(&mut self, count: usize) -> Result<(), SnapshotError> { + self.entry_visits = self + .entry_visits + .checked_add(count) + .filter(|count| *count <= self.limits.prepare_entry_visits) + .ok_or_else(|| limit("metadata preparation work budget exceeded"))?; + Ok(()) + } + + fn add_edge( + &mut self, + parent: MetadataPageId, + child: MetadataPageId, + ) -> Result<(), SnapshotError> { + if !self.edges.contains(&(parent, child)) { + check_count( + self.edges.len().saturating_add(1), + self.limits.edges, + "metadata edges", + )?; + self.require_node(parent)?; + self.require_node(child)?; + self.edges.insert((parent, child)); + } + Ok(()) + } + + pub(crate) fn finish( + self, + root: MetadataPageId, + ) -> Result { + if self.failed { + return Err(integrity( + "metadata preparation failed; no group can be exposed", + )); + } + ValidatedMetadataDag::validate( + MetadataDagCandidate { + metadata_codec: METADATA_CODEC, + root, + pages: self.pages.into_values().collect(), + edges: self.edges.into_iter().collect(), + }, + self.limits, + ) + } +} + +fn validate_graph( + root: MetadataPageId, + pages: &BTreeMap, + edges: &BTreeSet<(MetadataPageId, MetadataPageId)>, +) -> Result<(), SnapshotError> { + if !pages.contains_key(&root) { + return Err(unavailable("metadata root payload is missing")); + } + let mut incoming: BTreeMap<_, usize> = pages.keys().map(|id| (*id, 0)).collect(); + let mut children: BTreeMap<_, Vec<_>> = BTreeMap::new(); + for &(parent, child) in edges { + if !pages.contains_key(&parent) || !pages.contains_key(&child) { + return Err(unavailable("metadata child payload is missing")); + } + *incoming + .get_mut(&child) + .ok_or_else(|| unavailable("metadata child is missing"))? += 1; + children.entry(parent).or_default().push(child); + } + let mut ready: VecDeque<_> = incoming + .iter() + .filter_map(|(id, count)| (*count == 0).then_some(*id)) + .collect(); + let mut visited = 0usize; + while let Some(id) = ready.pop_front() { + visited += 1; + for child in children.get(&id).into_iter().flatten() { + let count = incoming + .get_mut(child) + .ok_or_else(|| unavailable("metadata child is missing"))?; + *count -= 1; + if *count == 0 { + ready.push_back(*child); + } + } + } + if visited != pages.len() { + return Err(integrity("metadata graph contains a cycle")); + } + let mut reachable = BTreeSet::new(); + let mut pending = vec![root]; + while let Some(id) = pending.pop() { + if reachable.insert(id) { + pending.extend(children.get(&id).into_iter().flatten().copied()); + } + } + if reachable.len() != pages.len() { + return Err(integrity("metadata graph includes unreachable payloads")); + } + Ok(()) +} + +fn node_id(id: &MetadataPageId) -> String { + use std::fmt::Write; + let mut value = String::with_capacity(76); + value.push_str("page:sha256:"); + for byte in id { + let _ = write!(value, "{byte:02x}"); + } + value +} + +fn check_count(count: usize, maximum: usize, name: &str) -> Result<(), SnapshotError> { + if count > maximum { + return Err(limit(&format!("{name} budget exceeded"))); + } + Ok(()) +} +fn limit(message: &str) -> SnapshotError { + SnapshotError::new(SnapshotErrorCode::LimitExceeded, message) +} +fn integrity(message: &str) -> SnapshotError { + SnapshotError::new(SnapshotErrorCode::IntegrityError, message) +} +fn unavailable(message: &str) -> SnapshotError { + SnapshotError::new(SnapshotErrorCode::ObjectUnavailable, message) +} +fn codec_error(error: mst2_codec::CodecError) -> SnapshotError { + integrity(&format!("invalid metadata page: {error}")) +} + +#[cfg(test)] +#[path = "retention_dag_tests.rs"] +mod tests; diff --git a/src/ceres/snapshot/retention_dag_tests.rs b/src/ceres/snapshot/retention_dag_tests.rs new file mode 100644 index 00000000..3525a4b5 --- /dev/null +++ b/src/ceres/snapshot/retention_dag_tests.rs @@ -0,0 +1,345 @@ +use mst2_codec::metapage::{BranchChild, EntryKind}; + +use super::*; + +fn payload(entries: &[Entry]) -> MetadataPagePayload { + let bytes = Page::build(entries).unwrap(); + MetadataPagePayload { + id: page_id(&bytes), + size: bytes.len() as u64, + bytes, + } +} + +fn candidate(dag: &ValidatedMetadataDag) -> MetadataDagCandidate { + let ids: BTreeMap<_, _> = dag + .payloads() + .iter() + .map(|payload| (node_id(&payload.id), payload.id)) + .collect(); + MetadataDagCandidate { + metadata_codec: METADATA_CODEC, + root: dag.root(), + pages: dag.payloads().to_vec(), + edges: dag + .edges() + .iter() + .map(|edge| (ids[&edge.parent], ids[&edge.child])) + .collect(), + } +} + +fn nested_shared() -> ValidatedMetadataDag { + let empty = payload(&[]); + let mut wide: Vec<_> = (0..129) + .map(|index| { + Entry::file( + EntryKind::Regular, + format!("f{index:03}").as_bytes(), + index, + [index as u8; 32], + ) + }) + .collect(); + wide.push(Entry::dir(b"nested", empty.id)); + let shared = payload(&wide); + let root_entries = vec![ + Entry::dir(b"alpha", shared.id), + Entry::dir(b"beta", shared.id), + ]; + let root = payload(&root_entries); + let mut builder = MetadataDagBuilder::new(MetadataDagLimits::default()); + builder.add_directory(&root.bytes, &root_entries).unwrap(); + builder.add_directory(&shared.bytes, &wide).unwrap(); + builder.add_directory(&empty.bytes, &[]).unwrap(); + builder.add_directory(&shared.bytes, &wide).unwrap(); + builder.finish(root.id).unwrap() +} + +#[test] +fn complete_radix_nested_shared_closure_accounts_unique_payloads_and_edges() { + let dag = nested_shared(); + assert!( + dag.payloads() + .iter() + .any(|payload| matches!(Page::decode(&payload.bytes).unwrap().0, Page::Branch { .. })) + ); + assert!( + dag.payloads().len() > 3, + "internal radix pages must be retained too" + ); + assert_eq!( + dag.payload_bytes(), + dag.payloads() + .iter() + .map(|payload| payload.size) + .sum::() + ); + assert_eq!(dag.nodes().len(), dag.payloads().len()); + let mut ids = BTreeSet::new(); + for (node, payload) in dag.nodes().iter().zip(dag.payloads()) { + assert_eq!(node.kind, RetainedKind::Page); + assert_eq!(node.state, NodeState::Live); + assert_eq!(node.bytes, payload.bytes.len() as u64); + assert_eq!(node.id, node_id(&payload.id)); + assert!(ids.insert(node.id.clone())); + } + let edges: BTreeSet<_> = dag + .edges() + .iter() + .map(|edge| (&edge.parent, &edge.child)) + .collect(); + assert_eq!(edges.len(), dag.edges().len()); + assert!( + dag.edges() + .iter() + .all(|edge| ids.contains(&edge.parent) && ids.contains(&edge.child)) + ); + let root_id = node_id(&dag.root()); + assert_eq!( + dag.edges() + .iter() + .filter(|edge| edge.parent == root_id) + .count(), + 1, + "two names sharing one child acquire one edge" + ); +} + +#[test] +fn digest_and_advertised_size_are_verified_before_exposing_group() { + let dag = nested_shared(); + let mut wrong_id = candidate(&dag); + wrong_id.pages[0].id[0] ^= 1; + assert_eq!( + ValidatedMetadataDag::validate(wrong_id, MetadataDagLimits::default()) + .unwrap_err() + .code, + SnapshotErrorCode::DigestMismatch + ); + let mut wrong_size = candidate(&dag); + wrong_size.pages[0].size += 1; + assert_eq!( + ValidatedMetadataDag::validate(wrong_size, MetadataDagLimits::default()) + .unwrap_err() + .code, + SnapshotErrorCode::DigestMismatch + ); + let mut corrupt = candidate(&dag); + corrupt.pages[0].bytes[0] ^= 1; + assert_eq!( + ValidatedMetadataDag::validate(corrupt, MetadataDagLimits::default()) + .unwrap_err() + .code, + SnapshotErrorCode::DigestMismatch + ); +} + +#[test] +fn missing_child_and_missing_root_are_unavailable() { + let dag = nested_shared(); + let mut missing = candidate(&dag); + let index = missing + .pages + .iter() + .position(|payload| payload.id != missing.root) + .unwrap(); + missing.pages.remove(index); + assert_eq!( + ValidatedMetadataDag::validate(missing, MetadataDagLimits::default()) + .unwrap_err() + .code, + SnapshotErrorCode::ObjectUnavailable + ); + let mut missing_root = candidate(&dag); + missing_root.root = [255; 32]; + assert_eq!( + ValidatedMetadataDag::validate(missing_root, MetadataDagLimits::default()) + .unwrap_err() + .code, + SnapshotErrorCode::ObjectUnavailable + ); +} + +#[test] +fn duplicate_payloads_and_edges_are_rejected() { + let dag = nested_shared(); + let mut duplicate = candidate(&dag); + duplicate.pages.push(duplicate.pages[0].clone()); + assert_eq!( + ValidatedMetadataDag::validate(duplicate, MetadataDagLimits::default()) + .unwrap_err() + .code, + SnapshotErrorCode::IntegrityError + ); + let mut duplicate_edge = candidate(&dag); + duplicate_edge.edges.push(duplicate_edge.edges[0]); + assert_eq!( + ValidatedMetadataDag::validate(duplicate_edge, MetadataDagLimits::default()) + .unwrap_err() + .code, + SnapshotErrorCode::IntegrityError + ); +} + +#[test] +fn cycles_wrong_edges_and_unreachable_objects_are_rejected() { + let dag = nested_shared(); + let mut cyclic = candidate(&dag); + cyclic.edges.push((cyclic.root, cyclic.root)); + let error = ValidatedMetadataDag::validate(cyclic, MetadataDagLimits::default()).unwrap_err(); + assert_eq!(error.code, SnapshotErrorCode::IntegrityError); + assert!(error.message.contains("cycle")); + let mut missing_edge = candidate(&dag); + missing_edge.edges.pop(); + assert_eq!( + ValidatedMetadataDag::validate(missing_edge, MetadataDagLimits::default()) + .unwrap_err() + .code, + SnapshotErrorCode::IntegrityError + ); + let mut orphan = candidate(&dag); + orphan.pages.push(payload(&[Entry::file( + EntryKind::Regular, + b"orphan", + 1, + [91; 32], + )])); + assert_eq!( + ValidatedMetadataDag::validate(orphan, MetadataDagLimits::default()) + .unwrap_err() + .code, + SnapshotErrorCode::IntegrityError + ); +} + +#[test] +fn correct_hashes_do_not_make_a_noncanonical_directory_valid() { + let left_entry = Entry::file(EntryKind::Regular, b"a", 1, [1; 32]); + let right_entry = Entry::file(EntryKind::Regular, b"b", 1, [2; 32]); + let left = payload(&[left_entry]); + let right = payload(&[right_entry]); + let bytes = Page::Branch { + prefix: Vec::new(), + terminal: None, + children: vec![ + BranchChild { + label: b'a', + subtree_entries: 1, + child_page_id: left.id, + }, + BranchChild { + label: b'b', + subtree_entries: 1, + child_page_id: right.id, + }, + ], + } + .encode() + .unwrap(); + let root = page_id(&bytes); + let group = MetadataDagCandidate { + metadata_codec: METADATA_CODEC, + root, + edges: vec![(root, left.id), (root, right.id)], + pages: vec![ + MetadataPagePayload { + id: root, + size: bytes.len() as u64, + bytes, + }, + left, + right, + ], + }; + let error = ValidatedMetadataDag::validate(group, MetadataDagLimits::default()).unwrap_err(); + assert_eq!(error.code, SnapshotErrorCode::IntegrityError); + assert!(error.message.contains("canonical")); +} + +#[test] +fn node_edge_payload_entry_and_work_budgets_reject_complete_groups() { + let dag = nested_shared(); + for limits in [ + MetadataDagLimits { + nodes: dag.nodes().len() - 1, + ..MetadataDagLimits::default() + }, + MetadataDagLimits { + edges: dag.edges().len() - 1, + ..MetadataDagLimits::default() + }, + MetadataDagLimits { + payload_bytes: dag.payload_bytes() - 1, + ..MetadataDagLimits::default() + }, + MetadataDagLimits { + entries: 1, + ..MetadataDagLimits::default() + }, + MetadataDagLimits { + prepare_entry_visits: 0, + ..MetadataDagLimits::default() + }, + ] { + assert_eq!( + ValidatedMetadataDag::validate(candidate(&dag), limits) + .unwrap_err() + .code, + SnapshotErrorCode::LimitExceeded + ); + } +} + +#[test] +fn failed_preparation_cannot_expose_a_partial_group() { + let empty = payload(&[]); + let root_entries = vec![Entry::dir(b"child", empty.id)]; + let root = payload(&root_entries); + let mut builder = MetadataDagBuilder::new(MetadataDagLimits { + nodes: 1, + ..MetadataDagLimits::default() + }); + assert_eq!( + builder + .add_directory(&root.bytes, &root_entries) + .unwrap_err() + .code, + SnapshotErrorCode::LimitExceeded + ); + assert_eq!( + builder.finish(root.id).unwrap_err().code, + SnapshotErrorCode::IntegrityError + ); + let mut builder = MetadataDagBuilder::new(MetadataDagLimits::default()); + builder.add_directory(&root.bytes, &root_entries).unwrap(); + assert_eq!( + builder.finish(root.id).unwrap_err().code, + SnapshotErrorCode::ObjectUnavailable + ); +} + +#[test] +fn codec_identity_and_canonical_entry_source_are_fixed() { + let dag = nested_shared(); + let mut wrong_codec = candidate(&dag); + wrong_codec.metadata_codec += 1; + assert_eq!( + ValidatedMetadataDag::validate(wrong_codec, MetadataDagLimits::default()) + .unwrap_err() + .code, + SnapshotErrorCode::ScopeInvalid + ); + let empty = payload(&[]); + let mut builder = MetadataDagBuilder::new(MetadataDagLimits::default()); + assert_eq!( + builder + .add_directory( + &empty.bytes, + &[Entry::file(EntryKind::Regular, b"different", 1, [1; 32])] + ) + .unwrap_err() + .code, + SnapshotErrorCode::IntegrityError + ); +} diff --git a/src/ceres/snapshot/retention_prepare_tests.rs b/src/ceres/snapshot/retention_prepare_tests.rs new file mode 100644 index 00000000..2a427795 --- /dev/null +++ b/src/ceres/snapshot/retention_prepare_tests.rs @@ -0,0 +1,92 @@ +use git_internal::hash::HashKind; + +use super::*; +use crate::ceres::snapshot::retention_dag::{MetadataDagBuilder, MetadataDagLimits}; + +fn dag() -> Arc { + let bytes = Page::build(&[]).unwrap(); + let mut builder = MetadataDagBuilder::new(MetadataDagLimits::default()); + builder.add_directory(&bytes, &[]).unwrap(); + Arc::new(builder.finish(page_id(&bytes)).unwrap()) +} + +fn key(kind: HashKind, scope: &str) -> NativeRetentionKey { + NativeRetentionKey { + projection: NativeProjectionKey::new( + ObjectHash::from_hex_for_kind(kind, &"a".repeat(64)).unwrap(), + ), + scope: scope.to_owned(), + } +} + +#[test] +fn retention_memo_preserves_tagged_source_scope_and_profile_identity() { + let cache = NativeProjectionCache::default(); + let original = key(HashKind::Sha256, "/scope"); + let prepared = dag(); + cache.insert_retention_dag(original.clone(), Arc::clone(&prepared)); + assert!(Arc::ptr_eq( + &prepared, + &cache.retention_dag(&original).unwrap() + )); + assert!( + cache + .retention_dag(&key(HashKind::Blake3, "/scope")) + .is_none() + ); + assert!( + cache + .retention_dag(&key(HashKind::Sha256, "/different")) + .is_none() + ); + let mut different = original; + different.projection.projection_revision += 1; + assert!(cache.retention_dag(&different).is_none()); +} + +#[test] +fn retention_memo_limits_total_residency_and_eviction_keeps_active_arc() { + let cache = NativeProjectionCache::default(); + let first_key = key(HashKind::Sha256, "/first"); + let first = dag(); + let overhead = std::mem::size_of::() + + std::mem::size_of::>() + + first_key.scope.capacity() + + first_key.projection.tree_oid.capacity(); + let one_entry_bytes = first.residency_bytes().unwrap() + overhead; + cache.insert_retention_dag_with_budget( + first_key.clone(), + Arc::clone(&first), + one_entry_bytes, + 16, + ); + assert!(cache.retention_dag(&first_key).is_some()); + let next_key = key(HashKind::Sha256, "/other"); + let next = dag(); + cache.insert_retention_dag_with_budget( + next_key.clone(), + Arc::clone(&next), + one_entry_bytes, + 16, + ); + assert!(cache.retention_dag(&first_key).is_none()); + assert!(Arc::ptr_eq(&next, &cache.retention_dag(&next_key).unwrap())); + assert_eq!(first.payloads()[0].bytes, Page::build(&[]).unwrap()); + assert!(cache.state.lock().unwrap().retained_dag_bytes <= one_entry_bytes); + let skipped_key = key(HashKind::Sha256, "/too-large"); + cache.insert_retention_dag_with_budget(skipped_key.clone(), dag(), 1, 16); + assert!(cache.retention_dag(&skipped_key).is_none()); + assert!(cache.retention_dag(&next_key).is_some()); +} + +#[test] +fn retention_memo_entry_cap_bounds_distinct_roots() { + let cache = NativeProjectionCache::default(); + let first = key(HashKind::Sha256, "/one"); + let second = key(HashKind::Sha256, "/two"); + cache.insert_retention_dag_with_budget(first.clone(), dag(), 64 * 1024 * 1024, 1); + cache.insert_retention_dag_with_budget(second.clone(), dag(), 64 * 1024 * 1024, 1); + assert!(cache.retention_dag(&first).is_none()); + assert!(cache.retention_dag(&second).is_some()); + assert_eq!(cache.state.lock().unwrap().retention_dags.len(), 1); +} diff --git a/src/ceres/snapshot/runtime.rs b/src/ceres/snapshot/runtime.rs index 770c4725..b8b4bce3 100644 --- a/src/ceres/snapshot/runtime.rs +++ b/src/ceres/snapshot/runtime.rs @@ -11,7 +11,7 @@ use std::{ collections::HashMap, - sync::{Mutex, OnceLock}, + sync::{Mutex, MutexGuard, OnceLock}, time::{Duration, SystemTime, UNIX_EPOCH}, }; @@ -101,16 +101,16 @@ impl Mst2Runtime { st.1 } - /// Insert a context and create its first lease. The lease also becomes - /// a retention root covering the snapshot's derived metadata root - /// (spec 10 §5: active leases pin what a fixed view needs). + /// Acquire the complete in-memory retention group before exposing a + /// context or lease. These placeholder anchors do not yet represent + /// the complete durable page/chunk graph required by T06. pub fn insert_context( &self, built: BuiltDescriptor, commit_oid: &str, root_tree_oid: &str, lease_seconds: u64, - ) -> SnapshotContext { + ) -> Result { let lease_id = Uuid::new_v4().to_string(); let expires = now_unix() + lease_seconds.clamp(1, 3600); let snapshot_id = built.snapshot_id.clone(); @@ -122,23 +122,9 @@ impl Mst2Runtime { lease_expires_at_unix: expires, authorization_epoch: 1, }; - self.contexts.lock().unwrap().insert( - snapshot_id.clone(), - ContextData { - built: built.clone(), - commit_oid: commit_oid.to_string(), - root_tree_oid: root_tree_oid.to_string(), - }, - ); - self.leases.lock().unwrap().insert( - lease_id.clone(), - LeaseRec { - snapshot_id: snapshot_id.clone(), - expires_at_unix: expires, - }, - ); - // Retention root for the lease: the metadata-root page node and the - // snapshot's chunk projections stay reachable while it is active. + // Serialize registration with release/expiry/renewal. No path holds + // contexts while acquiring leases, so this lock order cannot cycle. + let mut leases = self.leases.lock().unwrap(); let root = crate::ceres::snapshot::retention::RetentionRoot::Lease(lease_id.clone()); let page_node = crate::ceres::snapshot::retention::RetentionNode { id: format!("page:{}", built.metadata_root), @@ -152,15 +138,24 @@ impl Mst2Runtime { state: crate::ceres::snapshot::retention::NodeState::Live, bytes: 0, }; - if let Err(e) = self - .retention - .pin_root(&root, &[page_node, projection_node]) - { - // Coverage failure is surfaced, not swallowed: a lease that - // cannot pin its view must not pretend to. - tracing::warn!(error = %e, "retention pin for new lease failed"); - } - ctx + self.retention + .pin_root(&root, &[page_node, projection_node])?; + self.contexts.lock().unwrap().insert( + snapshot_id.clone(), + ContextData { + built, + commit_oid: commit_oid.to_string(), + root_tree_oid: root_tree_oid.to_string(), + }, + ); + leases.insert( + lease_id, + LeaseRec { + snapshot_id, + expires_at_unix: expires, + }, + ); + Ok(ctx) } /// Look up a snapshot context by snapshot_id. It is servable while at @@ -194,7 +189,7 @@ impl Mst2Runtime { /// as it goes. fn active_lease(&self, snapshot_id: &str) -> Result<(String, u64), SnapshotError> { let mut leases = self.leases.lock().unwrap(); - leases.retain(|_, r| r.expires_at_unix >= now_unix()); + self.prune_expired_leases(&mut leases, now_unix()); leases .iter() .find(|(_, r)| r.snapshot_id == snapshot_id) @@ -207,6 +202,22 @@ impl Mst2Runtime { }) } + /// The leases lock serializes expiry with renewal and registration; + /// releasing one expired lease never removes another lease's coverage. + /// The current in-memory store's release operation is infallible. + fn prune_expired_leases(&self, leases: &mut HashMap, now: u64) { + leases.retain(|lease_id, rec| { + if rec.expires_at_unix >= now { + return true; + } + self.retention + .release(&crate::ceres::snapshot::retention::RetentionRoot::Lease( + lease_id.clone(), + )); + false + }); + } + /// Validate the lease a request presented (spec 04 §1: identity lives in /// the `X-Mega-Snapshot-Lease` header, and a read path must check it — /// knowing the snapshot id alone is not a capability). Any lease that is @@ -216,8 +227,8 @@ impl Mst2Runtime { /// existed is answered identically: probing with lease ids must not /// distinguish released from forged. /// - /// Expired-row pruning happens in `active_lease`, not here: this runs on - /// every authenticated read and must stay a single key lookup. + /// Expired-row pruning happens in `active_lease` and collection: this + /// runs on every authenticated read and stays a single key lookup. pub fn validate_lease(&self, snapshot_id: &str, lease_id: &str) -> Result<(), SnapshotError> { let leases = self.leases.lock().unwrap(); match leases.get(lease_id) { @@ -237,8 +248,24 @@ impl Mst2Runtime { lease_id: &str, lease_seconds: u64, ) -> Result { - let now = now_unix(); - let mut leases = self.leases.lock().unwrap(); + Self::renew_lease_with_clock( + lease_id, + lease_seconds, + || self.leases.lock().unwrap(), + now_unix, + ) + } + + /// Sample the clock only after acquiring the lease lock. Separate + /// sources let tests force a deadline crossing during lock contention. + fn renew_lease_with_clock<'a>( + lease_id: &str, + lease_seconds: u64, + acquire_leases: impl FnOnce() -> MutexGuard<'a, HashMap>, + clock: impl FnOnce() -> u64, + ) -> Result { + let mut leases = acquire_leases(); + let now = clock(); let Some(rec) = leases.get_mut(lease_id) else { return Err(SnapshotError::new( SnapshotErrorCode::LeaseUnknown, @@ -268,7 +295,8 @@ impl Mst2Runtime { /// only drops the retention root so a later collection pass may /// reclaim derived pages no other lease/pin covers. pub fn release_lease(&self, lease_id: &str) -> bool { - let removed = self.leases.lock().unwrap().remove(lease_id).is_some(); + let mut leases = self.leases.lock().unwrap(); + let removed = leases.remove(lease_id).is_some(); self.retention .release(&crate::ceres::snapshot::retention::RetentionRoot::Lease( lease_id.to_string(), @@ -282,6 +310,7 @@ impl Mst2Runtime { pub fn collect_retention( &self, ) -> Result { + self.prune_expired_leases(&mut self.leases.lock().unwrap(), now_unix()); self.retention .collect(&crate::ceres::snapshot::retention::NoopReaper) } @@ -336,20 +365,24 @@ pub fn rfc3339(unix: u64) -> String { #[cfg(test)] mod tests { - use super::*; + use std::{ + sync::{ + Arc, TryLockError, + atomic::{AtomicU64, Ordering}, + mpsc::sync_channel, + }, + thread, + }; - #[test] - fn rfc3339_known_values() { - assert_eq!(rfc3339(0), "1970-01-01T00:00:00Z"); - assert_eq!(rfc3339(1_789_525_722), "2026-09-16T02:28:42Z"); - } + use mst2_codec::descriptor::ServingDescriptor; - #[tokio::test] - async fn lease_lifecycle_drives_retention() { - use crate::ceres::snapshot::retention::{ - NodeState, RetainedKind, RetentionRoot, RetentionStore as _, - }; - let r = Mst2Runtime { + use super::*; + use crate::ceres::snapshot::retention::{ + NodeState, RetainedKind, RetentionRoot, RetentionStore as _, + }; + + fn isolated_runtime() -> Mst2Runtime { + Mst2Runtime { hmac_key: blake3_key(), contexts: Mutex::new(HashMap::new()), leases: Mutex::new(HashMap::new()), @@ -357,9 +390,324 @@ mod tests { crate::ceres::snapshot::retention::mem::InMemoryRetentionStore::default(), ), tip_state: Mutex::new((String::new(), 0u64)), + } + } + + fn descriptor(root_byte: u8) -> BuiltDescriptor { + let descriptor = ServingDescriptor { + instance_uuid: [0x11; 16], + namespace_view_id: [0x22; 32], + scope: "/".into(), + metadata_root: [root_byte; 32], + }; + BuiltDescriptor { + instance_id: Uuid::from_bytes(descriptor.instance_uuid).to_string(), + snapshot_id: format!( + "sha256:{}", + crate::ceres::snapshot::view::hex(&descriptor.snapshot_id().unwrap()) + ), + metadata_root: format!( + "sha256:{}", + crate::ceres::snapshot::view::hex(&descriptor.metadata_root) + ), + descriptor, + } + } + + #[test] + fn pin_failure_exposes_no_context_lease_or_partial_graph_and_can_retry() { + let r = isolated_runtime(); + let built = descriptor(3); + r.retention.store().fail_next_retain_for_test(); + assert_eq!( + r.insert_context(built.clone(), "commit", "tree", 60) + .unwrap_err() + .code, + SnapshotErrorCode::Internal + ); + assert!(r.contexts.lock().unwrap().is_empty()); + assert!(r.leases.lock().unwrap().is_empty()); + assert!(r.retention.store().all_live().is_empty()); + assert_eq!( + r.context(&built.snapshot_id).unwrap_err().code, + SnapshotErrorCode::SnapshotGone + ); + + let ctx = r + .insert_context(built.clone(), "commit", "tree", 60) + .unwrap(); + r.validate_lease(&built.snapshot_id, &ctx.lease_id).unwrap(); + assert!( + r.retention + .is_retained(&format!("page:{}", built.metadata_root)) + ); + assert!(r.retention.is_retained("projection:commit")); + } + + #[test] + fn failed_resolve_preserves_an_existing_active_snapshot() { + let r = isolated_runtime(); + let old = descriptor(3); + let active = r + .insert_context(old.clone(), "old-commit", "old-tree", 60) + .unwrap(); + let new = descriptor(4); + r.retention.store().fail_next_retain_for_test(); + assert!( + r.insert_context(new.clone(), "new-commit", "new-tree", 60) + .is_err() + ); + assert_eq!(r.contexts.lock().unwrap().len(), 1); + assert_eq!(r.leases.lock().unwrap().len(), 1); + assert_eq!( + r.context(&old.snapshot_id).unwrap().lease_id, + active.lease_id + ); + assert_eq!( + r.context(&new.snapshot_id).unwrap_err().code, + SnapshotErrorCode::SnapshotGone + ); + assert!( + r.retention + .store() + .node(&format!("page:{}", new.metadata_root)) + .is_none() + ); + assert!(r.retention.store().node("projection:new-commit").is_none()); + } + + #[test] + fn resolve_cannot_create_a_lease_for_a_deleting_group() { + let r = isolated_runtime(); + let built = descriptor(3); + let ctx = r + .insert_context(built.clone(), "commit", "tree", 60) + .unwrap(); + assert!(r.release_lease(&ctx.lease_id)); + assert!(r.collect_retention().unwrap().reaped.is_empty()); + assert_eq!( + r.insert_context(built.clone(), "commit", "tree", 60) + .unwrap_err() + .code, + SnapshotErrorCode::ObjectUnavailable + ); + assert!(r.leases.lock().unwrap().is_empty()); + assert!( + r.retention.store().all_live().is_empty(), + "no new lease anchor leaked" + ); + assert_eq!( + r.context(&built.snapshot_id).unwrap_err().code, + SnapshotErrorCode::LeaseExpired + ); + assert_eq!( + r.retention + .store() + .node(&format!("page:{}", built.metadata_root)) + .unwrap() + .state, + NodeState::Deleting + ); + } + + #[test] + fn context_expiry_releases_only_the_expired_lease_on_a_shared_snapshot() { + let r = isolated_runtime(); + let built = descriptor(3); + let expired = r + .insert_context(built.clone(), "commit", "tree", 60) + .unwrap(); + let active = r + .insert_context(built.clone(), "commit", "tree", 60) + .unwrap(); + r.leases + .lock() + .unwrap() + .get_mut(&expired.lease_id) + .unwrap() + .expires_at_unix = 0; + + assert_eq!( + r.context(&built.snapshot_id).unwrap().lease_id, + active.lease_id + ); + assert!(!r.leases.lock().unwrap().contains_key(&expired.lease_id)); + assert!( + !r.retention + .store() + .root_covers(&format!("lease:{}", expired.lease_id)) + ); + assert!( + r.retention + .store() + .root_covers(&format!("lease:{}", active.lease_id)) + ); + assert!( + r.retention + .is_retained(&format!("page:{}", built.metadata_root)) + ); + r.validate_lease(&built.snapshot_id, &active.lease_id) + .unwrap(); + assert_eq!( + r.validate_lease(&built.snapshot_id, &expired.lease_id) + .unwrap_err() + .code, + SnapshotErrorCode::LeaseExpired + ); + assert_eq!( + r.collect_retention().unwrap().unreachable, + vec![format!("lease:{}", expired.lease_id)] + ); + assert_eq!( + r.context(&built.snapshot_id).unwrap().lease_id, + active.lease_id + ); + + assert!(r.release_lease(&active.lease_id)); + assert!( + !r.retention + .store() + .root_covers(&format!("page:{}", built.metadata_root)) + ); + assert!(!r.retention.store().root_covers("projection:commit")); + } + + #[test] + fn collection_prunes_expired_roots_without_a_context_read() { + let r = isolated_runtime(); + let built = descriptor(3); + let ctx = r + .insert_context(built.clone(), "commit", "tree", 60) + .unwrap(); + r.leases + .lock() + .unwrap() + .get_mut(&ctx.lease_id) + .unwrap() + .expires_at_unix = 0; + let report = r.collect_retention().unwrap(); + assert_eq!(report.unreachable.len(), 3, "anchor, page and projection"); + assert!(report.reaped.is_empty(), "NoopReaper stays fail-closed"); + assert!(r.leases.lock().unwrap().is_empty()); + for id in [ + format!("lease:{}", ctx.lease_id), + format!("page:{}", built.metadata_root), + "projection:commit".into(), + ] { + assert!(report.unreachable.contains(&id)); + assert!(!r.retention.store().root_covers(&id)); + assert_eq!( + r.retention.store().node(&id).unwrap().state, + NodeState::Deleting + ); + } + } + + #[test] + fn renewal_winning_before_expiry_keeps_its_root_and_expiry_winning_prevents_renewal() { + let r = isolated_runtime(); + let built = descriptor(3); + let ctx = r + .insert_context(built.clone(), "commit", "tree", 60) + .unwrap(); + let renewed = r.renew_lease(&ctx.lease_id, 60).unwrap(); + { + let mut leases = r.leases.lock().unwrap(); + // Advance past the old deadline, before the renewed deadline. + r.prune_expired_leases(&mut leases, ctx.lease_expires_at_unix + 1); + } + r.validate_lease(&built.snapshot_id, &ctx.lease_id).unwrap(); + assert!( + r.retention + .store() + .root_covers(&format!("lease:{}", ctx.lease_id)) + ); + { + let mut leases = r.leases.lock().unwrap(); + r.prune_expired_leases(&mut leases, renewed.expires_at_unix + 1); + } + assert!( + !r.retention + .store() + .root_covers(&format!("lease:{}", ctx.lease_id)) + ); + assert_eq!( + r.renew_lease(&ctx.lease_id, 60).unwrap_err().code, + SnapshotErrorCode::LeaseUnknown + ); + assert_eq!( + r.context(&built.snapshot_id).unwrap_err().code, + SnapshotErrorCode::LeaseExpired + ); + } + + #[test] + fn renewal_waiting_for_the_lease_lock_cannot_use_a_pre_expiry_clock_sample() { + let r = Arc::new(isolated_runtime()); + let ctx = r + .insert_context(descriptor(3), "commit", "tree", 60) + .unwrap(); + let clock = Arc::new(AtomicU64::new(9)); + let (blocked_tx, blocked_rx) = sync_channel(1); + let mut leases = r.leases.lock().unwrap(); + leases.get_mut(&ctx.lease_id).unwrap().expires_at_unix = 10; + let renewal = { + let r = Arc::clone(&r); + let clock = Arc::clone(&clock); + let lease_id = ctx.lease_id.clone(); + thread::spawn(move || { + Mst2Runtime::renew_lease_with_clock( + &lease_id, + 60, + || { + // Prove contention before allowing the clock to cross + // the deadline. A pre-lock sample would observe 9. + assert!(matches!(r.leases.try_lock(), Err(TryLockError::WouldBlock))); + blocked_tx.send(()).unwrap(); + r.leases.lock().unwrap() + }, + || clock.load(Ordering::SeqCst), + ) + }) }; - // Retain the snapshot's derived nodes directly, covered by the lease - // root that insert_context created. + blocked_rx + .recv_timeout(Duration::from_secs(5)) + .expect("renewal did not contend for the lease lock"); + clock.store(11, Ordering::SeqCst); + drop(leases); + + assert_eq!( + renewal.join().unwrap().unwrap_err().code, + SnapshotErrorCode::LeaseExpired + ); + assert_eq!( + r.leases + .lock() + .unwrap() + .get(&ctx.lease_id) + .unwrap() + .expires_at_unix, + 10 + ); + assert_eq!(r.leases.lock().unwrap().len(), 1, "no replacement lease"); + assert!(r.collect_retention().unwrap().reaped.is_empty()); + assert!( + !r.retention + .store() + .root_covers(&format!("lease:{}", ctx.lease_id)) + ); + } + + #[test] + fn rfc3339_known_values() { + assert_eq!(rfc3339(0), "1970-01-01T00:00:00Z"); + assert_eq!(rfc3339(1_789_525_722), "2026-09-16T02:28:42Z"); + } + + #[test] + fn lease_lifecycle_drives_retention() { + let r = isolated_runtime(); + // Cover a derived node directly with a manual test lease root. let lease_root = RetentionRoot::Lease("gc-lease-1".to_string()); let page = crate::ceres::snapshot::retention::RetentionNode { id: "page:sha256:test".to_string(), @@ -382,7 +730,7 @@ mod tests { assert_eq!(report.unreachable, vec!["page:sha256:test".to_string()]); assert_eq!(report.reclaimed_bytes, 100); assert!(report.reaped.is_empty(), "fail-closed reaper"); - // The node is DELETING, not gone: a re-resolve re-lifts it. + // The node stays DELETING; a new acquire must not resurrect it. assert_eq!( r.retention.store().node("page:sha256:test").unwrap().state, NodeState::Deleting diff --git a/src/commands/config.rs b/src/commands/config.rs index a4e2763a..f08a287d 100644 --- a/src/commands/config.rs +++ b/src/commands/config.rs @@ -1091,6 +1091,117 @@ mod tests { assert_eq!(load_mode(&matches), LoadMode::RawSources); } + #[test] + fn config_validate_accepts_mst2_file_source_and_checks_identity() { + use futures::FutureExt; + + let lock = env_lock(); + // The repository test environment still carries this removed setting. + // Keep the file-source regression isolated for both load and validate. + let _mail = crate::config::testing::EnvVarGuard::remove(&lock, "MEGA_MAIL__ENABLED"); + let _enabled = crate::config::testing::EnvVarGuard::remove(&lock, "MEGA_MST2__ENABLED"); + let _identity = + crate::config::testing::EnvVarGuard::remove(&lock, "MEGA_MST2__INSTANCE_UUID"); + let _publication = + crate::config::testing::EnvVarGuard::remove(&lock, "MEGA_MST2__PUBLICATION_ENABLED"); + let _token = crate::config::testing::EnvVarGuard::remove(&lock, "MEGA_MST2__AUTH_TOKEN"); + let dir = tempfile::tempdir().expect("temp dir"); + let config_path = dir.path().join("config.toml"); + let content = format!( + "{}\n[mst2]\nenabled = true\ninstance_uuid = \"12345678-1234-4234-9234-123456789abc\"\npublication_enabled = false\nauth_token = \"mst2-cli-test-token\"\n", + crate::config::template::config_init_template(dir.path()), + ); + fs::write(&config_path, &content).expect("write MST/2 config"); + let selected = + crate::config::loader::ConfigLoader::new(crate::config::loader::ConfigInput { + cli_path: Some(config_path.clone()), + ..Default::default() + }) + .load_readonly() + .expect("select explicit existing config"); + assert_eq!(selected.path, config_path); + let config = + load_config_for_validate(&selected.path, None).expect("CLI loader accepts MST/2 file"); + assert!(config.mst2.enabled); + assert_eq!( + config.mst2.auth_token.as_deref(), + Some("mst2-cli-test-token") + ); + let invalid_path = dir.path().join("invalid.toml"); + fs::write( + &invalid_path, + content.replace( + "12345678-1234-4234-9234-123456789abc", + "00000000-0000-0000-0000-000000000000", + ), + ) + .expect("write nil UUID config"); + let invalid_config = load_config_for_validate(&invalid_path, None) + .expect("syntactically valid MST/2 file loads before semantic validation"); + validate_config( + &config, + Some(&config_path), + None, + false, + false, + false, + "human", + ) + .now_or_never() + .expect("validation without secret resolution must complete synchronously") + .expect("ordinary CLI validation accepts MST/2"); + let message = validate_config( + &invalid_config, + Some(&invalid_path), + None, + false, + false, + false, + "human", + ) + .now_or_never() + .expect("validation without secret resolution must complete synchronously") + .expect_err("ordinary CLI validation must reject nil UUID") + .to_string(); + assert!(message.contains("mst2.instance_uuid"), "{message}"); + assert!(!message.contains("mst2-cli-test-token"), "{message}"); + } + + #[test] + fn config_validate_rejects_unknown_mst2_profile_field() { + let dir = tempfile::tempdir().expect("temp dir"); + let config_path = dir.path().join("config.toml"); + let profile_path = dir.path().join("config.test.toml"); + fs::write( + &config_path, + crate::config::template::config_init_template(dir.path()), + ) + .expect("write base config"); + fs::write( + &profile_path, + "[mst2]\nenabled = false\npublication_enabeld = true\n", + ) + .expect("write typo profile"); + let selected = + crate::config::loader::ConfigLoader::new(crate::config::loader::ConfigInput { + cli_path: Some(config_path), + cli_profile: Some("test".to_string()), + ..Default::default() + }) + .load_readonly() + .expect("select explicit existing profile"); + let message = load_config_for_validate( + &selected.path, + selected + .profile + .as_ref() + .map(|profile| profile.path.as_path()), + ) + .expect_err("MST/2 nested typo must fail in the CLI profile loader") + .to_string(); + assert!(message.contains("mst2.publication_enabeld"), "{message}"); + } + #[test] fn config_validate_accepts_deny_warnings_flag() { let matches = cli() diff --git a/src/commands/mod.rs b/src/commands/mod.rs index 92361218..4c7fc6d3 100644 --- a/src/commands/mod.rs +++ b/src/commands/mod.rs @@ -119,7 +119,7 @@ pub(crate) fn builtin_exec(cmd: &str) -> Option { pub(crate) fn load_mode(cmd: &str, args: &ArgMatches) -> Option { match cmd { "service" => match args.subcommand_name() { - Some("init") => Some(LoadMode::ParsedExistingConfig), + Some("init" | "native-publication-init") => Some(LoadMode::ParsedExistingConfig), _ => Some(LoadMode::FullAppContext), }, "debug" => Some(LoadMode::FullAppContext), diff --git a/src/commands/service/mod.rs b/src/commands/service/mod.rs index 666dfb7c..204fc9ce 100644 --- a/src/commands/service/mod.rs +++ b/src/commands/service/mod.rs @@ -17,12 +17,19 @@ use crate::{ pub mod http; pub mod init; pub mod multi; +pub mod native_publication_init; pub mod ssh; const CONFIG_RELOAD_POLL_INTERVAL: Duration = Duration::from_secs(5); pub fn cli() -> Command { - let subcommands = vec![init::cli(), http::cli(), ssh::cli(), multi::cli()]; + let subcommands = vec![ + init::cli(), + native_publication_init::cli(), + http::cli(), + ssh::cli(), + multi::cli(), + ]; Command::new("service") .about("Start different kinds of server: for example https or ssh") .subcommands(subcommands) @@ -30,6 +37,9 @@ pub fn cli() -> Command { #[tokio::main] pub(crate) async fn exec(ctx: CommandContext, args: &ArgMatches) -> MegaResult { + if let Some(("native-publication-init", subcommand_args)) = args.subcommand() { + return native_publication_init::exec(ctx, subcommand_args).await; + } let config_path = ctx.config_path.clone(); let config_profile_path = ctx.config_profile_path.clone(); let config = require_config(ctx, "service")?; @@ -96,7 +106,7 @@ pub(crate) async fn exec(ctx: CommandContext, args: &ArgMatches) -> MegaResult { .await }; - match setup.await { + let result = match setup.await { Err(error) => cleanup_tail(Err(error), emitter, None, signal_forwarder).await, Ok(reload_watcher) => { let service = context.clone(); @@ -117,7 +127,16 @@ pub(crate) async fn exec(ctx: CommandContext, args: &ArgMatches) -> MegaResult { ) .await } + }; + if let Some(sink) = &context.storage.projection_observation_sink { + let deadline = std::time::Instant::now() + std::time::Duration::from_secs(5); + if sink.shutdown(deadline).await.is_err() && result.is_ok() { + return Err(MegaError::Other( + "typed projection writer did not drain successfully".into(), + )); + } } + result } /// Runs the service body under the unified cleanup tail (WH-13). @@ -256,7 +275,10 @@ mod tests { .map(|cmd| cmd.get_name().to_owned()) .collect::>(); - assert_eq!(names, vec!["init", "http", "ssh", "multi"]); + assert_eq!( + names, + vec!["init", "native-publication-init", "http", "ssh", "multi"] + ); } #[test] diff --git a/src/commands/service/native_publication_init.rs b/src/commands/service/native_publication_init.rs new file mode 100644 index 00000000..483f647c --- /dev/null +++ b/src/commands/service/native_publication_init.rs @@ -0,0 +1,198 @@ +use std::sync::Arc; + +use clap::{Arg, ArgAction, ArgMatches, Command}; +use sea_orm_migration::MigratorTrait; + +use crate::{ + commands::{CommandContext, require_config}, + common::errors::{MegaError, MegaResult}, + config::{DbConfig, loader::ConfigSource}, + jupiter::{ + migration::Migrator, + storage::{ + base_storage::{BaseStorage, StorageConnector}, + init::{postgres_connection, read_only_database_connection}, + mono_storage::MonoStorage, + native_publication_storage::NativeRoot, + }, + }, +}; + +#[path = "native_publication_init_preflight.rs"] +mod preflight; +use preflight::InitializationTarget; + +pub fn cli() -> Command { + let mut command = Command::new("native-publication-init") + .about("Prepare native publication in a stopped, paused and drained SHA-1 deployment"); + for (name, help) in [ + ("instance", "Deployment UUID; must match mst2.instance_uuid"), + ( + "expected-root-commit", + "Exact current native root commit object ID", + ), + ( + "expected-root-tree", + "Exact current native root tree object ID", + ), + ] { + command = command.arg(Arg::new(name).long(name).required(true).help(help)); + } + for (name, help) in [ + ("yes", "Confirm preparation of the native publication head"), + ( + "writers-stopped", + "Confirm all old writer processes have been stopped", + ), + ] { + command = command.arg( + Arg::new(name) + .long(name) + .action(ArgAction::SetTrue) + .required(true) + .help(help), + ); + } + command +} + +pub(crate) async fn exec(ctx: CommandContext, args: &ArgMatches) -> MegaResult { + if !ctx + .config_summary + .as_ref() + .is_some_and(|summary| matches!(summary.source, ConfigSource::Cli | ConfigSource::Env)) + { + return Err(MegaError::Other( + "MST2_NATIVE_INIT_CONFIG_REQUIRED: name the deployment with --config or MEGA_CONFIG" + .into(), + )); + } + if !args.get_flag("yes") || !args.get_flag("writers-stopped") { + return Err(MegaError::Other( + "MST2_NATIVE_INIT_CONFIRMATION_REQUIRED: --yes and --writers-stopped are required" + .into(), + )); + } + let config = require_config(ctx, "service native-publication-init")?; + config.validate()?; + if !config.mst2.enabled || !config.mst2.publication_enabled { + return Err(MegaError::Other( + "MST2_NATIVE_INIT_DISABLED: mst2.enabled and mst2.publication_enabled must be true" + .into(), + )); + } + let argument = |name| { + args.get_one::(name) + .map(String::as_str) + .ok_or_else(|| MegaError::Other(format!("--{name} is required"))) + }; + let target = InitializationTarget::parse( + config.monorepo.object_format.as_str(), + config.mst2.instance_uuid.as_deref(), + argument("instance")?, + argument("expected-root-commit")?, + argument("expected-root-tree")?, + ) + .map_err(|error| MegaError::Other(error.to_string()))?; + ensure_schema_current(&config.database).await?; + let db = Arc::new(postgres_connection(&config.database).await?); + let mono = MonoStorage { + base: BaseStorage::new(db.clone()), + }; + let result = mono + .initialize_native_publication_for_maintenance( + &target.instance, + &NativeRoot { + commit: target.commit, + tree: target.tree, + }, + ) + .await + .map_err(|error| MegaError::Other(error.to_string())); + let close = db.close_by_ref().await; + result?; + close?; + tracing::info!( + instance = %target.instance, + state = "INITIALIZING", + "Native publication initialization prepared; queue remains paused" + ); + Ok(()) +} + +async fn ensure_schema_current(config: &DbConfig) -> MegaResult { + let db = read_only_database_connection(config).await?; + let pending = Migrator::get_pending_migrations_read_only(&db).await; + let _ = db.close().await; + let pending = pending.map_err(|error| { + MegaError::Other(format!( + "MST2_NATIVE_INIT_SCHEMA_MISMATCH: cannot confirm the current schema: {error}" + )) + })?; + if !pending.is_empty() { + return Err(MegaError::Other(format!( + "MST2_NATIVE_INIT_SCHEMA_MISMATCH: {} migrations are pending; this command does not migrate", + pending.len() + ))); + } + Ok(()) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::commands::{LoadMode, load_mode}; + + #[test] + fn maintenance_cli_requires_each_confirmation_and_fixed_target() { + let arguments = [ + "native-publication-init", + "--yes", + "--writers-stopped", + "--instance", + "instance", + "--expected-root-commit", + "commit", + "--expected-root-tree", + "tree", + ]; + assert!(cli().try_get_matches_from(arguments).is_ok()); + for remove in [1, 2, 3, 5, 7] { + let count = if remove <= 2 { 1 } else { 2 }; + let mut incomplete = arguments.to_vec(); + incomplete.drain(remove..remove + count); + assert!(cli().try_get_matches_from(incomplete).is_err()); + } + let mut service_arguments = vec!["service"]; + service_arguments.extend(arguments); + let args = super::super::cli() + .try_get_matches_from(service_arguments) + .unwrap(); + assert_eq!( + load_mode("service", &args), + Some(LoadMode::ParsedExistingConfig) + ); + } + + #[tokio::test] + async fn maintenance_command_refuses_unnamed_config_before_any_database_io() { + let args = cli() + .try_get_matches_from([ + "native-publication-init", + "--yes", + "--writers-stopped", + "--instance", + "instance", + "--expected-root-commit", + "commit", + "--expected-root-tree", + "tree", + ]) + .unwrap(); + let error = exec(CommandContext::default(), &args).await.unwrap_err(); + assert!( + matches!(&error, MegaError::Other(message) if message.starts_with("MST2_NATIVE_INIT_CONFIG_REQUIRED")), + "unexpected error: {error}" + ); + } +} diff --git a/src/commands/service/native_publication_init_preflight.rs b/src/commands/service/native_publication_init_preflight.rs new file mode 100644 index 00000000..360751b6 --- /dev/null +++ b/src/commands/service/native_publication_init_preflight.rs @@ -0,0 +1,162 @@ +#[derive(Debug, thiserror::Error, PartialEq, Eq)] +pub(super) enum InitializationTargetError { + #[error( + "MST2_NATIVE_INIT_UNSUPPORTED_OBJECT_FORMAT: {0}; this maintenance entry supports sha1 only" + )] + UnsupportedObjectFormat(String), + #[error("MST2_NATIVE_INIT_INVALID_INSTANCE: both deployment instances must be non-nil UUIDs")] + InvalidInstance, + #[error("MST2_NATIVE_INIT_INSTANCE_MISMATCH: --instance differs from mst2.instance_uuid")] + InstanceMismatch, + #[error( + "MST2_NATIVE_INIT_INVALID_ROOT: expected commit and tree must be canonical SHA-1 object IDs" + )] + InvalidRoot, +} + +#[derive(Debug, PartialEq, Eq)] +pub(super) struct InitializationTarget { + pub(super) instance: String, + pub(super) commit: String, + pub(super) tree: String, +} + +impl InitializationTarget { + pub(super) fn parse( + object_format: &str, + configured_instance: Option<&str>, + requested_instance: &str, + commit: &str, + tree: &str, + ) -> Result { + if object_format != "sha1" { + return Err(InitializationTargetError::UnsupportedObjectFormat( + object_format.into(), + )); + } + let parse_instance = |value: &str| { + uuid::Uuid::parse_str(value) + .ok() + .filter(|id| !id.is_nil()) + .ok_or(InitializationTargetError::InvalidInstance) + }; + let configured = + parse_instance(configured_instance.ok_or(InitializationTargetError::InvalidInstance)?)?; + let requested = parse_instance(requested_instance)?; + if configured != requested { + return Err(InitializationTargetError::InstanceMismatch); + } + if [commit, tree].iter().any(|value| { + value.len() != 40 + || !value + .bytes() + .all(|byte| byte.is_ascii_digit() || (b'a'..=b'f').contains(&byte)) + }) { + return Err(InitializationTargetError::InvalidRoot); + } + Ok(Self { + instance: configured.to_string(), + commit: commit.into(), + tree: tree.into(), + }) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + const INSTANCE: &str = "6ab219b0-4275-45ba-9d7b-7b0b633018cd"; + + #[test] + fn target_is_canonical_and_bound_to_the_configured_instance() { + let target = InitializationTarget::parse( + "sha1", + Some(INSTANCE), + &INSTANCE.to_uppercase(), + &"a".repeat(40), + &"b".repeat(40), + ) + .unwrap(); + assert_eq!(target.instance, INSTANCE); + assert_eq!(target.commit, "a".repeat(40)); + assert_eq!(target.tree, "b".repeat(40)); + } + + #[test] + fn target_rejects_other_formats_without_inferring_from_equal_width() { + for format in ["sha256", "blake3", "unknown"] { + assert_eq!( + InitializationTarget::parse( + format, + Some(INSTANCE), + INSTANCE, + &"a".repeat(64), + &"b".repeat(64), + ), + Err(InitializationTargetError::UnsupportedObjectFormat( + format.into() + )) + ); + } + } + + #[test] + fn target_rejects_nil_missing_invalid_or_different_instances() { + for instance in [ + None, + Some("invalid"), + Some("00000000-0000-0000-0000-000000000000"), + ] { + assert_eq!( + InitializationTarget::parse( + "sha1", + instance, + INSTANCE, + &"a".repeat(40), + &"b".repeat(40) + ), + Err(InitializationTargetError::InvalidInstance) + ); + } + for instance in ["invalid", "00000000-0000-0000-0000-000000000000"] { + assert_eq!( + InitializationTarget::parse( + "sha1", + Some(INSTANCE), + instance, + &"a".repeat(40), + &"b".repeat(40) + ), + Err(InitializationTargetError::InvalidInstance) + ); + } + assert_eq!( + InitializationTarget::parse( + "sha1", + Some(INSTANCE), + "11111111-2222-4333-8444-555555555555", + &"a".repeat(40), + &"b".repeat(40) + ), + Err(InitializationTargetError::InstanceMismatch) + ); + } + + #[test] + fn target_rejects_noncanonical_or_wrong_width_object_ids() { + for value in [ + "a".repeat(39), + "a".repeat(64), + "A".repeat(40), + "z".repeat(40), + ] { + for (commit, tree) in [(&value, &"b".repeat(40)), (&"a".repeat(40), &value)] { + assert_eq!( + InitializationTarget::parse("sha1", Some(INSTANCE), INSTANCE, commit, tree), + Err(InitializationTargetError::InvalidRoot) + ); + } + } + } +} diff --git a/src/config/model.rs b/src/config/model.rs index ec549858..db2c6220 100644 --- a/src/config/model.rs +++ b/src/config/model.rs @@ -957,6 +957,9 @@ impl Default for LFSSshConfig { /// MST/2 snapshot feature flags (default off; spec 00 §6 rollout). #[derive(Serialize, Deserialize, Debug, Clone, Default)] pub struct Mst2Config { + /// Startup-only typed projection diagnostics, independent of log.level. + #[serde(default)] + pub projection_observation_enabled: bool, /// Master switch for the `/api/v2/snapshots` surface. #[serde(default)] pub enabled: bool, diff --git a/src/config/reload.rs b/src/config/reload.rs index bbb6d47f..4674938d 100644 --- a/src/config/reload.rs +++ b/src/config/reload.rs @@ -669,6 +669,12 @@ fn collect_static_restart_fields( candidate: &Config, report: &mut ConfigReloadReport, ) { + if current.mst2.projection_observation_enabled != candidate.mst2.projection_observation_enabled + { + report + .restart_required_fields + .push("mst2.projection_observation_enabled"); + } if current.base_dir != candidate.base_dir { report.restart_required_fields.push("base_dir"); } @@ -1209,6 +1215,33 @@ mod tests { } } + #[test] + fn projection_writer_toggle_requires_restart_and_preserves_live_snapshot() { + let temp = tempfile::tempdir().unwrap(); + let mut current = isolated_config(temp.path().join("config")); + current.mst2.enabled = true; + current.mst2.publication_enabled = true; + current.mst2.instance_uuid = Some("12345678-1234-4234-9234-123456789abc".into()); + let handle = ConfigHandle::new(current); + let before = handle.snapshot().unwrap(); + let mut candidate = before.as_ref().clone(); + candidate.mst2.projection_observation_enabled = true; + let report = handle.reload(candidate).unwrap(); + assert_eq!( + report.restart_required_fields, + vec!["mst2.projection_observation_enabled"] + ); + assert!(!report.applied()); + assert!(Arc::ptr_eq(&before, &handle.snapshot().unwrap())); + assert!( + !handle + .snapshot() + .unwrap() + .mst2 + .projection_observation_enabled + ); + } + #[test] fn reload_applies_log_fields_and_preserves_restart_required_database_fields() { let temp_dir = tempfile::tempdir().expect("temp dir"); diff --git a/src/config/validate.rs b/src/config/validate.rs index 5d67fd35..bb14e9a6 100644 --- a/src/config/validate.rs +++ b/src/config/validate.rs @@ -9,9 +9,9 @@ use url::Url; use super::{ ArtifactGcConfig, BlameConfig, BuckConfig, CedarConfig, Config, DbConfig, GitConfig, - GithubSyncConfig, LFSConfig, LogConfig, MonoConfig, MonoObjectFormat, NotificationConfig, - OAuthConfig, PackConfig, PushAuth, PushPolicy, RedisConfig, VAULT_AUDIT_SINKS, VaultConfig, - normalize_token_path, + GithubSyncConfig, LFSConfig, LogConfig, MonoConfig, MonoObjectFormat, Mst2Config, + NotificationConfig, OAuthConfig, PackConfig, PushAuth, PushPolicy, RedisConfig, + VAULT_AUDIT_SINKS, VaultConfig, normalize_token_path, secret::{SecretRef, SecretResolver, is_secret_ref_value}, }; use crate::{ @@ -124,6 +124,7 @@ impl Config { validate_lfs_config(&self.lfs)?; validate_redis_config(&self.redis)?; validate_cedar_config(&self.cedar)?; + validate_mst2_config(&self.mst2)?; if let Some(buck_config) = &self.buck { validate_buck_config(buck_config)?; } @@ -150,6 +151,39 @@ impl Config { } } +fn validate_mst2_config(config: &Mst2Config) -> Result<(), MegaError> { + if config.projection_observation_enabled && (!config.enabled || !config.publication_enabled) { + return Err(MegaError::Other( + "mst2.projection_observation_enabled requires mst2.enabled and mst2.publication_enabled".into(), + )); + } + if !config.enabled { + return Ok(()); + } + let instance_uuid = config.instance_uuid.as_deref().ok_or_else(|| { + MegaError::Other("mst2.instance_uuid is required when mst2.enabled is true".to_string()) + })?; + let instance_uuid = uuid::Uuid::parse_str(instance_uuid).map_err(|_| { + MegaError::Other("mst2.instance_uuid must be a valid non-nil UUID".to_string()) + })?; + if instance_uuid.is_nil() { + return Err(MegaError::Other( + "mst2.instance_uuid must be a valid non-nil UUID".to_string(), + )); + } + if config + .auth_token + .as_deref() + .is_some_and(|token| token.trim().is_empty()) + { + return Err(MegaError::Other( + "mst2.auth_token must not be empty when configured and mst2.enabled is true" + .to_string(), + )); + } + Ok(()) +} + /// Validate `[oauth]` settings: website API base URL, session cookie names, and /// each CORS origin must be a browser Origin of the form `scheme://host[:port]` /// (http/https, no path/query/fragment) that also parses as an HTTP header value @@ -2093,6 +2127,7 @@ pub(crate) fn known_fields(path: &str) -> Option<&'static [&'static str]> { "object_storage", "oauth", "blame", + "mst2", "redis", "buck", "artifacts_gc", @@ -2157,6 +2192,13 @@ pub(crate) fn known_fields(path: &str) -> Option<&'static [&'static str]> { "enable_caching", ]), "redis" => Some(&["url"]), + "mst2" => Some(&[ + "enabled", + "instance_uuid", + "publication_enabled", + "projection_observation_enabled", + "auth_token", + ]), "buck" => Some(&[ "session_timeout", "max_file_size", @@ -2263,7 +2305,7 @@ mod tests { StorageEventsTargetConfig, ViewsConfig, secret::{SecretRef, SecretResolver}, template::config_init_template, - testing::{env_lock, isolated_config}, + testing::{EnvVarGuard, env_lock, isolated_config}, }; #[rustfmt::skip] use crate::orbit_api::factory::{GcsConfig, LocalConfig, ObjectStorageBackend, S3Config}; @@ -2279,6 +2321,148 @@ mod tests { assert!(err.to_string().contains(field), "{err}"); } + #[test] + fn typed_projection_writer_is_default_off_and_requires_native_publication() { + assert!(!Mst2Config::default().projection_observation_enabled); + let loaded: Mst2Config = toml::from_str("enabled = false").unwrap(); + assert!(!loaded.projection_observation_enabled); + let mut config = valid_config(); + config.mst2.projection_observation_enabled = true; + config.mst2.instance_uuid = Some("12345678-1234-4234-9234-123456789abc".into()); + for (enabled, publication) in [(false, false), (false, true), (true, false)] { + config.mst2.enabled = enabled; + config.mst2.publication_enabled = publication; + assert!( + config + .validate() + .unwrap_err() + .to_string() + .contains("projection_observation_enabled") + ); + } + config.mst2.enabled = true; + config.mst2.publication_enabled = true; + config.validate().unwrap(); + assert!(is_known_field_path("mst2.projection_observation_enabled")); + } + + #[test] + fn config_validate_mst2_checks_enabled_identity_and_configured_token() { + let mut config = valid_config(); + config.mst2.enabled = true; + for instance_uuid in [ + None, + Some(""), + Some("invalid-uuid"), + Some("00000000-0000-0000-0000-000000000000"), + ] { + config.mst2.instance_uuid = instance_uuid.map(str::to_owned); + config.mst2.auth_token = Some("mst2-private-token-must-not-appear".to_string()); + let message = config + .validate() + .expect_err("enabled MST/2 requires a non-nil UUID") + .to_string(); + assert!(message.contains("mst2.instance_uuid"), "{message}"); + assert!( + !message.contains("mst2-private-token-must-not-appear"), + "{message}" + ); + } + config.mst2.instance_uuid = Some("12345678-1234-4234-9234-123456789abc".to_string()); + for token in ["", " \t\n"] { + config.mst2.auth_token = Some(token.to_string()); + let message = config + .validate() + .expect_err("configured MST/2 token must not be blank") + .to_string(); + assert!(message.contains("mst2.auth_token"), "{message}"); + } + config.mst2.auth_token = None; + config + .validate() + .expect("existing unauthenticated lab mode remains valid"); + config.mst2.auth_token = Some("mst2-test-token".to_string()); + config.mst2.publication_enabled = true; + config + .validate() + .expect("authenticated publication config is valid"); + config.mst2.enabled = false; + config.mst2.publication_enabled = false; + config.mst2.instance_uuid = None; + config.mst2.auth_token = None; + config + .validate() + .expect("disabled MST/2 defaults remain valid"); + } + + #[test] + fn config_load_str_accepts_mst2_and_expands_file_token_without_reexpansion() { + let lock = env_lock(); + let _enabled = EnvVarGuard::remove(&lock, "MEGA_MST2__ENABLED"); + let _identity = EnvVarGuard::remove(&lock, "MEGA_MST2__INSTANCE_UUID"); + let _publication = EnvVarGuard::remove(&lock, "MEGA_MST2__PUBLICATION_ENABLED"); + let _token = EnvVarGuard::remove(&lock, "MEGA_MST2__AUTH_TOKEN"); + let dir = tempfile::tempdir().expect("temp dir"); + let token_file = dir.path().join("mst2-token"); + let token = "literal-${must_not_be_reexpanded}"; + std::fs::write(&token_file, format!("{token}\n")).expect("write private test token"); + let file_placeholder = format!("${{file:{}}}", token_file.display()); + let content = format!( + "{}\n[mst2]\nenabled = true\ninstance_uuid = \"12345678-1234-4234-9234-123456789abc\"\npublication_enabled = false\nauth_token = {}\n", + config_init_template(dir.path()), + toml::Value::String(file_placeholder), + ); + let loaded = + Config::load_str(&content).expect("ordinary config loader must recognize MST/2"); + assert!(loaded.mst2.enabled); + assert!(!loaded.mst2.publication_enabled); + assert_eq!(loaded.mst2.auth_token.as_deref(), Some(token)); + loaded + .validate() + .expect("expanded MST/2 config must validate"); + assert!( + known_unconsumed_fields(&toml::from_str(&content).expect("parse source")).is_empty() + ); + for field in [ + "enabled", + "instance_uuid", + "publication_enabled", + "auth_token", + ] { + assert!(is_known_field_path(&format!("mst2.{field}"))); + } + assert!(!is_known_field_path("mst2.auth_token.extra")); + assert!(!is_known_field_path("mst2.typo")); + std::fs::write(&token_file, "\n").expect("write empty private test token"); + let empty_token = + Config::load_str(&content).expect("file token still uses ordinary expansion"); + let message = empty_token + .validate() + .expect_err("expanded empty token must fail semantic validation") + .to_string(); + assert!(message.contains("mst2.auth_token"), "{message}"); + } + + #[test] + fn config_load_str_rejects_mst2_typo_and_unknown_nested_table() { + let content = r#" + [mst2] + enabled = false + publication_enabeld = true + [mst2.credentials] + auth_token = "private-value-must-not-appear" + "#; + let message = Config::load_str(content) + .expect_err("MST/2 unknown fields must fail closed") + .to_string(); + assert!(message.contains("mst2.publication_enabeld"), "{message}"); + assert!(message.contains("mst2.credentials"), "{message}"); + assert!( + !message.contains("private-value-must-not-appear"), + "{message}" + ); + } + #[test] fn config_validate_accepts_default_isolated_test_config() { valid_config() diff --git a/src/context/mod.rs b/src/context/mod.rs index dd97ed20..a4c05d22 100644 --- a/src/context/mod.rs +++ b/src/context/mod.rs @@ -295,6 +295,22 @@ impl AppContext { } }; + let mut storage = storage; + if config.mst2.projection_observation_enabled { + match crate::ceres::snapshot::projection_writer::ProjectionObservationSink::start( + &crate::config::mega_cache(), + ) { + Ok(sink) => storage.projection_observation_sink = Some(sink), + Err(_) => { + notification_shutdown.cancel(); + storage.storage_event_emitter.shutdown().await; + return Err(MegaError::Other( + "typed projection writer failed startup".into(), + )); + } + } + } + Ok(Self { storage, vault, diff --git a/src/jupiter/migration/m20261005_000100_add_mst2_publication_request_digest.rs b/src/jupiter/migration/m20261005_000100_add_mst2_publication_request_digest.rs new file mode 100644 index 00000000..8186b7c4 --- /dev/null +++ b/src/jupiter/migration/m20261005_000100_add_mst2_publication_request_digest.rs @@ -0,0 +1,48 @@ +use sea_orm_migration::prelude::*; + +#[derive(DeriveMigrationName)] +pub struct Migration; + +#[async_trait::async_trait] +impl MigrationTrait for Migration { + async fn up(&self, manager: &SchemaManager) -> Result<(), DbErr> { + manager + .get_connection() + .execute_unprepared( + r#"ALTER TABLE mst2_publication + ADD COLUMN IF NOT EXISTS request_digest text, + ADD COLUMN IF NOT EXISTS request_digest_version integer; + DO $$ BEGIN + ALTER TABLE mst2_publication + ADD CONSTRAINT mst2_publication_request_digest_valid CHECK ( + (request_digest IS NULL AND request_digest_version IS NULL) OR + (request_digest IS NOT NULL AND request_digest_version IS NOT NULL + AND request_digest_version > 0 + AND request_digest ~ '^sha256:[0-9a-f]{64}$') + ); + EXCEPTION WHEN duplicate_object THEN NULL; + END $$; + CREATE TABLE IF NOT EXISTS mst2_queue_noop_receipt ( + id bigserial PRIMARY KEY, + operation_id text NOT NULL UNIQUE, + namespace text NOT NULL, + request_digest text NOT NULL CHECK (request_digest ~ '^sha256:[0-9a-f]{64}$'), + request_digest_version integer NOT NULL CHECK (request_digest_version > 0), + writer_epoch bigint NOT NULL, + writer_kind text NOT NULL, + observed_sequence bigint NOT NULL CHECK (observed_sequence >= 0), + observed_root_commit text NOT NULL, + observed_root_tree text NOT NULL, + landed_commit_id text NOT NULL, + created_at timestamptz NOT NULL + );"#, + ) + .await?; + Ok(()) + } + + async fn down(&self, _manager: &SchemaManager) -> Result<(), DbErr> { + // Receipts remain immutable, including the absence of legacy digests. + Ok(()) + } +} diff --git a/src/jupiter/migration/m20261005_000100_add_mst2_retention_durability.rs b/src/jupiter/migration/m20261005_000100_add_mst2_retention_durability.rs new file mode 100644 index 00000000..8f223af1 --- /dev/null +++ b/src/jupiter/migration/m20261005_000100_add_mst2_retention_durability.rs @@ -0,0 +1,119 @@ +//! T06-B: durable retention counters and replayable GC operations. +//! +//! The original T06 graph migration intentionally supplied only the graph +//! rows. This forward-only migration adds the counter used by the atomic +//! LIVE→DELETING CAS and an idempotent operation log for crash recovery. The +//! runtime is not switched to this schema by this migration alone. + +use sea_orm_migration::{prelude::*, schema::*}; + +#[derive(DeriveMigrationName)] +pub struct Migration; + +#[async_trait::async_trait] +impl MigrationTrait for Migration { + async fn up(&self, manager: &SchemaManager) -> Result<(), DbErr> { + manager + .alter_table( + Table::alter() + .table(Mst2RetentionNode::Table) + .add_column( + big_integer(Mst2RetentionNode::IncomingRefs) + .not_null() + .default(0), + ) + .to_owned(), + ) + .await?; + + manager + .create_index( + Index::create() + .if_not_exists() + .name("idx_mst2_retention_node_gc") + .table(Mst2RetentionNode::Table) + .col(Mst2RetentionNode::State) + .col(Mst2RetentionNode::IncomingRefs) + .to_owned(), + ) + .await?; + + manager + .create_table( + Table::create() + .table(Mst2RetentionGcOp::Table) + .if_not_exists() + .col( + string(Mst2RetentionGcOp::OperationId) + .not_null() + .primary_key(), + ) + .col(string(Mst2RetentionGcOp::NodeId).not_null()) + .col(string(Mst2RetentionGcOp::Operation).not_null()) + .col( + string(Mst2RetentionGcOp::State) + .not_null() + .default("PENDING"), + ) + .col(integer(Mst2RetentionGcOp::Attempts).not_null().default(0)) + .col( + timestamp_with_time_zone(Mst2RetentionGcOp::CreatedAt) + .not_null() + .default(Expr::current_timestamp()), + ) + .col(timestamp_with_time_zone_null( + Mst2RetentionGcOp::CompletedAt, + )) + .to_owned(), + ) + .await?; + + manager + .create_index( + Index::create() + .if_not_exists() + .name("idx_mst2_retention_gc_op_pending") + .table(Mst2RetentionGcOp::Table) + .col(Mst2RetentionGcOp::State) + .col(Mst2RetentionGcOp::CreatedAt) + .to_owned(), + ) + .await?; + + manager + .create_index( + Index::create() + .if_not_exists() + .name("idx_mst2_retention_gc_op_node") + .table(Mst2RetentionGcOp::Table) + .col(Mst2RetentionGcOp::NodeId) + .to_owned(), + ) + .await + } + + async fn down(&self, _manager: &SchemaManager) -> Result<(), DbErr> { + // Forward-only: retention state and replay evidence are never + // dropped automatically. + Ok(()) + } +} + +#[derive(DeriveIden)] +enum Mst2RetentionNode { + Table, + State, + IncomingRefs, +} + +#[derive(DeriveIden)] +enum Mst2RetentionGcOp { + Table, + OperationId, + NodeId, + Operation, + State, + Attempts, + CreatedAt, + CompletedAt, +} diff --git a/src/jupiter/migration/m20261005_000200_add_mst2_native_head.rs b/src/jupiter/migration/m20261005_000200_add_mst2_native_head.rs new file mode 100644 index 00000000..e29118cf --- /dev/null +++ b/src/jupiter/migration/m20261005_000200_add_mst2_native_head.rs @@ -0,0 +1,72 @@ +use sea_orm_migration::prelude::*; + +#[derive(DeriveMigrationName)] +pub struct Migration; + +#[async_trait::async_trait] +impl MigrationTrait for Migration { + async fn up(&self, manager: &SchemaManager) -> Result<(), DbErr> { + manager + .get_connection() + .execute_unprepared( + r#" + ALTER TABLE mst2_publication + ADD COLUMN IF NOT EXISTS native_certificate_version integer; + ALTER TABLE push_queue + ADD COLUMN IF NOT EXISTS expected_native_sequence bigint, + ADD COLUMN IF NOT EXISTS expected_native_epoch bigint, + ADD COLUMN IF NOT EXISTS expected_native_certificate bigint; + CREATE TABLE IF NOT EXISTS mst2_native_publication ( + receipt_id bigint PRIMARY KEY REFERENCES mst2_publication(id), + namespace text NOT NULL CHECK (namespace = '/'), + instance_id text NOT NULL, + sequence bigint NOT NULL CHECK (sequence > 0), + writer_epoch bigint NOT NULL CHECK (writer_epoch > 0), + old_root_commit text NOT NULL, + old_root_tree text NOT NULL, + root_commit text NOT NULL, + root_tree text NOT NULL, + origin_path text NOT NULL, + origin_ref text NOT NULL, + old_path_commit text, + old_path_tree text, + path_commit text NOT NULL, + path_tree text NOT NULL, + CHECK ((old_path_commit IS NULL) = (old_path_tree IS NULL)), + CHECK (old_path_commit IS DISTINCT FROM path_commit), + UNIQUE (namespace, sequence) + ); + CREATE TABLE IF NOT EXISTS mst2_native_head ( + namespace text PRIMARY KEY CHECK (namespace = '/'), + instance_id text NOT NULL, + sequence bigint NOT NULL CHECK (sequence >= 0), + writer_epoch bigint NOT NULL CHECK (writer_epoch > 0), + root_commit text NOT NULL, + root_tree text NOT NULL, + state text NOT NULL CHECK (state IN ('INITIALIZING', 'READY')), + certificate_receipt_id bigint REFERENCES mst2_native_publication(receipt_id), + CHECK ((state = 'READY') = (certificate_receipt_id IS NOT NULL)) + ); + DO $$ BEGIN + ALTER TABLE mst2_publication ADD CONSTRAINT mst2_native_certificate_version_valid + CHECK (native_certificate_version IS NULL OR native_certificate_version > 0); + EXCEPTION WHEN duplicate_object THEN NULL; END $$; + DO $$ BEGIN + ALTER TABLE push_queue ADD CONSTRAINT mst2_expected_native_token_valid CHECK ( + (expected_native_sequence IS NULL AND expected_native_epoch IS NULL + AND expected_native_certificate IS NULL) OR + (expected_native_sequence IS NOT NULL AND expected_native_sequence >= 0 + AND expected_native_epoch IS NOT NULL AND expected_native_epoch > 0) + ); + EXCEPTION WHEN duplicate_object THEN NULL; END $$; + "#, + ) + .await?; + Ok(()) + } + + async fn down(&self, _manager: &SchemaManager) -> Result<(), DbErr> { + // Historical native certificates and queue fences are never erased. + Ok(()) + } +} diff --git a/src/jupiter/migration/m20261005_000200_harden_mst2_retention_graph.rs b/src/jupiter/migration/m20261005_000200_harden_mst2_retention_graph.rs new file mode 100644 index 00000000..c42bcd28 --- /dev/null +++ b/src/jupiter/migration/m20261005_000200_harden_mst2_retention_graph.rs @@ -0,0 +1,76 @@ +//! T06-B: backfill graph counters and reject invalid durable graph state. +//! +//! Applies after the additive schema migration. Existing edges must form a +//! valid DAG and counters are derived from unique edges before any collector +//! can use them. Constraints keep failed writers from silently dangling a +//! node or underflowing a counter. Completed GC receipts intentionally do not +//! reference the removed node through a foreign key. + +use sea_orm_migration::prelude::*; + +#[derive(DeriveMigrationName)] +pub struct Migration; + +#[async_trait::async_trait] +impl MigrationTrait for Migration { + async fn up(&self, manager: &SchemaManager) -> Result<(), DbErr> { + let conn = manager.get_connection(); + conn.execute_unprepared( + "LOCK TABLE mst2_retention_node, mst2_retention_edge, mst2_retention_root, \ + mst2_retention_gc_op IN ACCESS EXCLUSIVE MODE", + ) + .await?; + let cycle = conn + .query_one_raw(sea_orm::Statement::from_string( + sea_orm::DbBackend::Postgres, + "WITH RECURSIVE walk(start_id, node_id) AS ( \ + SELECT parent_id, child_id FROM mst2_retention_edge \ + UNION SELECT w.start_id, e.child_id FROM walk w \ + JOIN mst2_retention_edge e ON e.parent_id = w.node_id \ + ) SELECT start_id FROM walk WHERE start_id = node_id LIMIT 1", + )) + .await?; + if cycle.is_some() { + return Err(DbErr::Migration( + "MST/2 retention graph contains a cycle".into(), + )); + } + conn.execute_unprepared( + "UPDATE mst2_retention_node n SET incoming_refs = \ + (SELECT count(*) FROM mst2_retention_edge e WHERE e.child_id = n.node_id)", + ) + .await?; + conn.execute_unprepared( + "ALTER TABLE mst2_retention_node \ + ADD CONSTRAINT mst2_retention_node_state_check CHECK (state IN ('LIVE', 'DELETING')), \ + ADD CONSTRAINT mst2_retention_node_kind_check \ + CHECK (kind IN ('page', 'chunk_map', 'frame', 'verified_object')), \ + ADD CONSTRAINT mst2_retention_node_nonnegative CHECK (bytes >= 0 AND incoming_refs >= 0); \ + ALTER TABLE mst2_retention_edge \ + ADD CONSTRAINT mst2_retention_edge_parent_fk FOREIGN KEY (parent_id) \ + REFERENCES mst2_retention_node(node_id), \ + ADD CONSTRAINT mst2_retention_edge_child_fk FOREIGN KEY (child_id) \ + REFERENCES mst2_retention_node(node_id), \ + ADD CONSTRAINT mst2_retention_edge_no_self CHECK (parent_id <> child_id); \ + ALTER TABLE mst2_retention_root \ + ADD CONSTRAINT mst2_retention_root_node_fk FOREIGN KEY (node_id) \ + REFERENCES mst2_retention_node(node_id), \ + ADD CONSTRAINT mst2_retention_root_kind_check CHECK (root_kind IN ('lease', 'pin', 'prepare')); \ + ALTER TABLE mst2_retention_gc_op \ + ADD CONSTRAINT mst2_retention_gc_op_kind_check CHECK (operation IN ('MARK_DELETING', 'REMOVE')), \ + ADD CONSTRAINT mst2_retention_gc_op_state_check CHECK (state IN ('PENDING', 'APPLIED', 'FAILED')), \ + ADD CONSTRAINT mst2_retention_gc_op_attempts_check CHECK (attempts >= 0), \ + ADD CONSTRAINT mst2_retention_gc_op_completion_check \ + CHECK ((state = 'APPLIED') = (completed_at IS NOT NULL)); \ + CREATE UNIQUE INDEX idx_mst2_retention_gc_op_pending_node \ + ON mst2_retention_gc_op(node_id) WHERE state = 'PENDING'", + ) + .await + .map(|_| ()) + } + + async fn down(&self, _manager: &SchemaManager) -> Result<(), DbErr> { + // Forward-only: preserve retained nodes and GC recovery evidence. + Ok(()) + } +} diff --git a/src/jupiter/migration/m20261005_000300_add_mst2_metadata_install.rs b/src/jupiter/migration/m20261005_000300_add_mst2_metadata_install.rs new file mode 100644 index 00000000..d0e8d168 --- /dev/null +++ b/src/jupiter/migration/m20261005_000300_add_mst2_metadata_install.rs @@ -0,0 +1,76 @@ +//! Additive, forward-only native metadata installation records. + +use mst2_codec::metapage::{HEADER_LEN, PAGE_MAX_BYTES}; +use sea_orm_migration::prelude::*; + +#[derive(DeriveMigrationName)] +pub struct Migration; + +#[async_trait::async_trait] +impl MigrationTrait for Migration { + async fn up(&self, manager: &SchemaManager) -> Result<(), DbErr> { + manager.get_connection().execute_unprepared(&format!( + "CREATE TABLE mst2_metadata_storage_scope ( + singleton smallint PRIMARY KEY CHECK (singleton = 1), + storage_uuid text NOT NULL UNIQUE + ); + CREATE TABLE mst2_metadata_payload ( + page_id bytea PRIMARY KEY CHECK (octet_length(page_id) = 32), + metadata_codec smallint NOT NULL CHECK (metadata_codec = 1), + byte_size integer NOT NULL CHECK (byte_size BETWEEN {HEADER_LEN} AND {PAGE_MAX_BYTES}), + payload bytea NOT NULL CHECK (octet_length(payload) = byte_size), + created_at timestamptz NOT NULL DEFAULT now() + ); + CREATE OR REPLACE FUNCTION mst2_metadata_payload_immutable() RETURNS trigger LANGUAGE plpgsql AS $$ + BEGIN RAISE EXCEPTION 'MST2 metadata payloads cannot be updated or deleted'; END $$; + CREATE TRIGGER mst2_metadata_payload_immutable BEFORE UPDATE OR DELETE + ON mst2_metadata_payload FOR EACH ROW EXECUTE FUNCTION mst2_metadata_payload_immutable(); + CREATE TRIGGER mst2_metadata_scope_immutable BEFORE UPDATE OR DELETE + ON mst2_metadata_storage_scope FOR EACH ROW EXECUTE FUNCTION mst2_metadata_payload_immutable(); + CREATE TABLE mst2_metadata_prepare ( + prepare_id text PRIMARY KEY CHECK (prepare_id ~ '^[0-9a-f]{{8}}-[0-9a-f]{{4}}-4[0-9a-f]{{3}}-[89ab][0-9a-f]{{3}}-[0-9a-f]{{12}}$'), + operation_id text NOT NULL UNIQUE CHECK (octet_length(operation_id) BETWEEN 1 AND 255), + manifest_digest bytea NOT NULL CHECK (octet_length(manifest_digest) = 32), + canonical_plan bytea NOT NULL CHECK (octet_length(canonical_plan) <= 2097152), + source_domain text NOT NULL CHECK (source_domain = 'native-git'), + tagged_root_tree_oid text NOT NULL, + scope text NOT NULL CHECK (octet_length(scope) <= 4096), + schema_version smallint NOT NULL, + metadata_codec smallint NOT NULL CHECK (metadata_codec = 1), + materialization_policy smallint NOT NULL, + fs_semantics smallint NOT NULL, + access_projection smallint NOT NULL, + verification_revision integer NOT NULL, + projection_revision smallint NOT NULL, + metadata_root bytea NOT NULL CHECK (octet_length(metadata_root) = 32), + node_count integer NOT NULL CHECK (node_count BETWEEN 1 AND 4096), + edge_count integer NOT NULL CHECK (edge_count BETWEEN 0 AND 16384), + total_bytes bigint NOT NULL CHECK (total_bytes BETWEEN 0 AND 67108864), + state text NOT NULL CHECK (state IN ('PREPARING', 'COMMITTED')), + created_at timestamptz NOT NULL DEFAULT now(), + committed_at timestamptz, + CHECK ((state = 'COMMITTED') = (committed_at IS NOT NULL)) + ); + CREATE TABLE mst2_metadata_prepare_page ( + prepare_id text NOT NULL REFERENCES mst2_metadata_prepare(prepare_id), + page_id bytea NOT NULL CHECK (octet_length(page_id) = 32), + expected_size integer NOT NULL CHECK (expected_size BETWEEN {HEADER_LEN} AND {PAGE_MAX_BYTES}), + PRIMARY KEY (prepare_id, page_id) + )" + )).await?; + manager + .get_connection() + .execute_raw(sea_orm::Statement::from_sql_and_values( + sea_orm::DbBackend::Postgres, + "INSERT INTO mst2_metadata_storage_scope(singleton,storage_uuid) VALUES(1,$1)", + [uuid::Uuid::new_v4().to_string().into()], + )) + .await + .map(|_| ()) + } + + async fn down(&self, _manager: &SchemaManager) -> Result<(), DbErr> { + // Preserve immutable payloads, prepare pins and recovery evidence. + Ok(()) + } +} diff --git a/src/jupiter/migration/mod.rs b/src/jupiter/migration/mod.rs index 61d94c37..7f964e4a 100644 --- a/src/jupiter/migration/mod.rs +++ b/src/jupiter/migration/mod.rs @@ -139,6 +139,11 @@ mod m20260921_000100_fix_git_tag_unique; mod m20260923_000100_import_repo_cleanups; mod m20260923_000200_canonicalize_import_repo_paths; mod m20260925_000100_media_paging; +mod m20261005_000100_add_mst2_publication_request_digest; +mod m20261005_000100_add_mst2_retention_durability; +mod m20261005_000200_add_mst2_native_head; +mod m20261005_000200_harden_mst2_retention_graph; +mod m20261005_000300_add_mst2_metadata_install; mod m20261006_000100_add_view_tables; mod runner; pub use m20260905_000100_add_push_queue::ensure_queue_control_seed; @@ -267,6 +272,11 @@ impl MigratorTrait for Migrator { Box::new(m20260923_000100_import_repo_cleanups::Migration), Box::new(m20260923_000200_canonicalize_import_repo_paths::Migration), Box::new(m20260925_000100_media_paging::Migration), + Box::new(m20261005_000100_add_mst2_publication_request_digest::Migration), + Box::new(m20261005_000200_add_mst2_native_head::Migration), + Box::new(m20261005_000100_add_mst2_retention_durability::Migration), + Box::new(m20261005_000200_harden_mst2_retention_graph::Migration), + Box::new(m20261005_000300_add_mst2_metadata_install::Migration), Box::new(m20261006_000100_add_view_tables::Migration), ] } @@ -1178,13 +1188,18 @@ mod tests { async fn import_repo_alias_rows_canonicalized() { let names = migration_names(); assert_eq!( - &names[names.len() - 3..], + &names[names.len() - 8..], &[ "m20260923_000200_canonicalize_import_repo_paths".to_string(), "m20260925_000100_media_paging".to_string(), + "m20261005_000100_add_mst2_publication_request_digest".to_string(), + "m20261005_000200_add_mst2_native_head".to_string(), + "m20261005_000100_add_mst2_retention_durability".to_string(), + "m20261005_000200_harden_mst2_retention_graph".to_string(), + "m20261005_000300_add_mst2_metadata_install".to_string(), VIEW_MIGRATION_NAME.to_string(), ], - "view tables are registered last" + "native retention, metadata installation and view tables follow media paging" ); let db = alias_db().await; diff --git a/src/jupiter/service/native_publication_push_tests.rs b/src/jupiter/service/native_publication_push_tests.rs new file mode 100644 index 00000000..8f2f9f76 --- /dev/null +++ b/src/jupiter/service/native_publication_push_tests.rs @@ -0,0 +1,384 @@ +use bytes::Bytes; + +use crate::callisto::{ + mst2_native_head, mst2_native_publication, mst2_publication, mst2_publication_outbox, +}; +use sea_orm::{ColumnTrait, ConnectionTrait, EntityTrait, PaginatorTrait, QueryFilter, Statement}; +use crate::jupiter::utils::converter::FromMegaModel; + +const NATIVE_INSTANCE: &str = "6ab219b0-4275-45ba-9d7b-7b0b633018cd"; + +async fn native_fixture() -> (tempfile::TempDir, crate::jupiter::storage::Storage, git_internal::internal::object::commit::Commit, String) { + let temp = tempfile::tempdir().unwrap(); + let mut config = crate::config::testing::isolated_config(temp.path().join("config")); + config.monorepo.push_policy = PushPolicy::Trunk; + config.mst2.enabled = true; + config.mst2.publication_enabled = true; + config.mst2.instance_uuid = Some(NATIVE_INSTANCE.to_owned()); + let storage = crate::jupiter::tests::test_storage_with_config(temp.path(), config).await; + let storage = crate::jupiter::tests::with_test_vault(storage, temp.path()).await; + let name = format!("native-{}", uuid::Uuid::new_v4().simple()); + use git_internal::internal::object::{ + commit::Commit, + tree::{Tree, TreeItem, TreeItemMode}, + }; + let keep_oid = storage.git_service.save_object_from_raw(Bytes::from_static(b"keep")).await.unwrap(); + let old_oid = storage.git_service.save_object_from_raw(Bytes::from_static(b"old")).await.unwrap(); + let child = Tree::from_tree_items(vec![wh03_blob_item("x.txt", &old_oid)]).unwrap(); + let root_tree = Tree::from_tree_items(vec![ + wh03_blob_item(".gitkeep", &keep_oid), + TreeItem::new(TreeItemMode::Tree, child.id, name.clone()), + ]).unwrap(); + let root_commit = Commit::from_tree_id(root_tree.id, vec![], "root"); + let tip = Commit::from_tree_id(child.id, vec![], "path tip"); + let mono = storage.mono_storage(); + mono.save_mega_trees(vec![child.clone(), root_tree.clone()], root_commit.id, None).await.unwrap(); + mono.save_mega_commits(vec![root_commit.clone(), tip.clone()], None).await.unwrap(); + mono.save_refs(mega_refs::Model::new( + "/", MEGA_BRANCH_NAME.to_owned(), root_commit.id.to_string(), root_tree.id.to_string(), false, + ), None).await.unwrap(); + let path = format!("/{name}"); + mono.save_refs(mega_refs::Model::new( + path.clone(), MEGA_BRANCH_NAME.to_owned(), tip.id.to_string(), child.id.to_string(), false, + ), None).await.unwrap(); + storage.push_queue_service.push_queue_storage.set_control_flags(Some(true), None, None).await.unwrap(); + mono.initialize_native_publication_for_maintenance( + NATIVE_INSTANCE, + &crate::jupiter::storage::native_publication_storage::NativeRoot { + commit: root_commit.id.to_string(), tree: root_tree.id.to_string(), + }, + ).await.unwrap(); + assert!(mono.read_native_publication_head(NATIVE_INSTANCE).await.is_err()); + storage.push_queue_service.push_queue_storage.set_control_flags(Some(false), None, None).await.unwrap(); + (temp, storage, tip, path) +} + +async fn save_same_tree_commit(storage: &crate::jupiter::storage::Storage, parent: &git_internal::internal::object::commit::Commit) -> (String, PushPayload) { + let commit = git_internal::internal::object::commit::Commit::from_tree_id(parent.tree_id, vec![parent.id], "same native tree n1"); + storage.mono_storage().save_mega_commits(vec![commit.clone()], None).await.unwrap(); + let oid = commit.id.to_string(); + (oid.clone(), PushPayload { commits: vec![oid], fork_base: Some(parent.id.to_string()), n: 1 }) +} + +#[tokio::test] +async fn real_same_tree_push_advances_one_global_certificate_and_n0_replay_stays_fixed() { + let (_temp, storage, tip, path) = native_fixture().await; + let mono = storage.mono_storage(); + let before = mono.get_main_ref("/").await.unwrap().unwrap(); + let (new, payload) = save_same_tree_commit(&storage, &tip).await; + let id = wh03_enqueue_push(&storage, &path, &tip.id.to_string(), &new, &payload).await; + assert!(matches!(wh03_exec(&storage,id).await, ExecuteOutcome::Done { root_cas_writes:1,.. })); + let after = mono.get_main_ref("/").await.unwrap().unwrap(); + assert_eq!((after.ref_commit_hash,after.ref_tree_hash),(before.ref_commit_hash.clone(),before.ref_tree_hash.clone())); + let head = mono.read_native_publication_head(NATIVE_INSTANCE).await.unwrap(); + assert_eq!(head.token.sequence,1); + let receipt = mst2_publication::Entity::find().filter(mst2_publication::Column::OperationId.eq(format!("mst2:trunk-queue:{id}"))) + .one(mono.get_connection()).await.unwrap().unwrap(); + assert_eq!(receipt.namespace,path); + assert_eq!(receipt.new_oid,new); + assert_eq!(receipt.native_certificate_version,Some(1)); + assert_ne!(receipt.new_oid,head.root.commit); + let noop = PushPayload { commits:vec![],fork_base:None,n:0 }; + let noop_id = wh03_enqueue_push(&storage,&path,&new,&new,&noop).await; + assert!(matches!(wh03_exec(&storage,noop_id).await, ExecuteOutcome::Done { root_cas_writes:1,.. })); + assert_eq!(mono.read_native_publication_head(NATIVE_INSTANCE).await.unwrap().token,head.token); + mono.get_connection().execute_raw(Statement::from_sql_and_values(mono.get_connection().get_database_backend(), + "UPDATE push_queue SET status='Running',landed_commit_id=NULL WHERE id=$1",[id.into()], + )).await.unwrap(); + assert!(matches!(wh03_exec(&storage,id).await, ExecuteOutcome::Done { root_cas_writes:0,.. })); + assert_eq!(mono.read_native_publication_head(NATIVE_INSTANCE).await.unwrap().token,head.token); + assert_eq!(mst2_native_publication::Entity::find().count(mono.get_connection()).await.unwrap(),1); +} + +#[tokio::test] +async fn maintenance_cannot_reinitialize_published_history_after_head_loss() { + let (_temp, storage, tip, path) = native_fixture().await; + let mono = storage.mono_storage(); + let (new, payload) = save_same_tree_commit(&storage, &tip).await; + let id = wh03_enqueue_push(&storage, &path, &tip.id.to_string(), &new, &payload).await; + assert!(matches!(wh03_exec(&storage, id).await, ExecuteOutcome::Done { .. })); + let head = mono.read_native_publication_head(NATIVE_INSTANCE).await.unwrap(); + let certificates = mst2_native_publication::Entity::find().all(mono.get_connection()).await.unwrap(); + let receipts = mst2_publication::Entity::find().all(mono.get_connection()).await.unwrap(); + let outbox = mst2_publication_outbox::Entity::find().all(mono.get_connection()).await.unwrap(); + assert_eq!(certificates.len(), 1); + storage.push_queue_service.push_queue_storage.set_control_flags(Some(true), None, None).await.unwrap(); + mono.get_connection().execute_unprepared("DELETE FROM mst2_native_head").await.unwrap(); + for instance in [NATIVE_INSTANCE, "11111111-2222-4333-8444-555555555555"] { + let error = mono.initialize_native_publication_for_maintenance(instance, &head.root).await.unwrap_err(); + assert!(error.to_string().contains("native publication history exists")); + } + assert_eq!(mst2_native_head::Entity::find().count(mono.get_connection()).await.unwrap(), 0); + assert_eq!(mst2_native_publication::Entity::find().all(mono.get_connection()).await.unwrap(), certificates); + assert_eq!(mst2_publication::Entity::find().all(mono.get_connection()).await.unwrap(), receipts); + assert_eq!(mst2_publication_outbox::Entity::find().all(mono.get_connection()).await.unwrap(), outbox); + assert!(storage.push_queue_service.push_queue_storage.get_control().await.unwrap().paused); + mono.get_connection().execute_unprepared("DELETE FROM mst2_native_publication").await.unwrap(); + for instance in [NATIVE_INSTANCE, "11111111-2222-4333-8444-555555555555"] { + let error = mono.initialize_native_publication_for_maintenance(instance, &head.root).await.unwrap_err(); + assert!(error.to_string().contains("native publication history exists")); + } + assert_eq!(mst2_native_head::Entity::find().count(mono.get_connection()).await.unwrap(), 0); + assert_eq!(mst2_native_publication::Entity::find().count(mono.get_connection()).await.unwrap(), 0); + assert_eq!(mst2_publication::Entity::find().all(mono.get_connection()).await.unwrap(), receipts); + assert_eq!(mst2_publication_outbox::Entity::find().all(mono.get_connection()).await.unwrap(), outbox); + assert!(storage.push_queue_service.push_queue_storage.get_control().await.unwrap().paused); + let root = mono.get_main_ref("/").await.unwrap().unwrap(); + assert_eq!((root.ref_commit_hash, root.ref_tree_hash), (head.root.commit, head.root.tree)); +} + +#[tokio::test] +async fn real_merge_fails_closed_when_native_publication_is_enabled() { + let (_temp, storage, tip, path) = native_fixture().await; + let mono = storage.mono_storage(); + let root_before = mono.get_main_ref("/").await.unwrap().unwrap(); + let head_before = mst2_native_head::Entity::find_by_id("/") + .one(mono.get_connection()) + .await + .unwrap() + .unwrap(); + + let cl_link = format!("CL-NATIVE-GUARD-{}", uuid::Uuid::new_v4().simple()); + let EnqueueOutcome::Inserted { id } = storage + .push_queue_service + .enqueue(EnqueueRequest { + kind: PushQueueKindEnum::Merge, + operation_id: merge_operation_id(&cl_link), + path: path.clone(), + old_id: tip.id.to_string(), + new_id: tip.id.to_string(), + requester: Some("native-guard-test".to_owned()), + payload: serde_json::json!({ + "cl_link": cl_link, + "authz_principal": "native-guard-test", + "execution_actor": "native-guard-test" + }), + ref_name: None, + is_delete: false, + }) + .await + .unwrap() + else { + panic!("merge enqueue did not insert"); + }; + assert_eq!( + storage + .push_queue_service + .storage() + .claim_for_execution(id) + .await + .unwrap(), + ClaimOutcome::Claimed + ); + + let merge_ctx = MergeExecContext { + storage: storage.clone(), + git_object_cache: Arc::new(crate::ceres::api_service::cache::GitObjectCache { + connection: crate::jupiter::tests::test_redis_manager().await, + prefix: String::new(), + }), + abort_before_cl_status: false, + pause_after_apply: Duration::ZERO, + pause_after_apply_barrier: None, + }; + let outcome = storage + .push_queue_service + .execute_b3( + ExecuteRequest { + id, + ..Default::default() + }, + None, + Some(&merge_ctx), + None, + ) + .await + .unwrap(); + assert!(matches!( + outcome, + ExecuteOutcome::Failed { + ref failure, + ref message, + .. + } if failure == "Conflict" + && message == "MST2 native publication does not cover merge writer" + )); + assert_eq!(wh03_row_status(&storage, id).await, PushQueueStatusEnum::Failed); + + let root_after = mono.get_main_ref("/").await.unwrap().unwrap(); + assert_eq!(root_after.ref_commit_hash, root_before.ref_commit_hash); + assert_eq!(root_after.ref_tree_hash, root_before.ref_tree_hash); + let head_after = mst2_native_head::Entity::find_by_id("/") + .one(mono.get_connection()) + .await + .unwrap() + .unwrap(); + assert_eq!(head_after, head_before); + assert_eq!(mst2_publication::Entity::find().count(mono.get_connection()).await.unwrap(), 0); + assert_eq!( + mst2_native_publication::Entity::find() + .count(mono.get_connection()) + .await + .unwrap(), + 0 + ); + assert_eq!( + mst2_publication_outbox::Entity::find() + .count(mono.get_connection()) + .await + .unwrap(), + 0 + ); +} + +#[tokio::test] +async fn real_push_rejects_same_root_changed_publication_token_before_any_ref_write() { + let (_temp, storage, tip, path) = native_fixture().await; + let (new,payload)=save_same_tree_commit(&storage,&tip).await; + let id=wh03_enqueue_push(&storage,&path,&tip.id.to_string(),&new,&payload).await; + let mono=storage.mono_storage(); + // A separate writer transaction changes only the publication generation. + mono.get_connection().execute_unprepared("UPDATE mst2_native_head SET sequence=1 WHERE namespace='/'").await.unwrap(); + let outcome=wh03_exec(&storage,id).await; + assert!(matches!(outcome,ExecuteOutcome::Failed { ref failure,ref message,.. } + if failure=="Conflict" && message.contains("publication token is stale"))); + assert_eq!(mono.get_main_ref(&path).await.unwrap().unwrap().ref_commit_hash,tip.id.to_string()); + assert_eq!(mst2_native_publication::Entity::find().count(mono.get_connection()).await.unwrap(),0); + assert_eq!(mst2_publication::Entity::find().count(mono.get_connection()).await.unwrap(),0); +} + +async fn api_state(storage: crate::jupiter::storage::Storage) -> crate::api::MonoApiServiceState { + crate::api::MonoApiServiceState { + entity_store: storage.entity_store.clone(),storage, + session_store:crate::api::oauth::api_store::BrowserSessionStore::Anonymous, + git_object_cache:Arc::new(crate::ceres::api_service::cache::GitObjectCache { + connection:crate::jupiter::tests::test_redis_manager().await,prefix:String::new(), + }),listen_addr:"127.0.0.1:0".to_owned(), + } +} + +async fn http_resolve(state: crate::api::MonoApiServiceState,scope:&str) -> (u16,serde_json::Value) { + use tower::ServiceExt; + let app=crate::api::router::snapshot_router::routers(state.clone()).with_state(state); + let response=app.oneshot(axum::http::Request::builder().method("POST").uri("/snapshots/resolve") + .header("content-type","application/json").body(axum::body::Body::from(serde_json::json!({"target":{"kind":"latest"},"scope":scope}).to_string())).unwrap()).await.unwrap(); + let status=response.status().as_u16(); + let body=axum::body::to_bytes(response.into_body(),1_048_576).await.unwrap(); + (status,serde_json::from_slice(&body).unwrap()) +} + +#[tokio::test] +async fn http_resolve_captured_before_commit_never_mixes_new_sequence_with_old_descriptor() { + let (_temp,storage,tip,path)=native_fixture().await; + #[cfg(unix)] + let writer=crate::ceres::snapshot::projection_writer::ProjectionObservationSink::start(_temp.path()).unwrap(); + #[cfg(unix)] + let storage={ let mut storage=storage; storage.projection_observation_sink=Some(writer.clone()); storage }; + let (first,payload)=save_same_tree_commit(&storage,&tip).await; + let id=wh03_enqueue_push(&storage,&path,&tip.id.to_string(),&first,&payload).await; + assert!(matches!(wh03_exec(&storage,id).await,ExecuteOutcome::Done{..})); + let state=api_state(storage.clone()).await; + let ((status,v1),first_observations)=crate::ceres::snapshot::projection_observation::with_observations(http_resolve(state.clone(),"/")).await; + assert_eq!(status,200,"{v1}"); + assert_eq!(first_observations.len(),1); + let first_identity=first_observations[0].test_identity(); + assert_eq!(first_identity["snapshot_id"],v1["descriptor"]["snapshot_id"]); + assert_eq!(first_identity["metadata_root"],v1["descriptor"]["metadata_root"]); + assert_eq!(first_identity["namespace_view_id"],v1["descriptor"]["namespace_view_id"]); + assert_eq!(first_identity["instance_id"],v1["descriptor"]["instance_id"]); + assert_eq!(first_identity["native_publication_sequence"],v1["publication_sequence"]); + assert_eq!(first_identity["native_writer_epoch"],v1["writer_epoch"]); + assert!(first_identity["native_certificate_receipt_id"].as_u64().unwrap()>0); + assert!(!first_identity["request_id"].as_str().unwrap().is_empty()); + let (status,subscope)=http_resolve(state.clone(),&path).await; + assert_eq!(status,200,"{subscope}"); + assert_eq!(v1["publication_sequence"],subscope["publication_sequence"]); + assert_eq!(v1["writer_epoch"],subscope["writer_epoch"]); + let root_v1=storage.mono_storage().get_main_ref("/").await.unwrap().unwrap(); + assert_eq!(first_identity["root_commit_oid"],git_internal::hash::ObjectHash::from_hex_for_kind(git_internal::hash::get_hash_kind(),&root_v1.ref_commit_hash).unwrap().to_tagged_string()); + assert_eq!(first_identity["root_tree_oid"],git_internal::hash::ObjectHash::from_hex_for_kind(git_internal::hash::get_hash_kind(),&root_v1.ref_tree_hash).unwrap().to_tagged_string()); + let native_v1=storage.mono_storage().read_native_publication_head(storage.config().mst2.instance_uuid.as_deref().unwrap()).await.unwrap(); + assert_eq!(first_identity["native_certificate_receipt_id"].as_u64(),Some(native_v1.token.certificate.unwrap() as u64)); + let captured=Arc::new(tokio::sync::Barrier::new(2)); + let release=Arc::new(tokio::sync::Barrier::new(2)); + let task_state=state.clone();let task_captured=captured.clone();let task_release=release.clone(); + let mut reader=tokio::spawn(async move { + crate::ceres::snapshot::projection_observation::with_observations(crate::api::router::snapshot_router::with_native_resolve_barriers(task_captured,task_release,http_resolve(task_state,"/"))).await + }); + tokio::time::timeout(Duration::from_secs(5),captured.wait()).await.unwrap(); + let first_commit=storage.mono_storage().get_commit_by_hash(&first).await.unwrap().unwrap(); + let first_commit=git_internal::internal::object::commit::Commit::from_mega_model(first_commit); + let next_blob = storage.git_service.save_object_from_raw(Bytes::from_static(b"next projection")).await.unwrap(); + let (next,next_payload)=wh03_save_n1_commit(&storage,first_commit.id,&next_blob,"next projection").await; + let id=wh03_enqueue_push(&storage,&path,&first,&next,&next_payload).await; + assert!(matches!(wh03_exec(&storage,id).await,ExecuteOutcome::Done{..})); + tokio::time::timeout(Duration::from_secs(5),release.wait()).await.unwrap(); + let ((status,delayed),delayed_observations)=tokio::time::timeout(Duration::from_secs(10),&mut reader).await.unwrap().unwrap(); + assert_eq!(status,200,"{delayed}");assert_eq!(delayed["descriptor"],v1["descriptor"]); + assert_eq!(delayed["publication_sequence"],v1["publication_sequence"]); + assert_eq!(delayed_observations.len(),1); + let delayed_identity=delayed_observations[0].test_identity(); + for field in ["root_commit_oid","root_tree_oid","native_certificate_receipt_id","native_writer_epoch","native_publication_sequence","snapshot_id","metadata_root"] { + assert_eq!(delayed_identity[field],first_identity[field],"{field} mixed with a later publication"); + } + let ((status,v2),next_observations)=crate::ceres::snapshot::projection_observation::with_observations(http_resolve(state.clone(),"/")).await; + assert_eq!(status,200,"{v2}");assert_eq!(v2["publication_sequence"],"2"); + assert_ne!(v2["descriptor"]["snapshot_id"],v1["descriptor"]["snapshot_id"]); + assert_eq!(next_observations.len(),1); + let next_identity=next_observations[0].test_identity(); + assert_eq!(next_identity["native_publication_sequence"],v2["publication_sequence"]); + assert_eq!(next_identity["snapshot_id"],v2["descriptor"]["snapshot_id"]); + assert_ne!(next_identity["native_certificate_receipt_id"],first_identity["native_certificate_receipt_id"]); + let ((status,_),failed_observations)=crate::ceres::snapshot::projection_observation::with_observations(http_resolve(state,"/absent-observation-scope")).await; + assert_eq!(status,404); + assert!(failed_observations.is_empty(),"failed projection emitted success observation"); + #[cfg(unix)] + { + writer.shutdown(std::time::Instant::now()+Duration::from_secs(5)).await.unwrap(); + let directory=writer.test_directory(_temp.path()); + let records=std::fs::read_to_string(directory.join("records.jsonl")).unwrap(); + let records:Vec=records.lines().map(|line|serde_json::from_str(line).unwrap()).collect(); + assert_eq!(records.len(),4,"failed resolve must not be a successful observation"); + for (record,observation) in [(&records[0],&first_observations[0]),(&records[2],&delayed_observations[0]),(&records[3],&next_observations[0])] { + let typed=serde_json::to_value(observation.wire_record()).unwrap(); + assert_eq!(record["payload"],typed,"writer must preserve the full validated operation tuple"); + assert_eq!(record["payload"].as_object().unwrap().len(),38); + } + assert_ne!(records[0]["payload"]["request_id"],records[2]["payload"]["request_id"]); + let status:serde_json::Value=serde_json::from_slice(&std::fs::read(directory.join("status.json")).unwrap()).unwrap(); + assert_eq!(status["closed"],true); + assert_eq!(status["accepted_records"],4); + assert_eq!(status["written_records"],4); + assert_eq!(status["first_error_code"],0); + } +} + +#[cfg(unix)] +#[tokio::test] +async fn rejected_native_observation_source_does_not_change_ready_resolve_and_is_a_sticky_writer_failure() { + let (temp,mut storage,tip,path)=native_fixture().await; + let (first,payload)=save_same_tree_commit(&storage,&tip).await; + let id=wh03_enqueue_push(&storage,&path,&tip.id.to_string(),&first,&payload).await; + assert!(matches!(wh03_exec(&storage,id).await,ExecuteOutcome::Done{..})); + let head=storage.mono_storage().read_native_publication_head(NATIVE_INSTANCE).await.unwrap(); + assert_eq!(head.token.sequence,1); + assert!(head.token.certificate.unwrap()>0); + let writer=crate::ceres::snapshot::projection_writer::ProjectionObservationSink::start(temp.path()).unwrap(); + storage.projection_observation_sink=Some(writer.clone()); + let state=api_state(storage).await; + // Corrupt only the observation's captured certificate. The actual native + // head is READY; INITIALIZING heads remain correctly rejected with 503. + let ((status,response),observations)=crate::ceres::snapshot::projection_observation::with_observations( + crate::api::router::snapshot_router::with_rejected_native_observation_source(http_resolve(state,"/")) + ).await; + assert_eq!(status,200,"{response}"); + assert_eq!(response["publication_sequence"],"1"); + assert!(observations.is_empty()); + assert!(writer.shutdown(std::time::Instant::now()+Duration::from_secs(5)).await.is_err()); + let directory=writer.test_directory(temp.path()); + assert!(std::fs::read(directory.join("records.jsonl")).unwrap().is_empty()); + let status:serde_json::Value=serde_json::from_slice(&std::fs::read(directory.join("status.json")).unwrap()).unwrap(); + assert_eq!(status["first_error_code"],crate::ceres::snapshot::projection_writer::WriterFailure::ObservationBindingRejected as u8); + assert_eq!(status["accepted_records"],0); + assert_eq!(status["closed"],true); +} diff --git a/src/jupiter/service/push_queue_service.rs b/src/jupiter/service/push_queue_service.rs index 61dedbce..643cc859 100644 --- a/src/jupiter/service/push_queue_service.rs +++ b/src/jupiter/service/push_queue_service.rs @@ -31,6 +31,11 @@ use crate::{ blob_path_index::BlobPathIndexMode, git_db_storage::GitDbStorage, mono_storage::MonoStorage, + mst2_publication_storage::{ + PreparedPublication, PublicationPreparation, PublicationReceiptError, + PublicationRequest, + }, + native_publication_storage::PreparedNativePublication, push_queue_storage::{ ClaimOutcome, EnqueueOutcome, EnqueueParams, EnqueueRejectReason, PushQueueStorage, }, @@ -419,7 +424,7 @@ pub struct PushExecContext { pub storage: crate::jupiter::storage::Storage, pub git_object_cache: std::sync::Arc, /// Test-only: two-phase sync inside the B3 txn right before - /// `apply_push_in_txn` (after the baseline read): the executor signals + /// `apply_push_in_txn` or the n=0 root CAS (after the baseline read): the executor signals /// "race window open" on the enter barrier, then blocks on the release /// barrier until the test's queue-bypassing writer has committed — a /// deterministic apply-time root CAS miss (WH-03). @@ -428,6 +433,12 @@ pub struct PushExecContext { pub pre_apply_release_barrier: Option>, } +struct PushRefBaseline<'a> { + cur_commit: Option<&'a str>, + cur_tree: Option<&'a str>, + root: Option<&'a crate::callisto::mega_refs::Model>, +} + /// Context required to execute a claimed `kind=merge` round under B3. pub struct MergeExecContext { pub storage: crate::jupiter::storage::Storage, @@ -697,6 +708,11 @@ impl PushQueueService { self } + pub(crate) fn with_native_publication(mut self, enabled: bool) -> Self { + self.push_queue_storage = self.push_queue_storage.with_native_publication(enabled); + self + } + pub fn with_max_push_commits(mut self, max_push_commits: usize) -> Self { self.max_push_commits = max_push_commits; self @@ -1277,8 +1293,56 @@ impl PushQueueService { return Ok(ExecuteOutcome::HardStopped { id: req.id }); } + // Receipt replay precedes baseline/ref writes: the root may already + // have advanced since a committed operation lost its response. + let publication = if let Some(ctx) = push_ctx + && ctx.storage.config().mst2.publication_enabled + && row.kind != PushQueueKindEnum::Attach + && !(row.kind == PushQueueKindEnum::Merge && merge_ctx.is_some()) + { + let request = match PublicationRequest::from_trunk_queue(&row) { + Ok(request) => request, + Err(error) => { + return self + .b3_fail_merge(txn, req.id, "Conflict", error.to_string()) + .await; + } + }; + let preparation = match self + .mono_storage + .begin_publication_in_txn(&txn, request) + .await + { + Ok(preparation) => preparation, + Err(PublicationReceiptError::Database(error)) => return Err(error.into()), + Err(error) => { + return self + .b3_fail_merge(txn, req.id, "Conflict", error.to_string()) + .await; + } + }; + match preparation { + PublicationPreparation::Prepared(prepared) => Some(prepared), + PublicationPreparation::AlreadyCommitted(committed) => { + return self + .b3_replay_committed_operation(txn, req.id, committed.receipt.new_oid) + .await; + } + PublicationPreparation::AlreadyCommittedNoop(receipt) => { + return self + .b3_replay_committed_operation(txn, req.id, receipt.landed_commit_id) + .await; + } + } + } else { + None + }; + // Expected-root baseline compare (NULL-safe). - let root = self.mono_storage.get_main_ref_in_txn("/", &txn).await?; + let root = self + .mono_storage + .get_native_main_ref_in_txn("/", &txn) + .await?; let (cur_commit, cur_tree) = match &root { Some(r) => ( Some(r.ref_commit_hash.as_str()), @@ -1309,6 +1373,55 @@ impl PushQueueService { return Ok(ExecuteOutcome::BypassDetected { id: req.id }); } + // Native publication currently has no merge projection. Refuse the + // real merge writer before it can mutate refs, commits, or CL state; + // keeping this gate ahead of dispatch makes publication fail closed + // while the writer matrix is still incomplete. + if row.kind == PushQueueKindEnum::Merge + && merge_ctx.is_some_and(|ctx| ctx.storage.config().mst2.publication_enabled) + { + return self + .b3_fail_merge( + txn, + req.id, + "Conflict", + "MST2 native publication does not cover merge writer".into(), + ) + .await; + } + + let native_publication = if publication.is_some() && row.kind == PushQueueKindEnum::Push { + let ctx = push_ctx.ok_or_else(|| { + MegaError::Other("native publication requires the real push writer".into()) + })?; + let config = ctx.storage.config(); + let Some(instance) = config.mst2.instance_uuid.as_deref() else { + return self + .b3_fail_merge( + txn, + req.id, + "Conflict", + "native publication instance missing".into(), + ) + .await; + }; + match self + .mono_storage + .reserve_native_publication_in_txn(&txn, &row, instance) + .await + { + Ok(prepared) => Some(prepared), + Err(PublicationReceiptError::Database(error)) => return Err(error.into()), + Err(error) => { + return self + .b3_fail_merge(txn, req.id, "Conflict", error.to_string()) + .await; + } + } + } else { + None + }; + if req.force_conflict() { return self .b4_conflict_requeue_holding_lock(txn, req.id, &row) @@ -1359,7 +1472,18 @@ impl PushQueueService { } if let Some(ctx) = push_ctx { return self - .b3_execute_push(txn, &row, ctx, cur_commit, cur_tree, root.as_ref()) + .b3_execute_push( + txn, + &row, + ctx, + PushRefBaseline { + cur_commit, + cur_tree, + root: root.as_ref(), + }, + publication, + native_publication, + ) .await; } tracing::trace!(policy = ?self.push_policy, id = req.id, "B3 push stub"); @@ -1462,26 +1586,16 @@ impl PushQueueService { return Ok(self.outcome_claim_lost(req.id)); } - // T05 (spec 09 §1): record the publication receipt, outbox event and - // per-namespace sequence in the *same* transaction as the root CAS - // above, so the visible change, its sequence and its outbox commit - // atomically. Gated off by default (`mst2.publication_enabled`), so - // an unconfigured deployment keeps the exact prior behavior. - if let Some(ctx) = push_ctx - && ctx.storage.config().mst2.publication_enabled - { - let old_oid = cur_commit.unwrap_or(ZERO_ID); - let namespace = self.mono_storage.normalize_namespace(&row.path); + if let Some(prepared) = publication { self.mono_storage .record_publication_in_txn( &txn, - &row.operation_id, - &namespace, - old_oid, + prepared, + cur_commit.unwrap_or(ZERO_ID), &new_commit, - "trunk_push", ) - .await?; + .await + .map_err(|error| MegaError::Other(error.to_string()))?; } PushQueueStorage::notify_mono_write_queue(&txn).await?; @@ -1494,6 +1608,25 @@ impl PushQueueService { }) } + async fn b3_replay_committed_operation( + &self, + txn: sea_orm::DatabaseTransaction, + id: i64, + landed_commit_id: String, + ) -> Result { + if !PushQueueStorage::mark_done_if_running_in_txn(&txn, id, &landed_commit_id).await? { + txn.rollback().await?; + return Ok(self.outcome_claim_lost(id)); + } + PushQueueStorage::notify_mono_write_queue(&txn).await?; + txn.commit().await?; + Ok(ExecuteOutcome::Done { + id, + landed_commit_id, + root_cas_writes: 0, + }) + } + /// TP-08: attach kind under held `MONO_WRITE_LOCK` — sole root write is /// `attach_to_monorepo_parent_in_txn` (no generic CAS stub, no retry loop). /// @@ -2321,13 +2454,13 @@ impl PushQueueService { txn: sea_orm::DatabaseTransaction, row: &push_queue::Model, ctx: &PushExecContext, - cur_commit: Option<&str>, - cur_tree: Option<&str>, - root: Option<&crate::callisto::mega_refs::Model>, + baseline: PushRefBaseline<'_>, + publication: Option, + native_publication: Option, ) -> Result { let id = row.id; match self - .b3_execute_push_inner(txn, row, ctx, cur_commit, cur_tree, root) + .b3_execute_push_inner(txn, row, ctx, baseline, publication, native_publication) .await { Ok(outcome) => Ok(outcome), @@ -2370,10 +2503,15 @@ impl PushQueueService { txn: sea_orm::DatabaseTransaction, row: &push_queue::Model, ctx: &PushExecContext, - cur_commit: Option<&str>, - cur_tree: Option<&str>, - root: Option<&crate::callisto::mega_refs::Model>, + baseline: PushRefBaseline<'_>, + publication: Option, + native_publication: Option, ) -> Result { + let PushRefBaseline { + cur_commit, + cur_tree, + root, + } = baseline; use std::path::PathBuf; use git_internal::{ @@ -2423,7 +2561,7 @@ impl PushQueueService { let path_row = self .mono_storage - .get_main_ref_in_txn(&row.path, &txn) + .get_native_main_ref_in_txn(&row.path, &txn) .await?; if path_row.is_none() { @@ -2494,6 +2632,12 @@ impl PushQueueService { ) .await; } + if let Some(barrier) = &ctx.pre_apply_enter_barrier { + barrier.wait().await; + } + if let Some(barrier) = &ctx.pre_apply_release_barrier { + barrier.wait().await; + } PushQueueStorage::savepoint(&txn, "b3_kind").await?; let cas_ok = self .mono_storage @@ -2530,6 +2674,24 @@ impl PushQueueService { txn.rollback().await?; return Ok(self.outcome_claim_lost(row.id)); } + if let Some(prepared) = native_publication { + self.mono_storage + .finish_native_noop_in_txn(&txn, prepared) + .await + .map_err(|error| MegaError::Other(error.to_string()))?; + } + if let Some(prepared) = publication { + self.mono_storage + .record_noop_operation_in_txn( + &txn, + prepared, + &root.ref_commit_hash, + &root.ref_tree_hash, + &pref.ref_commit_hash, + ) + .await + .map_err(|error| MegaError::Other(error.to_string()))?; + } PushQueueStorage::notify_mono_write_queue(&txn).await?; txn.commit().await?; self.run_c_segment_index(row.id, &row.path).await; @@ -2799,26 +2961,23 @@ impl PushQueueService { return Ok(self.outcome_claim_lost(row.id)); } - // T05 (spec 09 §1/§7): record the publication receipt, outbox event - // and per-namespace sequence in the *same* transaction as the ref CAS - // above, so the visible change, its sequence and its outbox commit - // atomically (trunk push = the primary default-namespace writer of - // spec 09 §5). Gated off by default (`mst2.publication_enabled`): - // a deployment that has not passed the §9 shadow comparison keeps - // the exact prior behavior. Replaying the same operation id lands - // the receipt once (PUB-11). - if ctx.storage.config().mst2.publication_enabled { - let namespace = self.mono_storage.normalize_namespace(&normalized); - self.mono_storage + if let Some(prepared) = publication { + let committed = self + .mono_storage .record_publication_in_txn( &txn, - &row.operation_id, - &namespace, + prepared, cur_commit.unwrap_or(ZERO_ID), &landed_commit_id, - "trunk_push", ) - .await?; + .await + .map_err(|error| MegaError::Other(error.to_string()))?; + let native = native_publication + .ok_or_else(|| MegaError::Other("native push reservation missing".into()))?; + self.mono_storage + .record_native_publication_in_txn(&txn, native, &committed) + .await + .map_err(|error| MegaError::Other(error.to_string()))?; } PushQueueStorage::notify_mono_write_queue(&txn).await?; @@ -3398,10 +3557,10 @@ fn reject_to_error(reason: EnqueueRejectReason) -> MegaError { #[cfg(test)] mod tests { - use sea_orm::ConnectionTrait; use serde_json::json; use super::*; + include!("native_publication_push_tests.rs"); use crate::{ callisto::{mega_refs, sea_orm_active_enums::PushQueueFailureEnum}, jupiter::{ @@ -4830,6 +4989,503 @@ mod tests { .unwrap() } + #[tokio::test] + async fn mst2_receipt_real_push_replays_before_root_baseline_and_rejects_changed_request() { + use sea_orm::{ + ColumnTrait, ConnectionTrait, EntityTrait, PaginatorTrait, QueryFilter, Statement, + }; + + use crate::callisto::{mst2_publication, mst2_publication_outbox, mst2_queue_noop_receipt}; + + let temp = tempfile::tempdir().unwrap(); + let mut config = crate::config::testing::isolated_config(temp.path().join("config")); + config.monorepo.push_policy = PushPolicy::Trunk; + config.mst2.publication_enabled = true; + config.mst2.instance_uuid = Some("6ab219b0-4275-45ba-9d7b-7b0b633018cd".to_owned()); + let storage = crate::jupiter::tests::test_storage_with_config(temp.path(), config).await; + let storage = crate::jupiter::tests::with_test_vault(storage, temp.path()).await; + let (old_commit, path) = wh03_path_fixture(&storage, "receipt").await; + storage + .mono_storage() + .initialize_native_publication("6ab219b0-4275-45ba-9d7b-7b0b633018cd") + .await + .unwrap(); + let old_id = old_commit.id.to_string(); + let (new_id, payload) = wh03_save_n1_commit( + &storage, + old_commit.id, + "eeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeee", + "receipt new", + ) + .await; + let id = wh03_enqueue_push(&storage, &path, &old_id, &new_id, &payload).await; + assert!(matches!( + wh03_exec(&storage, id).await, + ExecuteOutcome::Done { + root_cas_writes: 1, + .. + } + )); + let mono = storage.mono_storage(); + let root_after = mono.get_main_ref("/").await.unwrap().unwrap(); + let operation_id = format!("mst2:trunk-queue:{id}"); + let receipt = mst2_publication::Entity::find() + .filter(mst2_publication::Column::OperationId.eq(&operation_id)) + .one(mono.get_connection()) + .await + .unwrap() + .unwrap(); + assert_eq!(receipt.request_digest_version, Some(1)); + assert!(receipt.request_digest.is_some()); + assert_eq!(receipt.new_oid, new_id); + assert_eq!( + mst2_publication_outbox::Entity::find() + .count(mono.get_connection()) + .await + .unwrap(), + 1 + ); + + let retry = |requester: Option, payload: JsonValue| EnqueueRequest { + kind: PushQueueKindEnum::Push, + operation_id: push_operation_id(&old_id, &new_id), + path: path.clone(), + old_id: old_id.clone(), + new_id: new_id.clone(), + requester, + payload, + ref_name: Some(MEGA_BRANCH_NAME.to_owned()), + is_delete: false, + }; + assert!( + matches!(storage.push_queue_service.enqueue(retry(None, payload.to_json())).await.unwrap(), + EnqueueOutcome::Replay { id: replay_id, .. } if replay_id == id) + ); + let error = storage + .push_queue_service + .enqueue(retry(Some("different-actor".to_owned()), payload.to_json())) + .await + .unwrap_err(); + assert!(error.to_string().contains("MST2_PUBLICATION_CONFLICT")); + let mut different = payload.to_json(); + different["fork_base"] = serde_json::json!("different-base"); + let error = storage + .push_queue_service + .enqueue(retry(None, different)) + .await + .unwrap_err(); + assert!(error.to_string().contains("MST2_PUBLICATION_CONFLICT")); + + // Model recovery of a Running queue row after COMMIT lost its response. + // Original claim baselines remain old, so a late replay would fail them. + mono.get_connection() + .execute_raw(Statement::from_sql_and_values( + mono.get_connection().get_database_backend(), + "UPDATE push_queue SET status = 'Running', landed_commit_id = NULL WHERE id = $1", + [id.into()], + )) + .await + .unwrap(); + assert_eq!( + wh03_exec(&storage, id).await, + ExecuteOutcome::Done { + id, + landed_commit_id: new_id.clone(), + root_cas_writes: 0, + } + ); + let root = mono.get_main_ref("/").await.unwrap().unwrap(); + assert_eq!(root.ref_commit_hash, root_after.ref_commit_hash); + assert_eq!(root.ref_tree_hash, root_after.ref_tree_hash); + let check_txn = mono.get_connection().begin().await.unwrap(); + assert!( + !PushQueueStorage::is_hard_stopped_in_txn(&check_txn) + .await + .unwrap() + ); + check_txn.rollback().await.unwrap(); + assert_eq!(mono.publication_sequence(&path).await.unwrap(), 1); + assert_eq!( + mst2_publication::Entity::find() + .count(mono.get_connection()) + .await + .unwrap(), + 1 + ); + + mono.get_connection().execute_raw(Statement::from_sql_and_values( + mono.get_connection().get_database_backend(), + "UPDATE push_queue SET status = 'Running', requester = 'changed-actor' WHERE id = $1", + [id.into()], + )).await.unwrap(); + assert!( + matches!(wh03_exec(&storage, id).await, ExecuteOutcome::Failed { ref failure, ref message, .. } + if failure == "Conflict" && message.contains("MST2_PUBLICATION_CONFLICT")) + ); + assert_eq!( + mono.get_main_ref("/") + .await + .unwrap() + .unwrap() + .ref_commit_hash, + root_after.ref_commit_hash + ); + assert_eq!(mono.publication_sequence(&path).await.unwrap(), 1); + assert_eq!( + mst2_publication_outbox::Entity::find() + .count(mono.get_connection()) + .await + .unwrap(), + 1 + ); + + // Distinct internal no-op operations do not create publications. + let n0_payload = PushPayload { + commits: Vec::new(), + fork_base: None, + n: 0, + }; + let mut noop_ids = Vec::new(); + for round in 0..3 { + let operation_id = format!("internal-noop-{round}"); + let outcome = storage + .push_queue_service + .enqueue(EnqueueRequest { + kind: PushQueueKindEnum::Push, + operation_id: operation_id.clone(), + path: path.clone(), + old_id: new_id.clone(), + new_id: new_id.clone(), + requester: None, + payload: n0_payload.to_json(), + ref_name: Some(MEGA_BRANCH_NAME.to_owned()), + is_delete: false, + }) + .await + .unwrap(); + let EnqueueOutcome::Inserted { id: n0_id } = outcome else { + panic!("{outcome:?}"); + }; + assert_eq!( + storage + .push_queue_service + .storage() + .claim_for_execution(n0_id) + .await + .unwrap(), + ClaimOutcome::Claimed + ); + assert!(matches!( + wh03_exec(&storage, n0_id).await, + ExecuteOutcome::Done { + root_cas_writes: 1, + .. + } + )); + let noop = mst2_queue_noop_receipt::Entity::find() + .filter( + mst2_queue_noop_receipt::Column::OperationId + .eq(format!("mst2:trunk-queue:{n0_id}")), + ) + .one(mono.get_connection()) + .await + .unwrap() + .unwrap(); + assert_eq!(noop.observed_sequence, 1); + assert_eq!(noop.observed_root_commit, root_after.ref_commit_hash); + assert_eq!(noop.observed_root_tree, root_after.ref_tree_hash); + assert_eq!(noop.landed_commit_id, new_id); + assert_eq!(mono.publication_sequence(&path).await.unwrap(), 1); + assert_eq!( + mst2_publication::Entity::find() + .count(mono.get_connection()) + .await + .unwrap(), + 1 + ); + assert_eq!( + mst2_publication_outbox::Entity::find() + .count(mono.get_connection()) + .await + .unwrap(), + 1 + ); + assert!(matches!(storage.push_queue_service.enqueue(EnqueueRequest { + kind: PushQueueKindEnum::Push, operation_id: operation_id.clone(), + path: path.clone(), old_id: new_id.clone(), new_id: new_id.clone(), + requester: None, payload: n0_payload.to_json(), + ref_name: Some(MEGA_BRANCH_NAME.to_owned()), is_delete: false, + }).await.unwrap(), EnqueueOutcome::Replay { id: replay_id, .. } if replay_id == n0_id)); + let error = storage + .push_queue_service + .enqueue(EnqueueRequest { + kind: PushQueueKindEnum::Push, + operation_id: operation_id.clone(), + path: path.clone(), + old_id: new_id.clone(), + new_id: new_id.clone(), + requester: Some("changed-actor".to_owned()), + payload: n0_payload.to_json(), + ref_name: Some(MEGA_BRANCH_NAME.to_owned()), + is_delete: false, + }) + .await + .unwrap_err(); + assert!(error.to_string().contains("MST2_PUBLICATION_CONFLICT")); + let mut changed_payload = n0_payload.to_json(); + changed_payload["fork_base"] = serde_json::json!("changed-intent"); + let error = storage + .push_queue_service + .enqueue(EnqueueRequest { + kind: PushQueueKindEnum::Push, + operation_id, + path: path.clone(), + old_id: new_id.clone(), + new_id: new_id.clone(), + requester: None, + payload: changed_payload, + ref_name: Some(MEGA_BRANCH_NAME.to_owned()), + is_delete: false, + }) + .await + .unwrap_err(); + assert!(error.to_string().contains("MST2_PUBLICATION_CONFLICT")); + noop_ids.push(n0_id); + } + let (later_id, later_payload) = wh03_save_n1_commit( + &storage, + new_id.parse().unwrap(), + "ffffffffffffffffffffffffffffffffffffffff", + "after no-op", + ) + .await; + let later_queue_id = + wh03_enqueue_push(&storage, &path, &new_id, &later_id, &later_payload).await; + assert!(matches!( + wh03_exec(&storage, later_queue_id).await, + ExecuteOutcome::Done { + root_cas_writes: 1, + .. + } + )); + let later_root = mono.get_main_ref("/").await.unwrap().unwrap(); + assert_ne!(later_root.ref_commit_hash, root_after.ref_commit_hash); + for &n0_id in &noop_ids { + mono.get_connection().execute_raw(Statement::from_sql_and_values( + mono.get_connection().get_database_backend(), + "UPDATE push_queue SET status = 'Running', landed_commit_id = NULL WHERE id = $1", + [n0_id.into()], + )).await.unwrap(); + assert_eq!( + wh03_exec(&storage, n0_id).await, + ExecuteOutcome::Done { + id: n0_id, + landed_commit_id: new_id.clone(), + root_cas_writes: 0, + } + ); + } + let root = mono.get_main_ref("/").await.unwrap().unwrap(); + assert_eq!(root.ref_commit_hash, later_root.ref_commit_hash); + assert_eq!(root.ref_tree_hash, later_root.ref_tree_hash); + assert_eq!(mono.publication_sequence(&path).await.unwrap(), 2); + assert_eq!( + mst2_publication::Entity::find() + .count(mono.get_connection()) + .await + .unwrap(), + 2 + ); + assert_eq!( + mst2_publication_outbox::Entity::find() + .count(mono.get_connection()) + .await + .unwrap(), + 2 + ); + assert_eq!( + mst2_queue_noop_receipt::Entity::find() + .count(mono.get_connection()) + .await + .unwrap(), + 3 + ); + let n0_id = noop_ids[0]; + mono.get_connection().execute_raw(Statement::from_sql_and_values( + mono.get_connection().get_database_backend(), + "UPDATE push_queue SET status = 'Running', requester = 'changed-actor' WHERE id = $1", + [n0_id.into()], + )).await.unwrap(); + assert!( + matches!(wh03_exec(&storage, n0_id).await, ExecuteOutcome::Failed { ref failure, ref message, .. } + if failure == "Conflict" && message.contains("MST2_PUBLICATION_CONFLICT")) + ); + assert_eq!( + mono.get_main_ref("/") + .await + .unwrap() + .unwrap() + .ref_commit_hash, + later_root.ref_commit_hash + ); + assert_eq!(mono.publication_sequence(&path).await.unwrap(), 2); + assert_eq!( + mst2_queue_noop_receipt::Entity::find() + .count(mono.get_connection()) + .await + .unwrap(), + 3 + ); + } + + #[tokio::test] + async fn mst2_noop_real_push_checks_epoch_baseline_and_net_zero_cas_fences() { + use sea_orm::{ConnectionTrait, EntityTrait, PaginatorTrait, Statement}; + + use crate::callisto::{mst2_publication, mst2_publication_outbox, mst2_queue_noop_receipt}; + + for fence in ["epoch", "baseline", "cas"] { + let temp = tempfile::tempdir().unwrap(); + let mut config = crate::config::testing::isolated_config(temp.path().join("config")); + config.monorepo.push_policy = PushPolicy::Trunk; + config.mst2.publication_enabled = true; + config.mst2.instance_uuid = Some("6ab219b0-4275-45ba-9d7b-7b0b633018cd".to_owned()); + let storage = + crate::jupiter::tests::test_storage_with_config(temp.path(), config).await; + let storage = crate::jupiter::tests::with_test_vault(storage, temp.path()).await; + let (tip, path) = wh03_path_fixture(&storage, &format!("noop-{fence}")).await; + storage + .mono_storage() + .initialize_native_publication("6ab219b0-4275-45ba-9d7b-7b0b633018cd") + .await + .unwrap(); + let tip_id = tip.id.to_string(); + let payload = PushPayload { + commits: Vec::new(), + fork_base: None, + n: 0, + }; + let id = wh03_enqueue_push(&storage, &path, &tip_id, &tip_id, &payload).await; + let mono = storage.mono_storage(); + let root_before = mono.get_main_ref("/").await.unwrap().unwrap(); + let outcome = if fence == "epoch" { + mono.get_connection().execute_raw(Statement::from_sql_and_values( + mono.get_connection().get_database_backend(), + "INSERT INTO mst2_namespace_seq (namespace, sequence, epoch) VALUES ($1, 0, 2)", + [path.clone().into()], + )).await.unwrap(); + wh03_exec(&storage, id).await + } else if fence == "baseline" { + mono.get_connection() + .execute_raw(Statement::from_sql_and_values( + mono.get_connection().get_database_backend(), + "UPDATE push_queue SET expected_commit_hash = $1 WHERE id = $2", + ["f".repeat(40).into(), id.into()], + )) + .await + .unwrap(); + wh03_exec(&storage, id).await + } else { + let enter = Arc::new(tokio::sync::Barrier::new(2)); + let release = Arc::new(tokio::sync::Barrier::new(2)); + let ctx = PushExecContext { + storage: storage.clone(), + git_object_cache: Arc::new(crate::ceres::api_service::cache::GitObjectCache { + connection: crate::jupiter::tests::test_redis_manager().await, + prefix: String::new(), + }), + pre_apply_enter_barrier: Some(Arc::clone(&enter)), + pre_apply_release_barrier: Some(Arc::clone(&release)), + }; + let exec_storage = storage.clone(); + let mut worker = tokio::spawn(async move { + exec_storage + .push_queue_service + .execute_b3( + ExecuteRequest { + id, + ..Default::default() + }, + None, + None, + Some(&ctx), + ) + .await + }); + let handshake = async { + tokio::time::timeout(Duration::from_secs(5), enter.wait()).await.map_err(|_| "no-op did not reach CAS".to_owned())?; + tokio::time::timeout(Duration::from_secs(5), mono.get_connection().execute_raw(Statement::from_sql_and_values( + mono.get_connection().get_database_backend(), + "UPDATE mega_refs SET ref_commit_hash = $1 WHERE path = '/' AND ref_name = $2", + ["f".repeat(40).into(), MEGA_BRANCH_NAME.into()], + ))).await.map_err(|_| "bypass writer timed out".to_owned())?.map_err(|error| error.to_string())?; + tokio::time::timeout(Duration::from_secs(5), release.wait()).await.map_err(|_| "no-op CAS release timed out".to_owned()) + }.await; + if let Err(error) = handshake { + worker.abort(); + let _ = tokio::time::timeout(Duration::from_secs(5), &mut worker).await; + panic!("{error}"); + } + match tokio::time::timeout(Duration::from_secs(10), &mut worker).await { + Ok(result) => result.unwrap().unwrap(), + Err(_) => { + worker.abort(); + let _ = tokio::time::timeout(Duration::from_secs(5), &mut worker).await; + panic!("no-op CAS worker did not finish"); + } + } + }; + if fence == "epoch" { + assert!( + matches!(outcome, ExecuteOutcome::Failed { ref failure, ref message, .. } + if failure == "Conflict" && message.contains("writer epoch is fenced")) + ); + } else { + assert_eq!(outcome, ExecuteOutcome::BypassDetected { id }); + } + let root = mono.get_main_ref("/").await.unwrap().unwrap(); + assert_eq!( + root.ref_commit_hash, + if fence == "cas" { + "f".repeat(40) + } else { + root_before.ref_commit_hash + } + ); + assert_eq!(root.ref_tree_hash, root_before.ref_tree_hash); + assert_eq!(mono.publication_sequence(&path).await.unwrap(), 0); + assert_eq!( + mst2_publication::Entity::find() + .count(mono.get_connection()) + .await + .unwrap(), + 0 + ); + assert_eq!( + mst2_publication_outbox::Entity::find() + .count(mono.get_connection()) + .await + .unwrap(), + 0 + ); + assert_eq!( + mst2_queue_noop_receipt::Entity::find() + .count(mono.get_connection()) + .await + .unwrap(), + 0 + ); + let txn = mono.get_connection().begin().await.unwrap(); + assert_eq!( + PushQueueStorage::is_hard_stopped_in_txn(&txn) + .await + .unwrap(), + fence != "epoch" + ); + txn.rollback().await.unwrap(); + } + } + async fn assert_view_signal(signal: &ViewSignal, context: &str) { tokio::time::timeout(Duration::from_secs(1), signal.notified()) .await diff --git a/src/jupiter/storage/mod.rs b/src/jupiter/storage/mod.rs index 85bc6cf9..655a7d02 100644 --- a/src/jupiter/storage/mod.rs +++ b/src/jupiter/storage/mod.rs @@ -16,6 +16,10 @@ pub mod issue_storage; pub mod lfs_db_storage; pub mod media_paging_storage; pub mod mono_storage; +pub(crate) mod mst2_publication_storage; +pub mod mst2_retention; +pub mod native_metadata_install; +pub(crate) mod native_publication_storage; pub mod notification_storage; pub mod object_storage; pub mod oci_db_storage; @@ -151,6 +155,11 @@ impl AppService { #[derive(Clone)] pub struct Storage { pub(crate) app_service: Arc, + /// Derived native projection memoization, scoped to this storage assembly. + /// Clones share it; independent databases/backends never share entries. + pub(crate) native_projection_cache: Arc, + pub(crate) projection_observation_sink: + Option>, pub cl_service: CLService, pub push_queue_service: PushQueueService, pub artifact_service: ArtifactService, @@ -224,7 +233,8 @@ impl Storage { }; let commit_binding_storage = CommitBindingStorage { base: base.clone() }; - let push_queue_storage = PushQueueStorage::new(base.clone()); + let push_queue_storage = PushQueueStorage::new(base.clone()) + .with_native_publication(config.mst2.publication_enabled); let buck_storage = BuckStorage { base: base.clone() }; let bots_storage = BotsStorage { base: base.clone() }; @@ -291,7 +301,8 @@ impl Storage { let push_queue_service = PushQueueService::new(base.clone(), config.monorepo.push_policy.clone()) .with_view_signal(view_runtime.signal()) - .with_max_push_commits(config.monorepo.max_push_commits); + .with_max_push_commits(config.monorepo.max_push_commits) + .with_native_publication(config.mst2.publication_enabled); let artifact_service = ArtifactService::new(base.clone(), object_store.clone()); let buck_service = BuckService::new( base.clone(), @@ -310,6 +321,8 @@ impl Storage { Ok(Storage { app_service: app_service.into(), + native_projection_cache: Arc::default(), + projection_observation_sink: None, config_handle, config, cl_service: CLService::new(base.clone()), @@ -685,6 +698,8 @@ impl Storage { Storage { app_service, + native_projection_cache: Arc::default(), + projection_observation_sink: None, // app_service: AppService::mock(), cl_service: CLService::mock(), push_queue_service: PushQueueService::new( diff --git a/src/jupiter/storage/mono_storage.rs b/src/jupiter/storage/mono_storage.rs index ff2fb954..335869ac 100644 --- a/src/jupiter/storage/mono_storage.rs +++ b/src/jupiter/storage/mono_storage.rs @@ -26,7 +26,7 @@ use sea_orm::{ use crate::{ callisto::{ mega_blob, mega_cl, mega_commit, mega_ref_tombstones, mega_refs, mega_tag, mega_tree, - mst2_publication, mst2_publication_outbox, mst2_verified_object, + mst2_verified_object, }, common::{ errors::MegaError, @@ -274,6 +274,24 @@ impl MonoStorage { Ok(result) } + pub(crate) async fn get_native_main_ref_in_txn( + &self, + path: &str, + txn: &DatabaseTransaction, + ) -> Result, MegaError> { + let rows = mega_refs::Entity::find() + .filter(mega_refs::Column::Path.eq(path)) + .filter(mega_refs::Column::RefName.eq(MEGA_BRANCH_NAME)) + .filter(mega_refs::Column::IsCl.eq(false)) + .all(txn) + .await?; + match rows.len() { + 0 => Ok(None), + 1 => Ok(rows.into_iter().next()), + _ => Err(MegaError::Other("selected native ref is ambiguous".into())), + } + } + pub async fn get_main_ref_in_txn( &self, path: &str, @@ -1362,6 +1380,7 @@ impl MonoStorage { updated_at = $3 WHERE path = '/' AND ref_name = $4 + AND is_cl = false AND ref_commit_hash IS NOT DISTINCT FROM $5 AND ref_tree_hash IS NOT DISTINCT FROM $6 "#, @@ -1872,112 +1891,6 @@ impl MonoStorage { } } - /// Record one namespace publication inside the caller's transaction - /// (spec 09 §1/§7): bump the per-namespace sequence, insert the - /// receipt and append the outbox event — atomically with the ref CAS - /// the same transaction performs. - /// - /// Idempotent per (namespace, operation_id): a retried writer gets - /// back the original sequence and neither the counter nor the outbox - /// advances twice (PUB-11). The uniqueness is scoped to the namespace - /// because push operation ids (`old→new`) are only unique within one - /// repo: two namespaces landing the same (old,new) pair are distinct - /// publications, not replays. - pub async fn record_publication_in_txn( - &self, - txn: &DatabaseTransaction, - operation_id: &str, - namespace: &str, - old_oid: &str, - new_oid: &str, - writer_kind: &str, - ) -> Result { - // One SELECT by operation id answers both questions: the same id - // under this namespace is a replay (PUB-11) and returns the original - // sequence; the same id under a different namespace means the id is - // not actually unique to this publication — refuse rather than - // silently double-bind it. - if let Some(existing) = mst2_publication::Entity::find() - .filter(mst2_publication::Column::OperationId.eq(operation_id)) - .one(txn) - .await? - { - if existing.namespace == namespace { - return Ok(existing.sequence); - } - return Err(MegaError::Other(format!( - "operation id {operation_id} already published under a different namespace" - ))); - } - // Upsert the sequence row and bump it atomically; the unique - // namespace key makes this a single-winner counter. - let backend = txn.get_database_backend(); - let bump = sea_orm::Statement::from_sql_and_values( - backend, - r#"INSERT INTO mst2_namespace_seq ("namespace", "sequence", "epoch") - VALUES ($1, 1, 1) - ON CONFLICT ("namespace") DO UPDATE SET "sequence" = "mst2_namespace_seq"."sequence" + 1 - RETURNING "sequence""#, - [namespace.into()], - ); - let row = txn - .query_one_raw(bump) - .await? - .ok_or_else(|| MegaError::Other("mst2_namespace_seq upsert returned no row".into()))?; - let sequence: i64 = row.try_get_by_index(0)?; - let now = chrono::Utc::now().fixed_offset(); - - // Concurrency-safe insert: the receipt is unique per - // (namespace, operation_id). The replay check above covers a - // committed prior publication; a truly concurrent same-operation - // writer loses at the outbox insert's unique operation_id below, - // rolling this transaction back rather than double-publishing. - let insert = mst2_publication::ActiveModel { - id: sea_orm::ActiveValue::NotSet, - operation_id: Set(operation_id.to_string()), - namespace: Set(namespace.to_string()), - sequence: Set(sequence), - old_oid: Set(old_oid.to_string()), - new_oid: Set(new_oid.to_string()), - writer_epoch: Set(1), - writer_kind: Set(writer_kind.to_string()), - created_at: Set(now), - }; - match mst2_publication::Entity::insert(insert) - .on_conflict( - OnConflict::columns([ - mst2_publication::Column::Namespace, - mst2_publication::Column::OperationId, - ]) - .do_nothing() - .to_owned(), - ) - .exec(txn) - .await - { - // RecordNotInserted means a receipt with this identity already - // exists; see the outbox insert below for how a racing writer - // of the same operation is resolved. - Ok(_) | Err(DbErr::RecordNotInserted) => {} - Err(e) => return Err(e.into()), - } - - // The replay check above already returned for a pre-existing receipt, - // so reaching here means this call owns `sequence`; the outbox row - // inserts normally. - mst2_publication_outbox::ActiveModel { - id: sea_orm::ActiveValue::NotSet, - operation_id: Set(operation_id.to_string()), - namespace: Set(namespace.to_string()), - sequence: Set(sequence), - state: Set("PENDING".to_string()), - created_at: Set(now), - } - .insert(txn) - .await?; - Ok(sequence) - } - /// Canonical publication namespace for a repository path: a leading /// slash, no trailing slash, root is `/`. Publication identities are /// keyed on this so `/p/` and `/p` never count as two namespaces. @@ -3491,7 +3404,7 @@ mod tests { // Rollback ⇒ nothing visible: the receipt commits only with the // ref CAS in the same transaction (spec 09 §1). let txn = mono.get_connection().begin().await.unwrap(); - mono.record_publication_in_txn(&txn, "op-roll", "/", "", "aaa", "trunk_push") + mono.record_test_publication_in_txn(&txn, "op-roll", "/", "", "aaa", "trunk_push") .await .unwrap(); txn.rollback().await.unwrap(); @@ -3511,7 +3424,7 @@ mod tests { for (op, old, new) in [("op-1", "", "a1"), ("op-2", "a1", "a2")] { let txn = mono.get_connection().begin().await.unwrap(); let seq = mono - .record_publication_in_txn(&txn, op, "/", old, new, "trunk_push") + .record_test_publication_in_txn(&txn, op, "/", old, new, "trunk_push") .await .unwrap(); txn.commit().await.unwrap(); @@ -3530,7 +3443,7 @@ mod tests { // adds nothing (PUB-11) — the push retry path depends on this. let txn = mono.get_connection().begin().await.unwrap(); let again = mono - .record_publication_in_txn(&txn, "op-1", "/", "", "a1", "trunk_push") + .record_test_publication_in_txn(&txn, "op-1", "/", "", "a1", "trunk_push") .await .unwrap(); txn.commit().await.unwrap(); @@ -3547,7 +3460,7 @@ mod tests { // Distinct namespaces keep independent sequences. let txn = mono.get_connection().begin().await.unwrap(); let other = mono - .record_publication_in_txn(&txn, "op-child", "/child", "", "c1", "trunk_push") + .record_test_publication_in_txn(&txn, "op-child", "/child", "", "c1", "trunk_push") .await .unwrap(); txn.commit().await.unwrap(); @@ -3564,13 +3477,13 @@ mod tests { // replay: refuse rather than silently double-bind the id (the // reviewer's P3.1 — a silent no-op left /b's counter at 0 forever). let txn = mono.get_connection().begin().await.unwrap(); - mono.record_publication_in_txn(&txn, "shared-op", "/", "", "a1", "trunk_push") + mono.record_test_publication_in_txn(&txn, "shared-op", "/", "", "a1", "trunk_push") .await .unwrap(); txn.commit().await.unwrap(); let txn = mono.get_connection().begin().await.unwrap(); let err = mono - .record_publication_in_txn(&txn, "shared-op", "/b", "", "a1", "trunk_push") + .record_test_publication_in_txn(&txn, "shared-op", "/b", "", "a1", "trunk_push") .await; assert!(err.is_err(), "cross-namespace reuse must be refused"); drop(txn); // the failed transaction rolls back on drop @@ -3593,7 +3506,7 @@ mod tests { let m = base.mono_storage(); let txn = m.get_connection().begin().await.unwrap(); let seq = m - .record_publication_in_txn( + .record_test_publication_in_txn( &txn, &format!("op-c{i}"), "/", diff --git a/src/jupiter/storage/mst2_publication_storage.rs b/src/jupiter/storage/mst2_publication_storage.rs new file mode 100644 index 00000000..a7111c44 --- /dev/null +++ b/src/jupiter/storage/mst2_publication_storage.rs @@ -0,0 +1,514 @@ +//! SQL receipt identity for the existing trunk queue adapter. +//! +//! This is not a complete namespace/projection publication implementation. + +use sea_orm::{ + ActiveModelTrait, ActiveValue::Set, ColumnTrait, ConnectionTrait, DatabaseTransaction, DbErr, + EntityTrait, QueryFilter, Statement, +}; +use sha2::{Digest, Sha256}; + +use crate::{ + callisto::{ + mst2_publication, mst2_publication_outbox, mst2_queue_noop_receipt, push_queue, + sea_orm_active_enums::PushQueueKindEnum, + }, + common::utils::MEGA_BRANCH_NAME, + jupiter::storage::mono_storage::MonoStorage, +}; + +const REQUEST_DIGEST_VERSION: i32 = 1; +const WRITER_EPOCH: i64 = 1; + +#[derive(Debug, thiserror::Error)] +pub(crate) enum PublicationReceiptError { + #[error("MST2_PUBLICATION_CONFLICT: {0}")] + Conflict(String), + #[error("MST2_PUBLICATION_LEGACY_RECEIPT: operation {0} has no supported request digest")] + LegacyReceipt(String), + #[error("MST2_PUBLICATION_INTEGRITY_ERROR: {0}")] + Integrity(String), + #[error(transparent)] + Database(#[from] DbErr), +} + +/// Constructed from a server-persisted queue row, never a client digest. +#[derive(Debug, Clone)] +pub(crate) struct PublicationRequest { + operation_id: String, + namespace: String, + writer_kind: String, + request_digest: String, + legacy_operation_id: Option, + is_noop: bool, +} + +impl PublicationRequest { + pub(crate) fn from_trunk_queue( + row: &push_queue::Model, + ) -> Result { + if row.id <= 0 || row.operation_id.is_empty() { + return Err(PublicationReceiptError::Integrity( + "invalid queue operation identity".into(), + )); + } + mst2_codec::descriptor::validate_scope(&row.path) + .map_err(|error| PublicationReceiptError::Integrity(error.to_string()))?; + let writer_kind = match &row.kind { + PushQueueKindEnum::Push => "trunk_push", + PushQueueKindEnum::Merge => "trunk_merge_stub", + PushQueueKindEnum::Attach => { + return Err(PublicationReceiptError::Integrity( + "attach is not covered by this adapter".into(), + )); + } + }; + let operation_id = format!("mst2:trunk-queue:{}", row.id); + let mut hash = request_hasher(&operation_id, &row.path, writer_kind); + hash_field(&mut hash, row.operation_id.as_bytes()); + hash_field(&mut hash, MEGA_BRANCH_NAME.as_bytes()); + hash_field(&mut hash, row.old_id.as_bytes()); + hash_field(&mut hash, row.new_id.as_bytes()); + hash.update([u8::from(row.requester.is_some())]); + if let Some(actor) = &row.requester { + hash_field(&mut hash, actor.as_bytes()); + } + hash_json(&mut hash, &row.payload); + Ok(Self { + operation_id, + namespace: row.path.clone(), + writer_kind: writer_kind.to_string(), + request_digest: format!("sha256:{}", hex::encode(hash.finalize())), + legacy_operation_id: Some(row.operation_id.clone()), + is_noop: row.kind == PushQueueKindEnum::Push + && row.old_id == row.new_id + && row.payload.get("n").and_then(serde_json::Value::as_u64) == Some(0), + }) + } + + #[cfg(test)] + fn for_test( + operation_id: &str, + namespace: &str, + old: &str, + new: &str, + writer_kind: &str, + ) -> Self { + let mut hash = request_hasher(operation_id, namespace, writer_kind); + hash_field(&mut hash, old.as_bytes()); + hash_field(&mut hash, new.as_bytes()); + Self { + operation_id: operation_id.to_owned(), + namespace: namespace.to_owned(), + writer_kind: writer_kind.to_owned(), + request_digest: format!("sha256:{}", hex::encode(hash.finalize())), + legacy_operation_id: None, + is_noop: false, + } + } +} + +fn request_hasher(operation_id: &str, namespace: &str, writer_kind: &str) -> Sha256 { + let mut hash = Sha256::new(); + hash.update(b"mega.mst2.trunk-queue-request\0"); + hash.update(REQUEST_DIGEST_VERSION.to_le_bytes()); + hash.update(WRITER_EPOCH.to_le_bytes()); + hash_field(&mut hash, operation_id.as_bytes()); + hash_field(&mut hash, namespace.as_bytes()); + hash_field(&mut hash, writer_kind.as_bytes()); + hash +} + +fn hash_field(hash: &mut Sha256, bytes: &[u8]) { + hash.update((bytes.len() as u64).to_le_bytes()); + hash.update(bytes); +} + +fn hash_json(hash: &mut Sha256, value: &serde_json::Value) { + use serde_json::Value; + match value { + Value::Null => hash.update([0]), + Value::Bool(value) => hash.update([1, u8::from(*value)]), + Value::Number(value) => { + hash.update([2]); + hash_field(hash, value.to_string().as_bytes()); + } + Value::String(value) => { + hash.update([3]); + hash_field(hash, value.as_bytes()); + } + Value::Array(values) => { + hash.update([4]); + hash.update((values.len() as u64).to_le_bytes()); + for value in values { + hash_json(hash, value); + } + } + Value::Object(values) => { + hash.update([5]); + hash.update((values.len() as u64).to_le_bytes()); + let mut keys: Vec<_> = values.keys().collect(); + keys.sort_unstable(); + for key in keys { + hash_field(hash, key.as_bytes()); + hash_json(hash, &values[key]); + } + } + } +} + +#[derive(Debug)] +pub(crate) struct CommittedPublication { + pub(crate) receipt: mst2_publication::Model, + pub(crate) outbox: mst2_publication_outbox::Model, +} + +#[derive(Debug)] +pub(crate) enum PublicationPreparation { + Prepared(PreparedPublication), + AlreadyCommitted(CommittedPublication), + AlreadyCommittedNoop(mst2_queue_noop_receipt::Model), +} + +/// The SQL transaction id prevents finalizing a reservation in another txn. +#[derive(Debug)] +pub(crate) struct PreparedPublication { + request: PublicationRequest, + sequence: i64, + transaction_id: i64, +} + +async fn validate_committed_publication( + txn: &DatabaseTransaction, + request: &PublicationRequest, + receipt: mst2_publication::Model, +) -> Result { + if receipt.namespace != request.namespace { + return Err(PublicationReceiptError::Conflict( + "operation belongs to another namespace".into(), + )); + } + if receipt.request_digest_version != Some(REQUEST_DIGEST_VERSION) + || receipt.request_digest.is_none() + { + return Err(PublicationReceiptError::LegacyReceipt( + request.operation_id.clone(), + )); + } + if receipt.writer_kind != request.writer_kind + || receipt.writer_epoch != WRITER_EPOCH + || receipt.request_digest.as_deref() != Some(request.request_digest.as_str()) + { + return Err(PublicationReceiptError::Conflict( + "operation request changed".into(), + )); + } + let outbox = mst2_publication_outbox::Entity::find() + .filter(mst2_publication_outbox::Column::OperationId.eq(&request.operation_id)) + .one(txn) + .await? + .ok_or_else(|| PublicationReceiptError::Integrity("receipt outbox missing".into()))?; + if outbox.namespace != receipt.namespace || outbox.sequence != receipt.sequence { + return Err(PublicationReceiptError::Integrity( + "receipt outbox identity mismatch".into(), + )); + } + Ok(CommittedPublication { receipt, outbox }) +} + +async fn reject_legacy_queue_receipt( + txn: &DatabaseTransaction, + legacy_operation_id: Option<&str>, + namespace: &str, +) -> Result<(), PublicationReceiptError> { + if let Some(legacy_id) = legacy_operation_id + && let Some(legacy) = mst2_publication::Entity::find() + .filter(mst2_publication::Column::OperationId.eq(legacy_id)) + .filter(mst2_publication::Column::Namespace.eq(namespace)) + .one(txn) + .await? + && (legacy.request_digest_version != Some(REQUEST_DIGEST_VERSION) + || legacy.request_digest.is_none()) + { + return Err(PublicationReceiptError::LegacyReceipt(legacy_id.to_owned())); + } + Ok(()) +} + +enum RecordedOperation { + Publication(mst2_publication::Model), + Noop(mst2_queue_noop_receipt::Model), +} + +async fn find_committed_operation( + txn: &DatabaseTransaction, + operation_id: &str, +) -> Result, PublicationReceiptError> { + let publication = mst2_publication::Entity::find() + .filter(mst2_publication::Column::OperationId.eq(operation_id)) + .one(txn) + .await?; + let noop = mst2_queue_noop_receipt::Entity::find() + .filter(mst2_queue_noop_receipt::Column::OperationId.eq(operation_id)) + .one(txn) + .await?; + match (publication, noop) { + (Some(_), Some(_)) => Err(PublicationReceiptError::Integrity( + "operation has both publication and no-op receipts".into(), + )), + (Some(receipt), None) => Ok(Some(RecordedOperation::Publication(receipt))), + (None, Some(receipt)) => Ok(Some(RecordedOperation::Noop(receipt))), + (None, None) => Ok(None), + } +} + +async fn validate_committed_operation( + txn: &DatabaseTransaction, + request: &PublicationRequest, + existing: RecordedOperation, +) -> Result { + match existing { + RecordedOperation::Publication(receipt) => Ok(PublicationPreparation::AlreadyCommitted( + validate_committed_publication(txn, request, receipt).await?, + )), + RecordedOperation::Noop(receipt) => { + if receipt.namespace != request.namespace + || receipt.writer_kind != request.writer_kind + || receipt.writer_epoch != WRITER_EPOCH + || !request.is_noop + || receipt.request_digest != request.request_digest + { + return Err(PublicationReceiptError::Conflict( + "operation request changed".into(), + )); + } + if receipt.request_digest_version != REQUEST_DIGEST_VERSION { + return Err(PublicationReceiptError::LegacyReceipt( + request.operation_id.clone(), + )); + } + if mst2_publication_outbox::Entity::find() + .filter(mst2_publication_outbox::Column::OperationId.eq(&request.operation_id)) + .one(txn) + .await? + .is_some() + { + return Err(PublicationReceiptError::Integrity( + "no-op operation has a publication outbox".into(), + )); + } + Ok(PublicationPreparation::AlreadyCommittedNoop(receipt)) + } + } +} + +impl MonoStorage { + /// Queue admission can replay Done without executing B3; audit that path too. + pub(crate) async fn validate_queue_publication_replay_in_txn( + &self, + txn: &DatabaseTransaction, + candidate: &push_queue::Model, + ) -> Result<(), PublicationReceiptError> { + let operation_id = format!("mst2:trunk-queue:{}", candidate.id); + let Some(existing) = find_committed_operation(txn, &operation_id).await? else { + reject_legacy_queue_receipt(txn, Some(&candidate.operation_id), &candidate.path) + .await?; + return Ok(()); + }; + let request = PublicationRequest::from_trunk_queue(candidate)?; + if let RecordedOperation::Publication(receipt) = &existing { + self.validate_native_historical_receipt_in_txn(txn, receipt) + .await?; + } + validate_committed_operation(txn, &request, existing).await?; + Ok(()) + } + + /// Lock before any business ref write, then recheck immutable receipt identity. + pub(crate) async fn begin_publication_in_txn( + &self, + txn: &DatabaseTransaction, + request: PublicationRequest, + ) -> Result { + txn.execute_raw(Statement::from_sql_and_values( + txn.get_database_backend(), + "INSERT INTO mst2_namespace_seq (namespace, sequence, epoch) VALUES ($1, 0, $2) \ + ON CONFLICT (namespace) DO NOTHING", + [request.namespace.clone().into(), WRITER_EPOCH.into()], + )) + .await?; + let row = txn + .query_one_raw(Statement::from_sql_and_values( + txn.get_database_backend(), + "SELECT sequence, epoch, txid_current() AS transaction_id \ + FROM mst2_namespace_seq WHERE namespace = $1 FOR UPDATE", + [request.namespace.clone().into()], + )) + .await? + .ok_or_else(|| PublicationReceiptError::Integrity("namespace row missing".into()))?; + let sequence: i64 = row.try_get("", "sequence")?; + let epoch: i64 = row.try_get("", "epoch")?; + if let Some(existing) = find_committed_operation(txn, &request.operation_id).await? { + if let RecordedOperation::Publication(receipt) = &existing { + self.validate_native_historical_receipt_in_txn(txn, receipt) + .await?; + } + return validate_committed_operation(txn, &request, existing).await; + } + if epoch != WRITER_EPOCH || sequence < 0 { + return Err(PublicationReceiptError::Conflict( + "writer epoch is fenced".into(), + )); + } + reject_legacy_queue_receipt( + txn, + request.legacy_operation_id.as_deref(), + &request.namespace, + ) + .await?; + Ok(PublicationPreparation::Prepared(PreparedPublication { + request, + sequence, + transaction_id: row.try_get("", "transaction_id")?, + })) + } + + /// Finalize only the request reserved before this transaction's business writes. + pub(crate) async fn record_publication_in_txn( + &self, + txn: &DatabaseTransaction, + prepared: PreparedPublication, + old_oid: &str, + new_oid: &str, + ) -> Result { + let request = prepared.request; + if request.is_noop { + return Err(PublicationReceiptError::Integrity( + "no-op requests do not publish".into(), + )); + } + let row = txn + .query_one_raw(Statement::from_sql_and_values( + txn.get_database_backend(), + "UPDATE mst2_namespace_seq SET sequence = sequence + 1 \ + WHERE namespace = $1 AND sequence = $2 AND epoch = $3 AND txid_current() = $4 \ + RETURNING sequence", + [ + request.namespace.clone().into(), + prepared.sequence.into(), + WRITER_EPOCH.into(), + prepared.transaction_id.into(), + ], + )) + .await? + .ok_or_else(|| { + PublicationReceiptError::Conflict( + "publication reservation is stale or belongs to another transaction".into(), + ) + })?; + let sequence: i64 = row.try_get("", "sequence")?; + let now = chrono::Utc::now().fixed_offset(); + let receipt = mst2_publication::ActiveModel { + id: sea_orm::ActiveValue::NotSet, + operation_id: Set(request.operation_id.clone()), + namespace: Set(request.namespace.clone()), + sequence: Set(sequence), + old_oid: Set(old_oid.to_owned()), + new_oid: Set(new_oid.to_owned()), + writer_epoch: Set(WRITER_EPOCH), + writer_kind: Set(request.writer_kind), + request_digest: Set(Some(request.request_digest)), + request_digest_version: Set(Some(REQUEST_DIGEST_VERSION)), + native_certificate_version: Set(None), + created_at: Set(now), + } + .insert(txn) + .await?; + #[cfg(all(test, unix))] + tests::crash_checkpoint("receipt-written"); + let outbox = mst2_publication_outbox::ActiveModel { + id: sea_orm::ActiveValue::NotSet, + operation_id: Set(request.operation_id), + namespace: Set(request.namespace), + sequence: Set(sequence), + state: Set("PENDING".to_owned()), + created_at: Set(now), + } + .insert(txn) + .await?; + #[cfg(all(test, unix))] + tests::crash_checkpoint("outbox-written"); + Ok(CommittedPublication { receipt, outbox }) + } + + pub(crate) async fn record_noop_operation_in_txn( + &self, + txn: &DatabaseTransaction, + prepared: PreparedPublication, + root_commit: &str, + root_tree: &str, + landed_commit_id: &str, + ) -> Result { + if !prepared.request.is_noop { + return Err(PublicationReceiptError::Integrity( + "request is not a trunk push no-op".into(), + )); + } + txn.query_one_raw(Statement::from_sql_and_values( + txn.get_database_backend(), + "SELECT sequence FROM mst2_namespace_seq \ + WHERE namespace = $1 AND sequence = $2 AND epoch = $3 AND txid_current() = $4 FOR UPDATE", + [prepared.request.namespace.clone().into(), prepared.sequence.into(), WRITER_EPOCH.into(), prepared.transaction_id.into()], + )).await?.ok_or_else(|| PublicationReceiptError::Conflict("no-op reservation is stale or belongs to another transaction".into()))?; + let request = prepared.request; + let receipt = mst2_queue_noop_receipt::ActiveModel { + id: sea_orm::ActiveValue::NotSet, + operation_id: Set(request.operation_id), + namespace: Set(request.namespace), + request_digest: Set(request.request_digest), + request_digest_version: Set(REQUEST_DIGEST_VERSION), + writer_epoch: Set(WRITER_EPOCH), + writer_kind: Set(request.writer_kind), + observed_sequence: Set(prepared.sequence), + observed_root_commit: Set(root_commit.to_owned()), + observed_root_tree: Set(root_tree.to_owned()), + landed_commit_id: Set(landed_commit_id.to_owned()), + created_at: Set(chrono::Utc::now().fixed_offset()), + } + .insert(txn) + .await?; + #[cfg(all(test, unix))] + tests::crash_checkpoint("noop-receipt-written"); + Ok(receipt) + } + + #[cfg(test)] + pub(crate) async fn record_test_publication_in_txn( + &self, + txn: &DatabaseTransaction, + operation_id: &str, + namespace: &str, + old: &str, + new: &str, + writer_kind: &str, + ) -> Result { + let request = PublicationRequest::for_test(operation_id, namespace, old, new, writer_kind); + let committed = match self.begin_publication_in_txn(txn, request).await? { + PublicationPreparation::AlreadyCommitted(committed) => committed, + PublicationPreparation::AlreadyCommittedNoop(_) => { + return Err(PublicationReceiptError::Integrity( + "test publication replayed a no-op".into(), + )); + } + PublicationPreparation::Prepared(prepared) => { + self.record_publication_in_txn(txn, prepared, old, new) + .await? + } + }; + Ok(committed.receipt.sequence) + } +} + +#[cfg(test)] +#[path = "mst2_publication_tests.rs"] +mod tests; diff --git a/src/jupiter/storage/mst2_publication_tests.rs b/src/jupiter/storage/mst2_publication_tests.rs new file mode 100644 index 00000000..0937a625 --- /dev/null +++ b/src/jupiter/storage/mst2_publication_tests.rs @@ -0,0 +1,1091 @@ +use std::sync::Arc; + +use sea_orm::{PaginatorTrait, TransactionTrait}; +use sea_orm_migration::MigratorTrait; + +use super::*; +use crate::jupiter::{ + migration::{Migrator, apply_migrations}, + storage::{ + base_storage::{BaseStorage, StorageConnector}, + push_queue_storage::{ClaimOutcome, EnqueueOutcome, EnqueueParams, PushQueueStorage}, + }, + tests::test_db_connection, +}; + +fn queue_row() -> push_queue::Model { + let now = chrono::Utc::now().fixed_offset(); + push_queue::Model { + id: 7, + kind: PushQueueKindEnum::Push, + operation_id: "old→new".to_owned(), + path: "/project".to_owned(), + old_id: "old".to_owned(), + new_id: "new".to_owned(), + payload: serde_json::json!({"commits": ["new"], "n": 1, "fork_base": "old"}), + landed_commit_id: None, + status: crate::callisto::sea_orm_active_enums::PushQueueStatusEnum::Running, + requester: Some("alice".to_owned()), + failure_type: None, + error_message: None, + heartbeat_at: now, + superseded_by: None, + expected_commit_hash: Some("root-before".to_owned()), + expected_tree_hash: Some("tree-before".to_owned()), + expected_native_sequence: None, + expected_native_epoch: None, + expected_native_certificate: None, + pending_action: None, + enqueued_at: now, + started_at: Some(now), + finished_at: None, + updated_at: now, + } +} + +fn noop_queue_row() -> push_queue::Model { + let mut row = queue_row(); + row.old_id = "tip".to_owned(); + row.new_id = "tip".to_owned(); + row.payload = serde_json::json!({"commits": [], "fork_base": null, "n": 0}); + row +} + +#[test] +fn queue_digest_is_versioned_and_binds_trusted_intent_not_retry_baseline() { + let row = queue_row(); + let original = PublicationRequest::from_trunk_queue(&row).unwrap(); + assert_eq!(original.operation_id, "mst2:trunk-queue:7"); + let mut retry = row.clone(); + retry.expected_commit_hash = Some("later-root".to_owned()); + retry.expected_tree_hash = Some("later-tree".to_owned()); + retry.started_at = None; + retry.payload = serde_json::from_str(r#"{"n":1,"fork_base":"old","commits":["new"]}"#).unwrap(); + assert_eq!( + PublicationRequest::from_trunk_queue(&retry) + .unwrap() + .request_digest, + original.request_digest + ); + for change in ["actor", "path", "old", "new", "payload", "identity"] { + let mut changed = row.clone(); + match change { + "actor" => changed.requester = Some("bob".to_owned()), + "path" => changed.path = "/other".to_owned(), + "old" => changed.old_id = "other-old".to_owned(), + "new" => changed.new_id = "other-new".to_owned(), + "payload" => changed.payload["commits"] = serde_json::json!(["new", "extra"]), + "identity" => changed.id += 1, + _ => unreachable!(), + } + assert_ne!( + PublicationRequest::from_trunk_queue(&changed) + .unwrap() + .request_digest, + original.request_digest, + "{change}" + ); + } +} + +async fn storage() -> (tempfile::TempDir, MonoStorage) { + let temp = tempfile::tempdir().unwrap(); + let db = test_db_connection(temp.path()).await; + apply_migrations(&db, false).await.unwrap(); + ( + temp, + MonoStorage { + base: BaseStorage::new(Arc::new(db)), + }, + ) +} + +async fn seed_root(mono: &MonoStorage) { + mono.get_connection().execute_raw(Statement::from_sql_and_values( + mono.get_connection().get_database_backend(), + "INSERT INTO mega_refs (id, path, ref_name, ref_commit_hash, ref_tree_hash, created_at, updated_at, is_cl) \ + VALUES (1, '/', $1, $2, $3, now(), now(), false)", + [MEGA_BRANCH_NAME.into(), "a".repeat(40).into(), "b".repeat(40).into()], + )).await.unwrap(); +} + +async fn prepared( + mono: &MonoStorage, + txn: &DatabaseTransaction, + request: PublicationRequest, +) -> PreparedPublication { + match mono.begin_publication_in_txn(txn, request).await.unwrap() { + PublicationPreparation::Prepared(prepared) => prepared, + PublicationPreparation::AlreadyCommitted(_) => panic!("fresh operation"), + PublicationPreparation::AlreadyCommittedNoop(_) => panic!("fresh no-op operation"), + } +} + +async fn advance(mono: &MonoStorage, txn: &DatabaseTransaction) -> bool { + mono.cas_update_root_main_ref_in_txn( + txn, + Some(&"a".repeat(40)), + Some(&"b".repeat(40)), + &"c".repeat(40), + &"d".repeat(40), + ) + .await + .unwrap() +} + +async fn counts(mono: &MonoStorage) -> (u64, u64) { + ( + mst2_publication::Entity::find() + .count(mono.get_connection()) + .await + .unwrap(), + mst2_publication_outbox::Entity::find() + .count(mono.get_connection()) + .await + .unwrap(), + ) +} + +#[tokio::test] +async fn merge_stub_admission_replays_same_receipt_and_rejects_changed_intent() { + let (_temp, mono) = storage().await; + seed_root(&mono).await; + let queue = PushQueueStorage::new(mono.base.clone()); + let payload = serde_json::json!({"steps": ["first"], "mode": "stub"}); + let admitted = queue + .enqueue_atomic(EnqueueParams { + kind: PushQueueKindEnum::Merge, + operation_id: "merge-stub-admission", + path: "/project", + old_id: "old", + new_id: "new", + requester: Some("alice"), + payload: payload.clone(), + }) + .await + .unwrap(); + let EnqueueOutcome::Inserted { id } = admitted else { + panic!("{admitted:?}"); + }; + // With no committed receipt, admission retains the existing adoption + // behavior and the original persisted intent remains the execution input. + assert_eq!( + queue + .enqueue_atomic(EnqueueParams { + kind: PushQueueKindEnum::Merge, + operation_id: "merge-stub-admission", + path: "/project", + old_id: "old", + new_id: "new", + requester: Some("legacy-retry"), + payload: serde_json::json!({"different": "no receipt yet"}), + }) + .await + .unwrap(), + EnqueueOutcome::Adopted { id } + ); + assert_eq!( + queue.claim_for_execution(id).await.unwrap(), + ClaimOutcome::Claimed + ); + let row = push_queue::Entity::find_by_id(id) + .one(mono.get_connection()) + .await + .unwrap() + .unwrap(); + assert_eq!(row.requester.as_deref(), Some("alice")); + assert_eq!(row.payload, payload); + let txn = mono.get_connection().begin().await.unwrap(); + let reservation = prepared( + &mono, + &txn, + PublicationRequest::from_trunk_queue(&row).unwrap(), + ) + .await; + assert!(advance(&mono, &txn).await); + let landed = "c".repeat(40); + assert!( + PushQueueStorage::mark_done_if_running_in_txn(&txn, id, &landed) + .await + .unwrap() + ); + let original = mono + .record_publication_in_txn(&txn, reservation, &"a".repeat(40), &landed) + .await + .unwrap(); + assert_eq!(original.receipt.writer_kind, "trunk_merge_stub"); + txn.commit().await.unwrap(); + assert_eq!( + queue + .enqueue_atomic(EnqueueParams { + kind: PushQueueKindEnum::Merge, + operation_id: "merge-stub-admission", + path: "/project", + old_id: "old", + new_id: "new", + requester: Some("alice"), + payload: payload.clone(), + }) + .await + .unwrap(), + EnqueueOutcome::Replay { + id, + landed_commit_id: Some(landed.clone()) + } + ); + for change in ["actor", "payload"] { + let (requester, retry_payload) = if change == "actor" { + ("mallory", payload.clone()) + } else { + ( + "alice", + serde_json::json!({"steps": ["first", "changed"], "mode": "stub"}), + ) + }; + let error = queue + .enqueue_atomic(EnqueueParams { + kind: PushQueueKindEnum::Merge, + operation_id: "merge-stub-admission", + path: "/project", + old_id: "old", + new_id: "new", + requester: Some(requester), + payload: retry_payload, + }) + .await + .unwrap_err(); + assert!( + error.to_string().contains("MST2_PUBLICATION_CONFLICT"), + "{change}" + ); + } + assert_eq!( + mst2_publication::Entity::find() + .all(mono.get_connection()) + .await + .unwrap(), + vec![original.receipt] + ); + assert_eq!( + mst2_publication_outbox::Entity::find() + .all(mono.get_connection()) + .await + .unwrap(), + vec![original.outbox] + ); + assert_eq!(mono.publication_sequence("/project").await.unwrap(), 1); + assert_eq!( + mono.get_main_ref("/") + .await + .unwrap() + .unwrap() + .ref_commit_hash, + landed + ); +} + +#[tokio::test] +async fn same_request_concurrently_returns_one_original_receipt_and_outbox() { + let (_temp, mono) = storage().await; + seed_root(&mono).await; + let request = PublicationRequest::for_test("same-op", "/", "a", "c", "test"); + let barrier = Arc::new(tokio::sync::Barrier::new(2)); + let mut workers = Vec::new(); + for _ in 0..2 { + let mono = mono.clone(); + let request = request.clone(); + let barrier = Arc::clone(&barrier); + workers.push(tokio::spawn(async move { + let txn = mono.get_connection().begin().await.unwrap(); + barrier.wait().await; + let (committed, wrote) = + match mono.begin_publication_in_txn(&txn, request).await.unwrap() { + PublicationPreparation::AlreadyCommitted(committed) => (committed, false), + PublicationPreparation::AlreadyCommittedNoop(_) => { + panic!("publication request") + } + PublicationPreparation::Prepared(prepared) => { + assert!(advance(&mono, &txn).await); + ( + mono.record_publication_in_txn(&txn, prepared, "a", "c") + .await + .unwrap(), + true, + ) + } + }; + txn.commit().await.unwrap(); + (committed, wrote) + })); + } + let mut results = Vec::new(); + for worker in workers { + results.push( + tokio::time::timeout(std::time::Duration::from_secs(30), worker) + .await + .unwrap() + .unwrap(), + ); + } + assert_eq!(results.iter().filter(|(_, wrote)| *wrote).count(), 1); + assert_eq!(results[0].0.receipt, results[1].0.receipt); + assert_eq!(results[0].0.outbox, results[1].0.outbox); + assert_eq!(mono.publication_sequence("/").await.unwrap(), 1); + assert_eq!(counts(&mono).await, (1, 1)); +} + +#[tokio::test] +async fn changed_request_conflicts_before_ref_writes() { + let (_temp, mono) = storage().await; + seed_root(&mono).await; + let txn = mono.get_connection().begin().await.unwrap(); + let request = PublicationRequest::from_trunk_queue(&queue_row()).unwrap(); + let reserved = prepared(&mono, &txn, request).await; + assert!(advance(&mono, &txn).await); + mono.record_publication_in_txn(&txn, reserved, "a", "c") + .await + .unwrap(); + txn.commit().await.unwrap(); + for change in ["actor", "new", "payload"] { + let mut changed = queue_row(); + match change { + "actor" => changed.requester = Some("mallory".to_owned()), + "new" => changed.new_id = "overwrite".to_owned(), + "payload" => changed.payload["commits"] = serde_json::json!(["injected"]), + _ => unreachable!(), + } + let txn = mono.get_connection().begin().await.unwrap(); + assert!(matches!( + mono.begin_publication_in_txn( + &txn, + PublicationRequest::from_trunk_queue(&changed).unwrap() + ) + .await, + Err(PublicationReceiptError::Conflict(_)) + )); + txn.rollback().await.unwrap(); + assert_eq!( + mono.get_main_ref("/") + .await + .unwrap() + .unwrap() + .ref_commit_hash, + "c".repeat(40) + ); + assert_eq!(mono.publication_sequence("/project").await.unwrap(), 1); + assert_eq!(counts(&mono).await, (1, 1)); + } +} + +#[tokio::test] +async fn different_operations_compete_on_actual_root_cas_without_extra_sequence() { + let (_temp, mono) = storage().await; + seed_root(&mono).await; + let barrier = Arc::new(tokio::sync::Barrier::new(2)); + let mut workers = Vec::new(); + for i in 0..2 { + let mono = mono.clone(); + let barrier = Arc::clone(&barrier); + workers.push(tokio::spawn(async move { + let txn = mono.get_connection().begin().await.unwrap(); + barrier.wait().await; + let request = + PublicationRequest::for_test(&format!("competitor-{i}"), "/", "a", "c", "test"); + let reserved = prepared(&mono, &txn, request).await; + if !advance(&mono, &txn).await { + txn.rollback().await.unwrap(); + return false; + } + mono.record_publication_in_txn(&txn, reserved, "a", "c") + .await + .unwrap(); + txn.commit().await.unwrap(); + true + })); + } + let mut winners = 0; + for worker in workers { + winners += usize::from( + tokio::time::timeout(std::time::Duration::from_secs(30), worker) + .await + .unwrap() + .unwrap(), + ); + } + assert_eq!(winners, 1); + assert_eq!(counts(&mono).await, (1, 1)); + assert_eq!(mono.publication_sequence("/").await.unwrap(), 1); +} + +#[tokio::test] +async fn reservation_cannot_be_finalized_from_another_transaction() { + let (_temp, mono) = storage().await; + // Keep the namespace/sequence unchanged after rollback: only the txn id differs. + mono.get_connection() + .execute_unprepared( + "INSERT INTO mst2_namespace_seq (namespace, sequence, epoch) VALUES ('/', 0, 1)", + ) + .await + .unwrap(); + let txn = mono.get_connection().begin().await.unwrap(); + let reserved = prepared( + &mono, + &txn, + PublicationRequest::for_test("txn-op", "/", "a", "c", "test"), + ) + .await; + txn.rollback().await.unwrap(); + let other = mono.get_connection().begin().await.unwrap(); + assert!(matches!( + mono.record_publication_in_txn(&other, reserved, "a", "c") + .await, + Err(PublicationReceiptError::Conflict(_)) + )); + other.rollback().await.unwrap(); + assert_eq!(counts(&mono).await, (0, 0)); + assert_eq!(mono.publication_sequence("/").await.unwrap(), 0); +} + +#[tokio::test] +async fn legacy_receipt_and_queue_alias_do_not_guess_request_digest() { + let (_temp, mono) = storage().await; + let row = queue_row(); + mono.get_connection().execute_raw(Statement::from_sql_and_values( + mono.get_connection().get_database_backend(), + "INSERT INTO mst2_publication (operation_id, namespace, sequence, old_oid, new_oid, writer_epoch, writer_kind, created_at) \ + VALUES ($1, $2, 1, 'old', 'new', 1, 'trunk_push', now())", + [row.operation_id.clone().into(), row.path.clone().into()], + )).await.unwrap(); + for request in [ + PublicationRequest::for_test(&row.operation_id, &row.path, "old", "new", "trunk_push"), + PublicationRequest::from_trunk_queue(&row).unwrap(), + ] { + let txn = mono.get_connection().begin().await.unwrap(); + assert!(matches!( + mono.begin_publication_in_txn(&txn, request).await, + Err(PublicationReceiptError::LegacyReceipt(_)) + )); + txn.rollback().await.unwrap(); + } + let txn = mono.get_connection().begin().await.unwrap(); + assert!(matches!( + mono.validate_queue_publication_replay_in_txn(&txn, &row) + .await, + Err(PublicationReceiptError::LegacyReceipt(_)) + )); + txn.rollback().await.unwrap(); + let legacy = mst2_publication::Entity::find() + .one(mono.get_connection()) + .await + .unwrap() + .unwrap(); + assert!(legacy.request_digest.is_none()); + assert!(legacy.request_digest_version.is_none()); + assert_eq!(counts(&mono).await, (1, 0)); +} + +#[tokio::test] +async fn incomplete_outbox_refuses_both_receipt_and_queue_replay() { + for corruption in ["missing", "sequence", "namespace"] { + let (_temp, mono) = storage().await; + seed_root(&mono).await; + let row = queue_row(); + let request = PublicationRequest::from_trunk_queue(&row).unwrap(); + let txn = mono.get_connection().begin().await.unwrap(); + let reserved = prepared(&mono, &txn, request.clone()).await; + assert!(advance(&mono, &txn).await); + mono.record_publication_in_txn(&txn, reserved, "a", "c") + .await + .unwrap(); + txn.commit().await.unwrap(); + let sql = match corruption { + "missing" => "DELETE FROM mst2_publication_outbox", + "sequence" => "UPDATE mst2_publication_outbox SET sequence = sequence + 1", + "namespace" => "UPDATE mst2_publication_outbox SET namespace = '/other'", + _ => unreachable!(), + }; + mono.get_connection().execute_unprepared(sql).await.unwrap(); + let before = counts(&mono).await; + let txn = mono.get_connection().begin().await.unwrap(); + assert!( + matches!( + mono.begin_publication_in_txn(&txn, request).await, + Err(PublicationReceiptError::Integrity(_)) + ), + "{corruption}" + ); + txn.rollback().await.unwrap(); + let txn = mono.get_connection().begin().await.unwrap(); + assert!( + matches!( + mono.validate_queue_publication_replay_in_txn(&txn, &row) + .await, + Err(PublicationReceiptError::Integrity(_)) + ), + "{corruption}" + ); + txn.rollback().await.unwrap(); + assert_eq!( + mono.get_main_ref("/") + .await + .unwrap() + .unwrap() + .ref_commit_hash, + "c".repeat(40) + ); + assert_eq!(mono.publication_sequence(&row.path).await.unwrap(), 1); + assert_eq!(counts(&mono).await, before); + } +} + +#[tokio::test] +async fn concurrent_noop_operations_keep_the_publication_sequence_and_outbox() { + let (_temp, mono) = storage().await; + seed_root(&mono).await; + let barrier = Arc::new(tokio::sync::Barrier::new(2)); + let mut workers = Vec::new(); + for _ in 0..2 { + let mono = mono.clone(); + let barrier = Arc::clone(&barrier); + workers.push(tokio::spawn(async move { + let txn = mono.get_connection().begin().await.unwrap(); + barrier.wait().await; + let (receipt, wrote) = match mono + .begin_publication_in_txn( + &txn, + PublicationRequest::from_trunk_queue(&noop_queue_row()).unwrap(), + ) + .await + .unwrap() + { + PublicationPreparation::Prepared(reserved) => { + assert!( + mono.cas_update_root_main_ref_in_txn( + &txn, + Some(&"a".repeat(40)), + Some(&"b".repeat(40)), + &"a".repeat(40), + &"b".repeat(40) + ) + .await + .unwrap() + ); + ( + mono.record_noop_operation_in_txn( + &txn, + reserved, + &"a".repeat(40), + &"b".repeat(40), + "tip", + ) + .await + .unwrap(), + true, + ) + } + PublicationPreparation::AlreadyCommittedNoop(receipt) => (receipt, false), + PublicationPreparation::AlreadyCommitted(_) => panic!("no-op request"), + }; + txn.commit().await.unwrap(); + (receipt, wrote) + })); + } + let mut results = Vec::new(); + for worker in workers { + results.push( + tokio::time::timeout(std::time::Duration::from_secs(30), worker) + .await + .unwrap() + .unwrap(), + ); + } + assert_eq!(results.iter().filter(|(_, wrote)| *wrote).count(), 1); + assert_eq!(results[0].0, results[1].0); + assert_eq!(results[0].0.observed_sequence, 0); + assert_eq!(mono.publication_sequence("/project").await.unwrap(), 0); + assert_eq!(counts(&mono).await, (0, 0)); + assert_eq!( + mst2_queue_noop_receipt::Entity::find() + .count(mono.get_connection()) + .await + .unwrap(), + 1 + ); + for change in ["actor", "payload"] { + let mut row = noop_queue_row(); + if change == "actor" { + row.requester = Some("other".to_owned()); + } else { + row.payload["fork_base"] = serde_json::json!("other"); + } + let txn = mono.get_connection().begin().await.unwrap(); + assert!(matches!( + mono.begin_publication_in_txn( + &txn, + PublicationRequest::from_trunk_queue(&row).unwrap() + ) + .await, + Err(PublicationReceiptError::Conflict(_)) + )); + txn.rollback().await.unwrap(); + let txn = mono.get_connection().begin().await.unwrap(); + assert!(matches!( + mono.validate_queue_publication_replay_in_txn(&txn, &row) + .await, + Err(PublicationReceiptError::Conflict(_)) + )); + txn.rollback().await.unwrap(); + } +} + +#[tokio::test] +async fn noop_receipt_replay_rejects_unsupported_version_and_publication_corruption() { + for corruption in ["version", "publication", "outbox"] { + let (_temp, mono) = storage().await; + let row = noop_queue_row(); + let request = PublicationRequest::from_trunk_queue(&row).unwrap(); + let txn = mono.get_connection().begin().await.unwrap(); + let reserved = prepared(&mono, &txn, request.clone()).await; + mono.record_noop_operation_in_txn(&txn, reserved, "root", "tree", "tip") + .await + .unwrap(); + txn.commit().await.unwrap(); + let sql = match corruption { + "version" => "UPDATE mst2_queue_noop_receipt SET request_digest_version = 2", + "publication" => { + "INSERT INTO mst2_publication (operation_id, namespace, sequence, old_oid, new_oid, writer_epoch, writer_kind, created_at) \ + VALUES ('mst2:trunk-queue:7', '/project', 1, 'a', 'b', 1, 'trunk_push', now())" + } + "outbox" => { + "INSERT INTO mst2_publication_outbox (operation_id, namespace, sequence, state, created_at) \ + VALUES ('mst2:trunk-queue:7', '/project', 1, 'PENDING', now())" + } + _ => unreachable!(), + }; + mono.get_connection().execute_unprepared(sql).await.unwrap(); + let publications_before = mst2_publication::Entity::find() + .all(mono.get_connection()) + .await + .unwrap(); + let outbox_before = mst2_publication_outbox::Entity::find() + .all(mono.get_connection()) + .await + .unwrap(); + for admission in [false, true] { + let txn = mono.get_connection().begin().await.unwrap(); + let result = if admission { + mono.validate_queue_publication_replay_in_txn(&txn, &row) + .await + } else { + mono.begin_publication_in_txn(&txn, request.clone()) + .await + .map(|_| ()) + }; + if corruption == "version" { + assert!(matches!( + result, + Err(PublicationReceiptError::LegacyReceipt(_)) + )); + } else { + assert!(matches!(result, Err(PublicationReceiptError::Integrity(_)))); + } + txn.rollback().await.unwrap(); + let namespace = mono + .get_connection() + .query_one_raw(Statement::from_sql_and_values( + mono.get_connection().get_database_backend(), + "SELECT sequence FROM mst2_namespace_seq WHERE namespace = $1", + [row.path.clone().into()], + )) + .await + .unwrap() + .unwrap(); + assert_eq!(namespace.try_get::("", "sequence").unwrap(), 0); + assert_eq!( + mst2_publication::Entity::find() + .all(mono.get_connection()) + .await + .unwrap(), + publications_before + ); + assert_eq!( + mst2_publication_outbox::Entity::find() + .all(mono.get_connection()) + .await + .unwrap(), + outbox_before + ); + } + assert_eq!( + mst2_queue_noop_receipt::Entity::find() + .count(mono.get_connection()) + .await + .unwrap(), + 1 + ); + } +} + +#[tokio::test] +async fn noop_reservation_cannot_move_to_another_transaction() { + let (_temp, mono) = storage().await; + mono.get_connection() + .execute_unprepared( + "INSERT INTO mst2_namespace_seq (namespace, sequence, epoch) VALUES ('/project', 0, 1)", + ) + .await + .unwrap(); + let txn = mono.get_connection().begin().await.unwrap(); + let reserved = prepared( + &mono, + &txn, + PublicationRequest::from_trunk_queue(&noop_queue_row()).unwrap(), + ) + .await; + txn.rollback().await.unwrap(); + let other = mono.get_connection().begin().await.unwrap(); + assert!(matches!( + mono.record_noop_operation_in_txn(&other, reserved, "root", "tree", "tip") + .await, + Err(PublicationReceiptError::Conflict(_)) + )); + other.rollback().await.unwrap(); + assert_eq!(mono.publication_sequence("/project").await.unwrap(), 0); + assert_eq!(counts(&mono).await, (0, 0)); + assert_eq!( + mst2_queue_noop_receipt::Entity::find() + .count(mono.get_connection()) + .await + .unwrap(), + 0 + ); +} + +#[tokio::test] +async fn migration_preserves_legacy_receipts_and_validates_digest_columns() { + let temp = tempfile::tempdir().unwrap(); + let db = test_db_connection(temp.path()).await; + let migrations = Migrator::migrations(); + let receipt_at = migrations + .iter() + .position(|migration| { + migration.name() == "m20261005_000100_add_mst2_publication_request_digest" + }) + .expect("publication receipt migration registered"); + let old_count = u32::try_from(receipt_at).unwrap(); + Migrator::up(&db, Some(old_count)).await.unwrap(); + db.execute_unprepared( + "INSERT INTO mst2_publication (operation_id, namespace, sequence, old_oid, new_oid, writer_epoch, writer_kind, created_at) \ + VALUES ('legacy', '/', 1, 'a', 'b', 1, 'test', now())" + ).await.unwrap(); + apply_migrations(&db, false).await.unwrap(); + let txn = db.begin().await.unwrap(); + migrations[receipt_at] + .up(&sea_orm_migration::SchemaManager::new(&txn)) + .await + .unwrap(); + txn.commit().await.unwrap(); + let legacy = mst2_publication::Entity::find() + .one(&db) + .await + .unwrap() + .unwrap(); + assert!(legacy.request_digest.is_none()); + assert!(legacy.request_digest_version.is_none()); + for (digest, version) in [ + (Some(format!("sha256:{}", "a".repeat(64))), Some(1)), + (Some(format!("sha256:{}", "A".repeat(64))), Some(1)), + (Some("broken".to_owned()), Some(1)), + (None, Some(1)), + (Some(format!("sha256:{}", "a".repeat(64))), None), + (Some(format!("sha256:{}", "a".repeat(64))), Some(0)), + ] { + let valid = digest.as_deref() == Some(format!("sha256:{}", "a".repeat(64)).as_str()) + && version == Some(1); + let txn = db.begin().await.unwrap(); + let result = txn.execute_raw(Statement::from_sql_and_values( + txn.get_database_backend(), + "UPDATE mst2_publication SET request_digest = $1, request_digest_version = $2 WHERE operation_id = 'legacy'", + [digest.into(), version.into()], + )).await; + assert_eq!(result.is_ok(), valid); + txn.rollback().await.unwrap(); + } + let legacy = mst2_publication::Entity::find() + .one(&db) + .await + .unwrap() + .unwrap(); + assert!(legacy.request_digest.is_none()); +} + +#[cfg(unix)] +pub(super) fn crash_checkpoint(phase: &str) { + if std::env::var("MEGA_MST2_RECEIPT_CRASH_PHASE") + .ok() + .as_deref() + == Some(phase) + { + unsafe { + libc::raise(libc::SIGKILL); + } + panic!("SIGKILL did not terminate the worker"); + } +} + +#[cfg(unix)] +#[test] +fn receipt_crash_worker() { + let Ok(url) = std::env::var("MEGA_MST2_RECEIPT_WORKER_DB") else { + return; + }; + tokio::runtime::Builder::new_current_thread() + .enable_all() + .build() + .unwrap() + .block_on(async { + let db = sea_orm::Database::connect(url).await.unwrap(); + let mono = MonoStorage { + base: BaseStorage::new(Arc::new(db)), + }; + let txn = mono.get_connection().begin().await.unwrap(); + let noop = std::env::var("MEGA_MST2_RECEIPT_WORKER_NOOP").is_ok(); + let request = if noop { + PublicationRequest::from_trunk_queue(&noop_queue_row()).unwrap() + } else { + PublicationRequest::for_test("process-op", "/", "a", "c", "test") + }; + let reserved = prepared(&mono, &txn, request).await; + if noop { + assert!( + mono.cas_update_root_main_ref_in_txn( + &txn, + Some(&"a".repeat(40)), + Some(&"b".repeat(40)), + &"a".repeat(40), + &"b".repeat(40) + ) + .await + .unwrap() + ); + } else { + assert!(advance(&mono, &txn).await); + } + crash_checkpoint("ref-written"); + if noop { + mono.record_noop_operation_in_txn( + &txn, + reserved, + &"a".repeat(40), + &"b".repeat(40), + "tip", + ) + .await + .unwrap(); + } else { + mono.record_publication_in_txn(&txn, reserved, "a", "c") + .await + .unwrap(); + } + txn.commit().await.unwrap(); + crash_checkpoint("commit-complete"); + }); +} + +#[cfg(unix)] +#[tokio::test] +async fn sigkill_recovery_and_commit_without_response_use_original_receipt() { + use std::os::unix::process::ExitStatusExt; + + for phase in [ + "ref-written", + "receipt-written", + "outbox-written", + "commit-complete", + ] { + let temp = tempfile::tempdir().unwrap(); + let (config, _schema) = crate::jupiter::tests::test_db_config(temp.path()).await; + let db = crate::jupiter::storage::init::database_connection(&config) + .await + .unwrap(); + let mono = MonoStorage { + base: BaseStorage::new(Arc::new(db)), + }; + seed_root(&mono).await; + let mut worker = std::process::Command::new(std::env::current_exe().unwrap()) + .args([ + "--exact", + "jupiter::storage::mst2_publication_storage::tests::receipt_crash_worker", + "--nocapture", + ]) + .env("MEGA_MST2_RECEIPT_WORKER_DB", &config.db_url) + .env("MEGA_MST2_RECEIPT_CRASH_PHASE", phase) + .env_remove("MEGA_MST2_RECEIPT_WORKER_NOOP") + .stdout(std::process::Stdio::null()) + .stderr(std::process::Stdio::inherit()) + .spawn() + .unwrap(); + let deadline = std::time::Instant::now() + std::time::Duration::from_secs(30); + let status = loop { + if let Some(status) = worker.try_wait().unwrap() { + break status; + } + if std::time::Instant::now() >= deadline { + worker.kill().unwrap(); + worker.wait().unwrap(); + panic!("receipt worker timed out at {phase}"); + } + tokio::time::sleep(std::time::Duration::from_millis(10)).await; + }; + assert_eq!(status.signal(), Some(libc::SIGKILL), "{phase}: {status}"); + let committed = phase == "commit-complete"; + let root = mono.get_main_ref("/").await.unwrap().unwrap(); + assert_eq!( + root.ref_commit_hash, + if committed { "c" } else { "a" }.repeat(40) + ); + assert_eq!(counts(&mono).await, if committed { (1, 1) } else { (0, 0) }); + assert_eq!( + mono.publication_sequence("/").await.unwrap(), + i64::from(committed) + ); + let retry_db = sea_orm::Database::connect(config.db_url.as_str()) + .await + .unwrap(); + let retry_mono = MonoStorage { + base: BaseStorage::new(Arc::new(retry_db)), + }; + let txn = retry_mono.get_connection().begin().await.unwrap(); + let request = PublicationRequest::for_test("process-op", "/", "a", "c", "test"); + match retry_mono + .begin_publication_in_txn(&txn, request) + .await + .unwrap() + { + PublicationPreparation::AlreadyCommitted(original) => { + assert!(committed); + assert_eq!(original.receipt.sequence, 1); + assert_eq!(original.receipt.new_oid, "c"); + assert_eq!(original.outbox.sequence, 1); + txn.commit().await.unwrap(); + } + PublicationPreparation::Prepared(_) => { + assert!(!committed); + txn.rollback().await.unwrap(); + } + PublicationPreparation::AlreadyCommittedNoop(_) => panic!("publication request"), + } + assert_eq!(counts(&mono).await, if committed { (1, 1) } else { (0, 0) }); + } +} + +#[cfg(unix)] +#[tokio::test] +async fn noop_sigkill_and_lost_commit_response_never_create_a_publication() { + use std::os::unix::process::ExitStatusExt; + + for phase in ["ref-written", "noop-receipt-written", "commit-complete"] { + let temp = tempfile::tempdir().unwrap(); + let (config, _schema) = crate::jupiter::tests::test_db_config(temp.path()).await; + let db = crate::jupiter::storage::init::database_connection(&config) + .await + .unwrap(); + let mono = MonoStorage { + base: BaseStorage::new(Arc::new(db)), + }; + seed_root(&mono).await; + let mut worker = std::process::Command::new(std::env::current_exe().unwrap()) + .args([ + "--exact", + "jupiter::storage::mst2_publication_storage::tests::receipt_crash_worker", + "--nocapture", + ]) + .env("MEGA_MST2_RECEIPT_WORKER_DB", &config.db_url) + .env("MEGA_MST2_RECEIPT_CRASH_PHASE", phase) + .env("MEGA_MST2_RECEIPT_WORKER_NOOP", "1") + .stdout(std::process::Stdio::null()) + .stderr(std::process::Stdio::inherit()) + .spawn() + .unwrap(); + let deadline = std::time::Instant::now() + std::time::Duration::from_secs(30); + let status = loop { + if let Some(status) = worker.try_wait().unwrap() { + break status; + } + if std::time::Instant::now() >= deadline { + worker.kill().unwrap(); + worker.wait().unwrap(); + panic!("no-op receipt worker timed out at {phase}"); + } + tokio::time::sleep(std::time::Duration::from_millis(10)).await; + }; + assert_eq!(status.signal(), Some(libc::SIGKILL), "{phase}: {status}"); + let committed = phase == "commit-complete"; + assert_eq!( + mono.get_main_ref("/") + .await + .unwrap() + .unwrap() + .ref_commit_hash, + "a".repeat(40) + ); + assert_eq!(counts(&mono).await, (0, 0)); + assert_eq!(mono.publication_sequence("/project").await.unwrap(), 0); + assert_eq!( + mst2_queue_noop_receipt::Entity::find() + .count(mono.get_connection()) + .await + .unwrap(), + u64::from(committed) + ); + let original = mst2_queue_noop_receipt::Entity::find() + .one(mono.get_connection()) + .await + .unwrap(); + if committed { + let txn = mono.get_connection().begin().await.unwrap(); + assert!(advance(&mono, &txn).await); + txn.commit().await.unwrap(); + } + // Reconnect after the worker died; committed no-op identity survives a later root. + let retry_db = sea_orm::Database::connect(config.db_url.as_str()) + .await + .unwrap(); + let retry_mono = MonoStorage { + base: BaseStorage::new(Arc::new(retry_db)), + }; + let txn = retry_mono.get_connection().begin().await.unwrap(); + match retry_mono + .begin_publication_in_txn( + &txn, + PublicationRequest::from_trunk_queue(&noop_queue_row()).unwrap(), + ) + .await + .unwrap() + { + PublicationPreparation::AlreadyCommittedNoop(receipt) => { + assert!(committed); + assert_eq!(Some(receipt.clone()), original); + assert_eq!(receipt.observed_sequence, 0); + assert_eq!(receipt.observed_root_commit, "a".repeat(40)); + assert_eq!(receipt.landed_commit_id, "tip"); + txn.commit().await.unwrap(); + } + PublicationPreparation::Prepared(_) => { + assert!(!committed); + txn.rollback().await.unwrap(); + } + PublicationPreparation::AlreadyCommitted(_) => panic!("no-op must not publish"), + } + assert_eq!(counts(&mono).await, (0, 0)); + assert_eq!(mono.publication_sequence("/project").await.unwrap(), 0); + assert_eq!( + mono.get_main_ref("/") + .await + .unwrap() + .unwrap() + .ref_commit_hash, + if committed { "c" } else { "a" }.repeat(40) + ); + } +} diff --git a/src/jupiter/storage/mst2_retention.rs b/src/jupiter/storage/mst2_retention.rs new file mode 100644 index 00000000..11c9c456 --- /dev/null +++ b/src/jupiter/storage/mst2_retention.rs @@ -0,0 +1,633 @@ +//! Asynchronous PostgreSQL retention graph repository (T06-B, spec 10 §6). +//! +//! This repository is not yet wired into the in-process snapshot runtime. +//! Graph mutations serialize on a schema-scoped transaction advisory lock; +//! callers can include them in a publication transaction. Each mutation uses +//! a savepoint, so a rejected group cannot leave partial data in that caller's +//! READ COMMITTED transaction. Bytes must be complete and verified before +//! retaining a node. +//! +//! A GC claim commits DELETING and a PENDING REMOVE intent together. A worker +//! may repeat the idempotent physical deletion after a crash, then call +//! `complete_gc`. The outgoing edges, child counters and APPLIED receipt commit +//! together, so replay cannot subtract references twice. No method here deletes +//! bytes, and shared Git raw deletion remains disabled until GitRetentionPort +//! is integrated. Physical workers and durable lease/publication wiring remain +//! separate integration work. Removed identities are tombstoned by their GC +//! receipts: rebuilding one requires future generation/fencing integration, +//! otherwise a stale physical worker could delete its newly reconstructed bytes. + +use std::collections::{BTreeMap, BTreeSet, VecDeque}; + +use sea_orm::{ + ColumnTrait, ConnectionTrait, DatabaseConnection, DatabaseTransaction, DbBackend, EntityTrait, + QueryFilter, QueryOrder, QuerySelect, Statement, TransactionTrait, +}; +use serde_json::json; + +use crate::{ + callisto::{mst2_retention_gc_op, mst2_retention_node}, + ceres::snapshot::{ + error::{SnapshotError, SnapshotErrorCode}, + retention::{NodeState, RetentionEdge, RetentionNode, RetentionRoot}, + }, +}; + +const MAX_NODES: usize = 4096; +const MAX_EDGES: usize = 16_384; +const MAX_ROOTS: usize = 16; +const MAX_PENDING_BATCH: u64 = 1000; +pub(crate) const RETENTION_LOCK_KEY: i32 = 1_296_717_362; + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum GcClaim { + /// This transaction marked the node and recorded its pending operation. + Marked, + /// The same operation is committed and awaits physical deletion/ack. + Pending, + /// The same operation already removed the node and released its edges. + Applied, + /// A root, an incoming edge, another claim or absence prevents collection. + Unavailable, +} + +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct GcCompletion { + pub replayed: bool, + /// Newly zero-reference children; roots must still be checked when claimed. + pub zero_reference_children: Vec, +} + +#[derive(Clone)] +pub struct PostgresRetentionRepository { + connection: DatabaseConnection, +} + +impl PostgresRetentionRepository { + pub fn new(connection: DatabaseConnection) -> Self { + Self { connection } + } + + pub async fn node( + &self, + id: &str, + ) -> Result, SnapshotError> { + mst2_retention_node::Entity::find_by_id(id.to_owned()) + .one(&self.connection) + .await + .map_err(internal) + } + + /// Bounded replay scan. PENDING intents survive process/worker replacement. + pub async fn pending_gc( + &self, + limit: u64, + ) -> Result, SnapshotError> { + if limit == 0 || limit > MAX_PENDING_BATCH { + return Err(limit_error("pending GC batch must be 1..=1000")); + } + mst2_retention_gc_op::Entity::find() + .filter(mst2_retention_gc_op::Column::State.eq("PENDING")) + .order_by_asc(mst2_retention_gc_op::Column::CreatedAt) + .order_by_asc(mst2_retention_gc_op::Column::OperationId) + .limit(limit) + .all(&self.connection) + .await + .map_err(internal) + } + + pub async fn retain_group( + &self, + nodes: &[RetentionNode], + edges: &[RetentionEdge], + roots: &[RetentionRoot], + ) -> Result<(), SnapshotError> { + let txn = self.connection.begin().await.map_err(internal)?; + let result = Self::retain_group_in_txn(&txn, nodes, edges, roots).await; + finish(txn, result).await + } + + /// Retain an immutable group atomically. Roots cover each supplied node; + /// callers select that group rather than scanning the reachable graph. + /// An existing parent's edges may only be replayed, not extended; a new + /// identity describes a new graph. + pub async fn retain_group_in_txn( + txn: &DatabaseTransaction, + nodes: &[RetentionNode], + edges: &[RetentionEdge], + roots: &[RetentionRoot], + ) -> Result<(), SnapshotError> { + let group = PreparedGroup::new(nodes, edges, roots)?; + let savepoint = begin_graph(txn).await?; + let result = retain_locked(&savepoint, &group).await; + finish(savepoint, result).await + } + + pub async fn release_root(&self, root: &RetentionRoot) -> Result<(), SnapshotError> { + let txn = self.connection.begin().await.map_err(internal)?; + let result = Self::release_root_in_txn(&txn, root).await; + finish(txn, result).await + } + + pub async fn release_root_in_txn( + txn: &DatabaseTransaction, + root: &RetentionRoot, + ) -> Result<(), SnapshotError> { + let (key, _) = root_identity(root)?; + let savepoint = begin_graph(txn).await?; + let result = savepoint + .execute_raw(statement( + "DELETE FROM mst2_retention_root WHERE root_key = $1", + [key.into()], + )) + .await + .map(|_| ()) + .map_err(internal); + finish(savepoint, result).await + } + + pub async fn mark_deleting( + &self, + operation_id: &str, + node_id: &str, + ) -> Result { + let txn = self.connection.begin().await.map_err(internal)?; + let result = Self::mark_deleting_in_txn(&txn, operation_id, node_id).await; + finish(txn, result).await + } + + /// The LIVE check, zero-reference check, CAS and replay intent share the + /// retention mutation lock. Acquisition either precedes this CAS or fails. + pub async fn mark_deleting_in_txn( + txn: &DatabaseTransaction, + operation_id: &str, + node_id: &str, + ) -> Result { + validate_id(operation_id)?; + validate_id(node_id)?; + let savepoint = begin_graph(txn).await?; + let result = mark_locked(&savepoint, operation_id, node_id).await; + finish(savepoint, result).await + } + + /// Acknowledge an idempotent, durable physical delete. Until this succeeds + /// DELETING parents retain every child. This changes graph rows only. + pub async fn complete_gc(&self, operation_id: &str) -> Result { + let txn = self.connection.begin().await.map_err(internal)?; + let result = Self::complete_gc_in_txn(&txn, operation_id).await; + finish(txn, result).await + } + + pub async fn complete_gc_in_txn( + txn: &DatabaseTransaction, + operation_id: &str, + ) -> Result { + validate_id(operation_id)?; + let savepoint = begin_graph(txn).await?; + let result = complete_locked(&savepoint, operation_id).await; + finish(savepoint, result).await + } +} + +struct PreparedGroup { + nodes_json: String, + edges_json: String, + roots_json: String, +} + +impl PreparedGroup { + fn new( + nodes: &[RetentionNode], + edges: &[RetentionEdge], + roots: &[RetentionRoot], + ) -> Result { + if nodes.len() > MAX_NODES || edges.len() > MAX_EDGES || roots.len() > MAX_ROOTS { + return Err(limit_error("retention group exceeds bounded batch limits")); + } + let mut unique_nodes = BTreeMap::new(); + for node in nodes { + validate_id(&node.id)?; + if node.state != NodeState::Live { + return Err(unavailable("retention acquisition requires LIVE nodes")); + } + let bytes = i64::try_from(node.bytes) + .map_err(|_| limit_error("retention bytes exceed signed database range"))?; + let definition = (node.kind.as_str(), bytes); + if unique_nodes + .insert(node.id.as_str(), definition) + .is_some_and(|old| old != definition) + { + return Err(integrity("conflicting retention node definitions")); + } + } + let mut unique_edges = BTreeSet::new(); + for edge in edges { + validate_id(&edge.parent)?; + validate_id(&edge.child)?; + unique_edges.insert((edge.parent.as_str(), edge.child.as_str())); + } + reject_cycle(&unique_edges)?; + let roots: BTreeSet<_> = roots.iter().map(root_identity).collect::>()?; + Ok(Self { + nodes_json: json!( + unique_nodes + .iter() + .map(|(id, (kind, bytes))| { + json!({"node_id": id, "kind": kind, "bytes": bytes}) + }) + .collect::>() + ) + .to_string(), + edges_json: json!( + unique_edges + .iter() + .map(|(parent, child)| { json!({"parent_id": parent, "child_id": child}) }) + .collect::>() + ) + .to_string(), + roots_json: json!( + roots + .iter() + .map(|(key, kind)| { json!({"root_key": key, "root_kind": kind}) }) + .collect::>() + ) + .to_string(), + }) + } +} + +fn reject_cycle(edges: &BTreeSet<(&str, &str)>) -> Result<(), SnapshotError> { + let mut incoming: BTreeMap<&str, usize> = BTreeMap::new(); + let mut children: BTreeMap<&str, Vec<&str>> = BTreeMap::new(); + for &(parent, child) in edges { + incoming.entry(parent).or_default(); + *incoming.entry(child).or_default() += 1; + children.entry(parent).or_default().push(child); + } + let mut ready: VecDeque<_> = incoming + .iter() + .filter_map(|(&id, &count)| (count == 0).then_some(id)) + .collect(); + let mut processed = 0; + while let Some(id) = ready.pop_front() { + processed += 1; + if let Some(children) = children.get(id) { + for child in children { + if let Some(count) = incoming.get_mut(child) { + *count -= 1; + if *count == 0 { + ready.push_back(child); + } + } + } + } + } + if processed != incoming.len() { + return Err(integrity("retention graph must be acyclic")); + } + Ok(()) +} + +async fn retain_locked( + txn: &DatabaseTransaction, + group: &PreparedGroup, +) -> Result<(), SnapshotError> { + if txn + .query_one_raw(statement( + "SELECT i.node_id FROM jsonb_to_recordset($1::jsonb) AS i(node_id text) \ + WHERE NOT EXISTS (SELECT 1 FROM mst2_retention_node n WHERE n.node_id = i.node_id) \ + AND EXISTS (SELECT 1 FROM mst2_retention_gc_op op WHERE op.node_id = i.node_id) LIMIT 1", + [group.nodes_json.clone().into()], + )) + .await + .map_err(internal)? + .is_some() + { + return Err(unavailable( + "removed node identity requires generation-fenced reconstruction", + )); + } + // Batch comparisons happen before any externally visible commit. An + // identity never changes kind, size or DELETING state on replay. + if txn + .query_one_raw(statement( + "SELECT n.node_id, n.state FROM mst2_retention_node n \ + JOIN jsonb_to_recordset($1::jsonb) AS i(node_id text, kind text, bytes bigint) \ + ON n.node_id = i.node_id \ + WHERE n.state <> 'LIVE' OR n.kind <> i.kind OR n.bytes <> i.bytes LIMIT 1", + [group.nodes_json.clone().into()], + )) + .await + .map_err(internal)? + .is_some() + { + return Err(unavailable( + "existing node is DELETING or conflicts with immutable identity", + )); + } + if txn + .query_one_raw(statement( + "SELECT e.parent_id FROM \ + jsonb_to_recordset($1::jsonb) AS e(parent_id text, child_id text) \ + JOIN mst2_retention_node n ON n.node_id = e.parent_id \ + WHERE NOT EXISTS (SELECT 1 FROM mst2_retention_edge old \ + WHERE old.parent_id = e.parent_id AND old.child_id = e.child_id) LIMIT 1", + [group.edges_json.clone().into()], + )) + .await + .map_err(internal)? + .is_some() + { + return Err(integrity( + "an existing retention parent's edges are immutable", + )); + } + txn.execute_raw(statement( + "INSERT INTO mst2_retention_node (node_id, kind, state, bytes, incoming_refs, created_at) \ + SELECT node_id, kind, 'LIVE', bytes, 0, now() FROM \ + jsonb_to_recordset($1::jsonb) AS i(node_id text, kind text, bytes bigint) \ + ON CONFLICT (node_id) DO NOTHING", + [group.nodes_json.clone().into()], + )) + .await + .map_err(internal)?; + if txn + .query_one_raw(statement( + "SELECT e.parent_id FROM \ + jsonb_to_recordset($1::jsonb) AS e(parent_id text, child_id text) \ + LEFT JOIN mst2_retention_node p ON p.node_id = e.parent_id \ + LEFT JOIN mst2_retention_node c ON c.node_id = e.child_id \ + WHERE p.node_id IS NULL OR c.node_id IS NULL \ + OR p.state <> 'LIVE' OR c.state <> 'LIVE' LIMIT 1", + [group.edges_json.clone().into()], + )) + .await + .map_err(internal)? + .is_some() + { + return Err(unavailable( + "retention edge requires two known LIVE endpoints", + )); + } + txn.execute_raw(statement( + "WITH added AS ( \ + INSERT INTO mst2_retention_edge (parent_id, child_id, created_at) \ + SELECT parent_id, child_id, now() FROM \ + jsonb_to_recordset($1::jsonb) AS e(parent_id text, child_id text) \ + ON CONFLICT (parent_id, child_id) DO NOTHING RETURNING child_id \ + ), delta AS (SELECT child_id, count(*) AS refs FROM added GROUP BY child_id) \ + UPDATE mst2_retention_node n SET incoming_refs = n.incoming_refs + delta.refs \ + FROM delta WHERE n.node_id = delta.child_id", + [group.edges_json.clone().into()], + )) + .await + .map_err(internal)?; + txn.execute_raw(statement( + "INSERT INTO mst2_retention_root (node_id, root_key, root_kind, created_at) \ + SELECT n.node_id, r.root_key, r.root_kind, now() FROM \ + jsonb_to_recordset($1::jsonb) AS n(node_id text) CROSS JOIN \ + jsonb_to_recordset($2::jsonb) AS r(root_key text, root_kind text) \ + ON CONFLICT (node_id, root_key) DO NOTHING", + [ + group.nodes_json.clone().into(), + group.roots_json.clone().into(), + ], + )) + .await + .map_err(internal)?; + Ok(()) +} + +async fn mark_locked( + txn: &DatabaseTransaction, + operation_id: &str, + node_id: &str, +) -> Result { + if let Some(op) = mst2_retention_gc_op::Entity::find_by_id(operation_id.to_owned()) + .one(txn) + .await + .map_err(internal)? + { + if op.node_id != node_id || op.operation != "REMOVE" { + return Err(integrity("GC operation id is bound to different work")); + } + return match op.state.as_str() { + "PENDING" => Ok(GcClaim::Pending), + "APPLIED" => Ok(GcClaim::Applied), + _ => Err(integrity("GC operation is not replayable")), + }; + } + let row = txn + .query_one_raw(statement( + "SELECT n.state, n.incoming_refs, \ + (SELECT count(*) FROM mst2_retention_edge e WHERE e.child_id = n.node_id) AS actual_refs \ + FROM mst2_retention_node n WHERE n.node_id = $1", + [node_id.into()], + )) + .await + .map_err(internal)?; + let Some(row) = row else { + return Ok(GcClaim::Unavailable); + }; + let refs: i64 = row.try_get("", "incoming_refs").map_err(internal)?; + let actual: i64 = row.try_get("", "actual_refs").map_err(internal)?; + if refs != actual || refs < 0 { + return Err(integrity("retention reference audit failed; GC stopped")); + } + if row.try_get::("", "state").map_err(internal)? != "LIVE" || refs != 0 { + return Ok(GcClaim::Unavailable); + } + let changed = txn + .execute_raw(statement( + "UPDATE mst2_retention_node n SET state = 'DELETING' \ + WHERE node_id = $1 AND state = 'LIVE' AND incoming_refs = 0 \ + AND NOT EXISTS (SELECT 1 FROM mst2_retention_root r WHERE r.node_id = n.node_id) \ + AND NOT EXISTS (SELECT 1 FROM mst2_retention_edge e WHERE e.child_id = n.node_id)", + [node_id.into()], + )) + .await + .map_err(internal)?; + if changed.rows_affected() == 0 { + return Ok(GcClaim::Unavailable); + } + txn.execute_raw(statement( + "INSERT INTO mst2_retention_gc_op (operation_id, node_id, operation, state, attempts, created_at) \ + VALUES ($1, $2, 'REMOVE', 'PENDING', 0, now())", + [operation_id.into(), node_id.into()], + )) + .await + .map_err(internal)?; + Ok(GcClaim::Marked) +} + +async fn complete_locked( + txn: &DatabaseTransaction, + operation_id: &str, +) -> Result { + let op = mst2_retention_gc_op::Entity::find_by_id(operation_id.to_owned()) + .one(txn) + .await + .map_err(internal)? + .ok_or_else(|| integrity("unknown GC operation"))?; + if op.operation != "REMOVE" { + return Err(integrity("GC operation is not a removal")); + } + if op.state == "APPLIED" { + return Ok(GcCompletion { + replayed: true, + zero_reference_children: Vec::new(), + }); + } + if op.state != "PENDING" { + return Err(integrity("GC operation is not replayable")); + } + let node = mst2_retention_node::Entity::find_by_id(op.node_id.clone()) + .one(txn) + .await + .map_err(internal)? + .ok_or_else(|| integrity("pending GC node is missing"))?; + if node.state != "DELETING" || node.incoming_refs != 0 { + return Err(integrity("pending GC node is not unreferenced DELETING")); + } + if txn + .query_one_raw(statement( + "SELECT node_id FROM mst2_retention_root WHERE node_id = $1 \ + UNION ALL SELECT child_id FROM mst2_retention_edge WHERE child_id = $1 LIMIT 1", + [op.node_id.clone().into()], + )) + .await + .map_err(internal)? + .is_some() + { + return Err(integrity("pending GC node gained a reference; GC stopped")); + } + // Audit before subtracting. Even a damaged stored counter must not cause + // a child's data to be treated as unreferenced. + if txn.query_one_raw(statement( + "SELECT c.node_id FROM mst2_retention_edge e \ + JOIN mst2_retention_node c ON c.node_id = e.child_id \ + WHERE e.parent_id = $1 AND (c.state <> 'LIVE' OR c.incoming_refs <= 0 OR \ + c.incoming_refs <> (SELECT count(*) FROM mst2_retention_edge i WHERE i.child_id = c.node_id)) \ + LIMIT 1", + [op.node_id.clone().into()], + )).await.map_err(internal)?.is_some() { + return Err(integrity("child reference audit failed; GC stopped")); + } + let rows = txn.query_all_raw(statement( + "WITH removed AS (DELETE FROM mst2_retention_edge WHERE parent_id = $1 RETURNING child_id), \ + delta AS (SELECT child_id, count(*) AS refs FROM removed GROUP BY child_id) \ + UPDATE mst2_retention_node n SET incoming_refs = n.incoming_refs - delta.refs \ + FROM delta WHERE n.node_id = delta.child_id RETURNING n.node_id, n.incoming_refs", + [op.node_id.clone().into()], + )).await.map_err(internal)?; + let mut zero_reference_children = Vec::new(); + for row in rows { + if row.try_get::("", "incoming_refs").map_err(internal)? == 0 { + zero_reference_children.push(row.try_get("", "node_id").map_err(internal)?); + } + } + zero_reference_children.sort(); + txn.execute_raw(statement( + "DELETE FROM mst2_retention_node WHERE node_id = $1 AND state = 'DELETING'", + [op.node_id.into()], + )) + .await + .map_err(internal)?; + txn.execute_raw(statement( + "UPDATE mst2_retention_gc_op SET state = 'APPLIED', completed_at = now(), attempts = attempts + 1 \ + WHERE operation_id = $1 AND state = 'PENDING'", + [operation_id.into()], + )).await.map_err(internal)?; + Ok(GcCompletion { + replayed: false, + zero_reference_children, + }) +} + +async fn begin_graph(txn: &DatabaseTransaction) -> Result { + if txn.get_database_backend() != DbBackend::Postgres { + return Err(internal("retention repository requires PostgreSQL")); + } + let isolation = txn + .query_one_raw(statement("SHOW transaction_isolation", [])) + .await + .map_err(internal)? + .ok_or_else(|| internal("transaction isolation query returned no row"))? + .try_get_by_index::(0) + .map_err(internal)?; + if isolation != "read committed" { + return Err(internal( + "retention mutations require READ COMMITTED transaction isolation", + )); + } + let savepoint = txn.begin().await.map_err(internal)?; + let result = savepoint + .execute_raw(statement( + "SELECT pg_advisory_xact_lock($1, hashtext(current_schema()))", + [RETENTION_LOCK_KEY.into()], + )) + .await + .map_err(internal); + if let Err(err) = result { + return finish(savepoint, Err(err)).await; + } + Ok(savepoint) +} + +async fn finish( + txn: DatabaseTransaction, + result: Result, +) -> Result { + match result { + Ok(value) => { + txn.commit().await.map_err(internal)?; + Ok(value) + } + Err(err) => { + txn.rollback().await.map_err(internal)?; + Err(err) + } + } +} + +fn statement(sql: &str, values: [sea_orm::Value; N]) -> Statement { + Statement::from_sql_and_values(DbBackend::Postgres, sql, values) +} + +fn validate_id(id: &str) -> Result<(), SnapshotError> { + if id.is_empty() || id.len() > 255 || id.contains('\0') { + return Err(limit_error( + "retention identifiers require 1..=255 bytes without NUL", + )); + } + Ok(()) +} + +fn root_identity(root: &RetentionRoot) -> Result<(String, &'static str), SnapshotError> { + let (id, kind) = match root { + RetentionRoot::Lease(id) => (id, "lease"), + RetentionRoot::Pin(id) => (id, "pin"), + RetentionRoot::Prepare(id) => (id, "prepare"), + }; + validate_id(id)?; + let key = format!("{kind}:{id}"); + validate_id(&key)?; + Ok((key, kind)) +} + +fn internal(error: impl std::fmt::Display) -> SnapshotError { + SnapshotError::new(SnapshotErrorCode::Internal, error.to_string()) +} +fn integrity(message: &str) -> SnapshotError { + SnapshotError::new(SnapshotErrorCode::IntegrityError, message) +} +fn unavailable(message: &str) -> SnapshotError { + SnapshotError::new(SnapshotErrorCode::ObjectUnavailable, message) +} +fn limit_error(message: &str) -> SnapshotError { + SnapshotError::new(SnapshotErrorCode::LimitExceeded, message) +} + +#[cfg(test)] +#[path = "mst2_retention_tests.rs"] +mod tests; diff --git a/src/jupiter/storage/mst2_retention_tests.rs b/src/jupiter/storage/mst2_retention_tests.rs new file mode 100644 index 00000000..a8f7c866 --- /dev/null +++ b/src/jupiter/storage/mst2_retention_tests.rs @@ -0,0 +1,629 @@ +use std::time::Duration; + +use sea_orm::{ + ConnectionTrait, DatabaseConnection, EntityTrait, IsolationLevel, PaginatorTrait, + TransactionTrait, +}; +use sea_orm_migration::MigratorTrait; + +use super::*; +use crate::{ + callisto::{mst2_retention_edge, mst2_retention_root}, + ceres::snapshot::retention::RetainedKind, + jupiter::{migration::Migrator, tests::test_db_connection}, +}; + +fn node(id: &str) -> RetentionNode { + RetentionNode { + id: id.into(), + kind: RetainedKind::Page, + state: NodeState::Live, + bytes: 7, + } +} + +fn edge(parent: &str, child: &str) -> RetentionEdge { + RetentionEdge { + parent: parent.into(), + child: child.into(), + } +} + +async fn fixture() -> (DatabaseConnection, PostgresRetentionRepository) { + let temp = tempfile::TempDir::new().expect("temp directory"); + let db = test_db_connection(temp.path()).await; + Migrator::up(&db, None) + .await + .expect("isolated schema migrations"); + let repository = PostgresRetentionRepository::new(db.clone()); + (db, repository) +} + +#[test] +fn t06b_rejects_cycles_conflicting_identities_and_unbounded_groups() { + let a = node("a"); + let b = node("b"); + assert_eq!( + PreparedGroup::new(&[a.clone(), b], &[edge("a", "b"), edge("b", "a")], &[]) + .err() + .expect("cycle rejected") + .code, + SnapshotErrorCode::IntegrityError + ); + let mut conflicting = a.clone(); + conflicting.bytes += 1; + assert!(PreparedGroup::new(&[a.clone(), conflicting], &[], &[]).is_err()); + assert_eq!( + PreparedGroup::new(&vec![a; MAX_NODES + 1], &[], &[]) + .err() + .expect("batch rejected") + .code, + SnapshotErrorCode::LimitExceeded + ); +} + +#[tokio::test] +async fn t06b_retains_unique_edges_and_independent_roots_once() { + let (db, repository) = fixture().await; + let roots = [ + RetentionRoot::Lease("one".into()), + RetentionRoot::Pin("two".into()), + ]; + let nodes = [node("parent"), node("child")]; + let edges = [edge("parent", "child"), edge("parent", "child")]; + for _ in 0..2 { + repository + .retain_group(&nodes, &edges, &roots) + .await + .expect("idempotent retain"); + } + assert_eq!( + repository + .node("child") + .await + .unwrap() + .unwrap() + .incoming_refs, + 1 + ); + assert_eq!( + mst2_retention_edge::Entity::find() + .count(&db) + .await + .unwrap(), + 1 + ); + assert_eq!( + mst2_retention_root::Entity::find() + .count(&db) + .await + .unwrap(), + 4 + ); + repository.release_root(&roots[0]).await.unwrap(); + repository.release_root(&roots[0]).await.unwrap(); + assert_eq!( + mst2_retention_root::Entity::find() + .count(&db) + .await + .unwrap(), + 2 + ); + assert_eq!( + repository + .mark_deleting("gc-parent", "parent") + .await + .unwrap(), + GcClaim::Unavailable + ); + repository.release_root(&roots[1]).await.unwrap(); + assert_eq!( + repository + .mark_deleting("gc-parent", "parent") + .await + .unwrap(), + GcClaim::Marked + ); + assert_eq!( + repository.mark_deleting("gc-child", "child").await.unwrap(), + GcClaim::Unavailable + ); +} + +#[tokio::test] +async fn t06b_group_failure_rolls_back_savepoint_even_if_caller_commits() { + let (db, repository) = fixture().await; + repository + .retain_group(&[node("old")], &[], &[]) + .await + .unwrap(); + assert_eq!( + repository.mark_deleting("old-gc", "old").await.unwrap(), + GcClaim::Marked + ); + let txn = db.begin().await.unwrap(); + let error = PostgresRetentionRepository::retain_group_in_txn( + &txn, + &[node("new")], + &[edge("new", "old")], + &[RetentionRoot::Lease("partial".into())], + ) + .await + .expect_err("DELETING child rejects whole group"); + assert_eq!(error.code, SnapshotErrorCode::ObjectUnavailable); + txn.commit().await.unwrap(); + assert!(repository.node("new").await.unwrap().is_none()); + assert_eq!( + repository.node("old").await.unwrap().unwrap().state, + "DELETING" + ); + assert_eq!( + mst2_retention_root::Entity::find() + .count(&db) + .await + .unwrap(), + 0 + ); + assert_eq!( + mst2_retention_edge::Entity::find() + .count(&db) + .await + .unwrap(), + 0 + ); +} + +#[tokio::test] +async fn t06b_publication_rollback_removes_retention_acquisition() { + let (db, repository) = fixture().await; + let txn = db.begin().await.unwrap(); + PostgresRetentionRepository::retain_group_in_txn( + &txn, + &[node("publication")], + &[], + &[RetentionRoot::Prepare("writer".into())], + ) + .await + .unwrap(); + txn.rollback().await.unwrap(); + assert!(repository.node("publication").await.unwrap().is_none()); + assert_eq!( + mst2_retention_root::Entity::find() + .count(&db) + .await + .unwrap(), + 0 + ); +} + +#[tokio::test] +async fn t06b_root_release_rollback_preserves_coverage() { + let (db, repository) = fixture().await; + let root = RetentionRoot::Lease("reader".into()); + repository + .retain_group(&[node("page")], &[], std::slice::from_ref(&root)) + .await + .unwrap(); + let txn = db.begin().await.unwrap(); + PostgresRetentionRepository::release_root_in_txn(&txn, &root) + .await + .unwrap(); + txn.rollback().await.unwrap(); + assert_eq!( + repository.mark_deleting("page-op", "page").await.unwrap(), + GcClaim::Unavailable + ); + assert_eq!( + mst2_retention_root::Entity::find() + .count(&db) + .await + .unwrap(), + 1 + ); +} + +#[tokio::test] +async fn t06b_rejects_snapshot_isolation_without_mutating_outer_transaction() { + let (db, repository) = fixture().await; + let txn = db + .begin_with_config(Some(IsolationLevel::RepeatableRead), None) + .await + .unwrap(); + let error = PostgresRetentionRepository::retain_group_in_txn( + &txn, + &[node("page")], + &[], + &[RetentionRoot::Lease("reader".into())], + ) + .await + .expect_err("old MVCC snapshots cannot safely follow an advisory lock"); + assert_eq!(error.code, SnapshotErrorCode::Internal); + assert!(error.message.contains("READ COMMITTED")); + txn.commit().await.unwrap(); + assert!(repository.node("page").await.unwrap().is_none()); + assert_eq!( + mst2_retention_root::Entity::find() + .count(&db) + .await + .unwrap(), + 0 + ); +} + +#[tokio::test] +async fn t06b_existing_parent_and_node_definition_are_immutable() { + let (db, repository) = fixture().await; + repository + .retain_group(&[node("parent"), node("child")], &[], &[]) + .await + .unwrap(); + let mut changed = node("parent"); + changed.bytes += 1; + assert!(repository.retain_group(&[changed], &[], &[]).await.is_err()); + assert!( + repository + .retain_group(&[node("parent")], &[edge("parent", "child")], &[]) + .await + .is_err() + ); + assert_eq!(repository.node("parent").await.unwrap().unwrap().bytes, 7); + assert_eq!( + mst2_retention_edge::Entity::find() + .count(&db) + .await + .unwrap(), + 0 + ); +} + +#[tokio::test] +async fn t06b_gc_replay_subtracts_edges_once_and_tombstones_removed_identity() { + let (db, repository) = fixture().await; + repository + .retain_group( + &[node("parent"), node("child")], + &[edge("parent", "child")], + &[], + ) + .await + .unwrap(); + assert_eq!( + repository + .mark_deleting("parent-op", "parent") + .await + .unwrap(), + GcClaim::Marked + ); + let restarted = PostgresRetentionRepository::new(db.clone()); + let pending = restarted.pending_gc(10).await.unwrap(); + assert_eq!(pending.len(), 1); + assert_eq!(pending[0].operation_id, "parent-op"); + assert_eq!( + restarted + .mark_deleting("parent-op", "parent") + .await + .unwrap(), + GcClaim::Pending + ); + assert_eq!( + restarted + .mark_deleting("another-op", "parent") + .await + .unwrap(), + GcClaim::Unavailable + ); + assert!( + restarted + .mark_deleting("parent-op", "different") + .await + .is_err() + ); + + // Simulated crash/rollback during bookkeeping: the PENDING intent and + // child's reference survive and can be replayed after reaper replacement. + let txn = db.begin().await.unwrap(); + let provisional = PostgresRetentionRepository::complete_gc_in_txn(&txn, "parent-op") + .await + .unwrap(); + assert_eq!(provisional.zero_reference_children, vec!["child"]); + txn.rollback().await.unwrap(); + assert_eq!( + restarted + .node("child") + .await + .unwrap() + .unwrap() + .incoming_refs, + 1 + ); + assert_eq!(restarted.pending_gc(10).await.unwrap().len(), 1); + assert_eq!( + restarted.node("parent").await.unwrap().unwrap().state, + "DELETING" + ); + + let completed = restarted.complete_gc("parent-op").await.unwrap(); + assert!(!completed.replayed); + assert_eq!(completed.zero_reference_children, vec!["child"]); + assert!(restarted.node("parent").await.unwrap().is_none()); + assert_eq!( + restarted + .node("child") + .await + .unwrap() + .unwrap() + .incoming_refs, + 0 + ); + assert!(restarted.complete_gc("parent-op").await.unwrap().replayed); + assert_eq!( + restarted + .mark_deleting("parent-op", "parent") + .await + .unwrap(), + GcClaim::Applied + ); + assert!(restarted.pending_gc(10).await.unwrap().is_empty()); + let receipt = mst2_retention_gc_op::Entity::find_by_id("parent-op".to_owned()) + .one(&db) + .await + .unwrap() + .unwrap(); + assert_eq!(receipt.attempts, 1); + assert!(receipt.completed_at.is_some()); + + // Old pending workers must not be able to delete same-id reconstructed + // bytes. Reconstruction requires a future generation/fencing protocol. + assert_eq!( + restarted + .retain_group(&[node("parent")], &[], &[]) + .await + .unwrap_err() + .code, + SnapshotErrorCode::ObjectUnavailable + ); + assert!(restarted.complete_gc("parent-op").await.unwrap().replayed); + assert!(restarted.node("parent").await.unwrap().is_none()); +} + +#[tokio::test] +async fn t06b_counter_audit_stops_gc_and_preserves_pending_edges() { + let (db, repository) = fixture().await; + repository + .retain_group( + &[node("parent"), node("child")], + &[edge("parent", "child")], + &[], + ) + .await + .unwrap(); + db.execute_unprepared( + "UPDATE mst2_retention_node SET incoming_refs = 0 WHERE node_id = 'child'", + ) + .await + .unwrap(); + assert_eq!( + repository + .mark_deleting("child-op", "child") + .await + .unwrap_err() + .code, + SnapshotErrorCode::IntegrityError + ); + assert_eq!( + repository + .mark_deleting("parent-op", "parent") + .await + .unwrap(), + GcClaim::Marked + ); + assert_eq!( + repository.complete_gc("parent-op").await.unwrap_err().code, + SnapshotErrorCode::IntegrityError + ); + assert_eq!( + mst2_retention_edge::Entity::find() + .count(&db) + .await + .unwrap(), + 1 + ); + assert_eq!(repository.pending_gc(10).await.unwrap().len(), 1); + assert_eq!( + repository.node("child").await.unwrap().unwrap().state, + "LIVE" + ); + db.execute_unprepared( + "UPDATE mst2_retention_node SET incoming_refs = 1 WHERE node_id = 'child'", + ) + .await + .unwrap(); + repository.complete_gc("parent-op").await.unwrap(); + assert_eq!( + repository + .node("child") + .await + .unwrap() + .unwrap() + .incoming_refs, + 0 + ); +} + +#[tokio::test] +async fn t06b_root_acquisition_wins_before_gc_cas() { + let (db, repository) = fixture().await; + repository + .retain_group(&[node("page")], &[], &[]) + .await + .unwrap(); + let root_txn = db.begin().await.unwrap(); + PostgresRetentionRepository::retain_group_in_txn( + &root_txn, + &[node("page")], + &[], + &[RetentionRoot::Lease("winner".into())], + ) + .await + .unwrap(); + let (started_tx, started_rx) = tokio::sync::oneshot::channel(); + let gc_repository = repository.clone(); + let mut gc = tokio::spawn(async move { + started_tx.send(()).unwrap(); + gc_repository.mark_deleting("gc-op", "page").await + }); + started_rx.await.unwrap(); + assert!( + tokio::time::timeout(Duration::from_millis(50), &mut gc) + .await + .is_err() + ); + root_txn.commit().await.unwrap(); + assert_eq!( + tokio::time::timeout(Duration::from_secs(5), gc) + .await + .unwrap() + .unwrap() + .unwrap(), + GcClaim::Unavailable + ); + assert_eq!( + repository.node("page").await.unwrap().unwrap().state, + "LIVE" + ); + assert!(repository.pending_gc(10).await.unwrap().is_empty()); +} + +#[tokio::test] +async fn t06b_gc_cas_wins_and_new_root_cannot_reference_deleting_node() { + let (db, repository) = fixture().await; + repository + .retain_group(&[node("page")], &[], &[]) + .await + .unwrap(); + let gc_txn = db.begin().await.unwrap(); + assert_eq!( + PostgresRetentionRepository::mark_deleting_in_txn(&gc_txn, "gc-op", "page") + .await + .unwrap(), + GcClaim::Marked + ); + let (started_tx, started_rx) = tokio::sync::oneshot::channel(); + let acquire_repository = repository.clone(); + let mut acquire = tokio::spawn(async move { + started_tx.send(()).unwrap(); + acquire_repository + .retain_group(&[node("page")], &[], &[RetentionRoot::Lease("late".into())]) + .await + }); + started_rx.await.unwrap(); + assert!( + tokio::time::timeout(Duration::from_millis(50), &mut acquire) + .await + .is_err() + ); + gc_txn.commit().await.unwrap(); + assert_eq!( + tokio::time::timeout(Duration::from_secs(5), acquire) + .await + .unwrap() + .unwrap() + .unwrap_err() + .code, + SnapshotErrorCode::ObjectUnavailable + ); + assert_eq!( + mst2_retention_root::Entity::find() + .count(&db) + .await + .unwrap(), + 0 + ); + assert_eq!(repository.pending_gc(10).await.unwrap().len(), 1); +} + +#[tokio::test] +async fn t06b_forward_migration_backfills_old_graph_and_enforces_constraints() { + let temp = tempfile::TempDir::new().unwrap(); + let db = test_db_connection(temp.path()).await; + let at = Migrator::migrations() + .iter() + .position(|migration| migration.name() == "m20261005_000200_harden_mst2_retention_graph") + .expect("hardening migration registered"); + Migrator::up(&db, Some(at.try_into().unwrap())) + .await + .unwrap(); + db.execute_unprepared( + "INSERT INTO mst2_retention_node (node_id, kind, state, bytes, created_at) \ + VALUES ('parent', 'page', 'LIVE', 7, now()), ('child', 'page', 'LIVE', 7, now()); \ + INSERT INTO mst2_retention_edge (parent_id, child_id, created_at) VALUES ('parent', 'child', now())", + ) + .await + .unwrap(); + Migrator::up(&db, None).await.unwrap(); + let repository = PostgresRetentionRepository::new(db.clone()); + assert_eq!( + repository + .node("child") + .await + .unwrap() + .unwrap() + .incoming_refs, + 1 + ); + assert!( + db.execute_unprepared( + "INSERT INTO mst2_retention_edge (parent_id, child_id, created_at) VALUES ('missing', 'child', now())" + ) + .await + .is_err() + ); + assert!( + db.execute_unprepared( + "UPDATE mst2_retention_node SET incoming_refs = -1 WHERE node_id = 'child'" + ) + .await + .is_err() + ); + let completion_error = db + .execute_unprepared( + "INSERT INTO mst2_retention_gc_op (operation_id, node_id, operation, state, attempts, created_at) \ + VALUES ('bad', 'child', 'REMOVE', 'APPLIED', 0, now())", + ) + .await + .expect_err("APPLIED receipt without completed_at must violate its completion constraint"); + assert!( + completion_error + .to_string() + .contains("mst2_retention_gc_op_completion_check"), + "unexpected constraint failure: {completion_error}" + ); +} + +#[tokio::test] +async fn t06b_forward_migration_refuses_existing_cycles() { + let temp = tempfile::TempDir::new().unwrap(); + let db = test_db_connection(temp.path()).await; + let at = Migrator::migrations() + .iter() + .position(|migration| migration.name() == "m20261005_000200_harden_mst2_retention_graph") + .unwrap(); + Migrator::up(&db, Some(at.try_into().unwrap())) + .await + .unwrap(); + db.execute_unprepared( + "INSERT INTO mst2_retention_node (node_id, kind, state, bytes, created_at) \ + VALUES ('a', 'page', 'LIVE', 7, now()), ('b', 'page', 'LIVE', 7, now()); \ + INSERT INTO mst2_retention_edge (parent_id, child_id, created_at) VALUES ('a', 'b', now()), ('b', 'a', now())", + ) + .await + .unwrap(); + assert!(Migrator::up(&db, None).await.is_err()); + assert_eq!( + mst2_retention_edge::Entity::find() + .count(&db) + .await + .unwrap(), + 2 + ); +} diff --git a/src/jupiter/storage/native_metadata_install.rs b/src/jupiter/storage/native_metadata_install.rs new file mode 100644 index 00000000..7f7d80ed --- /dev/null +++ b/src/jupiter/storage/native_metadata_install.rs @@ -0,0 +1,802 @@ +//! Immutable metadata payload CAS and durable native preparation receipts. +//! No runtime, publication, lease or physical collector is enabled here. + +use std::{ + collections::{BTreeMap, BTreeSet}, + time::Duration, +}; + +use mst2_codec::metapage::{HEADER_LEN, PAGE_MAX_BYTES, Page, page_id}; +use sea_orm::{ + ColumnTrait, ConnectionTrait, DatabaseConnection, DatabaseTransaction, DbBackend, EntityTrait, + IsolationLevel, QueryFilter, QuerySelect, Statement, TransactionTrait, +}; +use serde_json::json; + +use super::mst2_retention::{PostgresRetentionRepository, RETENTION_LOCK_KEY}; +use crate::{ + callisto::{ + mst2_metadata_payload, mst2_metadata_prepare, mst2_metadata_prepare_page, + mst2_retention_edge, mst2_retention_node, mst2_retention_root, + }, + ceres::snapshot::{ + error::{SnapshotError, SnapshotErrorCode}, + metadata_install::MetadataInstallPlan, + pages::PreparedNativeMetadataRetention, + retention::RetentionRoot, + retention_dag::{ + MetadataDagCandidate, MetadataDagLimits, MetadataPagePayload, ValidatedMetadataDag, + }, + }, +}; + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum MetadataCommitPhase { + Intent, + Payload, + Finalize, +} + +#[derive(Debug, thiserror::Error)] +pub enum MetadataInstallError { + #[error(transparent)] + Rejected(#[from] SnapshotError), + #[error("native metadata {phase:?} commit outcome is unknown for operation {operation_id}")] + CommitUncertain { + operation_id: String, + manifest_digest: [u8; 32], + phase: MetadataCommitPhase, + }, +} + +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct MetadataPrepareIntent { + prepare_id: String, + operation_id: String, + manifest_digest: [u8; 32], +} + +impl MetadataPrepareIntent { + pub fn prepare_id(&self) -> &str { + &self.prepare_id + } + pub fn operation_id(&self) -> &str { + &self.operation_id + } + pub fn manifest_digest(&self) -> [u8; 32] { + self.manifest_digest + } +} + +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct PreparedMetadataReceipt { + intent: MetadataPrepareIntent, + metadata_root: [u8; 32], + payload_bytes: u64, +} + +impl PreparedMetadataReceipt { + pub fn intent(&self) -> &MetadataPrepareIntent { + &self.intent + } + pub fn metadata_root(&self) -> [u8; 32] { + self.metadata_root + } + pub fn payload_bytes(&self) -> u64 { + self.payload_bytes + } +} + +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum MetadataPrepareObservation { + Absent, + Preparing(MetadataPrepareIntent), + Committed(PreparedMetadataReceipt), +} + +struct StoredPlan { + record: mst2_metadata_prepare::Model, + plan: MetadataInstallPlan, +} + +impl StoredPlan { + fn intent(&self) -> Result { + Ok(MetadataPrepareIntent { + prepare_id: self.record.prepare_id.clone(), + operation_id: self.record.operation_id.clone(), + manifest_digest: self + .record + .manifest_digest + .as_slice() + .try_into() + .map_err(|_| integrity("invalid stored metadata manifest digest"))?, + }) + } + fn receipt(&self) -> Result { + Ok(PreparedMetadataReceipt { + intent: self.intent()?, + metadata_root: self.plan.root, + payload_bytes: self.plan.total_bytes, + }) + } +} + +#[derive(Clone)] +pub struct PostgresMetadataInstallRepository { + connection: DatabaseConnection, + barrier_timeout: Duration, + storage_scope: PrimaryStorageScope, +} + +#[derive(Debug, Clone, PartialEq, Eq)] +struct PrimaryStorageScope { + storage_uuid: String, + database: String, + database_oid: i64, + schema: String, + schema_oid: i64, + server_address: Option, + server_port: Option, +} + +impl PostgresMetadataInstallRepository { + /// The connection must target the deployment's primary PostgreSQL database. + pub async fn new(connection: DatabaseConnection) -> Result { + let storage_scope = read_storage_scope(&connection).await?; + Ok(Self { + connection, + barrier_timeout: Duration::from_secs(5), + storage_scope, + }) + } + + pub async fn begin_intent( + &self, + operation_id: &str, + prepared: &PreparedNativeMetadataRetention, + ) -> Result { + validate_operation_id(operation_id)?; + let plan = prepared.install_plan()?; + let digest = plan.digest()?; + let txn = self.transaction().await?; + let result = async { + self.barrier(&txn).await?; + if let Some(stored) = load_plan(&txn, operation_id, &digest).await? { + return stored.intent(); + } + let id = uuid::Uuid::new_v4().to_string(); + let identity = &plan.identity; + txn.execute_raw(statement( + "INSERT INTO mst2_metadata_prepare (prepare_id, operation_id, manifest_digest, canonical_plan, + source_domain, tagged_root_tree_oid, scope, schema_version, metadata_codec, materialization_policy, + fs_semantics, access_projection, verification_revision, projection_revision, metadata_root, + node_count, edge_count, total_bytes, state) + VALUES ($1,$2,$3,$4,$5,$6,$7,$8,$9,$10,$11,$12,$13,$14,$15,$16,$17,$18,'PREPARING')", + [id.clone().into(), operation_id.into(), digest.to_vec().into(), plan.encode()?.into(), + identity.source_domain.clone().into(), identity.tagged_root_tree_oid.clone().into(), identity.scope.clone().into(), + (identity.schema_version as i16).into(), (identity.metadata_codec as i16).into(), + (identity.materialization_policy as i16).into(), (identity.fs_semantics as i16).into(), + (identity.access_projection as i16).into(), identity.verification_revision.into(), + (identity.projection_revision as i16).into(), plan.root.to_vec().into(), (plan.pages.len() as i32).into(), + (plan.edges.len() as i32).into(), (plan.total_bytes as i64).into()], + )).await.map_err(internal)?; + let pages: Vec<_> = plan.pages.iter().map(|(id,size)| json!({"page_id":hex::encode(id),"size":size})).collect(); + txn.execute_raw(statement( + "INSERT INTO mst2_metadata_prepare_page (prepare_id,page_id,expected_size) + SELECT $1,decode(p.page_id,'hex'),p.size FROM jsonb_to_recordset($2::jsonb) AS p(page_id text,size integer)", + [id.clone().into(), serde_json::to_string(&pages).map_err(internal)?.into()], + )).await.map_err(internal)?; + Ok(MetadataPrepareIntent { prepare_id:id, operation_id:operation_id.into(), manifest_digest:digest }) + }.await; + commit( + txn, + result, + operation_id, + digest, + MetadataCommitPhase::Intent, + ) + .await + } + + pub async fn install_page( + &self, + intent: &MetadataPrepareIntent, + payload: &MetadataPagePayload, + ) -> Result<(), MetadataInstallError> { + validate_payload(payload)?; + let txn = self.transaction().await?; + let result = async { + self.barrier(&txn).await?; + let stored = require_plan(&txn, intent).await?; + if stored.plan.pages.get(&payload.id) != Some(&payload.size) { + return Err(integrity( + "metadata payload is not a member of this fixed installation", + )); + } + txn.execute_raw(statement( + "INSERT INTO mst2_metadata_payload(page_id,metadata_codec,byte_size,payload) + VALUES ($1,$2,$3,$4) ON CONFLICT(page_id) DO NOTHING", + [ + payload.id.to_vec().into(), + (stored.plan.identity.metadata_codec as i16).into(), + (payload.size as i32).into(), + payload.bytes.clone().into(), + ], + )) + .await + .map_err(internal)?; + let existing = mst2_metadata_payload::Entity::find_by_id(payload.id.to_vec()) + .one(&txn) + .await + .map_err(internal)? + .ok_or_else(|| integrity("installed metadata payload disappeared"))?; + if existing.payload != payload.bytes + || existing.byte_size as u64 != payload.size + || existing.metadata_codec != stored.plan.identity.metadata_codec as i16 + { + return Err(integrity( + "immutable metadata payload identity conflicts with stored bytes", + )); + } + Ok(()) + } + .await; + commit( + txn, + result, + &intent.operation_id, + intent.manifest_digest, + MetadataCommitPhase::Payload, + ) + .await + } + + /// Verify all stored bytes outside the graph transaction. Immutable payloads + /// cannot change; LIVE state and coverage are checked again under the lock. + pub async fn load_installed_dag( + &self, + intent: &MetadataPrepareIntent, + ) -> Result { + let stored = require_plan(&self.connection, intent).await?; + load_installed_dag(&self.connection, &stored).await + } + + pub async fn finalize( + &self, + intent: &MetadataPrepareIntent, + ) -> Result { + let dag = self.load_installed_dag(intent).await?; + let txn = self.transaction().await?; + let result = self.finalize_in_txn(&txn, intent, &dag).await; + commit( + txn, + result, + &intent.operation_id, + intent.manifest_digest, + MetadataCommitPhase::Finalize, + ) + .await + } + + // This provisional result is private: only finalize's successful outer + // commit or the recovery barrier can issue a durable receipt to a caller. + async fn finalize_in_txn( + &self, + txn: &DatabaseTransaction, + intent: &MetadataPrepareIntent, + dag: &ValidatedMetadataDag, + ) -> Result { + self.barrier(txn).await?; + let stored = require_plan(txn, intent).await?; + let expected: BTreeSet<_> = dag + .payloads() + .iter() + .map(|page| (page.id, page.size)) + .collect(); + if dag.root() != stored.plan.root + || expected + != stored + .plan + .pages + .iter() + .map(|(id, size)| (*id, *size)) + .collect() + || dag + .edges() + .iter() + .map(|edge| (edge.parent.clone(), edge.child.clone())) + .collect::>() + != stored + .plan + .edges + .iter() + .map(|(parent, child)| (node_id(parent), node_id(child))) + .collect() + { + return Err(integrity( + "validated installed DAG differs from durable preparation plan", + )); + } + check_payload_coverage(txn, &stored).await?; + if stored.record.state == "COMMITTED" { + verify_graph(txn, &stored).await?; + return stored.receipt(); + } + PostgresRetentionRepository::retain_group_in_txn( + txn, + dag.nodes(), + dag.edges(), + &[RetentionRoot::Prepare(intent.prepare_id.clone())], + ) + .await?; + verify_graph(txn, &stored).await?; + let result = txn + .execute_raw(statement( + "UPDATE mst2_metadata_prepare SET state='COMMITTED',committed_at=now() + WHERE prepare_id=$1 AND manifest_digest=$2 AND state='PREPARING'", + [ + intent.prepare_id.clone().into(), + intent.manifest_digest.to_vec().into(), + ], + )) + .await + .map_err(internal)?; + if result.rows_affected() != 1 { + return Err(integrity("metadata prepare state changed during finalize")); + } + stored.receipt() + } + + /// Call with a fresh connection to the same primary after a commit error. + /// Lock acquisition is the completion barrier; absence before it proves nothing. + pub async fn inspect_prepare( + &self, + fresh_primary: &DatabaseConnection, + operation_id: &str, + digest: [u8; 32], + phase: MetadataCommitPhase, + ) -> Result { + validate_operation_id(operation_id)?; + let txn = fresh_primary + .begin_with_config(Some(IsolationLevel::ReadCommitted), None) + .await + .map_err(|_| uncertain(operation_id, digest, phase))?; + if self.barrier(&txn).await.is_err() { + let _ = txn.rollback().await; + return Err(uncertain(operation_id, digest, phase)); + } + let result = async { + match load_plan(&txn, operation_id, &digest).await? { + None => Ok(MetadataPrepareObservation::Absent), + Some(stored) if stored.record.state == "COMMITTED" => { + // Recovery rechecks the bounded bytes/DAG under the barrier, + // so corruption cannot turn a stored state into a receipt. + // This may read 64 MiB; normal finalization verifies outside + // its short graph transaction instead. + load_installed_dag(&txn, &stored).await?; + check_payload_coverage(&txn, &stored).await?; + verify_graph(&txn, &stored).await?; + Ok(MetadataPrepareObservation::Committed(stored.receipt()?)) + } + Some(stored) => Ok(MetadataPrepareObservation::Preparing(stored.intent()?)), + } + } + .await; + txn.rollback() + .await + .map_err(|_| uncertain(operation_id, digest, phase))?; + result.map_err(|error: SnapshotError| { + if error.code == SnapshotErrorCode::Internal { + uncertain(operation_id, digest, phase) + } else { + MetadataInstallError::Rejected(error) + } + }) + } + + async fn transaction(&self) -> Result { + if self.connection.get_database_backend() != DbBackend::Postgres { + return Err(internal("metadata installation requires PostgreSQL")); + } + self.connection + .begin_with_config(Some(IsolationLevel::ReadCommitted), None) + .await + .map_err(internal) + } + + async fn barrier(&self, txn: &DatabaseTransaction) -> Result<(), SnapshotError> { + if txn.get_database_backend() != DbBackend::Postgres { + return Err(internal("metadata installation requires PostgreSQL")); + } + if read_storage_scope(txn).await? != self.storage_scope { + return Err(internal( + "metadata recovery connection is outside the captured primary storage scope", + )); + } + let isolation = txn + .query_one_raw(statement("SHOW transaction_isolation", [])) + .await + .map_err(internal)? + .ok_or_else(|| internal("missing transaction isolation"))? + .try_get_by_index::(0) + .map_err(internal)?; + if isolation != "read committed" { + return Err(internal("metadata installation requires READ COMMITTED")); + } + let recovery = txn + .query_one_raw(statement("SELECT pg_is_in_recovery()", [])) + .await + .map_err(internal)? + .ok_or_else(|| internal("missing primary status"))? + .try_get_by_index::(0) + .map_err(internal)?; + if recovery { + return Err(internal( + "metadata installation recovery requires the primary", + )); + } + let timeout = format!("{}ms", self.barrier_timeout.as_millis().clamp(1, 5000)); + txn.execute_raw(statement( + "SELECT set_config('lock_timeout',$1,true)", + [timeout.into()], + )) + .await + .map_err(internal)?; + txn.execute_raw(statement( + "SELECT pg_advisory_xact_lock($1,hashtext(current_schema()))", + [RETENTION_LOCK_KEY.into()], + )) + .await + .map_err(internal)?; + Ok(()) + } +} + +async fn read_storage_scope( + connection: &C, +) -> Result { + if connection.get_database_backend() != DbBackend::Postgres { + return Err(internal("metadata storage scope requires PostgreSQL")); + } + let row = connection + .query_one_raw(statement( + "SELECT s.storage_uuid, current_database() AS database, d.oid::bigint AS database_oid, + current_schema() AS schema, n.oid::bigint AS schema_oid, + inet_server_addr()::text AS server_address, inet_server_port() AS server_port, + pg_is_in_recovery() AS replica FROM mst2_metadata_storage_scope s + JOIN pg_catalog.pg_database d ON d.datname=current_database() + JOIN pg_catalog.pg_namespace n ON n.nspname=current_schema() WHERE s.singleton=1", + [], + )) + .await + .map_err(internal)? + .ok_or_else(|| internal("metadata primary storage scope is missing"))?; + if row.try_get::("", "replica").map_err(internal)? { + return Err(internal("metadata storage scope requires the primary")); + } + let storage_uuid: String = row.try_get("", "storage_uuid").map_err(internal)?; + let id = uuid::Uuid::parse_str(&storage_uuid) + .map_err(|_| internal("invalid metadata storage UUID"))?; + if id.is_nil() || id.to_string() != storage_uuid { + return Err(internal("noncanonical metadata storage UUID")); + } + Ok(PrimaryStorageScope { + storage_uuid, + database: row.try_get("", "database").map_err(internal)?, + database_oid: row.try_get("", "database_oid").map_err(internal)?, + schema: row.try_get("", "schema").map_err(internal)?, + schema_oid: row.try_get("", "schema_oid").map_err(internal)?, + server_address: row.try_get("", "server_address").map_err(internal)?, + server_port: row.try_get("", "server_port").map_err(internal)?, + }) +} + +async fn load_plan( + connection: &C, + operation_id: &str, + digest: &[u8; 32], +) -> Result, SnapshotError> { + let Some(record) = mst2_metadata_prepare::Entity::find() + .filter(mst2_metadata_prepare::Column::OperationId.eq(operation_id)) + .one(connection) + .await + .map_err(internal)? + else { + return Ok(None); + }; + if record.manifest_digest.as_slice() != digest { + return Err(SnapshotError::new( + SnapshotErrorCode::Conflict, + "metadata operation ID is bound to a different manifest", + )); + } + let plan = MetadataInstallPlan::decode(&record.canonical_plan, digest)?; + let identity = &plan.identity; + if record.source_domain != identity.source_domain + || record.tagged_root_tree_oid != identity.tagged_root_tree_oid + || record.scope != identity.scope + || record.schema_version != identity.schema_version as i16 + || record.metadata_codec != identity.metadata_codec as i16 + || record.materialization_policy != identity.materialization_policy as i16 + || record.fs_semantics != identity.fs_semantics as i16 + || record.access_projection != identity.access_projection as i16 + || record.verification_revision != identity.verification_revision + || record.projection_revision != identity.projection_revision as i16 + || record.metadata_root.as_slice() != plan.root + || record.node_count as usize != plan.pages.len() + || record.edge_count as usize != plan.edges.len() + || record.total_bytes as u64 != plan.total_bytes + || !["PREPARING", "COMMITTED"].contains(&record.state.as_str()) + || (record.state == "COMMITTED") != record.committed_at.is_some() + { + return Err(integrity( + "stored metadata preparation fields disagree with their canonical plan", + )); + } + let id = uuid::Uuid::parse_str(&record.prepare_id) + .map_err(|_| integrity("invalid stored preparation identity"))?; + if id.is_nil() || id.to_string() != record.prepare_id { + return Err(integrity("noncanonical stored preparation identity")); + } + let rows = mst2_metadata_prepare_page::Entity::find() + .filter(mst2_metadata_prepare_page::Column::PrepareId.eq(&record.prepare_id)) + .limit((MetadataDagLimits::default().nodes + 1) as u64) + .all(connection) + .await + .map_err(internal)?; + let mut coverage = BTreeSet::new(); + for row in rows { + let page: [u8; 32] = row + .page_id + .as_slice() + .try_into() + .map_err(|_| integrity("invalid stored preparation page ID"))?; + coverage.insert((page, row.expected_size as u64)); + } + if coverage != plan.pages.iter().map(|(id, size)| (*id, *size)).collect() { + return Err(integrity( + "stored metadata preparation coverage differs from its canonical plan", + )); + } + Ok(Some(StoredPlan { record, plan })) +} + +async fn require_plan( + connection: &C, + intent: &MetadataPrepareIntent, +) -> Result { + validate_operation_id(&intent.operation_id)?; + let stored = load_plan(connection, &intent.operation_id, &intent.manifest_digest) + .await? + .ok_or_else(|| unavailable("metadata preparation intent is not durable"))?; + if stored.record.prepare_id != intent.prepare_id { + return Err(integrity("metadata intent identity mismatch")); + } + Ok(stored) +} + +async fn load_installed_dag( + connection: &C, + stored: &StoredPlan, +) -> Result { + let rows = mst2_metadata_payload::Entity::find() + .filter( + mst2_metadata_payload::Column::PageId + .is_in(stored.plan.pages.keys().map(|id| id.to_vec())), + ) + .all(connection) + .await + .map_err(internal)?; + let mut pages = Vec::with_capacity(rows.len()); + for row in rows { + let id: [u8; 32] = row + .page_id + .as_slice() + .try_into() + .map_err(|_| integrity("invalid stored page ID"))?; + if row.metadata_codec != stored.plan.identity.metadata_codec as i16 + || stored.plan.pages.get(&id) != Some(&(row.byte_size as u64)) + { + return Err(integrity( + "stored page profile or size disagrees with its manifest", + )); + } + let page = MetadataPagePayload { + id, + size: row.byte_size as u64, + bytes: row.payload, + }; + validate_payload(&page)?; + pages.push(page); + } + if pages.len() != stored.plan.pages.len() { + return Err(unavailable( + "native metadata installation has missing payloads", + )); + } + ValidatedMetadataDag::validate( + MetadataDagCandidate { + metadata_codec: stored.plan.identity.metadata_codec, + root: stored.plan.root, + pages, + edges: stored.plan.edges.iter().copied().collect(), + }, + MetadataDagLimits::default(), + ) +} + +async fn check_payload_coverage( + connection: &C, + stored: &StoredPlan, +) -> Result<(), SnapshotError> { + let missing=connection.query_one_raw(statement( + "SELECT p.page_id FROM mst2_metadata_prepare_page p LEFT JOIN mst2_metadata_payload b ON b.page_id=p.page_id + WHERE p.prepare_id=$1 AND (b.page_id IS NULL OR b.byte_size<>p.expected_size OR b.metadata_codec<>$2 + OR octet_length(b.payload)<>p.expected_size) LIMIT 1", + [stored.record.prepare_id.clone().into(),stored.record.metadata_codec.into()], + )).await.map_err(internal)?; + if missing.is_some() { + return Err(unavailable( + "durable metadata payload coverage is incomplete", + )); + } + Ok(()) +} + +async fn verify_graph( + connection: &C, + stored: &StoredPlan, +) -> Result<(), SnapshotError> { + let sizes: BTreeMap<_, _> = stored + .plan + .pages + .iter() + .map(|(id, size)| (node_id(id), *size)) + .collect(); + let ids: Vec<_> = sizes.keys().cloned().collect(); + let nodes = mst2_retention_node::Entity::find() + .filter(mst2_retention_node::Column::NodeId.is_in(ids.clone())) + .all(connection) + .await + .map_err(internal)?; + if nodes.len() != ids.len() { + return Err(unavailable("prepared metadata graph has missing nodes")); + } + for node in nodes { + if node.state != "LIVE" || node.kind != "page" { + return Err(unavailable("prepared metadata graph is not LIVE")); + } + if sizes.get(&node.node_id) != Some(&(node.bytes as u64)) { + return Err(integrity( + "prepared metadata graph bytes differ from its plan", + )); + } + } + let edges = mst2_retention_edge::Entity::find() + .filter(mst2_retention_edge::Column::ParentId.is_in(ids.clone())) + .limit((MetadataDagLimits::default().edges + 1) as u64) + .all(connection) + .await + .map_err(internal)?; + let actual: BTreeSet<_> = edges + .into_iter() + .map(|edge| (edge.parent_id, edge.child_id)) + .collect(); + let expected = stored + .plan + .edges + .iter() + .map(|(parent, child)| (node_id(parent), node_id(child))) + .collect(); + if actual != expected { + return Err(integrity( + "prepared metadata retention edges differ from its plan", + )); + } + let roots = mst2_retention_root::Entity::find() + .filter( + mst2_retention_root::Column::RootKey + .eq(format!("prepare:{}", stored.record.prepare_id)), + ) + .limit((MetadataDagLimits::default().nodes + 1) as u64) + .all(connection) + .await + .map_err(internal)?; + if roots.iter().any(|root| root.root_kind != "prepare") + || roots + .into_iter() + .map(|root| root.node_id) + .collect::>() + != ids.iter().cloned().collect() + { + return Err(unavailable("prepared metadata pin coverage is incomplete")); + } + let bad=connection.query_one_raw(statement( + "SELECT n.node_id FROM mst2_retention_node n WHERE n.node_id IN (SELECT jsonb_array_elements_text($1::jsonb)) + AND n.incoming_refs<>(SELECT count(*) FROM mst2_retention_edge e WHERE e.child_id=n.node_id) LIMIT 1", + [serde_json::to_string(&ids).map_err(internal)?.into()], + )).await.map_err(internal)?; + if bad.is_some() { + return Err(integrity("prepared metadata graph counter audit failed")); + } + Ok(()) +} + +fn validate_payload(payload: &MetadataPagePayload) -> Result<(), SnapshotError> { + if !(HEADER_LEN..=PAGE_MAX_BYTES).contains(&payload.bytes.len()) { + return Err(SnapshotError::new( + SnapshotErrorCode::LimitExceeded, + "metadata page exceeds protocol length bounds", + )); + } + if payload.size != payload.bytes.len() as u64 || page_id(&payload.bytes) != payload.id { + return Err(SnapshotError::new( + SnapshotErrorCode::DigestMismatch, + "metadata payload size or protocol page ID mismatch", + )); + } + Page::decode(&payload.bytes) + .map_err(|error| integrity(&format!("invalid canonical metadata page: {error}")))?; + Ok(()) +} + +fn validate_operation_id(operation: &str) -> Result<(), SnapshotError> { + if operation.is_empty() || operation.len() > 255 || operation.contains('\0') { + return Err(SnapshotError::new( + SnapshotErrorCode::InvalidRequest, + "metadata operation ID must be 1..=255 UTF8 bytes", + )); + } + Ok(()) +} + +async fn commit( + txn: DatabaseTransaction, + result: Result, + operation: &str, + digest: [u8; 32], + phase: MetadataCommitPhase, +) -> Result { + match result { + Ok(value) => { + txn.commit() + .await + .map_err(|_| uncertain(operation, digest, phase))?; + Ok(value) + } + Err(error) => { + txn.rollback().await.map_err(internal)?; + Err(error.into()) + } + } +} +fn uncertain( + operation: &str, + digest: [u8; 32], + phase: MetadataCommitPhase, +) -> MetadataInstallError { + MetadataInstallError::CommitUncertain { + operation_id: operation.into(), + manifest_digest: digest, + phase, + } +} +fn node_id(id: &[u8; 32]) -> String { + format!("page:sha256:{}", hex::encode(id)) +} +fn statement(sql: &str, values: [sea_orm::Value; N]) -> Statement { + Statement::from_sql_and_values(DbBackend::Postgres, sql, values) +} +fn internal(error: impl std::fmt::Display) -> SnapshotError { + SnapshotError::new(SnapshotErrorCode::Internal, error.to_string()) +} +fn integrity(message: &str) -> SnapshotError { + SnapshotError::new(SnapshotErrorCode::IntegrityError, message) +} +fn unavailable(message: &str) -> SnapshotError { + SnapshotError::new(SnapshotErrorCode::ObjectUnavailable, message) +} + +#[cfg(test)] +#[path = "native_metadata_install_tests.rs"] +mod tests; diff --git a/src/jupiter/storage/native_metadata_install_tests.rs b/src/jupiter/storage/native_metadata_install_tests.rs new file mode 100644 index 00000000..4b47fe38 --- /dev/null +++ b/src/jupiter/storage/native_metadata_install_tests.rs @@ -0,0 +1,1226 @@ +use std::sync::{ + Arc, + atomic::{AtomicBool, Ordering}, +}; + +use mst2_codec::metapage::{Entry, EntryKind}; +use sea_orm::{Database, PaginatorTrait}; +use sea_orm_migration::MigratorTrait; +use tokio::{ + io::{AsyncReadExt, AsyncWriteExt}, + net::{TcpListener, TcpStream}, + sync::{Notify, watch}, + task::{JoinHandle, JoinSet}, +}; + +use super::*; +use crate::{ + ceres::snapshot::retention_dag::MetadataDagBuilder, + jupiter::{ + migration::Migrator, + tests::{TestSchemaGuard, test_db_config}, + }, +}; + +fn prepared(scope: &str) -> PreparedNativeMetadataRetention { + let child_entries = [Entry::file(EntryKind::Regular, b"file", 3, [42; 32])]; + let child = Page::build(&child_entries).unwrap(); + let entries = [ + Entry::dir(b"one", page_id(&child)), + Entry::dir(b"two", page_id(&child)), + ]; + let root = Page::build(&entries).unwrap(); + let mut builder = MetadataDagBuilder::new(MetadataDagLimits::default()); + builder.add_directory(&child, &child_entries).unwrap(); + builder.add_directory(&root, &entries).unwrap(); + PreparedNativeMetadataRetention::test_installation( + Arc::new(builder.finish(page_id(&root)).unwrap()), + scope, + ) +} + +async fn fixture() -> ( + DatabaseConnection, + DatabaseConnection, + TestSchemaGuard, + String, +) { + let temp = tempfile::TempDir::new().unwrap(); + let (config, schema) = test_db_config(temp.path()).await; + let first = Database::connect(config.db_url.clone()).await.unwrap(); + Migrator::up(&first, None).await.unwrap(); + let second = Database::connect(config.db_url.clone()).await.unwrap(); + (first, second, schema, config.db_url) +} + +async fn install( + repository: &PostgresMetadataInstallRepository, + intent: &MetadataPrepareIntent, + prepared: &PreparedNativeMetadataRetention, +) { + for page in prepared.dag().payloads() { + repository.install_page(intent, page).await.unwrap(); + } +} + +fn rejected(error: MetadataInstallError) -> SnapshotErrorCode { + match error { + MetadataInstallError::Rejected(error) => error.code, + _ => panic!("expected definite rejection"), + } +} + +#[tokio::test] +async fn native_metadata_two_primary_connections_replay_one_receipt_without_recounting() { + let (first, second, _schema, _url) = fixture().await; + let a = PostgresMetadataInstallRepository::new(first.clone()) + .await + .unwrap(); + let b = PostgresMetadataInstallRepository::new(second.clone()) + .await + .unwrap(); + let prepared = prepared("/"); + let (ia, ib) = tokio::join!( + a.begin_intent("same", &prepared), + b.begin_intent("same", &prepared) + ); + let ia = ia.unwrap(); + assert_eq!(ia, ib.unwrap()); + for page in prepared.dag().payloads() { + let (left, right) = tokio::join!(a.install_page(&ia, page), b.install_page(&ia, page)); + left.unwrap(); + right.unwrap(); + } + let (ra, rb) = tokio::join!(a.finalize(&ia), b.finalize(&ia)); + assert_eq!(ra.unwrap(), rb.unwrap()); + assert_eq!( + mst2_metadata_prepare::Entity::find() + .count(&first) + .await + .unwrap(), + 1 + ); + assert_eq!( + mst2_metadata_payload::Entity::find() + .count(&first) + .await + .unwrap(), + 2 + ); + assert_eq!( + mst2_retention_edge::Entity::find() + .count(&first) + .await + .unwrap(), + 1 + ); + assert_eq!( + mst2_retention_root::Entity::find() + .count(&first) + .await + .unwrap(), + 2 + ); + let child = prepared.dag().edges()[0].child.clone(); + assert_eq!( + PostgresRetentionRepository::new(first) + .node(&child) + .await + .unwrap() + .unwrap() + .incoming_refs, + 1 + ); +} + +#[tokio::test] +async fn native_metadata_operation_conflict_and_multibyte_byte_limit_reject_without_writes() { + let (first, second, _schema, _url) = fixture().await; + let a = PostgresMetadataInstallRepository::new(first.clone()) + .await + .unwrap(); + let b = PostgresMetadataInstallRepository::new(second) + .await + .unwrap(); + a.begin_intent("bound", &prepared("/")).await.unwrap(); + assert_eq!( + rejected( + b.begin_intent("bound", &prepared("/scope")) + .await + .unwrap_err() + ), + SnapshotErrorCode::Conflict + ); + assert_eq!( + rejected( + a.begin_intent(&"é".repeat(128), &prepared("/")) + .await + .unwrap_err() + ), + SnapshotErrorCode::InvalidRequest + ); + assert_eq!( + mst2_metadata_prepare::Entity::find() + .count(&first) + .await + .unwrap(), + 1 + ); + assert_eq!( + mst2_metadata_payload::Entity::find() + .count(&first) + .await + .unwrap(), + 0 + ); + assert_eq!( + mst2_retention_root::Entity::find() + .count(&first) + .await + .unwrap(), + 0 + ); +} + +#[tokio::test] +async fn native_metadata_protocol_hash_and_immutable_bytes_conflicts_cannot_be_overwritten() { + let (first, _second, _schema, _url) = fixture().await; + let repository = PostgresMetadataInstallRepository::new(first.clone()) + .await + .unwrap(); + let prepared = prepared("/"); + let intent = repository.begin_intent("bytes", &prepared).await.unwrap(); + let original = &prepared.dag().payloads()[0]; + let mut invalid = original.clone(); + invalid.id[0] ^= 1; + assert_eq!( + rejected( + repository + .install_page(&intent, &invalid) + .await + .unwrap_err() + ), + SnapshotErrorCode::DigestMismatch + ); + let mut oversized = original.clone(); + oversized.bytes.resize(PAGE_MAX_BYTES + 1, 0); + oversized.size = oversized.bytes.len() as u64; + oversized.id = page_id(&oversized.bytes); + assert_eq!( + rejected( + repository + .install_page(&intent, &oversized) + .await + .unwrap_err() + ), + SnapshotErrorCode::LimitExceeded + ); + // A corrupt existing row is never replaced by a good upload. + let mut corrupt = original.bytes.clone(); + corrupt[0] ^= 1; + first.execute_raw(statement("INSERT INTO mst2_metadata_payload(page_id,metadata_codec,byte_size,payload) VALUES($1,1,$2,$3)", + [original.id.to_vec().into(),(original.size as i32).into(),corrupt.clone().into()])).await.unwrap(); + assert_eq!( + rejected( + repository + .install_page(&intent, original) + .await + .unwrap_err() + ), + SnapshotErrorCode::IntegrityError + ); + assert_eq!( + mst2_metadata_payload::Entity::find_by_id(original.id.to_vec()) + .one(&first) + .await + .unwrap() + .unwrap() + .payload, + corrupt + ); + assert!( + first + .execute_unprepared("UPDATE mst2_metadata_payload SET payload=payload") + .await + .is_err() + ); + assert!( + first + .execute_unprepared("DELETE FROM mst2_metadata_payload") + .await + .is_err() + ); + assert_eq!( + mst2_retention_root::Entity::find() + .count(&first) + .await + .unwrap(), + 0 + ); +} + +#[tokio::test] +async fn native_metadata_partial_installation_survives_repository_and_connection_restart() { + let (first, second, _schema, _url) = fixture().await; + let repository = PostgresMetadataInstallRepository::new(first.clone()) + .await + .unwrap(); + let prepared = prepared("/"); + let intent = repository.begin_intent("restart", &prepared).await.unwrap(); + repository + .install_page(&intent, &prepared.dag().payloads()[0]) + .await + .unwrap(); + assert_eq!( + mst2_retention_node::Entity::find() + .count(&first) + .await + .unwrap(), + 0 + ); + drop(repository); + first.close().await.unwrap(); + let restarted = PostgresMetadataInstallRepository::new(second.clone()) + .await + .unwrap(); + assert!(matches!( + restarted + .inspect_prepare( + &second, + intent.operation_id(), + intent.manifest_digest(), + MetadataCommitPhase::Payload + ) + .await + .unwrap(), + MetadataPrepareObservation::Preparing(_) + )); + assert_eq!( + restarted + .load_installed_dag(&intent) + .await + .unwrap_err() + .code, + SnapshotErrorCode::ObjectUnavailable + ); + install(&restarted, &intent, &prepared).await; + let receipt = restarted.finalize(&intent).await.unwrap(); + assert_eq!(receipt.metadata_root(), prepared.dag().root()); + assert_eq!(receipt.payload_bytes(), prepared.dag().payload_bytes()); +} + +#[tokio::test] +async fn native_metadata_half_installation_survives_real_process_kill() { + const CHILD_DB: &str = "MEGA_MST2_METADATA_CRASH_CHILD_DB"; + const CHECKPOINT: &str = "MEGA_MST2_METADATA_CRASH_CHECKPOINT"; + const OPERATION: &str = "process-crash"; + if let Ok(url) = std::env::var(CHILD_DB) { + let database = Database::connect(url).await.unwrap(); + let repository = PostgresMetadataInstallRepository::new(database) + .await + .unwrap(); + let prepared = prepared("/"); + let intent = repository.begin_intent(OPERATION, &prepared).await.unwrap(); + repository + .install_page(&intent, &prepared.dag().payloads()[0]) + .await + .unwrap(); + let checkpoint = std::path::PathBuf::from(std::env::var(CHECKPOINT).unwrap()); + let staging = checkpoint.with_extension("staging"); + std::fs::write( + &staging, + format!( + "{}\n{}", + intent.prepare_id(), + hex::encode(intent.manifest_digest()) + ), + ) + .unwrap(); + std::fs::rename(staging, checkpoint).unwrap(); + // Keep repository/connection alive until the parent kills this process. + std::future::pending::<()>().await; + unreachable!(); + } + let (first, second, _schema, url) = fixture().await; + let checkpoint_dir = tempfile::tempdir().unwrap(); + let checkpoint = checkpoint_dir.path().join("partial-installation"); + let full_name = concat!( + module_path!(), + "::native_metadata_half_installation_survives_real_process_kill" + ); + let test_name = full_name.split_once("::").unwrap().1; + let mut worker = tokio::process::Command::new(std::env::current_exe().unwrap()) + .args(["--exact", test_name, "--nocapture"]) + .env(CHILD_DB, url) + .env(CHECKPOINT, &checkpoint) + .stdout(std::process::Stdio::null()) + .stderr(std::process::Stdio::inherit()) + .kill_on_drop(true) + .spawn() + .unwrap(); + let marker = tokio::time::timeout(Duration::from_secs(60), async { + loop { + if checkpoint.exists() { + break std::fs::read_to_string(&checkpoint).unwrap(); + } + if let Some(status) = worker.try_wait().unwrap() { + panic!("metadata crash child exited before partial durable commit: {status}"); + } + tokio::time::sleep(Duration::from_millis(20)).await; + } + }) + .await + .expect("metadata crash child did not reach its durable checkpoint"); + worker.kill().await.unwrap(); + let status = worker.wait().await.unwrap(); + assert!(!status.success()); + #[cfg(unix)] + { + use std::os::unix::process::ExitStatusExt; + assert_eq!(status.signal(), Some(libc::SIGKILL)); + } + first.close().await.unwrap(); + let restarted = PostgresMetadataInstallRepository::new(second.clone()) + .await + .unwrap(); + let prepared = prepared("/"); + let digest = prepared.install_plan().unwrap().digest().unwrap(); + let original = match restarted + .inspect_prepare(&second, OPERATION, digest, MetadataCommitPhase::Payload) + .await + .unwrap() + { + MetadataPrepareObservation::Preparing(intent) => intent, + other => panic!("half installation exposed a false committed receipt: {other:?}"), + }; + let mut marker = marker.lines(); + assert_eq!(marker.next(), Some(original.prepare_id())); + assert_eq!(marker.next(), Some(hex::encode(digest).as_str())); + assert_eq!( + restarted.begin_intent(OPERATION, &prepared).await.unwrap(), + original + ); + assert_eq!( + mst2_metadata_payload::Entity::find() + .count(&second) + .await + .unwrap(), + 1 + ); + assert_eq!( + mst2_metadata_prepare_page::Entity::find() + .count(&second) + .await + .unwrap(), + prepared.dag().payloads().len() as u64 + ); + assert_eq!( + rejected(restarted.finalize(&original).await.unwrap_err()), + SnapshotErrorCode::ObjectUnavailable + ); + assert_eq!( + mst2_retention_root::Entity::find() + .count(&second) + .await + .unwrap(), + 0 + ); + assert_eq!( + mst2_retention_node::Entity::find() + .count(&second) + .await + .unwrap(), + 0 + ); + install(&restarted, &original, &prepared).await; + let receipt = restarted.finalize(&original).await.unwrap(); + assert_eq!(receipt.intent(), &original); + assert_eq!(restarted.finalize(&original).await.unwrap(), receipt); + assert_eq!( + mst2_retention_root::Entity::find() + .count(&second) + .await + .unwrap(), + prepared.dag().payloads().len() as u64 + ); + assert_eq!( + mst2_retention_edge::Entity::find() + .count(&second) + .await + .unwrap(), + 1 + ); +} + +async fn overwrite_installed_payload_for_test( + connection: &DatabaseConnection, + page: &MetadataPagePayload, + bytes: Vec, +) { + assert_eq!(bytes.len(), page.bytes.len()); + let txn = connection.begin().await.unwrap(); + txn.execute_unprepared( + "ALTER TABLE mst2_metadata_payload DISABLE TRIGGER mst2_metadata_payload_immutable", + ) + .await + .unwrap(); + txn.execute_raw(statement( + "UPDATE mst2_metadata_payload SET payload=$2 WHERE page_id=$1", + [page.id.to_vec().into(), bytes.into()], + )) + .await + .unwrap(); + txn.execute_unprepared( + "ALTER TABLE mst2_metadata_payload ENABLE TRIGGER mst2_metadata_payload_immutable", + ) + .await + .unwrap(); + txn.commit().await.unwrap(); +} + +#[tokio::test] +async fn native_metadata_committed_recovery_rejects_same_length_corruption_and_preserves_pins() { + let (first, second, _schema, _url) = fixture().await; + let repository = PostgresMetadataInstallRepository::new(first.clone()) + .await + .unwrap(); + let prepared = prepared("/"); + let intent = repository + .begin_intent("corruption", &prepared) + .await + .unwrap(); + install(&repository, &intent, &prepared).await; + let receipt = repository.finalize(&intent).await.unwrap(); + let page = &prepared.dag().payloads()[0]; + let counters = mst2_retention_node::Entity::find() + .all(&first) + .await + .unwrap() + .into_iter() + .map(|node| (node.node_id, node.incoming_refs)) + .collect::>(); + let mut corrupt = page.bytes.clone(); + corrupt[0] ^= 1; + overwrite_installed_payload_for_test(&first, page, corrupt).await; + assert_eq!( + rejected( + repository + .inspect_prepare( + &second, + intent.operation_id(), + intent.manifest_digest(), + MetadataCommitPhase::Finalize + ) + .await + .unwrap_err() + ), + SnapshotErrorCode::DigestMismatch + ); + assert_eq!( + rejected(repository.finalize(&intent).await.unwrap_err()), + SnapshotErrorCode::DigestMismatch + ); + assert_eq!( + mst2_retention_root::Entity::find() + .count(&second) + .await + .unwrap(), + prepared.dag().payloads().len() as u64 + ); + assert_eq!( + mst2_retention_node::Entity::find() + .all(&second) + .await + .unwrap() + .into_iter() + .map(|node| (node.node_id, node.incoming_refs)) + .collect::>(), + counters + ); + assert_eq!( + mst2_metadata_prepare::Entity::find_by_id(intent.prepare_id().to_owned()) + .one(&second) + .await + .unwrap() + .unwrap() + .state, + "COMMITTED" + ); + overwrite_installed_payload_for_test(&first, page, page.bytes.clone()).await; + assert_eq!( + repository + .inspect_prepare( + &second, + intent.operation_id(), + intent.manifest_digest(), + MetadataCommitPhase::Finalize + ) + .await + .unwrap(), + MetadataPrepareObservation::Committed(receipt) + ); +} + +#[tokio::test] +async fn native_metadata_outer_rollback_preserves_intent_but_exposes_no_graph_or_receipt() { + let (first, second, _schema, _url) = fixture().await; + let repository = PostgresMetadataInstallRepository::new(first.clone()) + .await + .unwrap(); + let prepared = prepared("/"); + let intent = repository + .begin_intent("rollback", &prepared) + .await + .unwrap(); + install(&repository, &intent, &prepared).await; + let dag = repository.load_installed_dag(&intent).await.unwrap(); + let txn = repository.transaction().await.unwrap(); + repository + .finalize_in_txn(&txn, &intent, &dag) + .await + .unwrap(); + assert_eq!( + mst2_retention_root::Entity::find() + .count(&second) + .await + .unwrap(), + 0 + ); + assert_eq!( + mst2_metadata_prepare::Entity::find_by_id(intent.prepare_id().to_owned()) + .one(&second) + .await + .unwrap() + .unwrap() + .state, + "PREPARING" + ); + txn.rollback().await.unwrap(); + assert!(matches!( + repository + .inspect_prepare( + &second, + intent.operation_id(), + intent.manifest_digest(), + MetadataCommitPhase::Finalize + ) + .await + .unwrap(), + MetadataPrepareObservation::Preparing(_) + )); + assert_eq!( + mst2_retention_node::Entity::find() + .count(&second) + .await + .unwrap(), + 0 + ); + repository.finalize(&intent).await.unwrap(); +} + +#[tokio::test] +async fn native_metadata_final_lock_rechecks_deleting_after_payload_verification() { + let (first, second, _schema, _url) = fixture().await; + let repository = PostgresMetadataInstallRepository::new(first.clone()) + .await + .unwrap(); + let prepared = prepared("/"); + let intent = repository + .begin_intent("deleting", &prepared) + .await + .unwrap(); + install(&repository, &intent, &prepared).await; + let dag = repository.load_installed_dag(&intent).await.unwrap(); + let graph = PostgresRetentionRepository::new(second.clone()); + graph + .retain_group(dag.nodes(), dag.edges(), &[]) + .await + .unwrap(); + let root = node_id(&dag.root()); + assert_eq!( + graph.mark_deleting("gc-root", &root).await.unwrap(), + super::super::mst2_retention::GcClaim::Marked + ); + let txn = repository.transaction().await.unwrap(); + assert_eq!( + repository + .finalize_in_txn(&txn, &intent, &dag) + .await + .unwrap_err() + .code, + SnapshotErrorCode::ObjectUnavailable + ); + txn.rollback().await.unwrap(); + assert_eq!( + mst2_retention_root::Entity::find() + .count(&second) + .await + .unwrap(), + 0 + ); + assert_eq!( + mst2_metadata_prepare::Entity::find_by_id(intent.prepare_id().to_owned()) + .one(&second) + .await + .unwrap() + .unwrap() + .state, + "PREPARING" + ); + assert_eq!(graph.node(&root).await.unwrap().unwrap().state, "DELETING"); +} + +#[tokio::test] +async fn native_metadata_receipt_replay_rejects_lost_pin_without_recreating_it() { + let (first, second, _schema, _url) = fixture().await; + let repository = PostgresMetadataInstallRepository::new(first.clone()) + .await + .unwrap(); + let prepared = prepared("/"); + let intent = repository + .begin_intent("lost-pin", &prepared) + .await + .unwrap(); + install(&repository, &intent, &prepared).await; + repository.finalize(&intent).await.unwrap(); + PostgresRetentionRepository::new(first) + .release_root(&RetentionRoot::Prepare(intent.prepare_id().into())) + .await + .unwrap(); + assert_eq!( + rejected( + repository + .inspect_prepare( + &second, + intent.operation_id(), + intent.manifest_digest(), + MetadataCommitPhase::Finalize + ) + .await + .unwrap_err() + ), + SnapshotErrorCode::ObjectUnavailable + ); + assert_eq!( + rejected(repository.finalize(&intent).await.unwrap_err()), + SnapshotErrorCode::ObjectUnavailable + ); + assert_eq!( + mst2_retention_root::Entity::find() + .count(&second) + .await + .unwrap(), + 0 + ); + assert_eq!( + mst2_metadata_payload::Entity::find() + .count(&second) + .await + .unwrap(), + 2 + ); +} + +#[tokio::test] +async fn native_metadata_recovery_lock_timeout_is_unknown_and_never_releases_pins() { + let (first, second, _schema, _url) = fixture().await; + let repository = PostgresMetadataInstallRepository::new(first.clone()) + .await + .unwrap(); + let prepared = prepared("/"); + let intent = repository.begin_intent("timeout", &prepared).await.unwrap(); + install(&repository, &intent, &prepared).await; + repository.finalize(&intent).await.unwrap(); + let blocker = repository.transaction().await.unwrap(); + repository.barrier(&blocker).await.unwrap(); + let mut recovery = PostgresMetadataInstallRepository::new(second.clone()) + .await + .unwrap(); + recovery.barrier_timeout = Duration::from_millis(25); + assert!(matches!( + recovery + .inspect_prepare( + &second, + intent.operation_id(), + intent.manifest_digest(), + MetadataCommitPhase::Finalize + ) + .await + .unwrap_err(), + MetadataInstallError::CommitUncertain { + phase: MetadataCommitPhase::Finalize, + .. + } + )); + assert_eq!( + mst2_retention_root::Entity::find() + .count(&second) + .await + .unwrap(), + 2 + ); + blocker.rollback().await.unwrap(); + assert!(matches!( + recovery + .inspect_prepare( + &second, + intent.operation_id(), + intent.manifest_digest(), + MetadataCommitPhase::Finalize + ) + .await + .unwrap(), + MetadataPrepareObservation::Committed(_) + )); +} + +#[tokio::test] +async fn native_metadata_stored_identity_and_plan_tampering_fail_closed_on_restart() { + let (first, second, _schema, _url) = fixture().await; + let repository = PostgresMetadataInstallRepository::new(first.clone()) + .await + .unwrap(); + let intent = repository + .begin_intent("profile", &prepared("/")) + .await + .unwrap(); + first + .execute_unprepared("UPDATE mst2_metadata_prepare SET projection_revision=2") + .await + .unwrap(); + assert_eq!( + rejected( + repository + .inspect_prepare( + &second, + intent.operation_id(), + intent.manifest_digest(), + MetadataCommitPhase::Intent + ) + .await + .unwrap_err() + ), + SnapshotErrorCode::IntegrityError + ); + first.execute_unprepared("UPDATE mst2_metadata_prepare SET projection_revision=1,canonical_plan=canonical_plan || decode('ff','hex')").await.unwrap(); + assert_eq!( + rejected( + repository + .inspect_prepare( + &second, + intent.operation_id(), + intent.manifest_digest(), + MetadataCommitPhase::Intent + ) + .await + .unwrap_err() + ), + SnapshotErrorCode::IntegrityError + ); + assert_eq!( + mst2_retention_root::Entity::find() + .count(&second) + .await + .unwrap(), + 0 + ); +} + +#[derive(Clone, Copy)] +enum Fault { + BeforeCommit, + AfterCommit, + PrepareRead, +} + +struct PgCommitFaultProxy { + url: String, + armed: Arc, + fired: Arc, + commit_observed: Arc, + changed: Arc, + task: JoinHandle<()>, +} + +impl PgCommitFaultProxy { + async fn start(original: &str, fault: Fault) -> Self { + let mut url = url::Url::parse(original).unwrap(); + let host = url.host_str().unwrap().to_owned(); + let port = url.port().unwrap_or(5432); + let listener = TcpListener::bind("127.0.0.1:0").await.unwrap(); + let address = listener.local_addr().unwrap(); + url.set_host(Some("127.0.0.1")).unwrap(); + url.set_port(Some(address.port())).unwrap(); + let pairs: Vec<_> = url + .query_pairs() + .filter(|(key, _)| key != "sslmode") + .map(|(key, value)| (key.into_owned(), value.into_owned())) + .collect(); + url.set_query(None); + url.query_pairs_mut() + .extend_pairs(pairs) + .append_pair("sslmode", "disable"); + let armed = Arc::new(AtomicBool::new(false)); + let fired = Arc::new(AtomicBool::new(false)); + let commit_observed = Arc::new(AtomicBool::new(false)); + let changed = Arc::new(Notify::new()); + let a = armed.clone(); + let f = fired.clone(); + let c = commit_observed.clone(); + let n = changed.clone(); + let task = tokio::spawn(async move { + let mut connections = JoinSet::new(); + loop { + tokio::select! { + accepted=listener.accept()=>{ + let (client,_)=accepted.unwrap();let host=host.clone(); + let a=a.clone();let f=f.clone();let c=c.clone();let n=n.clone(); + connections.spawn(async move { + let server=TcpStream::connect((host.as_str(),port)).await?; + forward_connection(client,server,fault,a,f,c,n).await + }); + } + _=connections.join_next(),if !connections.is_empty()=>{} + } + } + }); + Self { + url: url.to_string(), + armed, + fired, + commit_observed, + changed, + task, + } + } + async fn wait_for_fault(&self) { + tokio::time::timeout(Duration::from_secs(10), async { + loop { + let changed = self.changed.notified(); + if self.fired.load(Ordering::SeqCst) { + break; + } + changed.await; + } + }) + .await + .expect("proxy never observed the real COMMIT fault"); + } +} +impl Drop for PgCommitFaultProxy { + fn drop(&mut self) { + self.task.abort(); + } +} + +async fn read_frame( + reader: &mut R, +) -> std::io::Result<(u8, Vec)> { + let kind = reader.read_u8().await?; + let length = reader.read_u32().await?; + if !(4..=4 * 1024 * 1024).contains(&length) { + return Err(std::io::Error::other("invalid PostgreSQL frame length")); + } + let mut body = vec![0; length as usize - 4]; + reader.read_exact(&mut body).await?; + Ok((kind, body)) +} +async fn write_frame( + writer: &mut W, + kind: u8, + body: &[u8], +) -> std::io::Result<()> { + writer.write_u8(kind).await?; + writer.write_u32(body.len() as u32 + 4).await?; + writer.write_all(body).await?; + writer.flush().await +} + +async fn forward_connection( + mut client: TcpStream, + mut server: TcpStream, + fault: Fault, + armed: Arc, + fired: Arc, + commit_observed: Arc, + changed: Arc, +) -> std::io::Result<()> { + // sslmode=disable makes the first message a bounded StartupMessage. + let length = client.read_u32().await?; + if !(8..=65536).contains(&length) { + return Err(std::io::Error::other("invalid PostgreSQL startup length")); + } + let mut startup = vec![0; length as usize - 4]; + client.read_exact(&mut startup).await?; + server.write_u32(length).await?; + server.write_all(&startup).await?; + server.flush().await?; + let (mut cr, mut cw) = client.into_split(); + let (mut sr, mut sw) = server.into_split(); + let suppress = Arc::new(AtomicBool::new(false)); + let sender_suppress = suppress.clone(); + let (stop, mut stopped) = watch::channel(false); + let sender_stop = stop.clone(); + let sender_fired = fired.clone(); + let sender_changed = changed.clone(); + let to_server = async move { + loop { + let frame = tokio::select! {result=read_frame(&mut cr)=>result?, _=stopped.changed()=>return Ok::<_,std::io::Error>(())}; + let is_commit = frame.0 == b'Q' && frame.1.as_slice() == b"COMMIT\0"; + let fault_target = match fault { + Fault::PrepareRead => { + b"PQ".contains(&frame.0) + && frame + .1 + .windows(b"mst2_metadata_prepare".len()) + .any(|bytes| bytes == b"mst2_metadata_prepare") + } + _ => is_commit, + }; + if fault_target && armed.swap(false, Ordering::SeqCst) { + match fault { + Fault::BeforeCommit | Fault::PrepareRead => { + sender_fired.store(true, Ordering::SeqCst); + sender_changed.notify_one(); + sw.shutdown().await?; + sender_stop.send_replace(true); + return Ok(()); + } + Fault::AfterCommit => sender_suppress.store(true, Ordering::SeqCst), + } + } + write_frame(&mut sw, frame.0, &frame.1).await?; + } + }; + let mut stopped = stop.subscribe(); + let to_client = async move { + loop { + let frame = tokio::select! {result=read_frame(&mut sr)=>result?, _=stopped.changed()=>{cw.shutdown().await?;return Ok::<_,std::io::Error>(());}}; + if suppress.load(Ordering::SeqCst) { + if frame.0 == b'C' && frame.1.as_slice() == b"COMMIT\0" { + commit_observed.store(true, Ordering::SeqCst); + fired.store(true, Ordering::SeqCst); + changed.notify_one(); + cw.shutdown().await?; + stop.send_replace(true); + return Ok(()); + } + } else { + write_frame(&mut cw, frame.0, &frame.1).await?; + } + } + }; + tokio::try_join!(to_server, to_client)?; + Ok(()) +} + +async fn commit_fault_case(fault: Fault) { + let (direct, recovery, _schema, url) = fixture().await; + let proxy = PgCommitFaultProxy::start(&url, fault).await; + let mut options = sea_orm::ConnectOptions::new(proxy.url.clone()); + options.max_connections(1).min_connections(1); + let proxied = Database::connect(options).await.unwrap(); + let repository = PostgresMetadataInstallRepository::new(proxied) + .await + .unwrap(); + let prepared = prepared("/"); + let intent = repository + .begin_intent("commit-fault", &prepared) + .await + .unwrap(); + install(&repository, &intent, &prepared).await; + proxy.armed.store(true, Ordering::SeqCst); + let error = tokio::time::timeout(Duration::from_secs(15), repository.finalize(&intent)) + .await + .unwrap() + .unwrap_err(); + assert!(matches!( + error, + MetadataInstallError::CommitUncertain { + phase: MetadataCommitPhase::Finalize, + .. + } + )); + proxy.wait_for_fault().await; + let observed = repository + .inspect_prepare( + &recovery, + intent.operation_id(), + intent.manifest_digest(), + MetadataCommitPhase::Finalize, + ) + .await + .unwrap(); + match fault { + Fault::AfterCommit => { + assert!(proxy.commit_observed.load(Ordering::SeqCst)); + assert!(matches!(observed, MetadataPrepareObservation::Committed(_))); + assert_eq!( + mst2_retention_root::Entity::find() + .count(&direct) + .await + .unwrap(), + 2 + ); + assert_eq!( + mst2_retention_edge::Entity::find() + .count(&direct) + .await + .unwrap(), + 1 + ); + let restarted = PostgresMetadataInstallRepository::new(direct.clone()) + .await + .unwrap(); + restarted.finalize(&intent).await.unwrap(); + assert_eq!( + mst2_retention_edge::Entity::find() + .count(&direct) + .await + .unwrap(), + 1 + ); + } + Fault::BeforeCommit => { + assert!(!proxy.commit_observed.load(Ordering::SeqCst)); + assert!(matches!(observed, MetadataPrepareObservation::Preparing(_))); + assert_eq!( + mst2_retention_root::Entity::find() + .count(&direct) + .await + .unwrap(), + 0 + ); + assert_eq!( + mst2_retention_node::Entity::find() + .count(&direct) + .await + .unwrap(), + 0 + ); + PostgresMetadataInstallRepository::new(direct) + .await + .unwrap() + .finalize(&intent) + .await + .unwrap(); + } + Fault::PrepareRead => panic!("query fault uses its independent recovery test"), + } +} + +#[tokio::test] +async fn native_metadata_real_commit_response_loss_recovers_receipt_without_releasing_pin() { + commit_fault_case(Fault::AfterCommit).await; +} + +#[tokio::test] +async fn native_metadata_real_connection_loss_before_commit_recovers_preparing_without_false_success() + { + commit_fault_case(Fault::BeforeCommit).await; +} + +#[tokio::test] +async fn native_metadata_wrong_schema_primary_cannot_report_absent_for_another_committed_operation() +{ + let (first, second, _schema, _url) = fixture().await; + let (wrong_primary, _other, _wrong_schema, _wrong_url) = fixture().await; + let repository = PostgresMetadataInstallRepository::new(first.clone()) + .await + .unwrap(); + let prepared = prepared("/"); + let intent = repository.begin_intent("scope", &prepared).await.unwrap(); + install(&repository, &intent, &prepared).await; + repository.finalize(&intent).await.unwrap(); + assert!(matches!( + repository + .inspect_prepare( + &wrong_primary, + intent.operation_id(), + intent.manifest_digest(), + MetadataCommitPhase::Finalize + ) + .await + .unwrap_err(), + MetadataInstallError::CommitUncertain { .. } + )); + assert_eq!( + mst2_retention_root::Entity::find() + .count(&second) + .await + .unwrap(), + 2 + ); + assert_eq!( + mst2_metadata_prepare::Entity::find() + .count(&wrong_primary) + .await + .unwrap(), + 0 + ); + assert!(matches!( + repository + .inspect_prepare( + &second, + intent.operation_id(), + intent.manifest_digest(), + MetadataCommitPhase::Finalize + ) + .await + .unwrap(), + MetadataPrepareObservation::Committed(_) + )); +} + +#[tokio::test] +async fn native_metadata_query_loss_after_recovery_barrier_preserves_typed_unknown() { + let (first, second, _schema, url) = fixture().await; + let repository = PostgresMetadataInstallRepository::new(first.clone()) + .await + .unwrap(); + let prepared = prepared("/"); + let intent = repository + .begin_intent("read-fault", &prepared) + .await + .unwrap(); + install(&repository, &intent, &prepared).await; + repository.finalize(&intent).await.unwrap(); + let proxy = PgCommitFaultProxy::start(&url, Fault::PrepareRead).await; + let mut options = sea_orm::ConnectOptions::new(proxy.url.clone()); + options.max_connections(1).min_connections(1); + let fresh = Database::connect(options).await.unwrap(); + proxy.armed.store(true, Ordering::SeqCst); + assert!(matches!( + repository + .inspect_prepare( + &fresh, + intent.operation_id(), + intent.manifest_digest(), + MetadataCommitPhase::Finalize + ) + .await + .unwrap_err(), + MetadataInstallError::CommitUncertain { .. } + )); + proxy.wait_for_fault().await; + assert!(!proxy.commit_observed.load(Ordering::SeqCst)); + assert_eq!( + mst2_retention_root::Entity::find() + .count(&second) + .await + .unwrap(), + 2 + ); + assert!(matches!( + repository + .inspect_prepare( + &second, + intent.operation_id(), + intent.manifest_digest(), + MetadataCommitPhase::Finalize + ) + .await + .unwrap(), + MetadataPrepareObservation::Committed(_) + )); +} diff --git a/src/jupiter/storage/native_publication_storage.rs b/src/jupiter/storage/native_publication_storage.rs new file mode 100644 index 00000000..8039a3ea --- /dev/null +++ b/src/jupiter/storage/native_publication_storage.rs @@ -0,0 +1,656 @@ +//! Native `/` publication certificates; origin path receipts retain their v1 identity. + +use git_internal::hash::{ObjectHash, get_hash_kind}; +use sea_orm::{ + ActiveModelTrait, ActiveValue::Set, ColumnTrait, ConnectionTrait, DatabaseTransaction, + EntityTrait, QueryFilter, QueryResult, Statement, TransactionTrait, +}; + +use crate::{ + callisto::{mega_refs, mst2_native_publication, mst2_publication, push_queue}, + common::utils::MEGA_BRANCH_NAME, + jupiter::storage::{ + base_storage::StorageConnector, + mono_storage::MonoStorage, + mst2_publication_storage::{CommittedPublication, PublicationReceiptError}, + push_queue_storage::PushQueueStorage, + }, +}; + +const NATIVE_EPOCH: i64 = 1; + +// Used by both ordinary reads and the single UPDATE that captures a B2.5 token. +pub(crate) const NATIVE_OBSERVATION_CTES: &str = r#" + native_roots AS ( + SELECT count(*)::bigint AS root_count, + max(ref_commit_hash) AS root_commit, max(ref_tree_hash) AS root_tree + FROM mega_refs WHERE path = '/' AND ref_name = $1 AND is_cl = false + ), native_observation AS ( + SELECT roots.*, h.instance_id, h.sequence, h.writer_epoch, h.state, + h.root_commit AS head_commit, h.root_tree AS head_tree, + h.certificate_receipt_id, + c.receipt_id AS certificate_id, c.namespace AS certificate_namespace, + c.instance_id AS certificate_instance, c.sequence AS certificate_sequence, + c.writer_epoch AS certificate_epoch, + c.old_root_commit AS certificate_old_commit, + c.root_commit AS certificate_commit, c.root_tree AS certificate_tree, + c.origin_path, c.origin_ref, c.old_path_commit, c.path_commit, + r.id AS receipt_id, r.namespace AS receipt_namespace, + r.sequence AS receipt_sequence, r.operation_id, + r.old_oid, r.new_oid, r.writer_epoch AS receipt_epoch, r.writer_kind, + r.request_digest, r.request_digest_version, r.native_certificate_version, + o.id AS outbox_id, o.namespace AS outbox_namespace, o.sequence AS outbox_sequence + FROM native_roots roots + LEFT JOIN mst2_native_head h ON h.namespace = '/' + LEFT JOIN mst2_native_publication c ON c.receipt_id = h.certificate_receipt_id + LEFT JOIN mst2_publication r ON r.id = c.receipt_id + LEFT JOIN mst2_publication_outbox o ON o.operation_id = r.operation_id + ) +"#; + +#[derive(Clone, Debug, PartialEq, Eq)] +pub(crate) struct NativeRoot { + pub(crate) commit: String, + pub(crate) tree: String, +} + +#[derive(Clone, Debug, PartialEq, Eq)] +pub(crate) struct NativePublicationToken { + pub(crate) sequence: i64, + pub(crate) epoch: i64, + pub(crate) certificate: Option, +} + +#[derive(Clone, Debug)] +pub(crate) struct NativePublicationHead { + pub(crate) root: NativeRoot, + pub(crate) instance_id: String, + pub(crate) token: NativePublicationToken, + ready: bool, +} + +#[derive(Clone, Debug)] +pub(crate) struct NativeObservation { + pub(crate) root: Option, + pub(crate) head: Option, +} + +#[derive(Debug)] +pub(crate) struct PreparedNativePublication { + head: NativePublicationHead, + path: String, + old_path: Option, + transaction_id: i64, +} + +fn integrity(message: &str) -> PublicationReceiptError { + PublicationReceiptError::Integrity(message.to_owned()) +} + +fn canonical_instance(instance: &str) -> Result { + uuid::Uuid::parse_str(instance) + .map(|id| id.to_string()) + .map_err(|_| integrity("invalid native deployment instance")) +} + +fn validate_root(root: &NativeRoot) -> Result<(), PublicationReceiptError> { + for oid in [&root.commit, &root.tree] { + ObjectHash::from_hex_for_kind(get_hash_kind(), oid) + .map_err(|_| integrity("invalid native object identity"))?; + } + Ok(()) +} + +pub(crate) fn decode_native_observation( + row: &QueryResult, +) -> Result { + let root_count: i64 = row.try_get("", "root_count")?; + if !(0..=1).contains(&root_count) { + return Err(integrity("native root is ambiguous")); + } + let root = if root_count == 1 { + let root = NativeRoot { + commit: row.try_get("", "root_commit")?, + tree: row.try_get("", "root_tree")?, + }; + validate_root(&root)?; + Some(root) + } else { + None + }; + let instance: Option = row.try_get("", "instance_id")?; + let Some(instance_id) = instance else { + return Ok(NativeObservation { root, head: None }); + }; + if canonical_instance(&instance_id)? != instance_id { + return Err(integrity("native head instance is not canonical")); + } + let head_root = NativeRoot { + commit: row.try_get("", "head_commit")?, + tree: row.try_get("", "head_tree")?, + }; + validate_root(&head_root)?; + if root.as_ref() != Some(&head_root) { + return Err(integrity("native root bypassed its publication head")); + } + let token = NativePublicationToken { + sequence: row.try_get("", "sequence")?, + epoch: row.try_get("", "writer_epoch")?, + certificate: row.try_get("", "certificate_receipt_id")?, + }; + if token.sequence < 0 || token.epoch != NATIVE_EPOCH { + return Err(PublicationReceiptError::Conflict( + "native writer epoch is fenced".into(), + )); + } + let state: String = row.try_get("", "state")?; + let ready = match state.as_str() { + "INITIALIZING" if token.certificate.is_none() => false, + "READY" if token.certificate.is_some() => { + let certificate_id: Option = row.try_get("", "certificate_id")?; + let receipt_id: Option = row.try_get("", "receipt_id")?; + let certificate_sequence: Option = row.try_get("", "certificate_sequence")?; + let certificate_epoch: Option = row.try_get("", "certificate_epoch")?; + let certificate_namespace: Option = row.try_get("", "certificate_namespace")?; + let certificate_instance: Option = row.try_get("", "certificate_instance")?; + let certificate_commit: Option = row.try_get("", "certificate_commit")?; + let certificate_tree: Option = row.try_get("", "certificate_tree")?; + let marker: Option = row.try_get("", "native_certificate_version")?; + let version: Option = row.try_get("", "request_digest_version")?; + let digest: Option = row.try_get("", "request_digest")?; + let writer_kind: Option = row.try_get("", "writer_kind")?; + let receipt_epoch: Option = row.try_get("", "receipt_epoch")?; + let origin_path: Option = row.try_get("", "origin_path")?; + let origin_ref: Option = row.try_get("", "origin_ref")?; + let receipt_namespace: Option = row.try_get("", "receipt_namespace")?; + let old_oid: Option = row.try_get("", "old_oid")?; + let old_root: Option = row.try_get("", "certificate_old_commit")?; + let new_oid: Option = row.try_get("", "new_oid")?; + let path_commit: Option = row.try_get("", "path_commit")?; + let old_path: Option = row.try_get("", "old_path_commit")?; + let outbox: Option = row.try_get("", "outbox_id")?; + let outbox_namespace: Option = row.try_get("", "outbox_namespace")?; + let outbox_sequence: Option = row.try_get("", "outbox_sequence")?; + let receipt_sequence: Option = row.try_get("", "receipt_sequence")?; + let digest_valid = digest.as_deref().is_some_and(|value| { + value.len() == 71 + && value.starts_with("sha256:") + && value[7..] + .bytes() + .all(|byte| byte.is_ascii_digit() || (b'a'..=b'f').contains(&byte)) + }); + if certificate_id != token.certificate + || receipt_id != token.certificate + || certificate_sequence != Some(token.sequence) + || certificate_epoch != Some(token.epoch) + || certificate_namespace.as_deref() != Some("/") + || certificate_instance.as_deref() != Some(instance_id.as_str()) + || certificate_commit.as_deref() != Some(head_root.commit.as_str()) + || certificate_tree.as_deref() != Some(head_root.tree.as_str()) + || marker != Some(1) + || version != Some(1) + || !digest_valid + || writer_kind.as_deref() != Some("trunk_push") + || receipt_epoch != Some(token.epoch) + || origin_ref.as_deref() != Some(MEGA_BRANCH_NAME) + || origin_path.is_none() + || origin_path != receipt_namespace + || new_oid.is_none() + || new_oid != path_commit + || old_oid.is_none() + || old_oid != old_root + || old_path == path_commit + || outbox.is_none() + || outbox_namespace != receipt_namespace + || outbox_sequence != receipt_sequence + { + return Err(integrity( + "native publication association is incomplete or inconsistent", + )); + } + true + } + _ => return Err(integrity("invalid native head state")), + }; + Ok(NativeObservation { + root, + head: Some(NativePublicationHead { + root: head_root, + instance_id, + token, + ready, + }), + }) +} + +async fn observe( + connection: &C, +) -> Result { + let rows = connection + .query_all_raw(Statement::from_sql_and_values( + connection.get_database_backend(), + format!("WITH {NATIVE_OBSERVATION_CTES} SELECT * FROM native_observation"), + [MEGA_BRANCH_NAME.into()], + )) + .await?; + if rows.len() != 1 { + return Err(integrity("ambiguous native publication observation")); + } + decode_native_observation(&rows[0]) +} + +async fn selected_ref( + connection: &C, + path: &str, +) -> Result, PublicationReceiptError> { + let refs = mega_refs::Entity::find() + .filter(mega_refs::Column::Path.eq(path)) + .filter(mega_refs::Column::RefName.eq(MEGA_BRANCH_NAME)) + .filter(mega_refs::Column::IsCl.eq(false)) + .all(connection) + .await?; + match refs.as_slice() { + [] => Ok(None), + [row] => { + let root = NativeRoot { + commit: row.ref_commit_hash.clone(), + tree: row.ref_tree_hash.clone(), + }; + validate_root(&root)?; + Ok(Some(root)) + } + _ => Err(integrity("selected native ref is ambiguous")), + } +} + +impl MonoStorage { + pub(crate) async fn read_native_publication_head( + &self, + instance: &str, + ) -> Result { + let observation = observe(self.get_connection()).await?; + let head = observation + .head + .ok_or_else(|| integrity("native publication is not initialized"))?; + if !head.ready { + return Err(integrity("native publication is not ready")); + } + if head.instance_id != canonical_instance(instance)? { + return Err(integrity("native instance changed")); + } + Ok(head) + } + + /// The command confirms stopped writers; this transaction checks the + /// paused admission gate, drained queue and exact native root. + pub(crate) async fn initialize_native_publication_for_maintenance( + &self, + instance: &str, + expected_root: &NativeRoot, + ) -> Result<(), PublicationReceiptError> { + let instance = canonical_instance(instance)?; + if uuid::Uuid::parse_str(&instance).is_ok_and(|id| id.is_nil()) { + return Err(integrity("native deployment instance must not be nil")); + } + for oid in [&expected_root.commit, &expected_root.tree] { + if oid.len() != 40 + || !oid + .bytes() + .all(|byte| byte.is_ascii_digit() || (b'a'..=b'f').contains(&byte)) + { + return Err(integrity( + "maintenance requires canonical SHA-1 root identities", + )); + } + } + let txn = self.get_connection().begin().await?; + if !PushQueueStorage::try_mono_write_lock(&txn) + .await + .map_err(|error| integrity(&error.to_string()))? + { + return Err(PublicationReceiptError::Conflict( + "native initialization refused while a writer holds the mono write lock".into(), + )); + } + let control = txn + .query_one_raw(Statement::from_string( + txn.get_database_backend(), + "SELECT paused, hard_stopped FROM queue_control WHERE id=1 FOR UPDATE NOWAIT" + .to_owned(), + )) + .await? + .ok_or_else(|| { + integrity( + "queue_control row missing; maintenance cannot establish admission exclusion", + ) + })?; + if !control.try_get::("", "paused")? { + return Err(integrity( + "queue admission must be paused before native initialization", + )); + } + if control.try_get::("", "hard_stopped")? { + return Err(integrity( + "hard-stopped queue must be investigated before native initialization", + )); + } + let history = txn + .query_one_raw(Statement::from_string( + txn.get_database_backend(), + "SELECT EXISTS(SELECT 1 FROM mst2_native_publication) OR \ + EXISTS(SELECT 1 FROM mst2_publication WHERE native_certificate_version IS NOT NULL) AS present" + .to_owned(), + )) + .await? + .ok_or_else(|| integrity("native history observation missing"))?; + if history.try_get::("", "present")? { + return Err(integrity( + "native publication history exists; initialization cannot repair or replace it", + )); + } + let objects = txn + .query_one_raw(Statement::from_sql_and_values( + txn.get_database_backend(), + "SELECT (SELECT count(*) FROM mega_commit WHERE commit_id=$1) AS commits, \ + (SELECT max(tree) FROM mega_commit WHERE commit_id=$1) AS commit_tree, \ + (SELECT count(*) FROM mega_tree WHERE tree_id=$2) AS trees", + [ + expected_root.commit.clone().into(), + expected_root.tree.clone().into(), + ], + )) + .await? + .ok_or_else(|| integrity("native object observation missing"))?; + if objects.try_get::("", "commits")? != 1 + || objects + .try_get::>("", "commit_tree")? + .as_deref() + != Some(expected_root.tree.as_str()) + || objects.try_get::("", "trees")? != 1 + { + return Err(integrity( + "native root requires a unique stored commit with the expected tree and a unique stored tree", + )); + } + self.initialize_native_publication_in_txn(&txn, &instance, Some(expected_root)) + .await?; + txn.commit().await?; + Ok(()) + } + + #[cfg(test)] + pub(crate) async fn initialize_native_publication( + &self, + instance: &str, + ) -> Result<(), PublicationReceiptError> { + let instance = canonical_instance(instance)?; + let txn = self.get_connection().begin().await?; + PushQueueStorage::acquire_mono_write_lock(&txn) + .await + .map_err(|error| integrity(&error.to_string()))?; + txn.execute_unprepared("SELECT id FROM queue_control WHERE id=1 FOR UPDATE") + .await?; + self.initialize_native_publication_in_txn(&txn, &instance, None) + .await?; + txn.commit().await?; + Ok(()) + } + + async fn initialize_native_publication_in_txn( + &self, + txn: &DatabaseTransaction, + instance: &str, + expected_root: Option<&NativeRoot>, + ) -> Result<(), PublicationReceiptError> { + let observation = observe(txn).await?; + if observation.head.is_some() { + return Err(integrity("native head already initialized")); + } + let root = observation + .root + .ok_or_else(|| integrity("native root missing"))?; + if expected_root.is_some_and(|expected| *expected != root) { + return Err(PublicationReceiptError::Conflict( + "native root differs from the maintenance command's expected commit/tree".into(), + )); + } + let row = txn.query_one_raw(Statement::from_string(txn.get_database_backend(), + "SELECT count(*)::bigint AS active FROM push_queue WHERE status IN ('Queued', 'Running')".to_owned())).await? + .ok_or_else(|| integrity("queue observation missing"))?; + if row.try_get::("", "active")? != 0 { + return Err(integrity( + "queue must be drained before native initialization", + )); + } + let row = txn + .query_one_raw(Statement::from_string( + txn.get_database_backend(), + r#" + SELECT min(sequence) AS minimum, max(sequence) AS maximum FROM ( + SELECT sequence FROM mst2_namespace_seq UNION ALL SELECT sequence FROM mst2_publication + UNION ALL SELECT observed_sequence AS sequence FROM mst2_queue_noop_receipt + ) counters + "# + .to_owned(), + )) + .await? + .ok_or_else(|| integrity("native sequence floor missing"))?; + let minimum: Option = row.try_get("", "minimum")?; + let floor = row.try_get::>("", "maximum")?.unwrap_or(0); + if minimum.is_some_and(|value| value < 0) || floor == i64::MAX { + return Err(integrity("native sequence floor is invalid or exhausted")); + } + txn.execute_raw(Statement::from_sql_and_values(txn.get_database_backend(), + "INSERT INTO mst2_native_head (namespace, instance_id, sequence, writer_epoch, root_commit, root_tree, state) \ + VALUES ('/', $1, $2, $3, $4, $5, 'INITIALIZING')", + [instance.into(), floor.into(), NATIVE_EPOCH.into(), root.commit.into(), root.tree.into()], + )).await?; + Ok(()) + } + + pub(crate) async fn reserve_native_publication_in_txn( + &self, + txn: &DatabaseTransaction, + queue: &push_queue::Model, + instance: &str, + ) -> Result { + let owner = txn.query_one_raw(Statement::from_string(txn.get_database_backend(), + "SELECT txid_current() AS owner FROM mst2_native_head WHERE namespace = '/' FOR UPDATE".to_owned())) + .await?.ok_or_else(|| integrity("native publication is not initialized"))?; + let observation = observe(txn).await?; + let head = observation + .head + .ok_or_else(|| integrity("native publication is not initialized"))?; + if head.instance_id != canonical_instance(instance)? { + return Err(integrity("native instance changed")); + } + if queue.expected_native_sequence != Some(head.token.sequence) + || queue.expected_native_epoch != Some(head.token.epoch) + || queue.expected_native_certificate != head.token.certificate + || queue.expected_commit_hash.as_deref() != Some(head.root.commit.as_str()) + || queue.expected_tree_hash.as_deref() != Some(head.root.tree.as_str()) + { + return Err(PublicationReceiptError::Conflict( + "claimed native publication token is stale or missing".into(), + )); + } + let old_path = selected_ref(txn, &queue.path).await?; + Ok(PreparedNativePublication { + head, + path: queue.path.clone(), + old_path, + transaction_id: owner.try_get("", "owner")?, + }) + } + + async fn check_native_reservation( + &self, + txn: &DatabaseTransaction, + prepared: &PreparedNativePublication, + ) -> Result<(), PublicationReceiptError> { + let row = txn.query_one_raw(Statement::from_sql_and_values(txn.get_database_backend(), + "SELECT sequence FROM mst2_native_head WHERE namespace = '/' AND sequence = $1 AND writer_epoch = $2 \ + AND certificate_receipt_id IS NOT DISTINCT FROM $3 AND root_commit = $4 AND root_tree = $5 \ + AND txid_current() = $6 FOR UPDATE", + [prepared.head.token.sequence.into(), prepared.head.token.epoch.into(), prepared.head.token.certificate.into(), + prepared.head.root.commit.clone().into(), prepared.head.root.tree.clone().into(), prepared.transaction_id.into()], + )).await?; + if row.is_none() { + return Err(PublicationReceiptError::Conflict( + "native reservation is stale or belongs to another transaction".into(), + )); + } + Ok(()) + } + + pub(crate) async fn finish_native_noop_in_txn( + &self, + txn: &DatabaseTransaction, + prepared: PreparedNativePublication, + ) -> Result<(), PublicationReceiptError> { + self.check_native_reservation(txn, &prepared).await?; + if selected_ref(txn, "/").await?.as_ref() != Some(&prepared.head.root) + || selected_ref(txn, &prepared.path).await? != prepared.old_path + { + return Err(integrity("native no-op changed selected refs")); + } + Ok(()) + } + + pub(crate) async fn record_native_publication_in_txn( + &self, + txn: &DatabaseTransaction, + prepared: PreparedNativePublication, + committed: &CommittedPublication, + ) -> Result<(), PublicationReceiptError> { + self.check_native_reservation(txn, &prepared).await?; + let root = selected_ref(txn, "/") + .await? + .ok_or_else(|| integrity("published native root missing"))?; + let path = selected_ref(txn, &prepared.path) + .await? + .ok_or_else(|| integrity("published native path missing"))?; + if prepared + .old_path + .as_ref() + .is_some_and(|old| old.commit == path.commit) + || committed.receipt.namespace != prepared.path + || committed.receipt.new_oid != path.commit + || committed.receipt.old_oid != prepared.head.root.commit + || committed.receipt.writer_kind != "trunk_push" + || committed.receipt.writer_epoch != NATIVE_EPOCH + || committed.receipt.native_certificate_version.is_some() + || committed.outbox.operation_id != committed.receipt.operation_id + || committed.outbox.namespace != committed.receipt.namespace + || committed.outbox.sequence != committed.receipt.sequence + { + return Err(integrity( + "operation did not change its selected native ref", + )); + } + let resolved_path_tree = self + .resolve_path_tree_hash_in_txn(&root.tree, &prepared.path, txn) + .await + .map_err(|_| integrity("native path tree lookup failed"))? + .ok_or_else(|| integrity("native path tree is not materialized"))?; + if resolved_path_tree != path.tree { + return Err(integrity("native path tree does not match selected ref")); + } + let next = prepared + .head + .token + .sequence + .checked_add(1) + .ok_or_else(|| integrity("native sequence exhausted"))?; + let marked = txn.execute_raw(Statement::from_sql_and_values(txn.get_database_backend(), + "UPDATE mst2_publication SET native_certificate_version = 1 WHERE id = $1 AND native_certificate_version IS NULL", + [committed.receipt.id.into()], + )).await?; + if marked.rows_affected() != 1 { + return Err(integrity("native receipt marker changed")); + } + mst2_native_publication::ActiveModel { + receipt_id: Set(committed.receipt.id), + namespace: Set("/".to_owned()), + instance_id: Set(prepared.head.instance_id), + sequence: Set(next), + writer_epoch: Set(NATIVE_EPOCH), + old_root_commit: Set(prepared.head.root.commit.clone()), + old_root_tree: Set(prepared.head.root.tree.clone()), + root_commit: Set(root.commit.clone()), + root_tree: Set(root.tree.clone()), + origin_path: Set(prepared.path), + origin_ref: Set(MEGA_BRANCH_NAME.to_owned()), + old_path_commit: Set(prepared.old_path.as_ref().map(|old| old.commit.clone())), + old_path_tree: Set(prepared.old_path.map(|old| old.tree)), + path_commit: Set(path.commit), + path_tree: Set(path.tree), + } + .insert(txn) + .await?; + #[cfg(all(test, unix))] + tests::crash_checkpoint("native-certificate-written"); + let updated = txn.execute_raw(Statement::from_sql_and_values(txn.get_database_backend(), + "UPDATE mst2_native_head SET sequence = $1, root_commit = $2, root_tree = $3, state = 'READY', certificate_receipt_id = $4 \ + WHERE namespace = '/' AND sequence = $5 AND writer_epoch = $6 \ + AND certificate_receipt_id IS NOT DISTINCT FROM $7 AND root_commit = $8 AND root_tree = $9 AND txid_current() = $10", + [next.into(), root.commit.into(), root.tree.into(), committed.receipt.id.into(), prepared.head.token.sequence.into(), + NATIVE_EPOCH.into(), prepared.head.token.certificate.into(), prepared.head.root.commit.into(), + prepared.head.root.tree.into(), prepared.transaction_id.into()], + )).await?; + if updated.rows_affected() != 1 { + return Err(PublicationReceiptError::Conflict( + "native head CAS failed".into(), + )); + } + #[cfg(all(test, unix))] + tests::crash_checkpoint("native-head-written"); + Ok(()) + } + + pub(crate) async fn validate_native_historical_receipt_in_txn( + &self, + txn: &DatabaseTransaction, + receipt: &mst2_publication::Model, + ) -> Result<(), PublicationReceiptError> { + match receipt.native_certificate_version { + None => return Ok(()), + Some(1) => {} + _ => return Err(integrity("unsupported native certificate version")), + } + let certificate = mst2_native_publication::Entity::find_by_id(receipt.id) + .one(txn) + .await? + .ok_or_else(|| integrity("historical native certificate missing"))?; + if certificate.namespace != "/" + || certificate.sequence <= 0 + || certificate.writer_epoch != receipt.writer_epoch + || certificate.origin_path != receipt.namespace + || certificate.origin_ref != MEGA_BRANCH_NAME + || receipt.writer_kind != "trunk_push" + || certificate.old_root_commit != receipt.old_oid + || certificate.path_commit != receipt.new_oid + || certificate.old_path_commit.as_ref() == Some(&certificate.path_commit) + || canonical_instance(&certificate.instance_id)? != certificate.instance_id + { + return Err(integrity("historical native certificate identity mismatch")); + } + validate_root(&NativeRoot { + commit: certificate.root_commit, + tree: certificate.root_tree, + })?; + validate_root(&NativeRoot { + commit: certificate.old_root_commit, + tree: certificate.old_root_tree, + })?; + validate_root(&NativeRoot { + commit: certificate.path_commit, + tree: certificate.path_tree, + })?; + Ok(()) + } +} + +#[cfg(test)] +#[path = "native_publication_tests.rs"] +mod tests; diff --git a/src/jupiter/storage/native_publication_tests.rs b/src/jupiter/storage/native_publication_tests.rs new file mode 100644 index 00000000..c3e83323 --- /dev/null +++ b/src/jupiter/storage/native_publication_tests.rs @@ -0,0 +1,854 @@ +use std::sync::Arc; + +use git_internal::{ + hash::{ObjectHash, get_hash_kind}, + internal::object::tree::{TreeItem, TreeItemMode}, +}; +use sea_orm::{DatabaseTransaction, PaginatorTrait}; + +use super::*; +use crate::{ + callisto::{ + mst2_native_head, mst2_publication_outbox, sea_orm_active_enums::PushQueueKindEnum, + }, + jupiter::{ + migration::apply_migrations, + storage::{ + base_storage::BaseStorage, + mst2_publication_storage::{PublicationPreparation, PublicationRequest}, + push_queue_storage::{ClaimOutcome, EnqueueOutcome, EnqueueParams}, + }, + tests::test_db_connection, + }, +}; + +const INSTANCE: &str = "6ab219b0-4275-45ba-9d7b-7b0b633018cd"; + +async fn fixture() -> (tempfile::TempDir, MonoStorage, PushQueueStorage) { + let temp = tempfile::tempdir().unwrap(); + let db = test_db_connection(temp.path()).await; + apply_migrations(&db, false).await.unwrap(); + let base = BaseStorage::new(Arc::new(db)); + let mono = MonoStorage { base: base.clone() }; + for (id, path, commit, tree) in [(1_i64, "/", 'a', 'b'), (2, "/project", 'c', 'd')] { + mono.get_connection().execute_raw(Statement::from_sql_and_values( + mono.get_connection().get_database_backend(), + "INSERT INTO mega_refs (id,path,ref_name,ref_commit_hash,ref_tree_hash,created_at,updated_at,is_cl) \ + VALUES ($1,$2,$3,$4,$5,now(),now(),false)", + [id.into(), path.into(), MEGA_BRANCH_NAME.into(), commit.to_string().repeat(40).into(), tree.to_string().repeat(40).into()], + )).await.unwrap(); + } + let child_tree = "d".repeat(40); + let root_tree = "b".repeat(40); + let project = TreeItem::new( + TreeItemMode::Tree, + ObjectHash::from_hex_for_kind(get_hash_kind(), &child_tree).unwrap(), + "project".to_owned(), + ); + for (id, tree, sub_trees) in [ + (1_i64, child_tree, Vec::new()), + (2_i64, root_tree, project.to_data()), + ] { + mono.get_connection().execute_raw(Statement::from_sql_and_values( + mono.get_connection().get_database_backend(), + "INSERT INTO mega_tree (id,tree_id,sub_trees,size,created_at,pack_id,pack_offset,commit_id) \ + VALUES ($1,$2,$3,0,now(),$4,0,$5)", + [id.into(), tree.into(), sub_trees.into(), "".into(), "".into()], + )).await.unwrap(); + } + mono.initialize_native_publication(INSTANCE).await.unwrap(); + ( + temp, + mono, + PushQueueStorage::new(base).with_native_publication(true), + ) +} + +async fn enqueue_claim( + queue: &PushQueueStorage, + old: &str, + new: &str, + n: u64, +) -> push_queue::Model { + let operation = format!("{old}->{new}"); + let inserted = queue + .enqueue_atomic(EnqueueParams { + kind: PushQueueKindEnum::Push, + operation_id: &operation, + path: "/project", + old_id: old, + new_id: new, + requester: Some("alice"), + payload: serde_json::json!({"n":n,"commits":[new]}), + }) + .await + .unwrap(); + let EnqueueOutcome::Inserted { id } = inserted else { + panic!("fresh operation"); + }; + assert_eq!( + queue.claim_for_execution(id).await.unwrap(), + ClaimOutcome::Claimed + ); + queue.get_by_id(id).await.unwrap().unwrap() +} + +async fn publish_same_root( + mono: &MonoStorage, + txn: &DatabaseTransaction, + row: &push_queue::Model, +) -> i64 { + let request = PublicationRequest::from_trunk_queue(row).unwrap(); + let PublicationPreparation::Prepared(origin) = + mono.begin_publication_in_txn(txn, request).await.unwrap() + else { + panic!("fresh publication"); + }; + let native = mono + .reserve_native_publication_in_txn(txn, row, INSTANCE) + .await + .unwrap(); + assert!( + mono.cas_update_root_main_ref_in_txn( + txn, + row.expected_commit_hash.as_deref(), + row.expected_tree_hash.as_deref(), + row.expected_commit_hash.as_deref().unwrap(), + row.expected_tree_hash.as_deref().unwrap() + ) + .await + .unwrap() + ); + txn.execute_raw(Statement::from_sql_and_values(txn.get_database_backend(), + "UPDATE mega_refs SET ref_commit_hash = $1 WHERE path='/project' AND ref_name=$2 AND is_cl=false", + [row.new_id.clone().into(), MEGA_BRANCH_NAME.into()], + )).await.unwrap(); + let committed = mono + .record_publication_in_txn( + txn, + origin, + row.expected_commit_hash.as_deref().unwrap(), + &row.new_id, + ) + .await + .unwrap(); + mono.record_native_publication_in_txn(txn, native, &committed) + .await + .unwrap(); + txn.execute_raw(Statement::from_sql_and_values( + txn.get_database_backend(), + "UPDATE push_queue SET status='Done', landed_commit_id=$1 WHERE id=$2", + [row.new_id.clone().into(), row.id.into()], + )) + .await + .unwrap(); + committed.receipt.id +} + +#[tokio::test] +async fn same_root_changes_advance_global_head_and_historical_replay_ignores_latest() { + let (_temp, mono, queue) = fixture().await; + let root_before = mono.get_main_ref("/").await.unwrap().unwrap(); + let first = enqueue_claim(&queue, &"c".repeat(40), &"e".repeat(40), 1).await; + assert_eq!(first.expected_native_sequence, Some(0)); + assert_eq!(first.expected_native_epoch, Some(1)); + assert_eq!(first.expected_native_certificate, None); + let txn = mono.get_connection().begin().await.unwrap(); + let first_receipt = publish_same_root(&mono, &txn, &first).await; + txn.commit().await.unwrap(); + let first_head = mono.read_native_publication_head(INSTANCE).await.unwrap(); + assert_eq!(first_head.token.sequence, 1); + assert_eq!(first_head.root.commit, root_before.ref_commit_hash); + assert_eq!(first_head.root.tree, root_before.ref_tree_hash); + assert_eq!(first_head.token.certificate, Some(first_receipt)); + let second = enqueue_claim(&queue, &first.new_id, &"f".repeat(40), 1).await; + let txn = mono.get_connection().begin().await.unwrap(); + publish_same_root(&mono, &txn, &second).await; + txn.commit().await.unwrap(); + let txn = mono.get_connection().begin().await.unwrap(); + let replay = mono + .begin_publication_in_txn(&txn, PublicationRequest::from_trunk_queue(&first).unwrap()) + .await + .unwrap(); + assert!( + matches!(replay, PublicationPreparation::AlreadyCommitted(ref committed) if committed.receipt.id == first_receipt) + ); + txn.rollback().await.unwrap(); + assert_eq!( + mono.read_native_publication_head(INSTANCE) + .await + .unwrap() + .token + .sequence, + 2 + ); + assert_eq!( + mst2_native_publication::Entity::find() + .count(mono.get_connection()) + .await + .unwrap(), + 2 + ); + assert_eq!( + mst2_publication_outbox::Entity::find() + .count(mono.get_connection()) + .await + .unwrap(), + 2 + ); +} + +#[tokio::test] +async fn stale_same_root_claim_token_rejects_before_mutation_and_reset_clears_the_group() { + let (_temp, mono, queue) = fixture().await; + let row = enqueue_claim(&queue, &"c".repeat(40), &"e".repeat(40), 1).await; + // Independent transaction advances the head without changing either root OID. + mono.get_connection() + .execute_unprepared("UPDATE mst2_native_head SET sequence=1 WHERE namespace='/'") + .await + .unwrap(); + let txn = mono.get_connection().begin().await.unwrap(); + assert!(matches!( + mono.reserve_native_publication_in_txn(&txn, &row, INSTANCE) + .await, + Err(PublicationReceiptError::Conflict(_)) + )); + txn.rollback().await.unwrap(); + assert_eq!( + mono.get_main_ref("/project") + .await + .unwrap() + .unwrap() + .ref_commit_hash, + "c".repeat(40) + ); + assert!(queue.reset_running_to_queued(row.id).await.unwrap()); + let reset = queue.get_by_id(row.id).await.unwrap().unwrap(); + assert_eq!( + ( + reset.expected_commit_hash, + reset.expected_tree_hash, + reset.expected_native_sequence, + reset.expected_native_epoch, + reset.expected_native_certificate + ), + (None, None, None, None, None) + ); +} + +#[tokio::test] +async fn an_uncommitted_certificate_never_exposes_a_partial_head_to_another_connection() { + let (_temp, mono, queue) = fixture().await; + let first = enqueue_claim(&queue, &"c".repeat(40), &"e".repeat(40), 1).await; + let txn = mono.get_connection().begin().await.unwrap(); + publish_same_root(&mono, &txn, &first).await; + txn.commit().await.unwrap(); + let row = enqueue_claim(&queue, &first.new_id, &"f".repeat(40), 1).await; + let txn = mono.get_connection().begin().await.unwrap(); + publish_same_root(&mono, &txn, &row).await; + // The pool has two actual connections: one is held by txn, this read is on the other. + let before = tokio::time::timeout( + std::time::Duration::from_secs(5), + mono.read_native_publication_head(INSTANCE), + ) + .await + .unwrap() + .unwrap(); + assert_eq!(before.token.sequence, 1); + txn.commit().await.unwrap(); + let after = mono.read_native_publication_head(INSTANCE).await.unwrap(); + assert_eq!(after.token.sequence, 2); + assert_ne!(before.token.certificate, after.token.certificate); +} + +#[tokio::test] +async fn marked_history_fails_without_certificate_but_literal_v1_history_replays() { + let (_temp, mono, queue) = fixture().await; + let row = enqueue_claim(&queue, &"c".repeat(40), &"e".repeat(40), 1).await; + let txn = mono.get_connection().begin().await.unwrap(); + let receipt_id = publish_same_root(&mono, &txn, &row).await; + txn.commit().await.unwrap(); + mono.get_connection() + .execute_unprepared("DELETE FROM mst2_native_head; DELETE FROM mst2_native_publication;") + .await + .unwrap(); + let txn = mono.get_connection().begin().await.unwrap(); + assert!(matches!( + mono.begin_publication_in_txn(&txn, PublicationRequest::from_trunk_queue(&row).unwrap()) + .await, + Err(PublicationReceiptError::Integrity(_)) + )); + txn.rollback().await.unwrap(); + mono.get_connection() + .execute_raw(Statement::from_sql_and_values( + mono.get_connection().get_database_backend(), + "UPDATE mst2_publication SET native_certificate_version=NULL WHERE id=$1", + [receipt_id.into()], + )) + .await + .unwrap(); + let txn = mono.get_connection().begin().await.unwrap(); + assert!(matches!( + mono.begin_publication_in_txn(&txn, PublicationRequest::from_trunk_queue(&row).unwrap()) + .await + .unwrap(), + PublicationPreparation::AlreadyCommitted(_) + )); + txn.rollback().await.unwrap(); +} + +#[tokio::test] +async fn initialization_uses_only_a_counter_floor_and_cannot_reinitialize_or_serve_it() { + let (_temp, mono, queue) = fixture().await; + assert!(mono.read_native_publication_head(INSTANCE).await.is_err()); + assert!(mono.initialize_native_publication(INSTANCE).await.is_err()); + mono.get_connection().execute_unprepared("DELETE FROM mst2_native_head; INSERT INTO mst2_namespace_seq(namespace,sequence,epoch) VALUES('/older',19,1)").await.unwrap(); + mono.initialize_native_publication(INSTANCE).await.unwrap(); + let head = mst2_native_head::Entity::find() + .one(mono.get_connection()) + .await + .unwrap() + .unwrap(); + assert_eq!(head.sequence, 19); + assert_eq!(head.state, "INITIALIZING"); + assert_eq!(head.certificate_receipt_id, None); + let row = enqueue_claim(&queue, &"c".repeat(40), &"e".repeat(40), 1).await; + let txn = mono.get_connection().begin().await.unwrap(); + publish_same_root(&mono, &txn, &row).await; + txn.commit().await.unwrap(); + assert_eq!( + mono.read_native_publication_head(INSTANCE) + .await + .unwrap() + .token + .sequence, + 20 + ); +} + +#[tokio::test] +async fn reservation_owner_and_unchanged_selected_ref_cannot_issue_a_certificate() { + let (_temp, mono, queue) = fixture().await; + let row = enqueue_claim(&queue, &"c".repeat(40), &"e".repeat(40), 1).await; + let txn = mono.get_connection().begin().await.unwrap(); + let native = mono + .reserve_native_publication_in_txn(&txn, &row, INSTANCE) + .await + .unwrap(); + txn.rollback().await.unwrap(); + let other = mono.get_connection().begin().await.unwrap(); + assert!(matches!( + mono.finish_native_noop_in_txn(&other, native).await, + Err(PublicationReceiptError::Conflict(_)) + )); + other.rollback().await.unwrap(); + let txn = mono.get_connection().begin().await.unwrap(); + let native = mono + .reserve_native_publication_in_txn(&txn, &row, INSTANCE) + .await + .unwrap(); + let PublicationPreparation::Prepared(origin) = mono + .begin_publication_in_txn(&txn, PublicationRequest::from_trunk_queue(&row).unwrap()) + .await + .unwrap() + else { + panic!("fresh publication"); + }; + let committed = mono + .record_publication_in_txn( + &txn, + origin, + row.expected_commit_hash.as_deref().unwrap(), + &row.old_id, + ) + .await + .unwrap(); + assert!( + mono.record_native_publication_in_txn(&txn, native, &committed) + .await + .is_err() + ); + txn.rollback().await.unwrap(); + assert_eq!( + mst2_native_publication::Entity::find() + .count(mono.get_connection()) + .await + .unwrap(), + 0 + ); + assert_eq!( + mst2_publication_outbox::Entity::find() + .count(mono.get_connection()) + .await + .unwrap(), + 0 + ); +} + +#[tokio::test] +async fn missing_native_path_tree_keeps_head_initializing_and_rolls_back_receipt() { + let (_temp, mono, queue) = fixture().await; + let row = enqueue_claim(&queue, &"c".repeat(40), &"e".repeat(40), 1).await; + mono.get_connection() + .execute_raw(Statement::from_sql_and_values( + mono.get_connection().get_database_backend(), + "DELETE FROM mega_tree WHERE tree_id = $1", + ["d".repeat(40).into()], + )) + .await + .unwrap(); + + let txn = mono.get_connection().begin().await.unwrap(); + let request = PublicationRequest::from_trunk_queue(&row).unwrap(); + let PublicationPreparation::Prepared(origin) = + mono.begin_publication_in_txn(&txn, request).await.unwrap() + else { + panic!("fresh publication"); + }; + let native = mono + .reserve_native_publication_in_txn(&txn, &row, INSTANCE) + .await + .unwrap(); + assert!( + mono.cas_update_root_main_ref_in_txn( + &txn, + row.expected_commit_hash.as_deref(), + row.expected_tree_hash.as_deref(), + row.expected_commit_hash.as_deref().unwrap(), + row.expected_tree_hash.as_deref().unwrap(), + ) + .await + .unwrap() + ); + txn.execute_raw(Statement::from_sql_and_values( + txn.get_database_backend(), + "UPDATE mega_refs SET ref_commit_hash = $1 WHERE path='/project' AND ref_name=$2 AND is_cl=false", + [row.new_id.clone().into(), MEGA_BRANCH_NAME.into()], + )) + .await + .unwrap(); + let committed = mono + .record_publication_in_txn( + &txn, + origin, + row.expected_commit_hash.as_deref().unwrap(), + &row.new_id, + ) + .await + .unwrap(); + let error = mono + .record_native_publication_in_txn(&txn, native, &committed) + .await + .unwrap_err(); + assert!( + matches!(error, PublicationReceiptError::Integrity(message) if message.contains("path tree")) + ); + txn.rollback().await.unwrap(); + + assert!(mono.read_native_publication_head(INSTANCE).await.is_err()); + assert_eq!( + mst2_native_publication::Entity::find() + .count(mono.get_connection()) + .await + .unwrap(), + 0 + ); + assert_eq!( + mst2_publication_outbox::Entity::find() + .count(mono.get_connection()) + .await + .unwrap(), + 0 + ); +} + +#[cfg(unix)] +pub(super) fn crash_checkpoint(phase: &str) { + if std::env::var("MEGA_MST2_NATIVE_CRASH_PHASE") + .ok() + .as_deref() + == Some(phase) + { + unsafe { + libc::raise(libc::SIGKILL); + } + panic!("SIGKILL did not terminate the native worker"); + } +} + +async fn maintenance_fixture() -> (tempfile::TempDir, MonoStorage, PushQueueStorage, NativeRoot) { + use git_internal::{ + hash::HashKind, + internal::object::{ + blob::Blob, + commit::Commit, + tree::{Tree, TreeItem, TreeItemMode}, + }, + }; + + let (temp, mono, queue) = fixture().await; + mono.get_connection() + .execute_unprepared("DELETE FROM mst2_native_head") + .await + .unwrap(); + queue + .set_control_flags(Some(true), None, None) + .await + .unwrap(); + let blob = + Blob::from_content_bytes_with_kind(HashKind::Sha1, b"maintenance root".to_vec()).unwrap(); + let tree = Tree::from_tree_items_with_kind( + HashKind::Sha1, + vec![TreeItem::new( + TreeItemMode::Blob, + blob.id, + ".gitkeep".to_owned(), + )], + ) + .unwrap(); + let commit = + Commit::from_tree_id_with_kind(HashKind::Sha1, tree.id, vec![], "maintenance root") + .unwrap(); + mono.save_mega_trees(vec![tree.clone()], commit.id, None) + .await + .unwrap(); + mono.save_mega_commits(vec![commit.clone()], None) + .await + .unwrap(); + let root = NativeRoot { + commit: commit.id.to_string(), + tree: tree.id.to_string(), + }; + mono.get_connection().execute_raw(Statement::from_sql_and_values( + mono.get_connection().get_database_backend(), + "UPDATE mega_refs SET ref_commit_hash=$1,ref_tree_hash=$2 WHERE path='/' AND ref_name=$3 AND is_cl=false", + [root.commit.clone().into(), root.tree.clone().into(), MEGA_BRANCH_NAME.into()], + )).await.unwrap(); + (temp, mono, queue, root) +} + +async fn assert_no_maintenance_head(mono: &MonoStorage) { + assert_eq!( + mst2_native_head::Entity::find() + .count(mono.get_connection()) + .await + .unwrap(), + 0 + ); + assert_eq!( + mst2_native_publication::Entity::find() + .count(mono.get_connection()) + .await + .unwrap(), + 0 + ); + assert_eq!( + mst2_publication_outbox::Entity::find() + .count(mono.get_connection()) + .await + .unwrap(), + 0 + ); +} + +#[tokio::test] +async fn maintenance_initializes_only_a_floor_and_preserves_the_paused_gate() { + let (_temp, mono, queue, root) = maintenance_fixture().await; + mono.get_connection() + .execute_unprepared( + "INSERT INTO mst2_namespace_seq(namespace,sequence,epoch) VALUES('/older',23,1)", + ) + .await + .unwrap(); + mono.initialize_native_publication_for_maintenance(INSTANCE, &root) + .await + .unwrap(); + let before = mst2_native_head::Entity::find() + .one(mono.get_connection()) + .await + .unwrap() + .unwrap(); + assert_eq!(before.instance_id, INSTANCE); + assert_eq!(before.sequence, 23); + assert_eq!(before.writer_epoch, 1); + assert_eq!( + (&before.root_commit, &before.root_tree), + (&root.commit, &root.tree) + ); + assert_eq!(before.state, "INITIALIZING"); + assert_eq!(before.certificate_receipt_id, None); + assert!(queue.get_control().await.unwrap().paused); + assert!(mono.read_native_publication_head(INSTANCE).await.is_err()); + assert!( + mono.initialize_native_publication_for_maintenance(INSTANCE, &root) + .await + .is_err() + ); + assert!( + mono.initialize_native_publication_for_maintenance( + "11111111-2222-4333-8444-555555555555", + &root + ) + .await + .is_err() + ); + assert_eq!( + mst2_native_head::Entity::find() + .one(mono.get_connection()) + .await + .unwrap() + .unwrap(), + before + ); + assert_eq!( + selected_ref(mono.get_connection(), "/").await.unwrap(), + Some(root) + ); + assert_eq!( + mst2_native_publication::Entity::find() + .count(mono.get_connection()) + .await + .unwrap(), + 0 + ); + assert_eq!( + mst2_publication_outbox::Entity::find() + .count(mono.get_connection()) + .await + .unwrap(), + 0 + ); +} + +#[tokio::test] +async fn maintenance_rejects_unpaused_missing_or_hard_stopped_control_without_writes() { + for state in ["unpaused", "missing", "hard-stopped"] { + let (_temp, mono, queue, root) = maintenance_fixture().await; + match state { + "unpaused" => queue + .set_control_flags(Some(false), None, None) + .await + .unwrap(), + "hard-stopped" => queue + .set_control_flags(None, Some(true), None) + .await + .unwrap(), + "missing" => { + mono.get_connection() + .execute_unprepared("DELETE FROM queue_control") + .await + .unwrap(); + } + _ => unreachable!(), + } + assert!( + mono.initialize_native_publication_for_maintenance(INSTANCE, &root) + .await + .is_err(), + "{state}" + ); + assert_no_maintenance_head(&mono).await; + let control = crate::callisto::queue_control::Entity::find() + .one(mono.get_connection()) + .await + .unwrap(); + match state { + "missing" => assert!(control.is_none()), + "hard-stopped" => assert!(control.unwrap().hard_stopped), + _ => assert!(!control.unwrap().paused), + } + } +} + +#[tokio::test] +async fn maintenance_refuses_queued_and_running_work_until_drained() { + for status in ["Queued", "Running"] { + let (_temp, mono, queue, root) = maintenance_fixture().await; + queue + .set_control_flags(Some(false), None, None) + .await + .unwrap(); + let EnqueueOutcome::Inserted { id } = queue + .enqueue_atomic(EnqueueParams { + kind: PushQueueKindEnum::Push, + operation_id: "maintenance-pending", + path: "/project", + old_id: &"c".repeat(40), + new_id: &"e".repeat(40), + requester: None, + payload: serde_json::json!({"n":1,"commits":["e".repeat(40)]}), + }) + .await + .unwrap() + else { + panic!("fresh pending operation"); + }; + mono.get_connection() + .execute_raw(Statement::from_sql_and_values( + mono.get_connection().get_database_backend(), + "UPDATE push_queue SET status=$1::push_queue_status_enum WHERE id=$2", + [status.into(), id.into()], + )) + .await + .unwrap(); + queue + .set_control_flags(Some(true), None, None) + .await + .unwrap(); + assert!( + mono.initialize_native_publication_for_maintenance(INSTANCE, &root) + .await + .is_err(), + "{status}" + ); + assert_no_maintenance_head(&mono).await; + assert!(queue.get_control().await.unwrap().paused); + } +} + +#[tokio::test] +async fn maintenance_refuses_busy_writer_or_admission_locks_and_can_retry() { + for lock in ["writer", "admission"] { + let (_temp, mono, _queue, root) = maintenance_fixture().await; + let other = mono.get_connection().begin().await.unwrap(); + if lock == "writer" { + PushQueueStorage::acquire_mono_write_lock(&other) + .await + .unwrap(); + } else { + other + .execute_unprepared("SELECT id FROM queue_control WHERE id=1 FOR UPDATE") + .await + .unwrap(); + } + let result = tokio::time::timeout( + std::time::Duration::from_secs(5), + mono.initialize_native_publication_for_maintenance(INSTANCE, &root), + ) + .await; + assert!( + result + .expect("maintenance lock refusal must be bounded") + .is_err(), + "{lock}" + ); + assert_no_maintenance_head(&mono).await; + other.rollback().await.unwrap(); + mono.initialize_native_publication_for_maintenance(INSTANCE, &root) + .await + .unwrap(); + } +} + +#[tokio::test] +async fn maintenance_rejects_wrong_or_ambiguous_root_and_invalid_counters() { + for damage in [ + "commit", + "tree", + "missing", + "ambiguous", + "negative", + "exhausted", + ] { + let (_temp, mono, _queue, mut root) = maintenance_fixture().await; + match damage { + "commit" => root.commit = "e".repeat(40), + "tree" => root.tree = "e".repeat(40), + "missing" => { + mono.get_connection() + .execute_unprepared("DELETE FROM mega_refs WHERE path='/'") + .await + .unwrap(); + } + "ambiguous" => { + // Simulate a corrupted/old schema, independently of the normal + // uniqueness protection on selected refs. + mono.get_connection() + .execute_unprepared("DROP INDEX uniq_mref_path") + .await + .unwrap(); + mono.get_connection().execute_unprepared("INSERT INTO mega_refs(id,path,ref_name,ref_commit_hash,ref_tree_hash,created_at,updated_at,is_cl) SELECT 3,path,ref_name,ref_commit_hash,ref_tree_hash,created_at,updated_at,is_cl FROM mega_refs WHERE path='/'").await.unwrap(); + } + "negative" | "exhausted" => { + let floor = if damage == "negative" { + -1_i64 + } else { + i64::MAX + }; + mono.get_connection().execute_raw(Statement::from_sql_and_values( + mono.get_connection().get_database_backend(), + "INSERT INTO mst2_namespace_seq(namespace,sequence,epoch) VALUES('/older',$1,1)", [floor.into()], + )).await.unwrap(); + } + _ => unreachable!(), + } + assert!( + mono.initialize_native_publication_for_maintenance(INSTANCE, &root) + .await + .is_err(), + "{damage}" + ); + assert_no_maintenance_head(&mono).await; + } +} + +#[tokio::test] +async fn maintenance_rejects_nil_instance_and_non_sha1_expected_roots() { + let (_temp, mono, _queue, root) = maintenance_fixture().await; + for instance in ["invalid", "00000000-0000-0000-0000-000000000000"] { + assert!( + mono.initialize_native_publication_for_maintenance(instance, &root) + .await + .is_err() + ); + } + for commit in ["a".repeat(64), "A".repeat(40)] { + let wrong = NativeRoot { + commit, + tree: root.tree.clone(), + }; + assert!( + mono.initialize_native_publication_for_maintenance(INSTANCE, &wrong) + .await + .is_err() + ); + } + assert_no_maintenance_head(&mono).await; +} + +#[tokio::test] +async fn maintenance_refuses_missing_commit_tree_or_a_mismatched_commit_tree() { + for damage in ["commit", "tree", "wrong-tree"] { + let (_temp, mono, _queue, root) = maintenance_fixture().await; + let sql = match damage { + "commit" => "DELETE FROM mega_commit WHERE commit_id=$1", + "tree" => "DELETE FROM mega_tree WHERE tree_id=$1", + "wrong-tree" => "UPDATE mega_commit SET tree=repeat('e',40) WHERE commit_id=$1", + _ => unreachable!(), + }; + let id = if damage == "tree" { + &root.tree + } else { + &root.commit + }; + mono.get_connection() + .execute_raw(Statement::from_sql_and_values( + mono.get_connection().get_database_backend(), + sql, + [id.clone().into()], + )) + .await + .unwrap(); + assert!( + mono.initialize_native_publication_for_maintenance(INSTANCE, &root) + .await + .is_err(), + "{damage}" + ); + assert_no_maintenance_head(&mono).await; + assert_eq!( + selected_ref(mono.get_connection(), "/").await.unwrap(), + Some(root) + ); + } +} diff --git a/src/jupiter/storage/push_queue_storage.rs b/src/jupiter/storage/push_queue_storage.rs index 06878580..5be79779 100644 --- a/src/jupiter/storage/push_queue_storage.rs +++ b/src/jupiter/storage/push_queue_storage.rs @@ -8,7 +8,7 @@ use serde_json::Value as JsonValue; use crate::{ callisto::{ - authz_notify_outbox, mega_refs, push_queue, queue_control, + authz_notify_outbox, push_queue, queue_control, sea_orm_active_enums::{PushQueueKindEnum, PushQueueStatusEnum}, }, common::{errors::MegaError, utils::MEGA_BRANCH_NAME}, @@ -101,6 +101,7 @@ pub struct MonoWriteLockHolder { #[derive(Clone)] pub struct PushQueueStorage { base: BaseStorage, + native_publication_enabled: bool, } impl Deref for PushQueueStorage { @@ -123,7 +124,15 @@ pub struct EnqueueParams<'a> { impl PushQueueStorage { pub fn new(base: BaseStorage) -> Self { - Self { base } + Self { + base, + native_publication_enabled: false, + } + } + + pub(crate) fn with_native_publication(mut self, enabled: bool) -> Self { + self.native_publication_enabled = enabled; + self } /// B1: lock `queue_control`, conditional INSERT, classify on zero rows. @@ -241,8 +250,9 @@ impl PushQueueStorage { }); } // operation_states race: winner already committed — classify by read. + let replay_txn = self.get_connection().begin().await?; if let Some(existing) = push_queue::Entity::find() - .filter(push_queue::Column::Kind.eq(kind)) + .filter(push_queue::Column::Kind.eq(kind.clone())) .filter(push_queue::Column::Path.eq(path)) .filter(push_queue::Column::OperationId.eq(operation_id)) .filter(push_queue::Column::Status.is_in([ @@ -250,9 +260,23 @@ impl PushQueueStorage { PushQueueStatusEnum::Running, PushQueueStatusEnum::Done, ])) - .one(self.get_connection()) + .one(&replay_txn) .await? { + if matches!(&kind, PushQueueKindEnum::Push | PushQueueKindEnum::Merge) { + let mut candidate = existing.clone(); + candidate.old_id = old_id.to_owned(); + candidate.new_id = new_id.to_owned(); + candidate.requester = requester.map(str::to_owned); + candidate.payload = payload.clone(); + crate::jupiter::storage::mono_storage::MonoStorage { + base: self.base.clone(), + } + .validate_queue_publication_replay_in_txn(&replay_txn, &candidate) + .await + .map_err(|error| MegaError::Other(error.to_string()))?; + } + replay_txn.commit().await?; return match existing.status { PushQueueStatusEnum::Done => Ok(EnqueueOutcome::Replay { id: existing.id, @@ -264,6 +288,7 @@ impl PushQueueStorage { _ => unreachable!("filter restricts status"), }; } + replay_txn.rollback().await?; return Err(MegaError::Other(format!( "unique conflict without classifiable row: {msg}" ))); @@ -283,6 +308,29 @@ impl PushQueueStorage { Some(outcome) => { match &outcome { EnqueueOutcome::Adopted { .. } | EnqueueOutcome::Replay { .. } => { + if matches!(&kind, PushQueueKindEnum::Push | PushQueueKindEnum::Merge) { + let id = match &outcome { + EnqueueOutcome::Adopted { id } + | EnqueueOutcome::Replay { id, .. } => *id, + _ => unreachable!("adopt or replay"), + }; + let mut candidate = push_queue::Entity::find_by_id(id) + .one(&txn) + .await? + .ok_or_else(|| { + MegaError::Other("queue replay row missing".into()) + })?; + candidate.old_id = old_id.to_owned(); + candidate.new_id = new_id.to_owned(); + candidate.requester = requester.map(str::to_owned); + candidate.payload = payload.clone(); + crate::jupiter::storage::mono_storage::MonoStorage { + base: self.base.clone(), + } + .validate_queue_publication_replay_in_txn(&txn, &candidate) + .await + .map_err(|error| MegaError::Other(error.to_string()))?; + } txn.commit().await?; } EnqueueOutcome::Rejected { .. } => { @@ -722,66 +770,63 @@ impl PushQueueStorage { return Ok(ClaimOutcome::HardStopped); } - let claimed = txn - .execute_raw(Statement::from_sql_and_values( + // One statement snapshot supplies both the successful claim and its + // root/publication token. Updating the same queue row twice in a + // data-modifying CTE is deliberately avoided. + let sql = format!( + r#" + WITH {native_ctes}, claimed AS ( + UPDATE push_queue q + SET status = 'Running'::push_queue_status_enum, + started_at = now(), heartbeat_at = now(), updated_at = now(), + expected_commit_hash = observation.root_commit, + expected_tree_hash = observation.root_tree, + expected_native_sequence = observation.sequence, + expected_native_epoch = observation.writer_epoch, + expected_native_certificate = observation.certificate_receipt_id + FROM native_observation observation + WHERE q.id = $2 AND q.status = 'Queued'::push_queue_status_enum + AND (SELECT NOT hard_stopped FROM queue_control WHERE id = 1) + AND NOT EXISTS (SELECT 1 FROM push_queue WHERE status = 'Running'::push_queue_status_enum) + AND q.id = (SELECT min(id) FROM push_queue WHERE status = 'Queued'::push_queue_status_enum) + RETURNING q.id + ) + SELECT observation.* FROM native_observation observation + CROSS JOIN claimed + "#, + native_ctes = super::native_publication_storage::NATIVE_OBSERVATION_CTES + .replace("h.namespace = '/'", "h.namespace = '/' AND $3") + ); + let observations = txn + .query_all_raw(Statement::from_sql_and_values( DbBackend::Postgres, - r#" - UPDATE push_queue - SET status = 'Running'::push_queue_status_enum, - started_at = now(), - heartbeat_at = now(), - updated_at = now() - WHERE id = $1 - AND status = 'Queued'::push_queue_status_enum - AND (SELECT NOT hard_stopped FROM queue_control WHERE id = 1) - AND NOT EXISTS ( - SELECT 1 FROM push_queue - WHERE status = 'Running'::push_queue_status_enum - ) - AND id = ( - SELECT min(id) FROM push_queue - WHERE status = 'Queued'::push_queue_status_enum - ) - "#, - [Value::from(id)], + sql, + [ + MEGA_BRANCH_NAME.into(), + id.into(), + self.native_publication_enabled.into(), + ], )) .await?; - - if claimed.rows_affected() == 0 { + if observations.is_empty() { txn.rollback().await?; return Ok(ClaimOutcome::Missed); } - - // Snapshot root AFTER the claim UPDATE succeeds (still in this txn). - // Ordering matters: - // - Before UPDATE: a legitimate predecessor B3 can commit Done + new - // root in the gap → stale expected_* → false QueueBypassDetected. - // - FOR SHARE before UPDATE: blocks admissions while a writer holds - // the root row exclusively (stalls past wait_timeout). - // - After UPDATE: NOT EXISTS(Running) already held, so no legitimate - // B3 can still be writing; plain SELECT needs no row lock. - let root = mega_refs::Entity::find() - .filter(mega_refs::Column::Path.eq("/")) - .filter(mega_refs::Column::RefName.eq(MEGA_BRANCH_NAME.to_owned())) - .one(&txn) - .await?; - let (commit, tree) = match root { - Some(r) => (Some(r.ref_commit_hash), Some(r.ref_tree_hash)), - None => (None, None), - }; - - txn.execute_raw(Statement::from_sql_and_values( - DbBackend::Postgres, - r#" - UPDATE push_queue - SET expected_commit_hash = $2, - expected_tree_hash = $3, - updated_at = now() - WHERE id = $1 - "#, - [Value::from(id), Value::from(commit), Value::from(tree)], - )) - .await?; + if observations.len() != 1 { + return Err(MegaError::Other( + "ambiguous native claim observation".into(), + )); + } + if self.native_publication_enabled { + let observation = + super::native_publication_storage::decode_native_observation(&observations[0]) + .map_err(|error| MegaError::Other(error.to_string()))?; + if observation.head.is_none() { + return Err(MegaError::Other( + "native publication is not initialized".into(), + )); + } + } txn.commit().await?; Ok(ClaimOutcome::Claimed) @@ -899,6 +944,9 @@ impl PushQueueStorage { started_at = NULL, expected_commit_hash = NULL, expected_tree_hash = NULL, + expected_native_sequence = NULL, + expected_native_epoch = NULL, + expected_native_certificate = NULL, heartbeat_at = now(), updated_at = now() WHERE id = $1 @@ -985,6 +1033,9 @@ impl PushQueueStorage { started_at = NULL, expected_commit_hash = NULL, expected_tree_hash = NULL, + expected_native_sequence = NULL, + expected_native_epoch = NULL, + expected_native_certificate = NULL, pending_action = NULL, heartbeat_at = now(), updated_at = now() @@ -1363,9 +1414,12 @@ mod tests { use serde_json::json; use super::*; - use crate::jupiter::{ - migration::apply_migrations, storage::base_storage::StorageConnector, - tests::test_db_connection, + use crate::{ + callisto::mega_refs, + jupiter::{ + migration::apply_migrations, storage::base_storage::StorageConnector, + tests::test_db_connection, + }, }; async fn storage() -> (tempfile::TempDir, PushQueueStorage) { diff --git a/src/jupiter/tests.rs b/src/jupiter/tests.rs index ca684f5d..f79aeb41 100644 --- a/src/jupiter/tests.rs +++ b/src/jupiter/tests.rs @@ -359,7 +359,8 @@ pub async fn test_storage_with_config(temp_dir: impl AsRef, config: Config view_storage: ViewStorage::new(base.clone()), conversation_storage: ConversationStorage { base: base.clone() }, commit_binding_storage: CommitBindingStorage { base: base.clone() }, - push_queue_storage: PushQueueStorage::new(base.clone()), + push_queue_storage: PushQueueStorage::new(base.clone()) + .with_native_publication(config.mst2.publication_enabled), buck_storage: BuckStorage { base: base.clone() }, bots_storage: BotsStorage { base: base.clone() }, webhook_storage: WebhookStorage { base: base.clone() }, @@ -376,6 +377,8 @@ pub async fn test_storage_with_config(temp_dir: impl AsRef, config: Config Storage { app_service: Arc::new(svc), + native_projection_cache: Arc::default(), + projection_observation_sink: None, cl_service: CLService::mock(), push_queue_service: PushQueueService::new( base.clone(), @@ -383,7 +386,8 @@ pub async fn test_storage_with_config(temp_dir: impl AsRef, config: Config ) .with_view_signal(view_runtime.signal()) .with_timeouts(Duration::from_secs(30), Duration::from_millis(20)) - .with_max_push_commits(config.monorepo.max_push_commits), + .with_max_push_commits(config.monorepo.max_push_commits) + .with_native_publication(config.mst2.publication_enabled), artifact_service: ArtifactService::new(base.clone(), mock_object_storage()), buck_service: BuckService::mock(), config_handle: ConfigHandle::from_arc(config.clone()), diff --git a/tests/common/git_cli.rs b/tests/common/git_cli.rs index 9f5078a6..b2fa7c2a 100644 --- a/tests/common/git_cli.rs +++ b/tests/common/git_cli.rs @@ -1283,12 +1283,7 @@ pub fn ensure_git_cli_passwd_for_runtime_uid() { "unexpected git-cli id output: {id_pair}" ); - let script = format!( - "if getent passwd {uid} >/dev/null 2>&1; then exit 0; fi; \ - if ! getent group {gid} >/dev/null 2>&1; then addgroup -g {gid} gitcliruntime || true; fi; \ - group_name=\"$(getent group {gid} | cut -d: -f1)\"; \ - adduser -D -u {uid} -G \"$group_name\" -h /home/gitcli -s /bin/sh gitcliruntime" - ); + let script = git_cli_runtime_passwd_script(&uid, &gid); let mut command = docker_exec_base(); command .arg("-u") @@ -1303,6 +1298,81 @@ pub fn ensure_git_cli_passwd_for_runtime_uid() { ); } +fn git_cli_runtime_passwd_script(uid: &str, gid: &str) -> String { + assert!( + !uid.is_empty() + && !gid.is_empty() + && uid.bytes().all(|byte| byte.is_ascii_digit()) + && gid.bytes().all(|byte| byte.is_ascii_digit()), + "git-cli runtime uid/gid must be numeric" + ); + format!( + r#"passwd_matches() {{ + getent passwd {uid} | awk -F: -v uid={uid} -v gid={gid} \ + '$3 == uid && $4 == gid {{ matched++ }} END {{ exit matched != 1 }}' + }} + if getent passwd {uid} >/dev/null 2>&1; then passwd_matches; exit $?; fi + if ! getent group {gid} >/dev/null 2>&1; then addgroup -g {gid} gitcliruntime || true; fi + group_name="$(getent group {gid} | cut -d: -f1)" + test -n "$group_name" || exit 1 + if adduser -D -u {uid} -G "$group_name" -h /home/gitcli -s /bin/sh gitcliruntime; then + passwd_matches + else + status=$? + # Another SSH test can win the check/create race. Accept only the + # exact runtime identity, never an unrelated name or primary GID. + if passwd_matches; then exit 0; fi + exit "$status" + fi"# + ) +} + +#[cfg(target_os = "linux")] +#[test] +fn git_cli_passwd_creation_race_requires_exact_runtime_identity() { + let fixture = r#" + getent() { + if [ "$1" = group ]; then printf 'gitcliruntime:x:1001:\n'; return 0; fi + if [ "$SCENARIO" = existing ]; then printf 'gitcliruntime:x:1001:1001::/home/gitcli:/bin/sh\n'; return 0; fi + if [ "$SCENARIO" = existing_wrong_gid ]; then printf 'gitcliruntime:x:1001:1000::/home/gitcli:/bin/sh\n'; return 0; fi + if [ ! -f "$MOCK_STATE" ]; then return 2; fi + cat "$MOCK_STATE" + } + adduser() { + case "$SCENARIO" in + raced|created) printf 'gitcliruntime:x:1001:1001::/home/gitcli:/bin/sh\n' > "$MOCK_STATE" ;; + wrong_uid) printf 'gitcliruntime:x:1000:1001::/home/gitcli:/bin/sh\n' > "$MOCK_STATE" ;; + wrong_gid|created_wrong_gid) printf 'gitcliruntime:x:1001:1000::/home/gitcli:/bin/sh\n' > "$MOCK_STATE" ;; + esac + if [ "$SCENARIO" = created ] || [ "$SCENARIO" = created_wrong_gid ]; then return 0; fi + return 17 + } + "#; + let temp = tempfile::tempdir().expect("passwd race fixture"); + let script = format!( + "{fixture}\n{}", + git_cli_runtime_passwd_script("1001", "1001") + ); + for (scenario, expected) in [ + ("existing", 0), + ("existing_wrong_gid", 1), + ("raced", 0), + ("created", 0), + ("missing", 17), + ("wrong_uid", 17), + ("wrong_gid", 17), + ("created_wrong_gid", 1), + ] { + let mut command = Command::new("sh"); + command + .args(["-c", &script]) + .env("SCENARIO", scenario) + .env("MOCK_STATE", temp.path().join(scenario)); + let output = output_with_timeout(command, Duration::from_secs(5), "passwd race fixture"); + assert_eq!(output.status.code(), Some(expected), "{scenario}"); + } +} + /// Build ADR-GM-05 `GIT_SSH_COMMAND` for the selected runner (container paths under `/work`). #[allow( dead_code, diff --git a/tests/integration_git_cli.rs b/tests/integration_git_cli.rs index 41e5b509..8f35c974 100644 --- a/tests/integration_git_cli.rs +++ b/tests/integration_git_cli.rs @@ -20,7 +20,6 @@ use std::{ collections::BTreeMap, fs, io::{Read, Write}, - net::TcpStream, path::{Path, PathBuf}, process::{Child, Command, ExitStatus, Stdio}, sync::atomic::{AtomicUsize, Ordering}, @@ -204,34 +203,51 @@ impl ServiceProcess { service } - fn wait_until_listening( + fn wait_until_openapi_ready( &mut self, port: u16, timeout: Duration, stdout_path: &Path, stderr_path: &Path, ) { + let client = reqwest::blocking::Client::builder() + .timeout(Duration::from_secs(5)) + .redirect(reqwest::redirect::Policy::none()) + .build() + .expect("build HTTP readiness client"); + let url = format!("http://127.0.0.1:{port}/api/openapi.json"); let deadline = Instant::now() + timeout; loop { - if TcpStream::connect(("127.0.0.1", port)).is_ok() { - return; - } if let Some(status) = self.child.try_wait().expect("poll service") { self.reaped = true; panic!( - "service exited before binding port {port} (status {status})\nstdout:\n{}\nstderr:\n{}", + "service exited before OpenAPI was ready on port {port} (status {status})\nstdout:\n{}\nstderr:\n{}", read_log(stdout_path), read_log(stderr_path), ); } - if Instant::now() >= deadline { + let remaining = deadline.saturating_duration_since(Instant::now()); + if remaining.is_zero() { panic!( - "service did not bind port {port} within {timeout:?}\nstdout:\n{}\nstderr:\n{}", + "service OpenAPI not ready on port {port} within {timeout:?}\nstdout:\n{}\nstderr:\n{}", read_log(stdout_path), read_log(stderr_path), ); } - sleep(Duration::from_millis(200)); + if let Ok(response) = client + .get(&url) + .timeout(remaining.min(Duration::from_secs(5))) + .send() + && response.status() == reqwest::StatusCode::OK + && Instant::now() <= deadline + { + return; + } + sleep( + deadline + .saturating_duration_since(Instant::now()) + .min(Duration::from_millis(200)), + ); } } @@ -345,7 +361,7 @@ fn boot_service_http_with_env( .stderr(Stdio::from(create_log_file(&stderr_path))); let mut service = ServiceProcess::spawn(command); - service.wait_until_listening(port, Duration::from_secs(90), &stdout_path, &stderr_path); + service.wait_until_openapi_ready(port, Duration::from_secs(90), &stdout_path, &stderr_path); (service, port, stdout_path, stderr_path) } diff --git a/tests/integration_storage_events_runtime.rs b/tests/integration_storage_events_runtime.rs index 9443a443..ab8c4908 100644 --- a/tests/integration_storage_events_runtime.rs +++ b/tests/integration_storage_events_runtime.rs @@ -363,12 +363,6 @@ fn integration_storage_events_runtime_occupied_port_exits_through_cleanup() { // evidence that the cleanup tail ran. let receipt = wait_for_shutdown_receipt(&stdout_path, &stderr_path, Duration::from_secs(60)); assert_shutdown_receipt_sanitized(&receipt); - // The failure must be the port bind, not some unrelated startup error. - let combined = format!("{}\n{}", read_log(&stdout_path), read_log(&stderr_path)); - assert!( - combined.contains("failed to bind HTTP listener"), - "occupied-port case must fail at the HTTP bind:\n{combined}" - ); let status = service .wait_for_exit(Duration::from_secs(60)) .expect("occupied-port startup failure must exit on its own"); @@ -378,6 +372,13 @@ fn integration_storage_events_runtime_occupied_port_exits_through_cleanup() { read_log(&stdout_path), read_log(&stderr_path), ); + // main prints the startup error after cleanup returns; wait for exit + // before reading the final diagnostic that must identify the bind failure. + let combined = format!("{}\n{}", read_log(&stdout_path), read_log(&stderr_path)); + assert!( + combined.contains("failed to bind HTTP listener"), + "occupied-port case must fail at the HTTP bind:\n{combined}" + ); } #[test] From 142a5b13a06c7b41384b05e28924424f38330a68 Mon Sep 17 00:00:00 2001 From: Xiaoyang Han Date: Wed, 7 Oct 2026 17:34:54 +0800 Subject: [PATCH 02/50] test(mst2): inject and verify actual fixed-content storage faults Git SinglePut preserves existing immutable objects. In the isolated test fixture, remove the owned object before fault injection or repair, read back the exact bytes, and exclude setup reads from HTTP counters. Preserve all corruption, missing-body, retry and cache assertions. --- src/api/router/snapshot_content_tests.rs | 23 ++++++++++++++++++++++- 1 file changed, 22 insertions(+), 1 deletion(-) diff --git a/src/api/router/snapshot_content_tests.rs b/src/api/router/snapshot_content_tests.rs index 72590edb..f9643d92 100644 --- a/src/api/router/snapshot_content_tests.rs +++ b/src/api/router/snapshot_content_tests.rs @@ -536,12 +536,33 @@ impl Fixture { } async fn write_raw(&self, bytes: Vec) { + // Git SinglePut is create-if-absent. Remove only this isolated + // fixture's object so corruption and repair actually change its bytes. + let objects = &self.state.storage.git_service.obj_storage.inner; + let key = ObjectKey { + namespace: ObjectNamespace::Git, + key: self.oid.clone(), + }; + if objects.exists(&key).await.unwrap() { + objects.delete(&key).await.unwrap(); + } self.state .storage .git_service - .save_object_from_model(bytes, &self.oid) + .save_object_from_model(bytes.clone(), &self.oid) .await .unwrap(); + assert_eq!( + self.state + .storage + .git_service + .get_object_as_bytes(&self.oid) + .await + .unwrap(), + bytes + ); + // Fault setup is outside the HTTP read measurement window. + self.counts.reset(); } fn chunk_body(&self, path: &str, map_id: &str, index: &str) -> Value { From 723cd3ed8e383901ff78fa75949c86339d2f46f7 Mon Sep 17 00:00:00 2001 From: Xiaoyang Han Date: Wed, 7 Oct 2026 18:08:05 +0800 Subject: [PATCH 03/50] feat(mst2): persist native HTTP snapshot sessions and guard delivery (#56) Use primary PostgreSQL authority for fixed native snapshot sources and per-lease certificates. Atomically hand off installed metadata protection to durable sessions and leases, retain only roots on warm resolve, and retire session pins with the last lease. Recheck current lease, source and captured primary scope before each actual TreeFrame or shared raw block. Add real HTTP/PostgreSQL regressions; native validation and performance remain pending. --- src/api/router/snapshot_content.rs | 44 +- src/api/router/snapshot_content_tests.rs | 69 +- src/api/router/snapshot_router.rs | 320 ++++- src/api/router/snapshot_session_tests.rs | 1121 +++++++++++++++++ src/callisto/mod.rs | 3 + src/callisto/mst2_snapshot_context.rs | 24 + src/callisto/mst2_snapshot_lease.rs | 20 + ...61007_000100_add_mst2_snapshot_sessions.rs | 72 ++ src/jupiter/migration/mod.rs | 5 +- src/jupiter/storage/mod.rs | 5 + src/jupiter/storage/mst2_retention.rs | 47 + .../storage/native_metadata_install.rs | 303 ++++- .../storage/native_publication_storage.rs | 11 +- .../storage/native_snapshot_session.rs | 678 ++++++++++ src/jupiter/tests.rs | 1 + 15 files changed, 2606 insertions(+), 117 deletions(-) create mode 100644 src/api/router/snapshot_session_tests.rs create mode 100644 src/callisto/mst2_snapshot_context.rs create mode 100644 src/callisto/mst2_snapshot_lease.rs create mode 100644 src/jupiter/migration/m20261007_000100_add_mst2_snapshot_sessions.rs create mode 100644 src/jupiter/storage/native_snapshot_session.rs diff --git a/src/api/router/snapshot_content.rs b/src/api/router/snapshot_content.rs index 08158723..d1dd2a15 100644 --- a/src/api/router/snapshot_content.rs +++ b/src/api/router/snapshot_content.rs @@ -13,7 +13,9 @@ use futures::stream::StreamExt; use serde::Deserialize; use serde_json::json; -use super::{abs_view_path, internal, mst2_error_response, request::Mst2Bytes, treeframe_response}; +use super::{ + abs_view_path, guarded_treeframe_response, internal, mst2_error_response, request::Mst2Bytes, +}; use crate::ceres::snapshot::{ chunks::{ChunkProjection, get_or_project}, error::{SnapshotError, SnapshotErrorCode}, @@ -22,7 +24,6 @@ use crate::ceres::snapshot::{ resolve_abs_metadata, }, resolver::FsKind, - runtime::runtime, view::validate_scope_relative_path, }; @@ -226,9 +227,7 @@ pub(super) async fn blob_head( headers: HeaderMap, ) -> Result { ensure(&state)?; - let ctx = runtime() - .context(&snapshot_id) - .map_err(mst2_error_response)?; + let ctx = super::request_context(&state, &snapshot_id).map_err(mst2_error_response)?; validate_scope_relative_path(&q.path).map_err(mst2_error_response)?; if headers.contains_key("range") { return Err(mst2_error_response(SnapshotError::new( @@ -286,9 +285,7 @@ pub(super) async fn objects( Mst2Bytes(body): Mst2Bytes, ) -> Result { ensure(&state)?; - let ctx = runtime() - .context(&snapshot_id) - .map_err(mst2_error_response)?; + let ctx = super::request_context(&state, &snapshot_id).map_err(mst2_error_response)?; let req: ObjectsRequest = super::parse_json_body(&body)?; if req.items.is_empty() || req.items.len() > OBJECT_MAX_ITEMS { return Err(mst2_error_response(SnapshotError::new( @@ -365,7 +362,7 @@ pub(super) async fn objects( // then exactly one END. Encoding is negotiated per request. use crate::ceres::snapshot::frame_stream::FrameStream; let mut stream = FrameStream::new(1, encoding); - let mut out: Vec = Vec::new(); + let mut out: Vec> = Vec::new(); let mut frame: Vec<([u8; 32], Vec)> = Vec::new(); let mut frame_raw = 0usize; for (cid, data) in unique { @@ -377,7 +374,7 @@ pub(super) async fn objects( let bytes = stream .object(std::mem::take(&mut frame)) .map_err(mst2_error_response)?; - out.extend_from_slice(&bytes); + out.push(bytes); frame_raw = 0; } frame_raw += data.len(); @@ -385,18 +382,17 @@ pub(super) async fn objects( } if !frame.is_empty() { let bytes = stream.object(frame).map_err(mst2_error_response)?; - out.extend_from_slice(&bytes); + out.push(bytes); } - let end = stream.end( req.items.len() as u32, seen.len() as u32, logical_bytes, sha256_of(&body), ); - out.extend_from_slice(&end); + out.push(end); - treeframe_response(&snapshot_id, &body, out).map_err(mst2_error_response) + guarded_treeframe_response(&state, &ctx, &snapshot_id, &body, out).map_err(mst2_error_response) } #[derive(Deserialize, Debug)] @@ -467,9 +463,7 @@ pub(super) async fn chunk_map( Query(q): Query, ) -> Result { ensure(&state)?; - let ctx = runtime() - .context(&snapshot_id) - .map_err(mst2_error_response)?; + let ctx = super::request_context(&state, &snapshot_id).map_err(mst2_error_response)?; validate_scope_relative_path(&q.path).map_err(mst2_error_response)?; let handler = state .api_handler(std::path::Path::new("/")) @@ -513,9 +507,7 @@ pub(super) async fn chunk_map_pages( Query(q): Query, ) -> Result { ensure(&state)?; - let ctx = runtime() - .context(&snapshot_id) - .map_err(mst2_error_response)?; + let ctx = super::request_context(&state, &snapshot_id).map_err(mst2_error_response)?; validate_scope_relative_path(&q.path).map_err(mst2_error_response)?; // Canonical v3 uses `map_id` + `page_index`; retain parsing of the old // names only while the legacy client is still present in this checkout. @@ -622,9 +614,7 @@ pub(super) async fn chunks( Mst2Bytes(body): Mst2Bytes, ) -> Result { ensure(&state)?; - let ctx = runtime() - .context(&snapshot_id) - .map_err(mst2_error_response)?; + let ctx = super::request_context(&state, &snapshot_id).map_err(mst2_error_response)?; let req: ChunksRequest = super::parse_json_body(&body)?; if req.items.is_empty() || req.items.len() > CHUNKS_MAX_ITEMS { return Err(mst2_error_response(SnapshotError::new( @@ -707,7 +697,7 @@ pub(super) async fn chunks( use crate::ceres::snapshot::frame_stream::FrameStream; let mut stream = FrameStream::new(1, encoding); - let mut out: Vec = Vec::new(); + let mut out: Vec> = Vec::new(); for p in planned.iter() { // Re-verified slice (digest + length) from the staged projection. let bytes = p @@ -722,7 +712,7 @@ pub(super) async fn chunks( bytes.to_vec(), ) .map_err(mst2_error_response)?; - out.extend_from_slice(&frame); + out.push(frame); } let end = stream.end( req.items.len() as u32, @@ -730,9 +720,9 @@ pub(super) async fn chunks( logical_bytes, sha256_of(&body), ); - out.extend_from_slice(&end); + out.push(end); - treeframe_response(&snapshot_id, &body, out).map_err(mst2_error_response) + guarded_treeframe_response(&state, &ctx, &snapshot_id, &body, out).map_err(mst2_error_response) } /// Strict decimal-string parse for unsigned counts (spec 04 §1: no leading diff --git a/src/api/router/snapshot_content_tests.rs b/src/api/router/snapshot_content_tests.rs index f9643d92..aa697589 100644 --- a/src/api/router/snapshot_content_tests.rs +++ b/src/api/router/snapshot_content_tests.rs @@ -63,7 +63,10 @@ use crate::{ object_storage::{MegaObjectStorageWrapper, build_object_storage}, push_queue_storage::{ClaimOutcome, EnqueueOutcome}, }, - tests::{test_redis_manager, test_storage_with_config, with_test_vault}, + tests::{ + TestSchemaGuard, test_db_config, test_redis_manager, test_storage_with_config, + with_test_vault, + }, }, orbit_api::{ error::OrbitResult, @@ -76,6 +79,9 @@ use crate::{ const TOKEN: &str = "mst2-fixed-content-test"; +#[path = "snapshot_session_tests.rs"] +mod durable_sessions; + #[derive(Default)] struct ReadCounts { whole: AtomicUsize, @@ -214,6 +220,7 @@ struct Fixture { digest: [u8; 32], counts: Arc, _temp: tempfile::TempDir, + _schema: Option, } fn tree(items: Vec) -> Tree { @@ -301,6 +308,14 @@ async fn publish_native_push( impl Fixture { async fn new() -> Self { + Self::new_with_pg_config(false).await + } + + async fn new_with_pg_config(rebuildable: bool) -> Self { + Self::new_with_pg_config_and_directories(rebuildable, 0).await + } + + async fn new_with_pg_config_and_directories(rebuildable: bool, directory_count: usize) -> Self { let temp = tempfile::tempdir().unwrap(); let mut config = isolated_config(temp.path().join("config")); config.monorepo.push_policy = PushPolicy::Trunk; @@ -310,7 +325,25 @@ impl Fixture { config.mst2.auth_token = Some(TOKEN.to_string()); let backend = build_object_storage(&config.object_storage).await.unwrap(); let counts = Arc::new(ReadCounts::default()); - let mut storage = test_storage_with_config(temp.path(), config).await; + let (mut storage, schema) = if rebuildable { + let (database, schema) = test_db_config(temp.path()).await; + config.database = database; + let connection = crate::jupiter::storage::init::database_connection(&config.database) + .await + .unwrap(); + ( + crate::jupiter::storage::Storage::new_with_connection( + Arc::new(config), + Arc::new(connection), + backend.clone(), + ) + .await + .unwrap(), + Some(schema), + ) + } else { + (test_storage_with_config(temp.path(), config).await, None) + }; storage.git_service = GitService { obj_storage: MegaObjectStorageWrapper::new(Arc::new(CountingStorage { inner: backend, @@ -344,7 +377,7 @@ impl Fixture { let empty_oid = ObjectHash::from_hex_for_kind(HashKind::Sha1, &empty_oid).unwrap(); let empty_dir = tree(vec![]); let nested = tree(vec![item(TreeItemMode::Blob, blob_oid, "file")]); - let project = tree(vec![ + let mut project_items = vec![ item(TreeItemMode::Blob, blob_oid, "alias"), item(TreeItemMode::Tree, empty_dir.id, "directory"), item(TreeItemMode::Blob, empty_oid, "empty"), @@ -352,7 +385,24 @@ impl Fixture { item(TreeItemMode::Blob, blob_oid, "file"), item(TreeItemMode::Link, link_oid, "link"), item(TreeItemMode::Tree, nested.id, "nested"), - ]); + ]; + let mut extra_trees = Vec::new(); + for index in 0..directory_count { + // Distinct names make distinct canonical pages even though all + // directories share the same already-verified immutable blob. + let child = tree(vec![item( + TreeItemMode::Blob, + blob_oid, + &format!("file-{index:03}"), + )]); + project_items.push(item( + TreeItemMode::Tree, + child.id, + &format!("wide-{index:03}"), + )); + extra_trees.push(child); + } + let project = tree(project_items); let old_tip = Commit::from_tree_id_with_kind( HashKind::Sha1, project.id, @@ -379,13 +429,9 @@ impl Fixture { ) .unwrap(); let mono = storage.mono_storage(); - mono.save_mega_trees( - vec![empty_dir, nested, project, root.clone()], - commit.id, - None, - ) - .await - .unwrap(); + let mut trees = vec![empty_dir, nested, project, root.clone()]; + trees.extend(extra_trees); + mono.save_mega_trees(trees, commit.id, None).await.unwrap(); mono.save_mega_commits(vec![commit.clone(), old_tip.clone(), new_tip.clone()], None) .await .unwrap(); @@ -474,6 +520,7 @@ impl Fixture { raw, counts, _temp: temp, + _schema: schema, } } diff --git a/src/api/router/snapshot_router.rs b/src/api/router/snapshot_router.rs index 3cfa4963..4c2604d2 100644 --- a/src/api/router/snapshot_router.rs +++ b/src/api/router/snapshot_router.rs @@ -101,6 +101,14 @@ pub(crate) fn treeframe_response( snapshot_id: &str, request_body: &[u8], body: Vec, +) -> Result { + treeframe_response_body(snapshot_id, request_body, axum::body::Body::from(body)) +} + +fn treeframe_response_body( + snapshot_id: &str, + request_body: &[u8], + body: axum::body::Body, ) -> Result { let request_digest: [u8; 32] = Sha256::digest(request_body).into(); Response::builder() @@ -112,20 +120,58 @@ pub(crate) fn treeframe_response( ) .header("cache-control", "private, no-cache, no-transform") .header("vary", "Authorization, Accept") - .body(axum::body::Body::from(Bytes::from(body))) + .body(body) .map_err(|e| internal(format!("TreeFrame response build failed: {e}"))) } +fn guarded_treeframe_response( + state: &MonoApiServiceState, + context: &crate::ceres::snapshot::runtime::SnapshotContext, + snapshot_id: &str, + request_body: &[u8], + frames: Vec>, +) -> Result { + let headers = REQUEST_HEADERS.try_with(Clone::clone).map_err(|_| { + SnapshotError::new( + SnapshotErrorCode::Unauthenticated, + "request authentication context missing", + ) + })?; + let units = frames + .into_iter() + .map(Bytes::from) + .collect::>(); + let stream = futures::stream::unfold( + (units, state.clone(), context.clone(), headers), + |(mut units, state, context, headers)| async move { + let unit = units.pop_front()?; + if let Err(error) = revalidate_access(&state, &context, &headers).await { + units.clear(); + return Some((Err(error), (units, state, context, headers))); + } + Some((Ok(unit), (units, state, context, headers))) + }, + ); + treeframe_response_body( + snapshot_id, + request_body, + axum::body::Body::from_stream(stream), + ) +} + tokio::task_local! { /// Per-request id for the error envelope (spec 14 §5). Sourced from the /// global `TraceContext` so the envelope, the `X-Request-Id` response /// header and log spans all carry the same id. static REQUEST_ID: String; + static REQUEST_CONTEXT: Option; + static REQUEST_HEADERS: HeaderMap; } #[cfg(test)] tokio::task_local! { static NATIVE_RESOLVE_BARRIERS: (std::sync::Arc, std::sync::Arc); + static NATIVE_HANDOFF_BARRIERS: (std::sync::Arc, std::sync::Arc); static REJECT_NATIVE_OBSERVATION_SOURCE: bool; } @@ -147,6 +193,17 @@ pub(crate) async fn with_native_resolve_barriers( .await } +#[cfg(test)] +pub(crate) async fn with_native_handoff_barriers( + prepared: std::sync::Arc, + release: std::sync::Arc, + future: F, +) -> F::Output { + NATIVE_HANDOFF_BARRIERS + .scope((prepared, release), future) + .await +} + /// The request id of the in-flight request, for error envelopes. pub(crate) fn current_request_id() -> String { REQUEST_ID.try_with(|id| id.clone()).unwrap_or_default() @@ -158,7 +215,7 @@ const LEASE_HEADER: &str = "x-mega-snapshot-lease"; /// every endpoint except `capabilities` requires `Authorization: Bearer /// ` when the deployment configured one, and snapshot-bound endpoints /// must additionally present the lease they resolved -/// (`X-Mega-Snapshot-Lease`), validated against the in-memory lease table — +/// (`X-Mega-Snapshot-Lease`), validated against the session authority — /// knowing the snapshot id alone is not a capability. async fn snapshot_auth_middleware( State(state): State, @@ -178,10 +235,41 @@ async fn snapshot_auth_middleware( }); let mut response = REQUEST_ID .scope(id.to_string(), async { - if let Some(res) = auth_error(&state, req.headers(), req.uri().path()) { - return res; - } - next.run(req).await + let headers = req.headers().clone(); + let path = req.uri().path().to_owned(); + let context = match authenticate_request(&state, &headers, &path).await { + Ok(context) => context, + Err(response) => return response, + }; + REQUEST_HEADERS + .scope( + headers.clone(), + REQUEST_CONTEXT.scope(context.clone(), async { + let response = next.run(req).await; + if response.status().is_success() + || response.status() == StatusCode::NOT_MODIFIED + { + match context { + Some(context) => { + if let Err(error) = + revalidate_access(&state, &context, &headers).await + { + return mst2_error_response(error); + } + } + None => { + if let Err(response) = + authenticate_request(&state, &headers, &path).await + { + return response; + } + } + } + } + response + }), + ) + .await }) .await; if let Ok(value) = HeaderValue::from_str(&id) { @@ -245,7 +333,11 @@ fn unauthenticated(message: &'static str) -> Response { /// Auth decision for one request path (see [`snapshot_auth_middleware`]). /// `capabilities` stays open; lease routes need only the bearer; a /// snapshot-bound route (`/snapshots/sha256:…/…`) also needs its lease. -fn auth_error(state: &MonoApiServiceState, headers: &HeaderMap, path: &str) -> Option { +async fn authenticate_request( + state: &MonoApiServiceState, + headers: &HeaderMap, + path: &str, +) -> Result, Response> { let config = state.storage.config(); let token = config.mst2.auth_token.as_deref(); let path = path.split('?').next().unwrap_or(path); @@ -253,32 +345,105 @@ fn auth_error(state: &MonoApiServiceState, headers: &HeaderMap, path: &str) -> O // middleware runs — accept both the stripped and full forms. let rest = path .strip_prefix("/api/v2/snapshots/") - .or_else(|| path.strip_prefix("/snapshots/"))?; + .or_else(|| path.strip_prefix("/snapshots/")); + let Some(rest) = rest else { + return Ok(None); + }; if rest == "capabilities" { - return None; + return Ok(None); } if let Some(token) = token && !bearer_ok(headers, token) { - return Some(unauthenticated("missing or invalid bearer credentials")); + return Err(unauthenticated("missing or invalid bearer credentials")); } let mut segments = rest.split('/'); match segments.next() { // Lease management names the lease in the path, not a snapshot. - Some("resolve") | Some("leases") | Some("capabilities") | None => None, + Some("resolve") | Some("leases") | Some("capabilities") | None => Ok(None), Some(snapshot_id) => { let lease = headers.get(LEASE_HEADER).and_then(|v| v.to_str().ok()); match lease { - None => Some(unauthenticated("missing X-Mega-Snapshot-Lease header")), - Some(lease) => runtime() - .validate_lease(snapshot_id, lease) - .err() - .map(mst2_error_response), + None => Err(unauthenticated("missing X-Mega-Snapshot-Lease header")), + Some(lease) => state + .storage + .snapshot_context(snapshot_id, lease) + .await + .map(Some) + .map_err(mst2_error_response), } } } } +fn request_context( + state: &MonoApiServiceState, + snapshot_id: &str, +) -> Result { + if let Ok(Some(context)) = REQUEST_CONTEXT.try_with(Clone::clone) + && context.built.snapshot_id == snapshot_id + { + return Ok(context); + } + if !state.storage.config().mst2.publication_enabled { + return runtime().context(snapshot_id); + } + Err(SnapshotError::new( + SnapshotErrorCode::Unauthenticated, + "validated snapshot session missing", + )) +} + +async fn revalidate_access( + state: &MonoApiServiceState, + context: &crate::ceres::snapshot::runtime::SnapshotContext, + headers: &HeaderMap, +) -> Result<(), SnapshotError> { + let config = state.storage.config(); + if !config.mst2.enabled { + return Err(SnapshotError::new( + SnapshotErrorCode::SnapshotNotReady, + "snapshot surface disabled", + )); + } + if let Some(token) = config.mst2.auth_token.as_deref() + && !bearer_ok(headers, token) + { + return Err(SnapshotError::new( + SnapshotErrorCode::Unauthenticated, + "bearer credentials changed", + )); + } + let current = state + .storage + .snapshot_context(&context.built.snapshot_id, &context.lease_id) + .await?; + if current.built.descriptor != context.built.descriptor + || current.commit_oid != context.commit_oid + || current.root_tree_oid != context.root_tree_oid + || current.authorization_epoch != context.authorization_epoch + { + return Err(SnapshotError::new( + SnapshotErrorCode::IntegrityError, + "fixed session changed during request", + )); + } + Ok(()) +} + +async fn revalidate_request( + state: &MonoApiServiceState, + context: &crate::ceres::snapshot::runtime::SnapshotContext, +) -> Result<(), SnapshotError> { + let headers = REQUEST_HEADERS.try_with(Clone::clone).map_err(|_| { + SnapshotError::new( + SnapshotErrorCode::Unauthenticated, + "request authentication context missing", + ) + })?; + revalidate_access(state, context, &headers).await +} + fn mst2_error_response(err: SnapshotError) -> Response { let status = StatusCode::from_u16(err.code.http_status()).unwrap_or(StatusCode::INTERNAL_SERVER_ERROR); @@ -451,6 +616,7 @@ async fn resolve( let config = state.storage.config(); let mut native_source = None; + let mut selected_native_head = None; let (commit_oid, tree_oid, sequence, writer_epoch) = if config.mst2.publication_enabled { let Some(instance) = config.mst2.instance_uuid.as_deref() else { return Err(mst2_error_response(SnapshotError::new( @@ -511,6 +677,7 @@ async fn resolve( None } }; + selected_native_head = Some(head.clone()); ( head.root.commit, head.root.tree, @@ -576,9 +743,46 @@ async fn resolve( let built = build_descriptor(&config.mst2, &view, &req.scope, scope_page.page_id) .map_err(mst2_error_response)?; - let ctx = runtime() - .insert_context(built.clone(), &commit_oid, &tree_oid, req.lease_seconds) - .map_err(mst2_error_response)?; + let ctx = if let Some(head) = selected_native_head.as_ref() { + let sessions = state.storage.snapshot_sessions().await; + if let Some(context) = sessions + .open(head, &built, None, req.lease_seconds) + .await + .map_err(mst2_error_response)? + { + context + } else { + let prepared = crate::ceres::snapshot::pages::prepare_native_metadata_retention( + handler.as_ref(), + &root_tree, + &req.scope, + crate::ceres::snapshot::retention_dag::MetadataDagLimits::default(), + ) + .await + .map_err(mst2_error_response)?; + let receipt = sessions + .install(&built, &prepared) + .await + .map_err(mst2_error_response)?; + #[cfg(test)] + if let Ok((prepared, release)) = NATIVE_HANDOFF_BARRIERS.try_with(|value| value.clone()) + { + prepared.wait().await; + release.wait().await; + } + sessions + .open(head, &built, Some(&receipt), req.lease_seconds) + .await + .map_err(mst2_error_response)? + .ok_or_else(|| { + mst2_error_response(internal("durable session handoff returned no context")) + })? + } + } else { + runtime() + .insert_context(built.clone(), &commit_oid, &tree_oid, req.lease_seconds) + .map_err(mst2_error_response)? + }; if let Some(source) = native_source { let request_id = current_request_id(); @@ -611,6 +815,9 @@ async fn resolve( } } + revalidate_request(&state, &ctx) + .await + .map_err(mst2_error_response)?; let body = json!({ "descriptor": descriptor_json(&built), "publication_sequence": sequence, @@ -648,9 +855,7 @@ async fn descriptor_get( AxumPath(snapshot_id): AxumPath, ) -> Result { ensure_enabled(&state).map_err(mst2_error_response)?; - let ctx = runtime() - .context(&snapshot_id) - .map_err(mst2_error_response)?; + let ctx = request_context(&state, &snapshot_id).map_err(mst2_error_response)?; // Reading the descriptor back never touches latest (spec 04 §2). let body = json!({ "snapshot_id": snapshot_id, @@ -680,8 +885,10 @@ async fn lease_renew( } else { parse_json_body(&body)? }; - let renewed = runtime() - .renew_lease(&lease_id, req.lease_seconds.unwrap_or(600)) + let renewed = state + .storage + .snapshot_renew(&lease_id, req.lease_seconds.unwrap_or(600)) + .await .map_err(mst2_error_response)?; // Renewal never changes the version or authorization (spec 04 §4). let body = json!({ @@ -701,7 +908,11 @@ async fn lease_release( ensure_enabled(&state).map_err(mst2_error_response)?; // Idempotent: releasing an unknown/already-released lease still succeeds // (spec 04 §2). This never deletes Git content. - let released = runtime().release_lease(&lease_id); + let released = state + .storage + .snapshot_release(&lease_id) + .await + .map_err(mst2_error_response)?; Ok(Json(json!({ "lease_id": lease_id, "released": released })).into_response()) } @@ -732,9 +943,7 @@ async fn directory( Query(q): Query, ) -> Result { ensure_enabled(&state).map_err(mst2_error_response)?; - let ctx = runtime() - .context(&snapshot_id) - .map_err(mst2_error_response)?; + let ctx = request_context(&state, &snapshot_id).map_err(mst2_error_response)?; validate_scope_relative_path(&q.path).map_err(mst2_error_response)?; if !(1..=256).contains(&q.limit) { return Err(mst2_error_response(SnapshotError::new( @@ -939,9 +1148,7 @@ async fn blob( headers: HeaderMap, ) -> Result { ensure_enabled(&state).map_err(mst2_error_response)?; - let ctx = runtime() - .context(&snapshot_id) - .map_err(mst2_error_response)?; + let ctx = request_context(&state, &snapshot_id).map_err(mst2_error_response)?; validate_scope_relative_path(&q.path).map_err(mst2_error_response)?; if headers.contains_key("range") { // Spec 04 section 9: raw blob has no Range semantics this profile. @@ -985,11 +1192,42 @@ async fn blob( crate::ceres::snapshot::resolver::FsKind::Symlink => "symlink", crate::ceres::snapshot::resolver::FsKind::Directory => "directory", }; + revalidate_request(&state, &ctx) + .await + .map_err(mst2_error_response)?; + let headers = REQUEST_HEADERS.try_with(Clone::clone).map_err(|_| { + mst2_error_response(SnapshotError::new( + SnapshotErrorCode::Unauthenticated, + "request authentication context missing", + )) + })?; + let stream_state = state.0.clone(); + let stream_context = ctx.clone(); + let blocks = futures::stream::unfold( + ( + Bytes::from(raw), + 0usize, + stream_state, + stream_context, + headers, + ), + |(raw, offset, state, context, headers)| async move { + if offset >= raw.len() { + return None; + } + if let Err(error) = revalidate_access(&state, &context, &headers).await { + return Some((Err(error), (raw, usize::MAX, state, context, headers))); + } + let end = (offset + 1_048_576).min(raw.len()); + let bytes = raw.slice(offset..end); + Some((Ok(bytes), (raw, end, state, context, headers))) + }, + ); Response::builder() .header("etag", format!("\"sha256:{}\"", hex_of(&digest))) .header("cache-control", "private, no-cache, no-transform") .header("x-mega-fs-kind", fs_kind_str) - .body(axum::body::Body::from(Bytes::from(raw))) + .body(axum::body::Body::from_stream(blocks)) .map_err(|e| { mst2_error_response(SnapshotError::new( SnapshotErrorCode::Internal, @@ -1053,9 +1291,7 @@ async fn lookup( .api_handler(std::path::Path::new("/")) .await .map_err(internal)?; - let ctx = runtime() - .context(&snapshot_id) - .map_err(mst2_error_response)?; + let ctx = request_context(&state, &snapshot_id).map_err(mst2_error_response)?; let root_tree = handler .get_tree_by_hash(&ctx.root_tree_oid) .await @@ -1208,9 +1444,7 @@ async fn metadata_pages( for item in &req.items { validate_scope_relative_path(&item.directory_path).map_err(mst2_error_response)?; } - let ctx = runtime() - .context(&snapshot_id) - .map_err(mst2_error_response)?; + let ctx = request_context(&state, &snapshot_id).map_err(mst2_error_response)?; let handler = state .api_handler(std::path::Path::new("/")) @@ -1281,18 +1515,18 @@ async fn metadata_pages( .unwrap_or(crate::ceres::snapshot::frame_stream::Encoding::Identity); use crate::ceres::snapshot::frame_stream::FrameStream; let mut stream = FrameStream::new(1, encoding); - let mut out: Vec = Vec::new(); + let mut out: Vec> = Vec::new(); let mut frame: Vec<([u8; 32], Vec)> = Vec::new(); let mut frame_raw: usize = 0; let mut flush = |frame: &mut Vec<([u8; 32], Vec)>, raw: &mut usize, - out: &mut Vec| + out: &mut Vec>| -> Result<(), SnapshotError> { if frame.is_empty() { return Ok(()); } let bytes = stream.meta(std::mem::take(frame))?; - out.extend_from_slice(&bytes); + out.push(bytes); *raw = 0; Ok(()) }; @@ -1318,9 +1552,9 @@ async fn metadata_pages( logical_bytes, request_body_sha256, ); - out.extend_from_slice(&end); + out.push(end); - treeframe_response(&snapshot_id, &body, out).map_err(mst2_error_response) + guarded_treeframe_response(&state, &ctx, &snapshot_id, &body, out).map_err(mst2_error_response) } #[cfg(test)] diff --git a/src/api/router/snapshot_session_tests.rs b/src/api/router/snapshot_session_tests.rs new file mode 100644 index 00000000..43a9c440 --- /dev/null +++ b/src/api/router/snapshot_session_tests.rs @@ -0,0 +1,1121 @@ +use sea_orm::{DatabaseConnection, DbBackend, Statement, TransactionTrait}; +use tokio::sync::Barrier; + +use super::*; +use crate::{ + api::router::snapshot_router::{with_native_handoff_barriers, with_native_resolve_barriers}, + jupiter::storage::{ + mst2_retention::{GcClaim, PostgresRetentionRepository}, + push_queue_storage::PushQueueStorage, + }, +}; + +fn resolve_request(scope: &str) -> Request { + Request::builder() + .method("POST") + .uri("/api/v2/snapshots/resolve") + .header("authorization", format!("Bearer {TOKEN}")) + .header("content-type", "application/json") + .body(Body::from( + json!({"target":{"kind":"latest"},"scope":scope}).to_string(), + )) + .unwrap() +} + +async fn scalar(db: &DatabaseConnection, sql: &str) -> i64 { + db.query_one_raw(Statement::from_string(DbBackend::Postgres, sql.to_owned())) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap() +} + +async fn rebuilt(fixture: &Fixture) -> MonoApiServiceState { + let config = fixture.state.storage.config(); + let connection = crate::jupiter::storage::init::postgres_connection(&config.database) + .await + .unwrap(); + let storage = crate::jupiter::storage::Storage::new_with_connection( + config, + Arc::new(connection), + fixture.state.storage.git_service.obj_storage.clone(), + ) + .await + .unwrap(); + MonoApiServiceState { + storage, + git_object_cache: Arc::new(GitObjectCache { + connection: fixture.state.git_object_cache.connection.clone(), + prefix: uuid::Uuid::new_v4().to_string(), + }), + ..fixture.state.clone() + } +} + +fn app(state: &MonoApiServiceState) -> Router { + Router::new().nest("/api/v2", routers(state.clone()).with_state(state.clone())) +} + +async fn lease_control(fixture: &Fixture, lease: &str, method: &str, renew: bool) -> Value { + let suffix = if renew { "/renew" } else { "" }; + success_json( + fixture + .app + .clone() + .oneshot( + Request::builder() + .method(method) + .uri(format!("/api/v2/snapshots/leases/{lease}{suffix}")) + .header("authorization", format!("Bearer {TOKEN}")) + .header("content-type", "application/json") + .body(Body::empty()) + .unwrap(), + ) + .await + .unwrap(), + ) + .await +} + +async fn advance(fixture: &Fixture) { + let mono = fixture.state.storage.mono_storage(); + let old = mono.get_main_ref("/project").await.unwrap().unwrap(); + let old_oid = ObjectHash::from_hex_for_kind(HashKind::Sha1, &old.ref_commit_hash).unwrap(); + let oid = ObjectHash::from_hex_for_kind(HashKind::Sha1, &fixture.oid).unwrap(); + let next_tree = tree(vec![item(TreeItemMode::Blob, oid, "new-only")]); + let next = Commit::from_tree_id_with_kind( + HashKind::Sha1, + next_tree.id, + vec![old_oid], + "durable HTTP next view", + ) + .unwrap(); + mono.save_mega_trees(vec![next_tree], next.id, None) + .await + .unwrap(); + mono.save_mega_commits(vec![next.clone()], None) + .await + .unwrap(); + publish_native_push(&fixture.state.storage, "/project", old_oid, &next).await; +} + +#[tokio::test] +async fn mst2_durable_http_resolve_installs_complete_dag_and_warm_leases_share_it() { + let fixture = Fixture::new_with_pg_config(true).await; + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + let pages = scalar(db, "SELECT count(*) FROM mst2_metadata_payload").await; + assert!( + pages >= 3, + "scope includes nested and empty directory pages" + ); + assert_eq!( + scalar(db, "SELECT count(*) FROM mst2_metadata_prepare_page").await, + pages + ); + assert_eq!( + scalar(db, "SELECT count(*) FROM mst2_snapshot_context").await, + 1 + ); + assert_eq!( + scalar( + db, + "SELECT count(*) FROM mst2_snapshot_lease WHERE state='ACTIVE'" + ) + .await, + 1 + ); + assert_eq!( + scalar( + db, + "SELECT count(*) FROM mst2_retention_root WHERE root_kind='prepare'" + ) + .await, + 0 + ); + assert_eq!( + scalar( + db, + "SELECT count(*) FROM mst2_retention_root WHERE root_kind='lease'" + ) + .await, + 1 + ); + assert_eq!( + scalar( + db, + "SELECT count(*) FROM mst2_retention_root WHERE root_kind='pin'" + ) + .await, + 1 + ); + let edges = scalar(db, "SELECT count(*) FROM mst2_retention_edge").await; + let refs = scalar( + db, + "SELECT sum(incoming_refs)::bigint FROM mst2_retention_node", + ) + .await; + assert_eq!(edges, refs); + fixture.counts.reset(); + let again = success_json( + fixture + .app + .clone() + .oneshot(resolve_request("/project")) + .await + .unwrap(), + ) + .await; + assert_eq!(again["descriptor"]["snapshot_id"], fixture.snapshot); + assert_ne!(again["lease_id"], fixture.lease); + fixture.counts.assert(0, 0); + assert_eq!( + scalar(db, "SELECT count(*) FROM mst2_metadata_payload").await, + pages + ); + assert_eq!( + scalar(db, "SELECT count(*) FROM mst2_metadata_prepare").await, + 1 + ); + assert_eq!( + scalar(db, "SELECT count(*) FROM mst2_snapshot_context").await, + 1 + ); + assert_eq!( + scalar( + db, + "SELECT count(*) FROM mst2_retention_root WHERE root_kind='lease'" + ) + .await, + 2 + ); + assert_eq!( + scalar( + db, + "SELECT sum(incoming_refs)::bigint FROM mst2_retention_node" + ) + .await, + refs + ); + + let state = rebuilt(&fixture).await; + assert!(state.storage.native_snapshot_sessions.get().is_none()); + assert!(!Arc::ptr_eq( + &state.storage.native_projection_cache, + &fixture.state.storage.native_projection_cache + )); + let replacement = app(&state); + let descriptor = success_json( + replacement + .clone() + .oneshot(fixture.request("GET", "descriptor", Body::empty())) + .await + .unwrap(), + ) + .await; + assert_eq!(descriptor["snapshot_id"], fixture.snapshot); + assert_eq!(descriptor["lease_id"], fixture.lease); + assert_eq!(descriptor["descriptor"], again["descriptor"]); + for (method, path, body) in [ + ("GET", "directory?path=/nested", Body::empty()), + ( + "POST", + "lookup", + Body::from(json!({"paths":["/nested/file"]}).to_string()), + ), + ( + "POST", + "metadata/pages", + Body::from(json!({"items":[{"directory_path":"/nested"}]}).to_string()), + ), + ("HEAD", "blob?path=/file", Body::empty()), + ] { + let response = replacement + .clone() + .oneshot(fixture.request(method, path, body)) + .await + .unwrap(); + assert_eq!(response.status(), 200); + to_bytes(response.into_body(), usize::MAX).await.unwrap(); + } + fixture.counts.assert(0, 0); + let response = replacement + .oneshot(fixture.request("GET", "blob?path=/file", Body::empty())) + .await + .unwrap(); + assert_eq!(response.status(), 200); + assert_eq!( + to_bytes(response.into_body(), usize::MAX) + .await + .unwrap() + .as_ref(), + fixture.raw.as_slice() + ); + fixture.counts.assert(1, fixture.raw.len()); +} + +#[tokio::test] +async fn mst2_durable_http_old_sid_and_original_lease_survive_real_publication_and_fresh_service() { + let fixture = Fixture::new_with_pg_config(true).await; + let old = success_json(fixture.send("GET", "descriptor", Body::empty()).await).await; + advance(&fixture).await; + let next = success_json( + fixture + .app + .clone() + .oneshot(resolve_request("/project")) + .await + .unwrap(), + ) + .await; + assert_eq!(next["publication_sequence"], "2"); + assert_ne!(next["descriptor"]["snapshot_id"], fixture.snapshot); + let state = rebuilt(&fixture).await; + let replacement = app(&state); + let restored = success_json( + replacement + .clone() + .oneshot(fixture.request("GET", "descriptor", Body::empty())) + .await + .unwrap(), + ) + .await; + assert_eq!(restored, old); + let response = replacement + .clone() + .oneshot(fixture.request("GET", "blob?path=/file", Body::empty())) + .await + .unwrap(); + assert_eq!(response.status(), 200); + assert_eq!( + to_bytes(response.into_body(), usize::MAX) + .await + .unwrap() + .as_ref(), + fixture.raw.as_slice() + ); + error( + replacement + .oneshot(fixture.request("GET", "chunk-map?path=/new-only", Body::empty())) + .await + .unwrap(), + 404, + "PATH_NOT_FOUND", + false, + ) + .await; + let new_request = Request::builder() + .uri(format!( + "/api/v2/snapshots/{}/blob?path=/new-only", + next["descriptor"]["snapshot_id"].as_str().unwrap() + )) + .header("authorization", format!("Bearer {TOKEN}")) + .header("x-mega-snapshot-lease", next["lease_id"].as_str().unwrap()) + .body(Body::empty()) + .unwrap(); + let response = app(&state).oneshot(new_request).await.unwrap(); + assert_eq!(response.status(), 200); + assert_eq!( + to_bytes(response.into_body(), usize::MAX) + .await + .unwrap() + .as_ref(), + fixture.raw.as_slice() + ); +} + +#[tokio::test] +async fn mst2_durable_http_renew_release_expiry_and_last_lease_retire_protection() { + let fixture = Fixture::new_with_pg_config(true).await; + let another = success_json( + fixture + .app + .clone() + .oneshot(resolve_request("/project")) + .await + .unwrap(), + ) + .await; + let other = another["lease_id"].as_str().unwrap(); + let renewed = lease_control(&fixture, &fixture.lease, "POST", true).await; + assert_eq!(renewed["snapshot_id"], fixture.snapshot); + let state = rebuilt(&fixture).await; + let restored = success_json( + app(&state) + .oneshot(fixture.request("GET", "descriptor", Body::empty())) + .await + .unwrap(), + ) + .await; + assert_eq!(restored["lease_expires_at"], renewed["lease_expires_at"]); + let first = lease_control(&fixture, &fixture.lease, "DELETE", false).await; + let second = lease_control(&fixture, &fixture.lease, "DELETE", false).await; + assert_eq!(first["released"], true); + assert_eq!(second["released"], false); + error( + app(&state) + .oneshot(fixture.request("GET", "descriptor", Body::empty())) + .await + .unwrap(), + 410, + "LEASE_EXPIRED", + false, + ) + .await; + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + assert_eq!( + scalar( + db, + "SELECT count(*) FROM mst2_retention_root WHERE root_kind='lease'" + ) + .await, + 1 + ); + assert_eq!( + scalar( + db, + "SELECT count(*) FROM mst2_retention_root WHERE root_kind='pin'" + ) + .await, + 1 + ); + let mut request = fixture.request("HEAD", "blob?path=/file", Body::empty()); + request + .headers_mut() + .insert("x-mega-snapshot-lease", other.parse().unwrap()); + assert_eq!(app(&state).oneshot(request).await.unwrap().status(), 200); + db.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "UPDATE mst2_snapshot_lease SET expires_at_unix=0 WHERE lease_id=$1", + [other.into()], + )) + .await + .unwrap(); + let renew = Request::builder() + .method("POST") + .uri(format!("/api/v2/snapshots/leases/{other}/renew")) + .header("authorization", format!("Bearer {TOKEN}")) + .body(Body::empty()) + .unwrap(); + error( + app(&state).oneshot(renew).await.unwrap(), + 410, + "LEASE_EXPIRED", + false, + ) + .await; + assert_eq!( + scalar( + db, + "SELECT count(*) FROM mst2_snapshot_lease WHERE state='EXPIRED'" + ) + .await, + 1 + ); + assert_eq!( + scalar(db, "SELECT count(*) FROM mst2_retention_root").await, + 0 + ); + let stored = + crate::callisto::mst2_snapshot_context::Entity::find_by_id(fixture.snapshot.clone()) + .one(db) + .await + .unwrap() + .unwrap(); + let node = format!("page:sha256:{}", hex::encode(stored.metadata_root)); + assert_eq!( + PostgresRetentionRepository::new(db.clone()) + .mark_deleting("last-lease-gc", &node) + .await + .unwrap(), + GcClaim::Marked + ); + assert_eq!( + scalar(db, "SELECT count(*) FROM mst2_metadata_payload").await, + scalar(db, "SELECT count(*) FROM mst2_metadata_prepare_page").await, + "release does not delete payload or Git bytes" + ); + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn mst2_durable_http_live_to_deleting_winner_rejects_paused_resolve_without_leak() { + let fixture = Fixture::new_with_pg_config(true).await; + let captured = Arc::new(Barrier::new(2)); + let release = Arc::new(Barrier::new(2)); + let resolving = { + let app = fixture.app.clone(); + let captured = captured.clone(); + let release = release.clone(); + tokio::spawn(async move { + with_native_resolve_barriers( + captured, + release, + app.oneshot(resolve_request("/project")), + ) + .await + .unwrap() + }) + }; + captured.wait().await; + lease_control(&fixture, &fixture.lease, "DELETE", false).await; + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + let context = + crate::callisto::mst2_snapshot_context::Entity::find_by_id(fixture.snapshot.clone()) + .one(db) + .await + .unwrap() + .unwrap(); + let node = format!("page:sha256:{}", hex::encode(context.metadata_root)); + let txn = db.begin().await.unwrap(); + assert_eq!( + PostgresRetentionRepository::mark_deleting_in_txn(&txn, "http-gc-winner", &node) + .await + .unwrap(), + GcClaim::Marked + ); + let txn_id = scalar(db, "SELECT count(*) FROM mst2_snapshot_lease").await; + release.wait().await; + txn.commit().await.unwrap(); + error(resolving.await.unwrap(), 503, "OBJECT_UNAVAILABLE", false).await; + assert_eq!( + scalar(db, "SELECT count(*) FROM mst2_snapshot_lease").await, + txn_id + ); + assert_eq!( + scalar(db, "SELECT count(*) FROM mst2_retention_root").await, + 0 + ); +} + +#[tokio::test] +async fn mst2_durable_http_lease_winner_blocks_gc_until_the_last_release() { + let fixture = Fixture::new_with_pg_config(true).await; + let another = success_json( + fixture + .app + .clone() + .oneshot(resolve_request("/project")) + .await + .unwrap(), + ) + .await; + let other = another["lease_id"].as_str().unwrap(); + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + let stored = + crate::callisto::mst2_snapshot_context::Entity::find_by_id(fixture.snapshot.clone()) + .one(db) + .await + .unwrap() + .unwrap(); + let node = format!("page:sha256:{}", hex::encode(stored.metadata_root)); + let retention = PostgresRetentionRepository::new(db.clone()); + assert_eq!( + retention + .mark_deleting("lease-winner-first", &node) + .await + .unwrap(), + GcClaim::Unavailable + ); + lease_control(&fixture, &fixture.lease, "DELETE", false).await; + assert_eq!( + retention + .mark_deleting("lease-winner-second", &node) + .await + .unwrap(), + GcClaim::Unavailable + ); + lease_control(&fixture, other, "DELETE", false).await; + assert_eq!( + scalar(db, "SELECT count(*) FROM mst2_retention_root").await, + 0 + ); + assert_eq!( + retention + .mark_deleting("lease-winner-final", &node) + .await + .unwrap(), + GcClaim::Marked + ); + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn mst2_durable_http_renew_waits_for_lock_before_checking_database_deadline() { + let fixture = Fixture::new_with_pg_config(true).await; + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + let held = db.begin().await.unwrap(); + held.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT pg_advisory_xact_lock($1,hashtext(current_schema()))", + [crate::jupiter::storage::mst2_retention::RETENTION_LOCK_KEY.into()], + )) + .await + .unwrap(); + let renewing = { + let app = fixture.app.clone(); + let lease = fixture.lease.clone(); + tokio::spawn(async move { + app.oneshot( + Request::builder() + .method("POST") + .uri(format!("/api/v2/snapshots/leases/{lease}/renew")) + .header("authorization", format!("Bearer {TOKEN}")) + .body(Body::empty()) + .unwrap(), + ) + .await + .unwrap() + }) + }; + tokio::time::timeout(Duration::from_secs(30), async { + loop { + let blocked = scalar( + db, + "SELECT count(*) FROM pg_locks l JOIN pg_stat_activity a ON a.pid=l.pid + WHERE l.locktype='advisory' AND NOT l.granted AND a.datname=current_database() + AND a.application_name=current_schema()", + ) + .await; + if blocked > 0 { + break; + } + tokio::task::yield_now().await; + } + }) + .await + .expect("HTTP renewal did not wait on the retention lock"); + held.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "UPDATE mst2_snapshot_lease SET expires_at_unix=0 WHERE lease_id=$1", + [fixture.lease.clone().into()], + )) + .await + .unwrap(); + held.commit().await.unwrap(); + error(renewing.await.unwrap(), 410, "LEASE_EXPIRED", false).await; + assert_eq!( + scalar( + db, + "SELECT count(*) FROM mst2_snapshot_lease WHERE state='EXPIRED'" + ) + .await, + 1 + ); + assert_eq!( + scalar(db, "SELECT count(*) FROM mst2_retention_root").await, + 0 + ); + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn mst2_durable_http_publication_advance_rejects_mixed_resolve_then_retries_current() { + let fixture = Fixture::new_with_pg_config(true).await; + let captured = Arc::new(Barrier::new(2)); + let release = Arc::new(Barrier::new(2)); + let resolving = { + let app = fixture.app.clone(); + let captured = captured.clone(); + let release = release.clone(); + tokio::spawn(async move { + with_native_resolve_barriers( + captured, + release, + app.oneshot(resolve_request("/project")), + ) + .await + .unwrap() + }) + }; + captured.wait().await; + advance(&fixture).await; + release.wait().await; + error(resolving.await.unwrap(), 503, "SNAPSHOT_NOT_READY", true).await; + let mono = fixture.state.storage.mono_storage(); + assert_eq!( + scalar( + mono.get_connection(), + "SELECT count(*) FROM mst2_snapshot_lease" + ) + .await, + 1 + ); + let next = success_json( + fixture + .app + .clone() + .oneshot(resolve_request("/project")) + .await + .unwrap(), + ) + .await; + assert_eq!(next["publication_sequence"], "2"); + assert_ne!(next["descriptor"]["snapshot_id"], fixture.snapshot); + let state = rebuilt(&fixture).await; + assert_eq!( + app(&state) + .oneshot(fixture.request("HEAD", "blob?path=/file", Body::empty())) + .await + .unwrap() + .status(), + 200 + ); +} + +#[tokio::test] +async fn mst2_durable_http_install_fault_never_hands_off_or_returns_success() { + let fixture = Fixture::new_with_pg_config(true).await; + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + db.execute_unprepared( + "CREATE FUNCTION reject_http_page() RETURNS trigger LANGUAGE plpgsql AS $$ + BEGIN RAISE EXCEPTION 'forced HTTP install interruption'; END $$; + CREATE TRIGGER reject_http_page BEFORE INSERT ON mst2_metadata_payload + FOR EACH ROW EXECUTE FUNCTION reject_http_page() ", + ) + .await + .unwrap(); + error( + fixture + .app + .clone() + .oneshot(resolve_request("/project/nested")) + .await + .unwrap(), + 503, + "TEMPORARY_UNAVAILABLE", + true, + ) + .await; + assert_eq!( + scalar(db, "SELECT count(*) FROM mst2_snapshot_context").await, + 1 + ); + assert_eq!( + scalar(db, "SELECT count(*) FROM mst2_snapshot_lease").await, + 1 + ); + assert_eq!( + scalar( + db, + "SELECT count(*) FROM mst2_retention_root WHERE root_kind='prepare'" + ) + .await, + 0 + ); + assert_eq!( + scalar( + db, + "SELECT count(*) FROM mst2_metadata_prepare WHERE state='PREPARING'" + ) + .await, + 1 + ); + assert_eq!( + fixture + .send("HEAD", "blob?path=/file", Body::empty()) + .await + .status(), + 200 + ); + db.execute_unprepared( + "DROP TRIGGER reject_http_page ON mst2_metadata_payload; DROP FUNCTION reject_http_page() ", + ) + .await + .unwrap(); + let success = success_json( + fixture + .app + .clone() + .oneshot(resolve_request("/project/nested")) + .await + .unwrap(), + ) + .await; + assert_eq!(success["descriptor"]["scope"], "/project/nested"); + assert_eq!( + scalar( + db, + "SELECT count(*) FROM mst2_metadata_prepare WHERE state='PREPARING'" + ) + .await, + 0 + ); + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn mst2_durable_http_frame_and_raw_delivery_recheck_after_release() { + for raw in [false, true] { + let fixture = Fixture::new_with_pg_config(true).await; + let response = if raw { + fixture.send("GET", "blob?path=/file", Body::empty()).await + } else { + fixture + .send( + "POST", + "metadata/pages", + Body::from(json!({"items":[{"directory_path":"/"}]}).to_string()), + ) + .await + }; + assert_eq!(response.status(), 200); + let mut stream = response.into_body().into_data_stream(); + let first = stream.next().await.unwrap().unwrap(); + assert!(!first.is_empty()); + if raw { + assert_eq!(first.len(), 1_048_576); + } + lease_control(&fixture, &fixture.lease, "DELETE", false).await; + assert!( + stream.next().await.unwrap().is_err(), + "revocation must suppress the next raw block or END frame" + ); + assert!(stream.next().await.is_none()); + fixture.counts.assert( + if raw { 1 } else { 0 }, + if raw { fixture.raw.len() } else { 0 }, + ); + } +} + +#[tokio::test] +async fn mst2_durable_http_current_state_wrong_lease_and_corrupt_source_reject_warm_reads() { + let fixture = Fixture::new_with_pg_config(true).await; + let warm = success_json(fixture.send("GET", "descriptor", Body::empty()).await).await; + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + for (sid, lease) in [ + (fixture.snapshot.clone(), uuid::Uuid::new_v4().to_string()), + (format!("sha256:{}", "1".repeat(64)), fixture.lease.clone()), + ] { + let request = Request::builder() + .method("GET") + .uri(format!("/api/v2/snapshots/{sid}/descriptor")) + .header("authorization", format!("Bearer {TOKEN}")) + .header("x-mega-snapshot-lease", lease) + .header("if-none-match", format!("\"{}\"", fixture.digest_string())) + .body(Body::empty()) + .unwrap(); + error( + fixture.app.clone().oneshot(request).await.unwrap(), + 410, + "LEASE_EXPIRED", + false, + ) + .await; + } + db.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "UPDATE mst2_snapshot_context SET authorization_epoch=2 WHERE snapshot_id=$1", + [fixture.snapshot.clone().into()], + )) + .await + .unwrap(); + let request = fixture.request("GET", "descriptor", Body::empty()); + error( + fixture.app.clone().oneshot(request).await.unwrap(), + 403, + "SCOPE_FORBIDDEN", + false, + ) + .await; + db.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "UPDATE mst2_snapshot_context SET authorization_epoch=1 WHERE snapshot_id=$1", + [fixture.snapshot.clone().into()], + )) + .await + .unwrap(); + db.execute_unprepared("UPDATE mst2_native_publication SET writer_epoch=2") + .await + .unwrap(); + error( + fixture.send("GET", "descriptor", Body::empty()).await, + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + let state = rebuilt(&fixture).await; + error( + app(&state) + .oneshot(fixture.request("GET", "descriptor", Body::empty())) + .await + .unwrap(), + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + assert_eq!(warm["snapshot_id"], fixture.snapshot); + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn mst2_durable_http_large_dag_batches_handoff_without_scanning_pages_or_edges() { + let fixture = Fixture::new_with_pg_config_and_directories(true, 80).await; + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + assert!(scalar(db, "SELECT count(*) FROM mst2_metadata_payload").await > 64); + db.execute_unprepared( + "CREATE TABLE http_install_batch_count(singleton integer PRIMARY KEY,batches bigint NOT NULL); + INSERT INTO http_install_batch_count VALUES(1,0); + CREATE FUNCTION count_http_install_batch() RETURNS trigger LANGUAGE plpgsql AS $$ + BEGIN UPDATE http_install_batch_count SET batches=batches+1 WHERE singleton=1; RETURN NULL; END $$; + CREATE TRIGGER count_http_install_batch AFTER INSERT ON mst2_metadata_payload + FOR EACH STATEMENT EXECUTE FUNCTION count_http_install_batch()", + ).await.unwrap(); + let prepared = Arc::new(Barrier::new(2)); + let release = Arc::new(Barrier::new(2)); + let resolving = { + let app = fixture.app.clone(); + let prepared = prepared.clone(); + let release = release.clone(); + tokio::spawn(async move { + with_native_handoff_barriers(prepared, release, app.oneshot(resolve_request("/"))) + .await + .unwrap() + }) + }; + prepared.wait().await; + let members = scalar( + db, + "SELECT count(*) FROM mst2_metadata_prepare_page pp JOIN mst2_metadata_prepare p + ON p.prepare_id=pp.prepare_id WHERE p.scope='/'", + ) + .await; + assert!(members > 64); + assert_eq!( + scalar(db, "SELECT batches FROM http_install_batch_count").await, + (members + 63) / 64 + ); + let writer = db.begin().await.unwrap(); + assert!( + PushQueueStorage::try_mono_write_lock(&writer) + .await + .unwrap(), + "full projection and installation completed without retaining the mono writer lock" + ); + writer.rollback().await.unwrap(); + let blocked_scans = db.begin().await.unwrap(); + blocked_scans.execute_unprepared( + "LOCK TABLE mst2_metadata_prepare_page,mst2_metadata_payload,mst2_retention_edge IN ACCESS EXCLUSIVE MODE", + ).await.unwrap(); + release.wait().await; + let response = tokio::time::timeout(Duration::from_secs(5), resolving) + .await + .expect("handoff must not read page membership, payload, or edge tables") + .unwrap(); + let resolved = success_json(response).await; + let writer = db.begin().await.unwrap(); + assert!( + PushQueueStorage::try_mono_write_lock(&writer) + .await + .unwrap() + ); + writer.rollback().await.unwrap(); + blocked_scans.rollback().await.unwrap(); + assert_eq!(resolved["descriptor"]["scope"], "/"); + assert_eq!( + scalar(db, "SELECT batches FROM http_install_batch_count").await, + (members + 63) / 64 + ); + fixture.counts.assert(0, 0); + let warm = success_json( + fixture + .app + .clone() + .oneshot(resolve_request("/")) + .await + .unwrap(), + ) + .await; + assert_eq!( + warm["descriptor"]["snapshot_id"], + resolved["descriptor"]["snapshot_id"] + ); + assert_eq!( + scalar(db, "SELECT batches FROM http_install_batch_count").await, + (members + 63) / 64 + ); + advance(&fixture).await; + let state = rebuilt(&fixture).await; + let directory = success_json( + app(&state) + .oneshot(fixture.request("GET", "directory?path=/wide-079", Body::empty())) + .await + .unwrap(), + ) + .await; + assert_eq!(directory["entries"][0]["name"], "file-079"); + assert_eq!( + app(&state) + .oneshot(fixture.request("HEAD", "blob?path=/wide-079/file-079", Body::empty())) + .await + .unwrap() + .status(), + 200 + ); + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn mst2_durable_http_handoff_rejects_changed_plan_or_receipt_summary() { + for mutation in [ + "canonical_plan=canonical_plan||decode('00','hex')", + "projection_revision=projection_revision+1", + "total_bytes=total_bytes+1", + ] { + let fixture = Fixture::new_with_pg_config_and_directories(true, 80).await; + let prepared = Arc::new(Barrier::new(2)); + let release = Arc::new(Barrier::new(2)); + let resolving = { + let app = fixture.app.clone(); + let prepared = prepared.clone(); + let release = release.clone(); + tokio::spawn(async move { + with_native_handoff_barriers(prepared, release, app.oneshot(resolve_request("/"))) + .await + .unwrap() + }) + }; + prepared.wait().await; + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + db.execute_unprepared(&format!( + "UPDATE mst2_metadata_prepare SET {mutation} WHERE scope='/'" + )) + .await + .unwrap(); + release.wait().await; + error(resolving.await.unwrap(), 502, "INTEGRITY_ERROR", false).await; + assert_eq!( + scalar(db, "SELECT count(*) FROM mst2_snapshot_context").await, + 1 + ); + assert_eq!( + scalar(db, "SELECT count(*) FROM mst2_snapshot_lease").await, + 1 + ); + assert_eq!( + scalar( + db, + "SELECT count(*) FROM mst2_retention_root WHERE root_kind='lease'" + ) + .await, + 1 + ); + assert!( + scalar( + db, + "SELECT count(*) FROM mst2_retention_root WHERE root_kind='prepare'" + ) + .await + > 64 + ); + assert_eq!( + fixture + .send("HEAD", "blob?path=/file", Body::empty()) + .await + .status(), + 200 + ); + fixture.counts.assert(0, 0); + } +} + +async fn overwrite_primary_scope_for_test(db: &DatabaseConnection, storage_uuid: String) { + // Fault injection owns this isolated schema. Restore the production + // immutable trigger before commit; failed injection rolls back its DDL. + let txn = db.begin().await.unwrap(); + txn.execute_unprepared( + "ALTER TABLE mst2_metadata_storage_scope DISABLE TRIGGER mst2_metadata_scope_immutable", + ) + .await + .unwrap(); + let changed = txn + .execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "UPDATE mst2_metadata_storage_scope SET storage_uuid=$1 WHERE singleton=1", + [storage_uuid.into()], + )) + .await + .unwrap(); + assert_eq!(changed.rows_affected(), 1); + txn.execute_unprepared( + "ALTER TABLE mst2_metadata_storage_scope ENABLE TRIGGER mst2_metadata_scope_immutable", + ) + .await + .unwrap(); + txn.commit().await.unwrap(); +} + +#[tokio::test] +async fn mst2_durable_http_warm_reads_renew_and_frame_delivery_reject_primary_scope_drift() { + let fixture = Fixture::new_with_pg_config(true).await; + success_json(fixture.send("GET", "descriptor", Body::empty()).await).await; + let response = fixture + .send( + "POST", + "metadata/pages", + Body::from(json!({"items":[{"directory_path":"/nested"}]}).to_string()), + ) + .await; + assert_eq!(response.status(), 200); + let mut body = response.into_body().into_data_stream(); + assert!(body.next().await.unwrap().is_ok()); + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + let original: String = db + .query_one_raw(Statement::from_string( + DbBackend::Postgres, + "SELECT storage_uuid FROM mst2_metadata_storage_scope WHERE singleton=1".to_owned(), + )) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap(); + overwrite_primary_scope_for_test(db, uuid::Uuid::new_v4().to_string()).await; + error( + fixture.send("GET", "descriptor", Body::empty()).await, + 500, + "INTERNAL", + true, + ) + .await; + error( + fixture + .app + .clone() + .oneshot( + Request::builder() + .method("POST") + .uri(format!("/api/v2/snapshots/leases/{}/renew", fixture.lease)) + .header("authorization", format!("Bearer {TOKEN}")) + .body(Body::empty()) + .unwrap(), + ) + .await + .unwrap(), + 500, + "INTERNAL", + true, + ) + .await; + assert!( + body.next().await.unwrap().is_err(), + "cached validation cannot emit END after scope drift" + ); + assert!(body.next().await.is_none()); + overwrite_primary_scope_for_test(db, original).await; + success_json(fixture.send("GET", "descriptor", Body::empty()).await).await; + fixture.counts.assert(0, 0); +} diff --git a/src/callisto/mod.rs b/src/callisto/mod.rs index ca18ec38..c8d90192 100644 --- a/src/callisto/mod.rs +++ b/src/callisto/mod.rs @@ -92,3 +92,6 @@ pub mod ssh_keys; pub mod user_notification_preferences; pub mod user_notification_settings; pub mod vault; + +pub mod mst2_snapshot_context; +pub mod mst2_snapshot_lease; diff --git a/src/callisto/mst2_snapshot_context.rs b/src/callisto/mst2_snapshot_context.rs new file mode 100644 index 00000000..5e7acefc --- /dev/null +++ b/src/callisto/mst2_snapshot_context.rs @@ -0,0 +1,24 @@ +use sea_orm::entity::prelude::*; + +#[derive(Clone, Debug, PartialEq, DeriveEntityModel, Eq)] +#[sea_orm(table_name = "mst2_snapshot_context")] +pub struct Model { + #[sea_orm(primary_key, auto_increment = false)] + pub snapshot_id: String, + pub canonical_descriptor: Vec, + pub instance_id: String, + pub commit_oid: String, + pub root_tree_oid: String, + pub metadata_root: Vec, + pub prepare_id: String, + pub publication_sequence: i64, + pub writer_epoch: i64, + pub certificate_receipt_id: Option, + pub authorization_epoch: i64, + pub state: String, + pub created_at: DateTimeWithTimeZone, +} + +#[derive(Copy, Clone, Debug, EnumIter, DeriveRelation)] +pub enum Relation {} +impl ActiveModelBehavior for ActiveModel {} diff --git a/src/callisto/mst2_snapshot_lease.rs b/src/callisto/mst2_snapshot_lease.rs new file mode 100644 index 00000000..2d5cc887 --- /dev/null +++ b/src/callisto/mst2_snapshot_lease.rs @@ -0,0 +1,20 @@ +use sea_orm::entity::prelude::*; + +#[derive(Clone, Debug, PartialEq, DeriveEntityModel, Eq)] +#[sea_orm(table_name = "mst2_snapshot_lease")] +pub struct Model { + #[sea_orm(primary_key, auto_increment = false)] + pub lease_id: String, + pub snapshot_id: String, + pub authorization_epoch: i64, + pub publication_sequence: i64, + pub writer_epoch: i64, + pub certificate_receipt_id: i64, + pub expires_at_unix: i64, + pub state: String, + pub created_at: DateTimeWithTimeZone, +} + +#[derive(Copy, Clone, Debug, EnumIter, DeriveRelation)] +pub enum Relation {} +impl ActiveModelBehavior for ActiveModel {} diff --git a/src/jupiter/migration/m20261007_000100_add_mst2_snapshot_sessions.rs b/src/jupiter/migration/m20261007_000100_add_mst2_snapshot_sessions.rs new file mode 100644 index 00000000..be67efa5 --- /dev/null +++ b/src/jupiter/migration/m20261007_000100_add_mst2_snapshot_sessions.rs @@ -0,0 +1,72 @@ +use sea_orm_migration::prelude::*; + +#[derive(DeriveMigrationName)] +pub struct Migration; + +#[async_trait::async_trait] +impl MigrationTrait for Migration { + async fn up(&self, manager: &SchemaManager) -> Result<(), DbErr> { + manager.get_connection().execute_unprepared( + "CREATE TABLE mst2_snapshot_context ( + snapshot_id text PRIMARY KEY, + canonical_descriptor bytea NOT NULL, + instance_id text NOT NULL, + commit_oid text NOT NULL, + root_tree_oid text NOT NULL, + metadata_root bytea NOT NULL CHECK (octet_length(metadata_root)=32), + prepare_id text NOT NULL UNIQUE REFERENCES mst2_metadata_prepare(prepare_id), + publication_sequence bigint NOT NULL CHECK (publication_sequence>=0), + writer_epoch bigint NOT NULL CHECK (writer_epoch>0), + certificate_receipt_id bigint REFERENCES mst2_native_publication(receipt_id), + authorization_epoch bigint NOT NULL DEFAULT 1 CHECK (authorization_epoch>0), + state text NOT NULL CHECK (state IN ('READY','DISABLED')), + created_at timestamptz NOT NULL DEFAULT clock_timestamp() + ); + CREATE TABLE mst2_snapshot_lease ( + lease_id text PRIMARY KEY, + snapshot_id text NOT NULL REFERENCES mst2_snapshot_context(snapshot_id), + authorization_epoch bigint NOT NULL CHECK (authorization_epoch>0), + publication_sequence bigint NOT NULL CHECK (publication_sequence>0), + writer_epoch bigint NOT NULL CHECK (writer_epoch>0), + certificate_receipt_id bigint NOT NULL REFERENCES mst2_native_publication(receipt_id), + expires_at_unix bigint NOT NULL CHECK (expires_at_unix>=0), + state text NOT NULL CHECK (state IN ('ACTIVE','RELEASED','EXPIRED')), + created_at timestamptz NOT NULL DEFAULT clock_timestamp() + ); + CREATE FUNCTION mst2_snapshot_lease_identity_immutable() RETURNS trigger LANGUAGE plpgsql AS $$ + BEGIN + IF NEW.lease_id<>OLD.lease_id OR NEW.snapshot_id<>OLD.snapshot_id + OR NEW.authorization_epoch<>OLD.authorization_epoch OR NEW.publication_sequence<>OLD.publication_sequence + OR NEW.writer_epoch<>OLD.writer_epoch OR NEW.certificate_receipt_id<>OLD.certificate_receipt_id + OR (OLD.state<>'ACTIVE' AND NEW.state<>OLD.state) THEN + RAISE EXCEPTION 'MST2 lease source cannot change or be revived'; + END IF; + RETURN NEW; + END $$; + CREATE TRIGGER mst2_snapshot_lease_identity_immutable BEFORE UPDATE ON mst2_snapshot_lease + FOR EACH ROW EXECUTE FUNCTION mst2_snapshot_lease_identity_immutable(); + CREATE INDEX mst2_snapshot_lease_active ON mst2_snapshot_lease(snapshot_id,expires_at_unix) + WHERE state='ACTIVE'; + CREATE INDEX mst2_snapshot_lease_expiry ON mst2_snapshot_lease(expires_at_unix) + WHERE state='ACTIVE'; + CREATE FUNCTION mst2_snapshot_identity_immutable() RETURNS trigger LANGUAGE plpgsql AS $$ + BEGIN + IF NEW.snapshot_id<>OLD.snapshot_id OR NEW.canonical_descriptor<>OLD.canonical_descriptor + OR NEW.instance_id<>OLD.instance_id OR NEW.commit_oid<>OLD.commit_oid + OR NEW.root_tree_oid<>OLD.root_tree_oid OR NEW.metadata_root<>OLD.metadata_root + OR NEW.prepare_id<>OLD.prepare_id OR NEW.publication_sequence<>OLD.publication_sequence + OR NEW.writer_epoch<>OLD.writer_epoch + OR NEW.certificate_receipt_id IS DISTINCT FROM OLD.certificate_receipt_id THEN + RAISE EXCEPTION 'MST2 fixed snapshot identity cannot be changed'; + END IF; + RETURN NEW; + END $$; + CREATE TRIGGER mst2_snapshot_identity_immutable BEFORE UPDATE ON mst2_snapshot_context + FOR EACH ROW EXECUTE FUNCTION mst2_snapshot_identity_immutable();" + ).await.map(|_| ()) + } + + async fn down(&self, _manager: &SchemaManager) -> Result<(), DbErr> { + Ok(()) + } +} diff --git a/src/jupiter/migration/mod.rs b/src/jupiter/migration/mod.rs index 7f964e4a..c102df51 100644 --- a/src/jupiter/migration/mod.rs +++ b/src/jupiter/migration/mod.rs @@ -145,6 +145,7 @@ mod m20261005_000200_add_mst2_native_head; mod m20261005_000200_harden_mst2_retention_graph; mod m20261005_000300_add_mst2_metadata_install; mod m20261006_000100_add_view_tables; +mod m20261007_000100_add_mst2_snapshot_sessions; mod runner; pub use m20260905_000100_add_push_queue::ensure_queue_control_seed; pub use runner::apply_migrations; @@ -278,6 +279,7 @@ impl MigratorTrait for Migrator { Box::new(m20261005_000200_harden_mst2_retention_graph::Migration), Box::new(m20261005_000300_add_mst2_metadata_install::Migration), Box::new(m20261006_000100_add_view_tables::Migration), + Box::new(m20261007_000100_add_mst2_snapshot_sessions::Migration), ] } } @@ -1188,7 +1190,7 @@ mod tests { async fn import_repo_alias_rows_canonicalized() { let names = migration_names(); assert_eq!( - &names[names.len() - 8..], + &names[names.len() - 9..], &[ "m20260923_000200_canonicalize_import_repo_paths".to_string(), "m20260925_000100_media_paging".to_string(), @@ -1198,6 +1200,7 @@ mod tests { "m20261005_000200_harden_mst2_retention_graph".to_string(), "m20261005_000300_add_mst2_metadata_install".to_string(), VIEW_MIGRATION_NAME.to_string(), + "m20261007_000100_add_mst2_snapshot_sessions".to_string(), ], "native retention, metadata installation and view tables follow media paging" ); diff --git a/src/jupiter/storage/mod.rs b/src/jupiter/storage/mod.rs index 655a7d02..b130aa5f 100644 --- a/src/jupiter/storage/mod.rs +++ b/src/jupiter/storage/mod.rs @@ -20,6 +20,7 @@ pub(crate) mod mst2_publication_storage; pub mod mst2_retention; pub mod native_metadata_install; pub(crate) mod native_publication_storage; +pub(crate) mod native_snapshot_session; pub mod notification_storage; pub mod object_storage; pub mod oci_db_storage; @@ -158,6 +159,8 @@ pub struct Storage { /// Derived native projection memoization, scoped to this storage assembly. /// Clones share it; independent databases/backends never share entries. pub(crate) native_projection_cache: Arc, + pub(crate) native_snapshot_sessions: + Arc>, pub(crate) projection_observation_sink: Option>, pub cl_service: CLService, @@ -322,6 +325,7 @@ impl Storage { Ok(Storage { app_service: app_service.into(), native_projection_cache: Arc::default(), + native_snapshot_sessions: Arc::default(), projection_observation_sink: None, config_handle, config, @@ -699,6 +703,7 @@ impl Storage { Storage { app_service, native_projection_cache: Arc::default(), + native_snapshot_sessions: Arc::default(), projection_observation_sink: None, // app_service: AppService::mock(), cl_service: CLService::mock(), diff --git a/src/jupiter/storage/mst2_retention.rs b/src/jupiter/storage/mst2_retention.rs index 11c9c456..b336b7a1 100644 --- a/src/jupiter/storage/mst2_retention.rs +++ b/src/jupiter/storage/mst2_retention.rs @@ -129,6 +129,53 @@ impl PostgresRetentionRepository { finish(txn, result).await } + /// Attach a root to an already verified immutable DAG root. Incoming edges + /// retain descendants; this does not copy every page for each lease. + pub(crate) async fn acquire_existing_roots_in_txn( + txn: &DatabaseTransaction, + node_id: &str, + roots: &[RetentionRoot], + ) -> Result<(), SnapshotError> { + validate_id(node_id)?; + if roots.is_empty() || roots.len() > MAX_ROOTS { + return Err(limit_error( + "existing-root acquisition requires 1..=16 roots", + )); + } + let identities = roots + .iter() + .map(root_identity) + .collect::, _>>()?; + let roots: Vec<_> = identities + .iter() + .map(|(key, kind)| json!({"key":key,"kind":kind})) + .collect(); + let encoded = serde_json::to_string(&roots).map_err(internal)?; + let savepoint = begin_graph(txn).await?; + let result=async { + let row=savepoint.query_one_raw(statement( + "SELECT state FROM mst2_retention_node WHERE node_id=$1 FOR UPDATE",[node_id.into()] + )).await.map_err(internal)?.ok_or_else(|| unavailable("metadata root is missing"))?; + if row.try_get::("","state").map_err(internal)?!="LIVE" { + return Err(unavailable("metadata root is not LIVE")); + } + if savepoint.query_one_raw(statement( + "SELECT r.node_id FROM mst2_retention_root r JOIN jsonb_to_recordset($2::jsonb) + AS p(key text,kind text) ON r.root_key=p.key WHERE r.node_id<>$1 OR r.root_kind<>p.kind LIMIT 1", + [node_id.into(),encoded.clone().into()], + )).await.map_err(internal)?.is_some() { + return Err(integrity("retention root is bound to another DAG or kind")); + } + savepoint.execute_raw(statement( + "INSERT INTO mst2_retention_root(node_id,root_key,root_kind) + SELECT $1,p.key,p.kind FROM jsonb_to_recordset($2::jsonb) AS p(key text,kind text) + ON CONFLICT(node_id,root_key) DO NOTHING",[node_id.into(),encoded.into()], + )).await.map_err(internal)?; + Ok(()) + }.await; + finish(savepoint, result).await + } + pub async fn release_root_in_txn( txn: &DatabaseTransaction, root: &RetentionRoot, diff --git a/src/jupiter/storage/native_metadata_install.rs b/src/jupiter/storage/native_metadata_install.rs index 7f7d80ed..659989c4 100644 --- a/src/jupiter/storage/native_metadata_install.rs +++ b/src/jupiter/storage/native_metadata_install.rs @@ -1,5 +1,5 @@ //! Immutable metadata payload CAS and durable native preparation receipts. -//! No runtime, publication, lease or physical collector is enabled here. +//! The native HTTP session authority installs batches and transfers prepare pins. use std::{ collections::{BTreeMap, BTreeSet}, @@ -9,7 +9,7 @@ use std::{ use mst2_codec::metapage::{HEADER_LEN, PAGE_MAX_BYTES, Page, page_id}; use sea_orm::{ ColumnTrait, ConnectionTrait, DatabaseConnection, DatabaseTransaction, DbBackend, EntityTrait, - IsolationLevel, QueryFilter, QuerySelect, Statement, TransactionTrait, + IsolationLevel, QueryFilter, QueryResult, QuerySelect, Statement, TransactionTrait, }; use serde_json::json; @@ -21,7 +21,7 @@ use crate::{ }, ceres::snapshot::{ error::{SnapshotError, SnapshotErrorCode}, - metadata_install::MetadataInstallPlan, + metadata_install::{MAX_PLAN_BYTES, MetadataInstallIdentity, MetadataInstallPlan}, pages::PreparedNativeMetadataRetention, retention::RetentionRoot, retention_dag::{ @@ -71,8 +71,12 @@ impl MetadataPrepareIntent { #[derive(Debug, Clone, PartialEq, Eq)] pub struct PreparedMetadataReceipt { intent: MetadataPrepareIntent, + identity: MetadataInstallIdentity, metadata_root: [u8; 32], payload_bytes: u64, + root_payload_bytes: u64, + node_count: usize, + edge_count: usize, } impl PreparedMetadataReceipt { @@ -115,8 +119,16 @@ impl StoredPlan { fn receipt(&self) -> Result { Ok(PreparedMetadataReceipt { intent: self.intent()?, + identity: self.plan.identity.clone(), metadata_root: self.plan.root, payload_bytes: self.plan.total_bytes, + root_payload_bytes: *self + .plan + .pages + .get(&self.plan.root) + .ok_or_else(|| integrity("metadata preparation root page is missing"))?, + node_count: self.plan.pages.len(), + edge_count: self.plan.edges.len(), }) } } @@ -150,6 +162,49 @@ impl PostgresMetadataInstallRepository { }) } + /// These columns are returned by the same query that authorizes a lease, + /// so a warm cache cannot authorize a stale replica or another schema. + pub(crate) fn verify_primary_scope_row(&self, row: &QueryResult) -> Result<(), SnapshotError> { + let actual = PrimaryStorageScope { + storage_uuid: row + .try_get::>("", "authority_storage_uuid") + .map_err(internal)? + .ok_or_else(|| internal("session authority storage scope is missing"))?, + database: row.try_get("", "authority_database").map_err(internal)?, + database_oid: row + .try_get("", "authority_database_oid") + .map_err(internal)?, + schema: row.try_get("", "authority_schema").map_err(internal)?, + schema_oid: row.try_get("", "authority_schema_oid").map_err(internal)?, + server_address: row + .try_get("", "authority_server_address") + .map_err(internal)?, + server_port: row.try_get("", "authority_server_port").map_err(internal)?, + }; + if row + .try_get::("", "authority_replica") + .map_err(internal)? + || actual != self.storage_scope + { + return Err(internal( + "session authority no longer targets its captured primary storage scope", + )); + } + Ok(()) + } + + pub(crate) async fn verify_primary_connection( + &self, + connection: &C, + ) -> Result<(), SnapshotError> { + if read_storage_scope(connection).await? != self.storage_scope { + return Err(internal( + "session mutation no longer targets its captured primary storage scope", + )); + } + Ok(()) + } + pub async fn begin_intent( &self, operation_id: &str, @@ -203,44 +258,62 @@ impl PostgresMetadataInstallRepository { intent: &MetadataPrepareIntent, payload: &MetadataPagePayload, ) -> Result<(), MetadataInstallError> { - validate_payload(payload)?; + self.install_pages(intent, std::slice::from_ref(payload)) + .await + } + + /// At most 64 pages (1 MiB of canonical payload) per transaction. + pub async fn install_pages( + &self, + intent: &MetadataPrepareIntent, + payloads: &[MetadataPagePayload], + ) -> Result<(), MetadataInstallError> { + if payloads.is_empty() || payloads.len() > 64 { + return Err(SnapshotError::new( + SnapshotErrorCode::LimitExceeded, + "metadata installation batch must contain 1..=64 pages", + ) + .into()); + } + let mut ids = BTreeSet::new(); + for payload in payloads { + validate_payload(payload)?; + if !ids.insert(payload.id) { + return Err(integrity("duplicate page in metadata installation batch").into()); + } + } let txn = self.transaction().await?; let result = async { self.barrier(&txn).await?; let stored = require_plan(&txn, intent).await?; - if stored.plan.pages.get(&payload.id) != Some(&payload.size) { - return Err(integrity( - "metadata payload is not a member of this fixed installation", - )); + for payload in payloads { + if stored.plan.pages.get(&payload.id) != Some(&payload.size) { + return Err(integrity("metadata payload is not a member of this fixed installation")); + } } + let pages: Vec<_> = payloads.iter().map(|p| json!({ + "page_id":hex::encode(p.id), "size":p.size, "payload":hex::encode(&p.bytes) + })).collect(); + let encoded = serde_json::to_string(&pages).map_err(internal)?; txn.execute_raw(statement( "INSERT INTO mst2_metadata_payload(page_id,metadata_codec,byte_size,payload) - VALUES ($1,$2,$3,$4) ON CONFLICT(page_id) DO NOTHING", - [ - payload.id.to_vec().into(), - (stored.plan.identity.metadata_codec as i16).into(), - (payload.size as i32).into(), - payload.bytes.clone().into(), - ], - )) - .await - .map_err(internal)?; - let existing = mst2_metadata_payload::Entity::find_by_id(payload.id.to_vec()) - .one(&txn) - .await - .map_err(internal)? - .ok_or_else(|| integrity("installed metadata payload disappeared"))?; - if existing.payload != payload.bytes - || existing.byte_size as u64 != payload.size - || existing.metadata_codec != stored.plan.identity.metadata_codec as i16 - { - return Err(integrity( - "immutable metadata payload identity conflicts with stored bytes", - )); + SELECT decode(p.page_id,'hex'),$1,p.size,decode(p.payload,'hex') + FROM jsonb_to_recordset($2::jsonb) AS p(page_id text,size integer,payload text) + ON CONFLICT(page_id) DO NOTHING", + [(stored.plan.identity.metadata_codec as i16).into(),encoded.clone().into()], + )).await.map_err(internal)?; + let bad = txn.query_one_raw(statement( + "SELECT p.page_id FROM jsonb_to_recordset($2::jsonb) AS p(page_id text,size integer,payload text) + LEFT JOIN mst2_metadata_payload b ON b.page_id=decode(p.page_id,'hex') + WHERE b.page_id IS NULL OR b.metadata_codec<>$1 OR b.byte_size<>p.size + OR b.payload<>decode(p.payload,'hex') LIMIT 1", + [(stored.plan.identity.metadata_codec as i16).into(),encoded.into()], + )).await.map_err(internal)?; + if bad.is_some() { + return Err(integrity("immutable metadata payload identity conflicts with stored bytes")); } Ok(()) - } - .await; + }.await; commit( txn, result, @@ -251,6 +324,156 @@ impl PostgresMetadataInstallRepository { .await } + pub(crate) async fn verify_receipt_in_txn( + &self, + txn: &DatabaseTransaction, + receipt: &PreparedMetadataReceipt, + tagged_root_tree_oid: &str, + scope: &str, + ) -> Result<(), SnapshotError> { + self.barrier(txn).await?; + let identity = &receipt.identity; + if identity.tagged_root_tree_oid != tagged_root_tree_oid || identity.scope != scope { + return Err(integrity( + "metadata receipt belongs to another fixed source", + )); + } + // The private receipt can only be issued after full validation and a + // definitive commit. Hash the bounded canonical plan in PostgreSQL, + // bind all redundant identity/summary columns to that receipt, and + // check its still-protected LIVE root under the retention barrier. + // This hashes <=2 MiB; it does not transfer or scan pages/edges here. + let row = txn.query_one_raw(statement( + "SELECT p.operation_id,p.manifest_digest, + CASE WHEN octet_length(p.canonical_plan)<=$2 THEN sha256(p.canonical_plan) END AS plan_digest, + p.source_domain,p.tagged_root_tree_oid,p.scope,p.schema_version,p.metadata_codec, + p.materialization_policy,p.fs_semantics,p.access_projection,p.verification_revision, + p.projection_revision,p.metadata_root,p.node_count,p.edge_count,p.total_bytes, + p.state,p.committed_at IS NOT NULL AS committed, + n.kind AS root_kind,n.state AS root_state,n.bytes AS root_bytes, + EXISTS(SELECT 1 FROM mst2_retention_root r WHERE r.node_id=n.node_id + AND r.root_key='prepare:'||p.prepare_id AND r.root_kind='prepare') AS prepare_covered + FROM mst2_metadata_prepare p LEFT JOIN mst2_retention_node n + ON n.node_id='page:sha256:'||encode(p.metadata_root,'hex') WHERE p.prepare_id=$1", + [receipt.intent.prepare_id.clone().into(), (MAX_PLAN_BYTES as i32).into()], + )).await.map_err(internal)?.ok_or_else(|| unavailable("metadata preparation is missing"))?; + if row + .try_get::("", "operation_id") + .map_err(internal)? + != receipt.intent.operation_id + || row + .try_get::>("", "manifest_digest") + .map_err(internal)? + != receipt.intent.manifest_digest + || row + .try_get::>>("", "plan_digest") + .map_err(internal)? + .as_deref() + != Some(receipt.intent.manifest_digest.as_slice()) + || row + .try_get::("", "source_domain") + .map_err(internal)? + != identity.source_domain + || row + .try_get::("", "tagged_root_tree_oid") + .map_err(internal)? + != identity.tagged_root_tree_oid + || row.try_get::("", "scope").map_err(internal)? != identity.scope + || row.try_get::("", "schema_version").map_err(internal)? + != identity.schema_version as i16 + || row.try_get::("", "metadata_codec").map_err(internal)? + != identity.metadata_codec as i16 + || row + .try_get::("", "materialization_policy") + .map_err(internal)? + != identity.materialization_policy as i16 + || row.try_get::("", "fs_semantics").map_err(internal)? + != identity.fs_semantics as i16 + || row + .try_get::("", "access_projection") + .map_err(internal)? + != identity.access_projection as i16 + || row + .try_get::("", "verification_revision") + .map_err(internal)? + != identity.verification_revision + || row + .try_get::("", "projection_revision") + .map_err(internal)? + != identity.projection_revision as i16 + || row + .try_get::>("", "metadata_root") + .map_err(internal)? + != receipt.metadata_root + || row.try_get::("", "node_count").map_err(internal)? != receipt.node_count as i32 + || row.try_get::("", "edge_count").map_err(internal)? != receipt.edge_count as i32 + || row.try_get::("", "total_bytes").map_err(internal)? + != receipt.payload_bytes as i64 + || row.try_get::("", "state").map_err(internal)? != "COMMITTED" + || !row.try_get::("", "committed").map_err(internal)? + { + return Err(integrity( + "metadata preparation differs from its definitive receipt", + )); + } + if row + .try_get::>("", "root_kind") + .map_err(internal)? + .as_deref() + != Some("page") + || row + .try_get::>("", "root_state") + .map_err(internal)? + .as_deref() + != Some("LIVE") + || row + .try_get::>("", "root_bytes") + .map_err(internal)? + != Some(receipt.root_payload_bytes as i64) + || !row + .try_get::("", "prepare_covered") + .map_err(internal)? + { + return Err(unavailable( + "metadata receipt root is not LIVE and protected by its prepare", + )); + } + Ok(()) + } + + pub(crate) async fn restore_session_dag(&self, prepare_id: &str) -> Result<(), SnapshotError> { + let record = mst2_metadata_prepare::Entity::find_by_id(prepare_id.to_owned()) + .one(&self.connection) + .await + .map_err(internal)? + .ok_or_else(|| unavailable("snapshot preparation is missing"))?; + let digest = record + .manifest_digest + .as_slice() + .try_into() + .map_err(|_| integrity("invalid metadata preparation digest"))?; + let stored = load_plan(&self.connection, &record.operation_id, &digest) + .await? + .ok_or_else(|| unavailable("snapshot preparation disappeared"))?; + if stored.record.state != "COMMITTED" { + return Err(unavailable("snapshot preparation is not committed")); + } + load_installed_dag(&self.connection, &stored).await?; + let txn = self.transaction().await?; + let result = async { + self.barrier(&txn).await?; + verify_graph(&txn, &stored).await + } + .await; + match result { + Ok(()) => txn.commit().await.map_err(internal), + Err(error) => { + txn.rollback().await.map_err(internal)?; + Err(error) + } + } + } + /// Verify all stored bytes outside the graph transaction. Immutable payloads /// cannot change; LIVE state and coverage are checked again under the lock. pub async fn load_installed_dag( @@ -702,7 +925,21 @@ async fn verify_graph( .all(connection) .await .map_err(internal)?; - if roots.iter().any(|root| root.root_kind != "prepare") + if roots.is_empty() { + let transferred = connection + .query_one_raw(statement( + "SELECT s.snapshot_id FROM mst2_snapshot_context s + JOIN mst2_retention_root r ON r.root_key='pin:session:'||s.snapshot_id + AND r.node_id='page:sha256:'||encode(s.metadata_root,'hex') AND r.root_kind='pin' + WHERE s.prepare_id=$1 LIMIT 1", + [stored.record.prepare_id.clone().into()], + )) + .await + .map_err(internal)?; + if transferred.is_none() { + return Err(unavailable("metadata protection handoff is incomplete")); + } + } else if roots.iter().any(|root| root.root_kind != "prepare") || roots .into_iter() .map(|root| root.node_id) diff --git a/src/jupiter/storage/native_publication_storage.rs b/src/jupiter/storage/native_publication_storage.rs index 8039a3ea..2dbff900 100644 --- a/src/jupiter/storage/native_publication_storage.rs +++ b/src/jupiter/storage/native_publication_storage.rs @@ -223,7 +223,7 @@ pub(crate) fn decode_native_observation( }) } -async fn observe( +pub(crate) async fn observe( connection: &C, ) -> Result { let rows = connection @@ -268,7 +268,14 @@ impl MonoStorage { &self, instance: &str, ) -> Result { - let observation = observe(self.get_connection()).await?; + Self::read_native_publication_head_from(self.get_connection(), instance).await + } + + pub(crate) async fn read_native_publication_head_from( + connection: &C, + instance: &str, + ) -> Result { + let observation = observe(connection).await?; let head = observation .head .ok_or_else(|| integrity("native publication is not initialized"))?; diff --git a/src/jupiter/storage/native_snapshot_session.rs b/src/jupiter/storage/native_snapshot_session.rs new file mode 100644 index 00000000..415228c8 --- /dev/null +++ b/src/jupiter/storage/native_snapshot_session.rs @@ -0,0 +1,678 @@ +//! Durable native HTTP sessions. Deployment bearer authentication is enforced +//! by the router; leases retain data and do not grant per-principal permission. + +use std::{collections::HashMap, sync::Arc}; + +use mst2_codec::descriptor::ServingDescriptor; +use sea_orm::{ + ConnectionTrait, DatabaseConnection, DatabaseTransaction, DbBackend, EntityTrait, + IsolationLevel, QueryResult, Statement, TransactionTrait, +}; +use tokio::sync::{Mutex, OnceCell}; + +use super::{ + mono_storage::MonoStorage, + mst2_retention::{PostgresRetentionRepository, RETENTION_LOCK_KEY}, + native_metadata_install::{ + MetadataInstallError, PostgresMetadataInstallRepository, PreparedMetadataReceipt, + }, + native_publication_storage::{NativePublicationHead, decode_native_observation}, + push_queue_storage::PushQueueStorage, +}; +use crate::{ + callisto::mst2_snapshot_context, + ceres::snapshot::{ + descriptor::BuiltDescriptor, + error::{SnapshotError, SnapshotErrorCode}, + pages::PreparedNativeMetadataRetention, + retention::RetentionRoot, + runtime::{LeaseRenewed, SnapshotContext}, + view::{SnapshotView, hex}, + }, +}; + +pub(crate) struct PostgresNativeSessionRepository { + connection: DatabaseConnection, + installer: OnceCell, + restored: Mutex>>>, +} + +impl PostgresNativeSessionRepository { + pub(crate) fn new(connection: DatabaseConnection) -> Self { + Self { + connection, + installer: OnceCell::new(), + restored: Mutex::new(HashMap::new()), + } + } + + async fn installer(&self) -> Result<&PostgresMetadataInstallRepository, SnapshotError> { + self.installer + .get_or_try_init(|| PostgresMetadataInstallRepository::new(self.connection.clone())) + .await + } + + pub(crate) async fn install( + &self, + built: &BuiltDescriptor, + prepared: &PreparedNativeMetadataRetention, + ) -> Result { + if prepared.dag().root() != built.descriptor.metadata_root + || prepared.scope() != built.descriptor.scope + { + return Err(integrity( + "prepared DAG differs from the serving descriptor", + )); + } + let installer = self.installer().await?; + let intent = installer + .begin_intent(&format!("http:{}", built.snapshot_id), prepared) + .await + .map_err(install_error)?; + for pages in prepared.dag().payloads().chunks(64) { + installer + .install_pages(&intent, pages) + .await + .map_err(install_error)?; + } + installer.finalize(&intent).await.map_err(install_error) + } + + /// Projection and payload installation happen before this boundary. + /// Cold handoff hashes a <=2 MiB plan, without a page/edge database scan. + /// The formal native head and LIVE root are selected with lease creation. + pub(crate) async fn open( + &self, + expected: &NativePublicationHead, + built: &BuiltDescriptor, + receipt: Option<&PreparedMetadataReceipt>, + lease_seconds: u64, + ) -> Result, SnapshotError> { + let installer = self.installer().await?; + let txn = self.transaction().await?; + let result = async { + PushQueueStorage::acquire_mono_write_lock(&txn).await.map_err(internal)?; + installer.verify_primary_connection(&txn).await?; + let current = MonoStorage::read_native_publication_head_from(&txn, &expected.instance_id) + .await.map_err(|_| not_ready("native publication is not ready"))?; + if current.root != expected.root || current.token != expected.token { + return Err(not_ready("native publication advanced during preparation; retry resolve")); + } + retention_lock(&txn).await?; + let existing = mst2_snapshot_context::Entity::find_by_id(built.snapshot_id.clone()) + .one(&txn).await.map_err(internal)?; + let prepare_id = if let Some(existing) = existing { + if existing.canonical_descriptor != built.descriptor.encode().map_err(internal)? + || existing.commit_oid != expected.root.commit || existing.root_tree_oid != expected.root.tree + || existing.instance_id != expected.instance_id + { + return Err(integrity("immutable session source conflicts with selected publication")); + } + if existing.state != "READY" || existing.authorization_epoch != 1 { + return Err(forbidden()); + } + existing.prepare_id + } else { + let Some(receipt) = receipt else { return Ok(None); }; + if receipt.metadata_root() != built.descriptor.metadata_root { + return Err(integrity("metadata receipt root differs from descriptor")); + } + let prepare_id = receipt.intent().prepare_id().to_owned(); + let tagged_tree=git_internal::hash::ObjectHash::from_hex_for_kind( + git_internal::hash::get_hash_kind(),&expected.root.tree).map_err(internal)?.to_tagged_string(); + self.installer().await?.verify_receipt_in_txn(&txn, receipt,&tagged_tree,&built.descriptor.scope).await?; + txn.execute_raw(statement( + "INSERT INTO mst2_snapshot_context(snapshot_id,canonical_descriptor,instance_id,commit_oid, + root_tree_oid,metadata_root,prepare_id,publication_sequence,writer_epoch,certificate_receipt_id, + authorization_epoch,state) VALUES($1,$2,$3,$4,$5,$6,$7,$8,$9,$10,1,'READY')", + [built.snapshot_id.clone().into(),built.descriptor.encode().map_err(internal)?.into(), + expected.instance_id.clone().into(),expected.root.commit.clone().into(),expected.root.tree.clone().into(), + built.descriptor.metadata_root.to_vec().into(),prepare_id.clone().into(),expected.token.sequence.into(), + expected.token.epoch.into(),expected.token.certificate.into()], + )).await.map_err(internal)?; + prepare_id + }; + let node = root_node(built); + let lease_id = uuid::Uuid::new_v4().to_string(); + PostgresRetentionRepository::acquire_existing_roots_in_txn(&txn,&node,&[ + RetentionRoot::Pin(format!("session:{}",built.snapshot_id)),RetentionRoot::Lease(lease_id.clone()) + ]).await?; + PostgresRetentionRepository::release_root_in_txn(&txn,&RetentionRoot::Prepare(prepare_id)).await?; + let row = txn.query_one_raw(statement( + "INSERT INTO mst2_snapshot_lease(lease_id,snapshot_id,authorization_epoch,expires_at_unix,state, + publication_sequence,writer_epoch,certificate_receipt_id) + VALUES($1,$2,1,floor(extract(epoch FROM clock_timestamp()))::bigint+$3,'ACTIVE',$4,$5,$6) + RETURNING expires_at_unix", + [lease_id.clone().into(),built.snapshot_id.clone().into(),(lease_seconds.clamp(1,3600) as i64).into(), + expected.token.sequence.into(),expected.token.epoch.into(),expected.token.certificate.into()], + )).await.map_err(internal)?.ok_or_else(|| internal("lease insert returned no deadline"))?; + expire_locked(&txn).await?; + Ok(Some(SnapshotContext { built:built.clone(),commit_oid:expected.root.commit.clone(), + root_tree_oid:expected.root.tree.clone(),lease_id, + lease_expires_at_unix:row.try_get::("","expires_at_unix").map_err(internal)? as u64, + authorization_epoch:1 })) + }.await; + let context = finish(txn, result).await?; + if let Some(context) = &context + && receipt.is_some() + { + let _ = self + .verification_cell(&context.built.snapshot_id) + .await + .set(()); + } + Ok(context) + } + + pub(crate) async fn context( + &self, + snapshot_id: &str, + lease_id: &str, + instance_id: &str, + ) -> Result { + let installer = self.installer().await?; + let row = self + .connection + .query_one_raw(statement( + SESSION_SQL, + [snapshot_id.into(), lease_id.into()], + )) + .await + .map_err(internal)? + .ok_or_else(expired)?; + installer.verify_primary_scope_row(&row)?; + let context = match decode_context(&row, snapshot_id, lease_id, instance_id) { + Ok(context) => context, + Err(error) => { + if error.code == SnapshotErrorCode::LeaseExpired { + let txn = self.transaction().await?; + let result = async { + retention_lock(&txn).await?; + expire_specific_locked(&txn, lease_id).await + } + .await; + finish(txn, result).await?; + } + return Err(error); + } + }; + let prepare_id: String = row.try_get("", "prepare_id").map_err(internal)?; + let cell = self.verification_cell(snapshot_id).await; + cell.get_or_try_init(|| async { + self.installer() + .await? + .restore_session_dag(&prepare_id) + .await + }) + .await?; + Ok(context) + } + + pub(crate) async fn renew( + &self, + lease_id: &str, + seconds: u64, + instance: &str, + ) -> Result { + let txn = self.transaction().await?; + let result = async { + retention_lock(&txn).await?; + let lease = txn.query_one_raw(statement( + "SELECT snapshot_id FROM mst2_snapshot_lease WHERE lease_id=$1 FOR UPDATE", + [lease_id.into()], + )).await.map_err(internal)?.ok_or_else(|| SnapshotError::new(SnapshotErrorCode::LeaseUnknown,"unknown lease_id"))?; + let sid: String = lease.try_get("","snapshot_id").map_err(internal)?; + let row = txn.query_one_raw(statement(SESSION_SQL,[sid.clone().into(),lease_id.into()])) + .await.map_err(internal)?.ok_or_else(expired)?; + self.installer().await?.verify_primary_scope_row(&row)?; + if let Err(error)=decode_context(&row,&sid,lease_id,instance) { + if error.code==SnapshotErrorCode::LeaseExpired { + expire_specific_locked(&txn,lease_id).await?; + return Ok(Err(error)); + } + return Err(error); + } + let updated = txn.query_one_raw(statement( + "UPDATE mst2_snapshot_lease SET expires_at_unix=greatest(expires_at_unix, + floor(extract(epoch FROM clock_timestamp()))::bigint)+$2 WHERE lease_id=$1 + AND state='ACTIVE' AND expires_at_unix>=floor(extract(epoch FROM clock_timestamp()))::bigint + RETURNING expires_at_unix", + [lease_id.into(),(seconds.clamp(1,3600) as i64).into()], + )).await.map_err(internal)?; + let Some(updated)=updated else { + expire_specific_locked(&txn,lease_id).await?; + return Ok(Err(expired())); + }; + Ok(Ok(LeaseRenewed { lease_id:lease_id.into(),snapshot_id:sid, + expires_at_unix:updated.try_get::("","expires_at_unix").map_err(internal)? as u64 })) + }.await; + finish(txn, result).await? + } + + pub(crate) async fn release(&self, lease_id: &str) -> Result { + let installer = self.installer().await?; + let txn = self.transaction().await?; + let result = async { + retention_lock(&txn).await?; + installer.verify_primary_connection(&txn).await?; + let changed=txn.query_one_raw(statement( + "UPDATE mst2_snapshot_lease SET state='RELEASED' WHERE lease_id=$1 AND state='ACTIVE' RETURNING snapshot_id", + [lease_id.into()], + )).await.map_err(internal)?; + PostgresRetentionRepository::release_root_in_txn(&txn,&RetentionRoot::Lease(lease_id.into())).await?; + expire_locked(&txn).await?; + if let Some(row)=&changed { + let sid:String=row.try_get("","snapshot_id").map_err(internal)?; + retire_session_pin_locked(&txn,&sid).await?; + } + Ok(changed.is_some()) + }.await; + finish(txn, result).await + } + + async fn verification_cell(&self, sid: &str) -> Arc> { + let mut restored = self.restored.lock().await; + if restored.len() >= 128 + && !restored.contains_key(sid) + && let Some(old) = restored.keys().next().cloned() + { + restored.remove(&old); + } + restored.entry(sid.to_owned()).or_default().clone() + } + + async fn transaction(&self) -> Result { + self.connection + .begin_with_config(Some(IsolationLevel::ReadCommitted), None) + .await + .map_err(internal) + } +} + +const SESSION_SQL: &str = "SELECT s.snapshot_id,s.canonical_descriptor,s.commit_oid,s.root_tree_oid, + (SELECT storage_uuid FROM mst2_metadata_storage_scope WHERE singleton=1) AS authority_storage_uuid, + current_database() AS authority_database, + (SELECT oid::bigint FROM pg_catalog.pg_database WHERE datname=current_database()) AS authority_database_oid, + current_schema() AS authority_schema, + (SELECT oid::bigint FROM pg_catalog.pg_namespace WHERE nspname=current_schema()) AS authority_schema_oid, + inet_server_addr()::text AS authority_server_address,inet_server_port() AS authority_server_port, + pg_is_in_recovery() AS authority_replica, + s.metadata_root,s.prepare_id,s.authorization_epoch,s.state AS session_state,l.lease_id, + l.authorization_epoch AS lease_epoch,l.expires_at_unix,l.state AS lease_state, + floor(extract(epoch FROM clock_timestamp()))::bigint AS db_now,n.state AS root_state, + EXISTS(SELECT 1 FROM mst2_retention_root rr WHERE rr.node_id=n.node_id AND rr.root_key='lease:'||l.lease_id + AND rr.root_kind='lease') AS lease_covered, + EXISTS(SELECT 1 FROM mst2_retention_root rr WHERE rr.node_id=n.node_id AND rr.root_key='pin:session:'||s.snapshot_id + AND rr.root_kind='pin') AS session_covered, + p.state AS prepare_state,p.metadata_root AS prepared_root,p.tagged_root_tree_oid,p.scope AS prepared_scope, + p.source_domain,p.schema_version,p.metadata_codec,p.materialization_policy,p.fs_semantics,p.access_projection, + p.verification_revision,p.projection_revision, + 1::bigint AS root_count,s.commit_oid AS root_commit,s.root_tree_oid AS root_tree,s.instance_id, + l.publication_sequence AS sequence,l.writer_epoch,'READY'::text AS state,s.commit_oid AS head_commit, + s.root_tree_oid AS head_tree,l.certificate_receipt_id,c.receipt_id AS certificate_id, + c.namespace AS certificate_namespace,c.instance_id AS certificate_instance,c.sequence AS certificate_sequence, + c.writer_epoch AS certificate_epoch,c.old_root_commit AS certificate_old_commit, + c.root_commit AS certificate_commit,c.root_tree AS certificate_tree,c.origin_path,c.origin_ref,c.old_path_commit,c.path_commit, + r.id AS receipt_id,r.namespace AS receipt_namespace,r.sequence AS receipt_sequence,r.operation_id,r.old_oid,r.new_oid, + r.writer_epoch AS receipt_epoch,r.writer_kind,r.request_digest,r.request_digest_version,r.native_certificate_version, + o.id AS outbox_id,o.namespace AS outbox_namespace,o.sequence AS outbox_sequence + FROM mst2_snapshot_context s JOIN mst2_snapshot_lease l ON l.snapshot_id=s.snapshot_id AND l.lease_id=$2 + LEFT JOIN mst2_retention_node n ON n.node_id='page:sha256:'||encode(s.metadata_root,'hex') + LEFT JOIN mst2_metadata_prepare p ON p.prepare_id=s.prepare_id + LEFT JOIN mst2_native_publication c ON c.receipt_id=l.certificate_receipt_id + LEFT JOIN mst2_publication r ON r.id=c.receipt_id + LEFT JOIN mst2_publication_outbox o ON o.operation_id=r.operation_id WHERE s.snapshot_id=$1"; + +fn decode_context( + row: &QueryResult, + sid: &str, + lease: &str, + instance: &str, +) -> Result { + let deadline: i64 = row.try_get("", "expires_at_unix").map_err(internal)?; + if row.try_get::("", "lease_state").map_err(internal)? != "ACTIVE" + || deadline < row.try_get::("", "db_now").map_err(internal)? + { + return Err(expired()); + } + let epoch: i64 = row.try_get("", "authorization_epoch").map_err(internal)?; + if row + .try_get::("", "session_state") + .map_err(internal)? + != "READY" + || epoch != 1 + || row.try_get::("", "lease_epoch").map_err(internal)? != epoch + { + return Err(forbidden()); + } + if row + .try_get::>("", "root_state") + .map_err(internal)? + .as_deref() + != Some("LIVE") + || !row.try_get::("", "lease_covered").map_err(internal)? + || !row + .try_get::("", "session_covered") + .map_err(internal)? + { + return Err(SnapshotError::new( + SnapshotErrorCode::ObjectUnavailable, + "snapshot root protection is unavailable", + )); + } + let head = decode_native_observation(row) + .map_err(|_| integrity("durable native source certificate is invalid"))? + .head + .ok_or_else(|| integrity("durable native source is missing"))?; + if head.instance_id != instance { + return Err(forbidden()); + } + let canonical: Vec = row.try_get("", "canonical_descriptor").map_err(internal)?; + let descriptor = ServingDescriptor::decode(&canonical) + .map_err(|_| integrity("invalid durable serving descriptor"))?; + let view = SnapshotView::from_commit(&head.root.commit, &head.root.tree); + let root: Vec = row.try_get("", "metadata_root").map_err(internal)?; + let prepared_root: Option> = row.try_get("", "prepared_root").map_err(internal)?; + let tagged_tree = git_internal::hash::ObjectHash::from_hex_for_kind( + git_internal::hash::get_hash_kind(), + &head.root.tree, + ) + .map_err(|_| integrity("invalid durable tree identity"))? + .to_tagged_string(); + if format!( + "sha256:{}", + hex(&descriptor.snapshot_id().map_err(internal)?) + ) != sid + || uuid::Uuid::from_bytes(descriptor.instance_uuid).to_string() != instance + || format!("sha256:{}", hex(&descriptor.namespace_view_id)) != view.view_id + || descriptor.metadata_root.as_slice() != root.as_slice() + || prepared_root.as_deref() != Some(root.as_slice()) + || row + .try_get::>("", "prepare_state") + .map_err(internal)? + .as_deref() + != Some("COMMITTED") + || row + .try_get::>("", "tagged_root_tree_oid") + .map_err(internal)? + .as_deref() + != Some(tagged_tree.as_str()) + || row + .try_get::>("", "prepared_scope") + .map_err(internal)? + .as_deref() + != Some(descriptor.scope.as_str()) + { + return Err(integrity( + "durable snapshot identity or source proof is inconsistent", + )); + } + let profile = [ + ("schema_version", mst2_codec::descriptor::SCHEMA_VERSION), + ("metadata_codec", mst2_codec::descriptor::METADATA_CODEC), + ( + "materialization_policy", + mst2_codec::descriptor::MATERIALIZATION_POLICY_GIT_RAW_V1, + ), + ( + "fs_semantics", + mst2_codec::descriptor::FS_SEMANTICS_LINUX_CODE_V1, + ), + ( + "access_projection", + mst2_codec::descriptor::ACCESS_PROJECTION_EXACT_FULL, + ), + ( + "projection_revision", + crate::ceres::snapshot::projection_observation::NATIVE_PROJECTION_REVISION, + ), + ]; + if profile.iter().any(|(column, value)| { + row.try_get::>("", column).ok().flatten() != Some(*value as i16) + }) || row + .try_get::>("", "source_domain") + .map_err(internal)? + .as_deref() + != Some("native-git") + || row + .try_get::>("", "verification_revision") + .map_err(internal)? + != Some(super::mono_storage::MST2_VERIFICATION_VERSION) + { + return Err(integrity( + "durable snapshot materialization profile changed", + )); + } + Ok(SnapshotContext { + built: BuiltDescriptor { + instance_id: instance.into(), + snapshot_id: sid.into(), + metadata_root: format!("sha256:{}", hex(&descriptor.metadata_root)), + descriptor, + }, + commit_oid: head.root.commit, + root_tree_oid: head.root.tree, + lease_id: lease.into(), + lease_expires_at_unix: deadline as u64, + authorization_epoch: epoch as u64, + }) +} + +async fn retention_lock(txn: &DatabaseTransaction) -> Result<(), SnapshotError> { + txn.execute_raw(statement( + "SELECT pg_advisory_xact_lock($1,hashtext(current_schema()))", + [RETENTION_LOCK_KEY.into()], + )) + .await + .map_err(internal) + .map(|_| ()) +} + +async fn retire_session_pin_locked( + txn: &DatabaseTransaction, + sid: &str, +) -> Result<(), SnapshotError> { + txn.execute_raw(statement( + "DELETE FROM mst2_retention_root r WHERE r.root_kind='pin' AND r.root_key='pin:session:'||$1 + AND NOT EXISTS(SELECT 1 FROM mst2_snapshot_lease l WHERE l.snapshot_id=$1 AND l.state='ACTIVE' + AND l.expires_at_unix>=floor(extract(epoch FROM clock_timestamp()))::bigint)",[sid.into()] + )).await.map_err(internal).map(|_| ()) +} + +async fn expire_specific_locked( + txn: &DatabaseTransaction, + lease: &str, +) -> Result<(), SnapshotError> { + let row=txn.query_one_raw(statement( + "UPDATE mst2_snapshot_lease SET state='EXPIRED' WHERE lease_id=$1 AND state='ACTIVE' + AND expires_at_unix("", "snapshot_id").map_err(internal)?, + ) + .await?; + } + Ok(()) +} + +async fn expire_locked(txn: &DatabaseTransaction) -> Result<(), SnapshotError> { + let rows = txn + .query_all_raw(statement( + "UPDATE mst2_snapshot_lease SET state='EXPIRED' WHERE lease_id IN + (SELECT lease_id FROM mst2_snapshot_lease WHERE state='ACTIVE' + AND expires_at_unix("","lease_id").map_err(internal)?, + "sid":row.try_get::("","snapshot_id").map_err(internal)? + })) + }) + .collect::, SnapshotError>>()?; + let encoded = serde_json::to_string(&rows).map_err(internal)?; + txn.execute_raw(statement( + "DELETE FROM mst2_retention_root r USING jsonb_to_recordset($1::jsonb) AS p(lease text,sid text) + WHERE r.root_kind='lease' AND r.root_key='lease:'||p.lease",[encoded.clone().into()] + )).await.map_err(internal)?; + txn.execute_raw(statement( + "DELETE FROM mst2_retention_root r USING jsonb_to_recordset($1::jsonb) AS p(lease text,sid text) + WHERE r.root_kind='pin' AND r.root_key='pin:session:'||p.sid + AND NOT EXISTS(SELECT 1 FROM mst2_snapshot_lease l WHERE l.snapshot_id=p.sid AND l.state='ACTIVE' + AND l.expires_at_unix>=floor(extract(epoch FROM clock_timestamp()))::bigint)",[encoded.into()] + )).await.map_err(internal).map(|_| ()) +} + +async fn finish( + txn: DatabaseTransaction, + result: Result, +) -> Result { + match result { + Ok(value) => { + txn.commit().await.map_err(|error| { + tracing::error!(error=%error,"native session commit outcome unknown"); + SnapshotError::new( + SnapshotErrorCode::TemporaryUnavailable, + "session commit outcome unknown; retry resolve", + ) + })?; + Ok(value) + } + Err(error) => { + txn.rollback().await.map_err(internal)?; + Err(error) + } + } +} +fn statement(sql: &str, values: [sea_orm::Value; N]) -> Statement { + Statement::from_sql_and_values(DbBackend::Postgres, sql, values) +} +fn root_node(built: &BuiltDescriptor) -> String { + format!("page:{}", built.metadata_root) +} +fn internal(error: impl std::fmt::Display) -> SnapshotError { + SnapshotError::new(SnapshotErrorCode::Internal, error.to_string()) +} +fn integrity(message: &str) -> SnapshotError { + SnapshotError::new(SnapshotErrorCode::IntegrityError, message) +} +fn not_ready(message: &str) -> SnapshotError { + SnapshotError::new(SnapshotErrorCode::SnapshotNotReady, message) +} +fn expired() -> SnapshotError { + SnapshotError::new( + SnapshotErrorCode::LeaseExpired, + "lease is not active for this snapshot; re-resolve", + ) +} +fn forbidden() -> SnapshotError { + SnapshotError::new( + SnapshotErrorCode::ScopeForbidden, + "snapshot serving state or authorization epoch changed", + ) +} +fn install_error(error: MetadataInstallError) -> SnapshotError { + match error { + MetadataInstallError::Rejected(error) if error.code == SnapshotErrorCode::Internal => { + tracing::error!(%error,"native metadata installation unavailable"); + SnapshotError::new( + SnapshotErrorCode::TemporaryUnavailable, + "native metadata installation unavailable", + ) + } + MetadataInstallError::Rejected(error) => error, + MetadataInstallError::CommitUncertain { + operation_id, + phase, + .. + } => { + tracing::error!(%operation_id,?phase,"native metadata commit outcome unknown"); + SnapshotError::new( + SnapshotErrorCode::TemporaryUnavailable, + "metadata commit outcome unknown; retry resolve", + ) + } + } +} + +impl super::Storage { + pub(crate) async fn snapshot_sessions(&self) -> &PostgresNativeSessionRepository { + use super::base_storage::StorageConnector; + self.native_snapshot_sessions + .get_or_init(|| async { + PostgresNativeSessionRepository::new(self.mono_storage().get_connection().clone()) + }) + .await + } + + pub(crate) async fn snapshot_context( + &self, + sid: &str, + lease: &str, + ) -> Result { + let config = self.config(); + if config.mst2.publication_enabled { + let instance = config + .mst2 + .instance_uuid + .as_deref() + .ok_or_else(|| not_ready("native instance missing"))?; + let instance = uuid::Uuid::parse_str(instance) + .map_err(|_| not_ready("invalid native instance"))? + .to_string(); + self.snapshot_sessions() + .await + .context(sid, lease, &instance) + .await + } else { + let runtime = crate::ceres::snapshot::runtime::runtime(); + runtime.validate_lease(sid, lease)?; + runtime.context(sid) + } + } + + pub(crate) async fn snapshot_renew( + &self, + lease: &str, + seconds: u64, + ) -> Result { + let config = self.config(); + if config.mst2.publication_enabled { + let instance = config + .mst2 + .instance_uuid + .as_deref() + .ok_or_else(|| not_ready("native instance missing"))?; + let instance = uuid::Uuid::parse_str(instance) + .map_err(|_| not_ready("invalid native instance"))? + .to_string(); + self.snapshot_sessions() + .await + .renew(lease, seconds, &instance) + .await + } else { + crate::ceres::snapshot::runtime::runtime().renew_lease(lease, seconds) + } + } + + pub(crate) async fn snapshot_release(&self, lease: &str) -> Result { + if self.config().mst2.publication_enabled { + self.snapshot_sessions().await.release(lease).await + } else { + Ok(crate::ceres::snapshot::runtime::runtime().release_lease(lease)) + } + } +} diff --git a/src/jupiter/tests.rs b/src/jupiter/tests.rs index f79aeb41..2b03da07 100644 --- a/src/jupiter/tests.rs +++ b/src/jupiter/tests.rs @@ -378,6 +378,7 @@ pub async fn test_storage_with_config(temp_dir: impl AsRef, config: Config Storage { app_service: Arc::new(svc), native_projection_cache: Arc::default(), + native_snapshot_sessions: Arc::default(), projection_observation_sink: None, cl_service: CLService::mock(), push_queue_service: PushQueueService::new( From 5217ab6d75e5067b9c8f2101e1fe9a50323932c7 Mon Sep 17 00:00:00 2001 From: Xiaoyang Han Date: Wed, 7 Oct 2026 18:25:26 +0800 Subject: [PATCH 04/50] fix(mst2): keep authentication errors typed until middleware response Resolve native all-target Clippy result_large_err by returning SnapshotError from authentication. Construct the identical HTTP error envelope at the two existing middleware exits inside the same request-ID scope. Preserve bearer and lease checks, codes, messages and every regression assertion; do not suppress the lint. --- src/api/router/snapshot_router.rs | 18 +++++++----------- 1 file changed, 7 insertions(+), 11 deletions(-) diff --git a/src/api/router/snapshot_router.rs b/src/api/router/snapshot_router.rs index 4c2604d2..f2c3515f 100644 --- a/src/api/router/snapshot_router.rs +++ b/src/api/router/snapshot_router.rs @@ -239,7 +239,7 @@ async fn snapshot_auth_middleware( let path = req.uri().path().to_owned(); let context = match authenticate_request(&state, &headers, &path).await { Ok(context) => context, - Err(response) => return response, + Err(error) => return mst2_error_response(error), }; REQUEST_HEADERS .scope( @@ -258,10 +258,10 @@ async fn snapshot_auth_middleware( } } None => { - if let Err(response) = + if let Err(error) = authenticate_request(&state, &headers, &path).await { - return response; + return mst2_error_response(error); } } } @@ -323,11 +323,8 @@ fn bearer_ok(headers: &HeaderMap, token: &str) -> bool { .is_some_and(|cred| cred == token) } -fn unauthenticated(message: &'static str) -> Response { - mst2_error_response(SnapshotError::new( - SnapshotErrorCode::Unauthenticated, - message, - )) +fn unauthenticated(message: &'static str) -> SnapshotError { + SnapshotError::new(SnapshotErrorCode::Unauthenticated, message) } /// Auth decision for one request path (see [`snapshot_auth_middleware`]). @@ -337,7 +334,7 @@ async fn authenticate_request( state: &MonoApiServiceState, headers: &HeaderMap, path: &str, -) -> Result, Response> { +) -> Result, SnapshotError> { let config = state.storage.config(); let token = config.mst2.auth_token.as_deref(); let path = path.split('?').next().unwrap_or(path); @@ -369,8 +366,7 @@ async fn authenticate_request( .storage .snapshot_context(snapshot_id, lease) .await - .map(Some) - .map_err(mst2_error_response), + .map(Some), } } } From 20dab4e39ae7e1910994ab5301af89b7f8754c20 Mon Sep 17 00:00:00 2001 From: Xiaoyang Han Date: Wed, 7 Oct 2026 18:54:55 +0800 Subject: [PATCH 05/50] feat(mst2): persist fixed metadata generations for native preparation (#57) Persist complete metadata lifetime bindings and private storage seals before payload installation. Verify exact generations at observed-DAG finalization and recovery; preserve existing v3 HTTP sessions and all original regression assertions. Isolate this repository until generation-aware collection and session adoption are implemented. Native Linux gates and performance remain pending. --- src/api/router/snapshot_content_tests.rs | 3 + .../snapshot_generation_upgrade_tests.rs | 178 ++++ src/callisto/mod.rs | 1 + src/callisto/mst2_metadata_lifetime.rs | 17 + src/callisto/mst2_metadata_payload.rs | 1 + src/callisto/mst2_metadata_prepare.rs | 4 + src/callisto/mst2_metadata_prepare_page.rs | 1 + ...07_000200_add_mst2_metadata_generations.rs | 91 ++ src/jupiter/migration/mod.rs | 8 +- .../native_metadata_generation_tests.rs | 906 ++++++++++++++++++ .../storage/native_metadata_generations.rs | 866 +++++++++++++++++ .../storage/native_metadata_install.rs | 25 +- 12 files changed, 2098 insertions(+), 3 deletions(-) create mode 100644 src/api/router/snapshot_generation_upgrade_tests.rs create mode 100644 src/callisto/mst2_metadata_lifetime.rs create mode 100644 src/jupiter/migration/m20261007_000200_add_mst2_metadata_generations.rs create mode 100644 src/jupiter/storage/native_metadata_generation_tests.rs create mode 100644 src/jupiter/storage/native_metadata_generations.rs diff --git a/src/api/router/snapshot_content_tests.rs b/src/api/router/snapshot_content_tests.rs index aa697589..3a7c8390 100644 --- a/src/api/router/snapshot_content_tests.rs +++ b/src/api/router/snapshot_content_tests.rs @@ -82,6 +82,9 @@ const TOKEN: &str = "mst2-fixed-content-test"; #[path = "snapshot_session_tests.rs"] mod durable_sessions; +#[path = "snapshot_generation_upgrade_tests.rs"] +mod generation_upgrade; + #[derive(Default)] struct ReadCounts { whole: AtomicUsize, diff --git a/src/api/router/snapshot_generation_upgrade_tests.rs b/src/api/router/snapshot_generation_upgrade_tests.rs new file mode 100644 index 00000000..c2afda08 --- /dev/null +++ b/src/api/router/snapshot_generation_upgrade_tests.rs @@ -0,0 +1,178 @@ +use sea_orm_migration::MigratorTrait; + +use super::*; +use crate::jupiter::migration::Migrator; + +async fn scalar(db: &sea_orm::DatabaseConnection, sql: &str) -> i64 { + db.query_one_raw(sea_orm::Statement::from_string( + sea_orm::DbBackend::Postgres, + sql, + )) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap() +} + +#[tokio::test] +async fn mst2_generation_additive_upgrade_preserves_legacy_v3_sid_lease_and_new_resolve() { + let fixture = Fixture::new_with_pg_config(true).await; + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + assert_eq!( + scalar(db, "SELECT count(*) FROM mst2_metadata_lifetime").await, + 0 + ); + assert_eq!( + scalar( + db, + "SELECT count(*) FROM mst2_metadata_prepare WHERE storage_seal IS NOT NULL" + ) + .await, + 0 + ); + assert_eq!( + scalar( + db, + "SELECT count(*) FROM mst2_metadata_payload WHERE generation IS NOT NULL" + ) + .await, + 0 + ); + let pages = scalar(db, "SELECT count(*) FROM mst2_metadata_payload").await; + let original = success_json(fixture.send("GET", "descriptor", Body::empty()).await).await; + // The unchanged v3 installer created these durable rows. Remove only the + // empty additive schema in this isolated fixture to reproduce the previous + // deployed schema; the SID, lease, CAS bytes and all original guards survive. + db.execute_unprepared( + "DROP TRIGGER mst2_metadata_generation_mapping_guard ON mst2_metadata_prepare_page; + DROP TRIGGER mst2_metadata_generation_seal_guard ON mst2_metadata_prepare; + DROP FUNCTION mst2_metadata_generation_mapping_guard(); + DROP FUNCTION mst2_metadata_generation_seal_guard(); + ALTER TABLE mst2_metadata_prepare DROP COLUMN canonical_bindings,DROP COLUMN bindings_digest, + DROP COLUMN primary_scope,DROP COLUMN storage_seal; + ALTER TABLE mst2_metadata_prepare_page DROP COLUMN generation; + ALTER TABLE mst2_metadata_payload DROP COLUMN generation; + DROP TABLE mst2_metadata_lifetime; + DROP FUNCTION mst2_metadata_lifetime_guard(); + DELETE FROM seaql_migrations WHERE version='m20261007_000200_add_mst2_metadata_generations'" + ).await.unwrap(); + assert_eq!( + scalar( + db, + "SELECT count(*) FROM mst2_snapshot_lease WHERE state='ACTIVE'" + ) + .await, + 1 + ); + assert_eq!( + scalar(db, "SELECT count(*) FROM mst2_snapshot_context").await, + 1 + ); + assert_eq!( + success_json(fixture.send("GET", "descriptor", Body::empty()).await).await, + original + ); + Migrator::up(db, None).await.unwrap(); + assert_eq!( + scalar( + db, + "SELECT count(*) FROM mst2_metadata_payload WHERE generation IS NULL" + ) + .await, + pages + ); + assert_eq!( + scalar( + db, + "SELECT count(*) FROM mst2_metadata_prepare WHERE storage_seal IS NULL" + ) + .await, + 1 + ); + assert_eq!( + success_json(fixture.send("GET", "descriptor", Body::empty()).await).await, + original + ); + + let config = fixture.state.storage.config(); + let connection = crate::jupiter::storage::init::postgres_connection(&config.database) + .await + .unwrap(); + let storage = crate::jupiter::storage::Storage::new_with_connection( + config, + Arc::new(connection), + fixture.state.storage.git_service.obj_storage.clone(), + ) + .await + .unwrap(); + let state = MonoApiServiceState { + storage, + ..fixture.state.clone() + }; + let app = Router::new().nest("/api/v2", routers(state.clone()).with_state(state)); + let restored = success_json( + app.clone() + .oneshot(fixture.request("GET", "descriptor", Body::empty())) + .await + .unwrap(), + ) + .await; + assert_eq!(restored, original); + let next = success_json( + app.clone() + .oneshot( + Request::builder() + .method("POST") + .uri("/api/v2/snapshots/resolve") + .header("authorization", format!("Bearer {TOKEN}")) + .header("content-type", "application/json") + .body(Body::from( + json!({"target":{"kind":"latest"},"scope":"/project"}).to_string(), + )) + .unwrap(), + ) + .await + .unwrap(), + ) + .await; + assert_eq!(next["descriptor"]["snapshot_id"], fixture.snapshot); + assert_ne!(next["lease_id"], fixture.lease); + assert_eq!( + scalar( + db, + "SELECT count(*) FROM mst2_snapshot_lease WHERE state='ACTIVE'" + ) + .await, + 2 + ); + let content = app + .oneshot(fixture.request("GET", "blob?path=/file", Body::empty())) + .await + .unwrap(); + assert_eq!(content.status(), 200); + assert_eq!( + to_bytes(content.into_body(), usize::MAX) + .await + .unwrap() + .as_ref(), + fixture.raw.as_slice() + ); + assert_eq!( + scalar(db, "SELECT count(*) FROM mst2_metadata_lifetime").await, + 0 + ); + assert_eq!( + scalar(db, "SELECT count(*) FROM mst2_metadata_payload").await, + pages + ); + assert_eq!( + scalar( + db, + "SELECT count(*) FROM mst2_metadata_prepare WHERE storage_seal IS NULL" + ) + .await, + 1 + ); +} diff --git a/src/callisto/mod.rs b/src/callisto/mod.rs index c8d90192..8eb43be5 100644 --- a/src/callisto/mod.rs +++ b/src/callisto/mod.rs @@ -66,6 +66,7 @@ pub mod mega_view_root_chain_scan; pub mod mega_webhook; pub mod mega_webhook_delivery; pub mod mega_webhook_event_type; +pub mod mst2_metadata_lifetime; pub mod mst2_metadata_payload; pub mod mst2_metadata_prepare; pub mod mst2_metadata_prepare_page; diff --git a/src/callisto/mst2_metadata_lifetime.rs b/src/callisto/mst2_metadata_lifetime.rs new file mode 100644 index 00000000..9807c27a --- /dev/null +++ b/src/callisto/mst2_metadata_lifetime.rs @@ -0,0 +1,17 @@ +use sea_orm::entity::prelude::*; + +#[derive(Clone, Debug, PartialEq, DeriveEntityModel, Eq)] +#[sea_orm(table_name = "mst2_metadata_lifetime")] +pub struct Model { + #[sea_orm(primary_key, auto_increment = false)] + pub page_id: Vec, + pub node_id: String, + pub generation: i64, + pub state: String, + pub metadata_codec: i16, + pub expected_size: i32, +} + +#[derive(Copy, Clone, Debug, EnumIter, DeriveRelation)] +pub enum Relation {} +impl ActiveModelBehavior for ActiveModel {} diff --git a/src/callisto/mst2_metadata_payload.rs b/src/callisto/mst2_metadata_payload.rs index 798aab27..9ec1a65b 100644 --- a/src/callisto/mst2_metadata_payload.rs +++ b/src/callisto/mst2_metadata_payload.rs @@ -5,6 +5,7 @@ use sea_orm::entity::prelude::*; pub struct Model { #[sea_orm(primary_key, auto_increment = false)] pub page_id: Vec, + pub generation: Option, pub metadata_codec: i16, pub byte_size: i32, pub payload: Vec, diff --git a/src/callisto/mst2_metadata_prepare.rs b/src/callisto/mst2_metadata_prepare.rs index 9401282b..98b017a8 100644 --- a/src/callisto/mst2_metadata_prepare.rs +++ b/src/callisto/mst2_metadata_prepare.rs @@ -8,6 +8,10 @@ pub struct Model { pub operation_id: String, pub manifest_digest: Vec, pub canonical_plan: Vec, + pub canonical_bindings: Option>, + pub bindings_digest: Option>, + pub primary_scope: Option>, + pub storage_seal: Option>, pub source_domain: String, pub tagged_root_tree_oid: String, pub scope: String, diff --git a/src/callisto/mst2_metadata_prepare_page.rs b/src/callisto/mst2_metadata_prepare_page.rs index 61299f7e..7946022e 100644 --- a/src/callisto/mst2_metadata_prepare_page.rs +++ b/src/callisto/mst2_metadata_prepare_page.rs @@ -8,6 +8,7 @@ pub struct Model { #[sea_orm(primary_key, auto_increment = false)] pub page_id: Vec, pub expected_size: i32, + pub generation: Option, } #[derive(Copy, Clone, Debug, EnumIter, DeriveRelation)] diff --git a/src/jupiter/migration/m20261007_000200_add_mst2_metadata_generations.rs b/src/jupiter/migration/m20261007_000200_add_mst2_metadata_generations.rs new file mode 100644 index 00000000..2c098c19 --- /dev/null +++ b/src/jupiter/migration/m20261007_000200_add_mst2_metadata_generations.rs @@ -0,0 +1,91 @@ +//! Fixed metadata lifetimes. Legacy rows remain explicitly unbound. + +use mst2_codec::metapage::{HEADER_LEN, PAGE_MAX_BYTES}; +use sea_orm_migration::prelude::*; + +#[derive(DeriveMigrationName)] +pub struct Migration; + +#[async_trait::async_trait] +impl MigrationTrait for Migration { + async fn up(&self, manager: &SchemaManager) -> Result<(), DbErr> { + manager.get_connection().execute_unprepared(&format!( + "CREATE TABLE mst2_metadata_lifetime ( + page_id bytea PRIMARY KEY CHECK (octet_length(page_id)=32), + node_id text NOT NULL UNIQUE CHECK (node_id='page:sha256:'||encode(page_id,'hex')), + generation bigint NOT NULL CHECK (generation>0), + state text NOT NULL CHECK (state IN ('RESERVED','LIVE','DELETING','REMOVED')), + metadata_codec smallint NOT NULL CHECK (metadata_codec=1), + expected_size integer NOT NULL CHECK (expected_size BETWEEN {HEADER_LEN} AND {PAGE_MAX_BYTES}), + UNIQUE(page_id,generation) + ); + CREATE INDEX idx_mst2_metadata_lifetime_state_page ON mst2_metadata_lifetime(state,page_id); + CREATE FUNCTION mst2_metadata_lifetime_guard() RETURNS trigger LANGUAGE plpgsql AS $$ + BEGIN + IF TG_OP='DELETE' THEN RAISE EXCEPTION 'metadata lifetime watermark cannot be deleted'; END IF; + IF NEW.page_id IS DISTINCT FROM OLD.page_id OR NEW.node_id IS DISTINCT FROM OLD.node_id + OR NEW.generation IS DISTINCT FROM OLD.generation + OR NEW.metadata_codec IS DISTINCT FROM OLD.metadata_codec + OR NEW.expected_size IS DISTINCT FROM OLD.expected_size THEN + RAISE EXCEPTION 'metadata lifetime identity is immutable'; + END IF; + IF NEW.state IS DISTINCT FROM OLD.state AND NOT (OLD.state='RESERVED' AND NEW.state='LIVE') THEN + RAISE EXCEPTION 'metadata lifetime transition requires generation collector'; + END IF; + RETURN NEW; + END $$; + CREATE TRIGGER mst2_metadata_lifetime_guard BEFORE UPDATE OR DELETE + ON mst2_metadata_lifetime FOR EACH ROW EXECUTE FUNCTION mst2_metadata_lifetime_guard(); + ALTER TABLE mst2_metadata_payload ADD COLUMN generation bigint CHECK (generation>0), + ADD CONSTRAINT mst2_metadata_payload_lifetime_fk FOREIGN KEY(page_id,generation) + REFERENCES mst2_metadata_lifetime(page_id,generation); + ALTER TABLE mst2_metadata_prepare_page ADD COLUMN generation bigint CHECK (generation>0), + ADD CONSTRAINT mst2_metadata_prepare_page_lifetime_fk FOREIGN KEY(page_id,generation) + REFERENCES mst2_metadata_lifetime(page_id,generation); + CREATE INDEX idx_mst2_metadata_prepare_page_lifetime + ON mst2_metadata_prepare_page(page_id,generation,prepare_id); + ALTER TABLE mst2_metadata_prepare + ADD COLUMN canonical_bindings bytea, + ADD COLUMN bindings_digest bytea CHECK (octet_length(bindings_digest)=32), + ADD COLUMN primary_scope bytea CHECK (octet_length(primary_scope) BETWEEN 1 AND 16384), + ADD COLUMN storage_seal bytea CHECK (octet_length(storage_seal)=32), + ADD CONSTRAINT mst2_metadata_prepare_generation_seal_check CHECK ( + (canonical_bindings IS NULL AND bindings_digest IS NULL AND primary_scope IS NULL AND storage_seal IS NULL) + OR (canonical_bindings IS NOT NULL AND octet_length(canonical_bindings) BETWEEN 60 AND 196620 + AND bindings_digest IS NOT NULL AND primary_scope IS NOT NULL AND storage_seal IS NOT NULL) + ); + CREATE FUNCTION mst2_metadata_generation_seal_guard() RETURNS trigger LANGUAGE plpgsql AS $$ + BEGIN + IF NEW.canonical_bindings IS DISTINCT FROM OLD.canonical_bindings + OR NEW.bindings_digest IS DISTINCT FROM OLD.bindings_digest + OR NEW.primary_scope IS DISTINCT FROM OLD.primary_scope + OR NEW.storage_seal IS DISTINCT FROM OLD.storage_seal THEN + RAISE EXCEPTION 'metadata generation seal cannot be rebound'; + END IF; + RETURN NEW; + END $$; + CREATE TRIGGER mst2_metadata_generation_seal_guard BEFORE UPDATE ON mst2_metadata_prepare + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_generation_seal_guard(); + CREATE FUNCTION mst2_metadata_generation_mapping_guard() RETURNS trigger LANGUAGE plpgsql AS $$ + DECLARE fixed boolean; preparing boolean; + BEGIN + SELECT storage_seal IS NOT NULL,state='PREPARING' INTO fixed,preparing + FROM mst2_metadata_prepare WHERE prepare_id=CASE WHEN TG_OP='DELETE' THEN OLD.prepare_id ELSE NEW.prepare_id END FOR UPDATE; + IF fixed AND (TG_OP<>'INSERT' OR NOT preparing) THEN + RAISE EXCEPTION 'metadata generation mappings are immutable'; + END IF; + IF TG_OP='UPDATE' AND OLD.prepare_id IS DISTINCT FROM NEW.prepare_id AND EXISTS ( + SELECT 1 FROM mst2_metadata_prepare WHERE prepare_id=OLD.prepare_id AND storage_seal IS NOT NULL + ) THEN RAISE EXCEPTION 'metadata generation mapping cannot leave its prepare'; END IF; + IF TG_OP='DELETE' THEN RETURN OLD; END IF; + RETURN NEW; + END $$; + CREATE TRIGGER mst2_metadata_generation_mapping_guard BEFORE INSERT OR UPDATE OR DELETE + ON mst2_metadata_prepare_page FOR EACH ROW EXECUTE FUNCTION mst2_metadata_generation_mapping_guard()" + )).await.map(|_| ()) + } + + async fn down(&self, _manager: &SchemaManager) -> Result<(), DbErr> { + Ok(()) + } +} diff --git a/src/jupiter/migration/mod.rs b/src/jupiter/migration/mod.rs index c102df51..6ceda285 100644 --- a/src/jupiter/migration/mod.rs +++ b/src/jupiter/migration/mod.rs @@ -146,6 +146,7 @@ mod m20261005_000200_harden_mst2_retention_graph; mod m20261005_000300_add_mst2_metadata_install; mod m20261006_000100_add_view_tables; mod m20261007_000100_add_mst2_snapshot_sessions; +mod m20261007_000200_add_mst2_metadata_generations; mod runner; pub use m20260905_000100_add_push_queue::ensure_queue_control_seed; pub use runner::apply_migrations; @@ -280,6 +281,7 @@ impl MigratorTrait for Migrator { Box::new(m20261005_000300_add_mst2_metadata_install::Migration), Box::new(m20261006_000100_add_view_tables::Migration), Box::new(m20261007_000100_add_mst2_snapshot_sessions::Migration), + Box::new(m20261007_000200_add_mst2_metadata_generations::Migration), ] } } @@ -1190,7 +1192,7 @@ mod tests { async fn import_repo_alias_rows_canonicalized() { let names = migration_names(); assert_eq!( - &names[names.len() - 9..], + &names[names.len() - 10..names.len() - 1], &[ "m20260923_000200_canonicalize_import_repo_paths".to_string(), "m20260925_000100_media_paging".to_string(), @@ -1204,6 +1206,10 @@ mod tests { ], "native retention, metadata installation and view tables follow media paging" ); + assert_eq!( + names.last().unwrap(), + "m20261007_000200_add_mst2_metadata_generations" + ); let db = alias_db().await; insert_repo(&db, 1, "/third-party//a").await; diff --git a/src/jupiter/storage/native_metadata_generation_tests.rs b/src/jupiter/storage/native_metadata_generation_tests.rs new file mode 100644 index 00000000..b565c091 --- /dev/null +++ b/src/jupiter/storage/native_metadata_generation_tests.rs @@ -0,0 +1,906 @@ +use std::sync::Arc; + +use mst2_codec::metapage::{Entry, EntryKind}; +use sea_orm::{Database, PaginatorTrait}; +use sea_orm_migration::MigratorTrait; + +use super::*; +use crate::{ + callisto::mst2_metadata_lifetime, + ceres::snapshot::retention_dag::MetadataDagBuilder, + jupiter::{ + migration::Migrator, + storage::mst2_retention::GcClaim, + tests::{TestSchemaGuard, test_db_config}, + }, +}; + +fn prepared(scope: &str) -> PreparedNativeMetadataRetention { + let entries = [Entry::file(EntryKind::Regular, b"file", 3, [42; 32])]; + let child = Page::build(&entries).unwrap(); + let root_entries = [ + Entry::dir(b"one", page_id(&child)), + Entry::dir(b"two", page_id(&child)), + ]; + let root = Page::build(&root_entries).unwrap(); + let mut builder = MetadataDagBuilder::new(MetadataDagLimits::default()); + builder.add_directory(&child, &entries).unwrap(); + builder.add_directory(&root, &root_entries).unwrap(); + PreparedNativeMetadataRetention::test_installation( + Arc::new(builder.finish(page_id(&root)).unwrap()), + scope, + ) +} + +async fn fixture() -> (DatabaseConnection, DatabaseConnection, TestSchemaGuard) { + let temp = tempfile::tempdir().unwrap(); + let (config, schema) = test_db_config(temp.path()).await; + let first = Database::connect(config.db_url.clone()).await.unwrap(); + Migrator::up(&first, None).await.unwrap(); + let second = Database::connect(config.db_url).await.unwrap(); + (first, second, schema) +} + +fn rejected(error: MetadataInstallError) -> SnapshotErrorCode { + match error { + MetadataInstallError::Rejected(error) => error.code, + _ => panic!("expected definitive rejection"), + } +} + +async fn scalar(db: &DatabaseConnection, sql: &str) -> i64 { + db.query_one_raw(statement(sql, [])) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap() +} + +async fn mappings(db: &DatabaseConnection, intent: &GenerationPrepareIntent) -> GenerationBindings { + let repository = PostgresMetadataGenerationRepository::new(db.clone()) + .await + .unwrap(); + repository + .require_fixed_plan(db, intent) + .await + .unwrap() + .bindings +} + +async fn wait_for_retention_waiter(txn: &DatabaseTransaction) { + let deadline = tokio::time::Instant::now() + Duration::from_secs(4); + loop { + let waiting = txn.query_one_raw(statement( + "SELECT EXISTS(SELECT 1 FROM pg_locks WHERE locktype='advisory' AND NOT granted + AND classid=$1::integer::oid AND objid=hashtext(current_schema())::oid + AND database=(SELECT oid FROM pg_database WHERE datname=current_database())) AS waiting", + [RETENTION_LOCK_KEY.into()], + )).await.unwrap().unwrap().try_get::("", "waiting").unwrap(); + if waiting { + break; + } + assert!( + tokio::time::Instant::now() < deadline, + "independent connection must wait on the schema retention barrier" + ); + tokio::task::yield_now().await; + } +} + +#[tokio::test] +async fn generation_begin_serializes_full_reserved_bindings_before_any_payload() { + let (first, second, _schema) = fixture().await; + let a = PostgresMetadataGenerationRepository::new(first.clone()) + .await + .unwrap(); + let b = PostgresMetadataGenerationRepository::new(second.clone()) + .await + .unwrap(); + let pages = prepared("/"); + let left = a.begin_intent("reserved-a", &pages).await.unwrap(); + let held = a.inner.transaction().await.unwrap(); + a.inner.barrier(&held).await.unwrap(); + let right_pages = prepared("/"); + let blocked = tokio::spawn(async move { b.begin_intent("reserved-b", &right_pages).await }); + wait_for_retention_waiter(&held).await; + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_payload").await, + 0 + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_retention_root").await, + 0 + ); + held.commit().await.unwrap(); + let right = blocked.await.unwrap().unwrap(); + assert_ne!(left.prepare_id(), right.prepare_id()); + assert_ne!(left.storage_seal, right.storage_seal); + assert_eq!(left.bindings_digest, right.bindings_digest); + assert_eq!( + mappings(&first, &left).await, + mappings(&second, &right).await + ); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_lifetime WHERE generation=1 AND state='RESERVED'" + ) + .await, + 2 + ); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_prepare_page WHERE generation=1" + ) + .await, + 4 + ); + let replay = a.begin_intent("reserved-a", &pages).await.unwrap(); + assert_eq!(left, replay); +} + +#[tokio::test] +async fn generation_partial_prepare_recovers_exact_complete_mapping_on_second_connection() { + let (first, second, _schema) = fixture().await; + let a = PostgresMetadataGenerationRepository::new(first.clone()) + .await + .unwrap(); + let pages = prepared("/"); + let intent = a.begin_intent("partial", &pages).await.unwrap(); + a.install_pages(&intent, &pages.dag().payloads()[..1]) + .await + .unwrap(); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_payload").await, + 1 + ); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_prepare_page WHERE generation=1" + ) + .await, + 2 + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_retention_node").await, + 0 + ); + drop(a); + let restarted = PostgresMetadataGenerationRepository::new(second.clone()) + .await + .unwrap(); + assert_eq!( + restarted + .inspect_prepare( + &second, + "partial", + intent.manifest_digest(), + MetadataCommitPhase::Payload + ) + .await + .unwrap(), + GenerationPrepareObservation::Preparing(intent.clone()) + ); + assert_eq!( + restarted + .observe_installed_dag(&intent) + .await + .unwrap_err() + .code, + SnapshotErrorCode::ObjectUnavailable + ); + restarted + .install_pages(&intent, pages.dag().payloads()) + .await + .unwrap(); + let receipt = restarted.finalize(&intent).await.unwrap(); + assert_eq!(receipt.intent(), &intent); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_lifetime WHERE state='LIVE' AND generation=1" + ) + .await, + 2 + ); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_retention_root WHERE root_kind='prepare'" + ) + .await, + 2 + ); +} + +#[tokio::test] +async fn generation_shared_live_prepare_pins_batch_before_generic_gc_can_claim() { + let (first, second, _schema) = fixture().await; + let a = PostgresMetadataGenerationRepository::new(first.clone()) + .await + .unwrap(); + let b = PostgresMetadataGenerationRepository::new(second.clone()) + .await + .unwrap(); + let pages = prepared("/"); + let old = a.begin_intent("live-old", &pages).await.unwrap(); + a.install_pages(&old, pages.dag().payloads()).await.unwrap(); + a.finalize(&old).await.unwrap(); + let shared = b.begin_intent("live-shared", &pages).await.unwrap(); + assert_eq!( + mappings(&first, &old).await, + mappings(&second, &shared).await + ); + let graph = PostgresRetentionRepository::new(first.clone()); + graph + .release_root(&RetentionRoot::Prepare(old.prepare_id().into())) + .await + .unwrap(); + assert_eq!( + graph + .mark_deleting("generic-cannot-claim-shared", &node_id(&pages.dag().root())) + .await + .unwrap(), + GcClaim::Unavailable + ); + assert_eq!( + scalar( + &second, + "SELECT count(*) FROM mst2_retention_root WHERE root_kind='prepare'" + ) + .await, + 2 + ); + b.install_pages(&shared, pages.dag().payloads()) + .await + .unwrap(); + b.finalize(&shared).await.unwrap(); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_payload").await, + 2 + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_retention_edge").await, + 1 + ); + assert_eq!( + scalar(&first, "SELECT max(generation) FROM mst2_metadata_lifetime").await, + 1 + ); +} + +#[tokio::test] +async fn generation_observation_is_bound_to_prepare_not_only_the_same_dag() { + let (first, second, _schema) = fixture().await; + let a = PostgresMetadataGenerationRepository::new(first.clone()) + .await + .unwrap(); + let b = PostgresMetadataGenerationRepository::new(second) + .await + .unwrap(); + let pages = prepared("/"); + let left = a.begin_intent("observation-left", &pages).await.unwrap(); + let right = b.begin_intent("observation-right", &pages).await.unwrap(); + a.install_pages(&left, pages.dag().payloads()) + .await + .unwrap(); + let observation = a.observe_installed_dag(&left).await.unwrap(); + assert_eq!( + rejected( + b.finalize_observation(&right, &observation) + .await + .unwrap_err() + ), + SnapshotErrorCode::IntegrityError + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_retention_node").await, + 0 + ); + let receipt = a.finalize_observation(&left, &observation).await.unwrap(); + assert_eq!(receipt, a.finalize(&left).await.unwrap()); +} + +#[tokio::test] +async fn generation_finalization_rechecks_changed_lifetime_after_lock_free_observation() { + let (first, second, _schema) = fixture().await; + let repository = PostgresMetadataGenerationRepository::new(first.clone()) + .await + .unwrap(); + let pages = prepared("/"); + let intent = repository + .begin_intent("state-change", &pages) + .await + .unwrap(); + repository + .install_pages(&intent, pages.dag().payloads()) + .await + .unwrap(); + let observed = repository.observe_installed_dag(&intent).await.unwrap(); + let txn = second.begin().await.unwrap(); + repository.inner.barrier(&txn).await.unwrap(); + // Fault injection only: G1 deliberately has no collector transition API. + txn.execute_unprepared( + "ALTER TABLE mst2_metadata_lifetime DISABLE TRIGGER mst2_metadata_lifetime_guard", + ) + .await + .unwrap(); + txn.execute_raw(statement( + "UPDATE mst2_metadata_lifetime SET state='DELETING' WHERE page_id=$1", + [pages.dag().root().to_vec().into()], + )) + .await + .unwrap(); + txn.execute_unprepared( + "ALTER TABLE mst2_metadata_lifetime ENABLE TRIGGER mst2_metadata_lifetime_guard", + ) + .await + .unwrap(); + txn.commit().await.unwrap(); + assert_eq!( + rejected( + repository + .finalize_observation(&intent, &observed) + .await + .unwrap_err() + ), + SnapshotErrorCode::ObjectUnavailable + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_retention_node").await, + 0 + ); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_prepare WHERE state='PREPARING'" + ) + .await, + 1 + ); + assert_eq!( + mst2_metadata_payload::Entity::find() + .count(&first) + .await + .unwrap(), + 2 + ); +} + +#[tokio::test] +async fn generation_deleting_removed_and_generic_tombstone_are_never_reopened() { + for state in ["DELETING", "REMOVED"] { + let (first, second, _schema) = fixture().await; + let repository = PostgresMetadataGenerationRepository::new(first.clone()) + .await + .unwrap(); + let pages = prepared("/"); + let old = repository.begin_intent("old", &pages).await.unwrap(); + let txn = second.begin().await.unwrap(); + repository.inner.barrier(&txn).await.unwrap(); + txn.execute_unprepared( + "ALTER TABLE mst2_metadata_lifetime DISABLE TRIGGER mst2_metadata_lifetime_guard", + ) + .await + .unwrap(); + txn.execute_raw(statement( + "UPDATE mst2_metadata_lifetime SET state=$1 WHERE page_id=$2", + [state.into(), pages.dag().root().to_vec().into()], + )) + .await + .unwrap(); + txn.execute_unprepared( + "ALTER TABLE mst2_metadata_lifetime ENABLE TRIGGER mst2_metadata_lifetime_guard", + ) + .await + .unwrap(); + txn.commit().await.unwrap(); + assert_eq!( + rejected( + repository + .begin_intent("fresh-disallowed", &pages) + .await + .unwrap_err() + ), + SnapshotErrorCode::ObjectUnavailable + ); + assert_eq!( + rejected( + repository + .install_pages(&old, pages.dag().payloads()) + .await + .unwrap_err() + ), + SnapshotErrorCode::ObjectUnavailable + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_prepare").await, + 1 + ); + assert_eq!( + scalar(&first, "SELECT max(generation) FROM mst2_metadata_lifetime").await, + 1 + ); + } + let (first, second, _schema) = fixture().await; + let pages = prepared("/"); + let graph = PostgresRetentionRepository::new(second.clone()); + graph + .retain_group(pages.dag().nodes(), pages.dag().edges(), &[]) + .await + .unwrap(); + assert_eq!( + graph + .mark_deleting("legacy-remove", &node_id(&pages.dag().root())) + .await + .unwrap(), + GcClaim::Marked + ); + graph.complete_gc("legacy-remove").await.unwrap(); + let repository = PostgresMetadataGenerationRepository::new(first.clone()) + .await + .unwrap(); + assert_eq!( + rejected( + repository + .begin_intent("no-resurrection", &pages) + .await + .unwrap_err() + ), + SnapshotErrorCode::ObjectUnavailable + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_prepare").await, + 0 + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_lifetime").await, + 0 + ); + assert_eq!( + graph + .retain_group(pages.dag().nodes(), pages.dag().edges(), &[]) + .await + .unwrap_err() + .code, + SnapshotErrorCode::ObjectUnavailable + ); +} + +#[tokio::test] +async fn generation_legacy_null_capability_remains_legacy_and_does_not_block_existing_installer() { + let (first, second, _schema) = fixture().await; + let legacy = PostgresMetadataInstallRepository::new(first.clone()) + .await + .unwrap(); + let fresh = PostgresMetadataGenerationRepository::new(second.clone()) + .await + .unwrap(); + let pages = prepared("/"); + let old = legacy + .begin_intent("legacy-operation", &pages) + .await + .unwrap(); + legacy + .install_pages(&old, pages.dag().payloads()) + .await + .unwrap(); + let original_receipt = legacy.finalize(&old).await.unwrap(); + assert_eq!( + rejected( + fresh + .begin_intent("legacy-operation", &pages) + .await + .unwrap_err() + ), + SnapshotErrorCode::ObjectUnavailable + ); + assert_eq!( + rejected( + fresh + .begin_intent("new-over-legacy-cas", &pages) + .await + .unwrap_err() + ), + SnapshotErrorCode::IntegrityError + ); + assert_eq!(legacy.finalize(&old).await.unwrap(), original_receipt); + let another = legacy + .begin_intent("legacy-new-resolve", &pages) + .await + .unwrap(); + legacy + .install_pages(&another, pages.dag().payloads()) + .await + .unwrap(); + legacy.finalize(&another).await.unwrap(); + assert_eq!( + scalar( + &second, + "SELECT count(*) FROM mst2_metadata_payload WHERE generation IS NULL" + ) + .await, + 2 + ); + assert_eq!( + scalar( + &second, + "SELECT count(*) FROM mst2_metadata_prepare WHERE storage_seal IS NULL" + ) + .await, + 2 + ); + assert_eq!( + mst2_metadata_lifetime::Entity::find() + .count(&first) + .await + .unwrap(), + 0 + ); +} + +#[tokio::test] +async fn generation_storage_seal_scope_and_complete_child_binding_cannot_be_substituted() { + let (first, second, _schema) = fixture().await; + let repository = PostgresMetadataGenerationRepository::new(first.clone()) + .await + .unwrap(); + let pages = prepared("/"); + let intent = repository + .begin_intent("seal-tamper", &pages) + .await + .unwrap(); + repository + .install_pages(&intent, pages.dag().payloads()) + .await + .unwrap(); + let mut wrong_bindings = intent.clone(); + wrong_bindings.bindings_digest[0] ^= 1; + let mut wrong_seal = intent.clone(); + wrong_seal.storage_seal[0] ^= 1; + let mut wrong_root = intent.clone(); + wrong_root.metadata_root[0] ^= 1; + for wrong in [wrong_bindings, wrong_seal, wrong_root] { + assert_eq!( + rejected( + repository + .install_pages(&wrong, pages.dag().payloads()) + .await + .unwrap_err() + ), + SnapshotErrorCode::IntegrityError + ); + } + let mut wrong_scope = intent.clone(); + wrong_scope.primary_scope[0] ^= 1; + assert_eq!( + repository + .observe_installed_dag(&wrong_scope) + .await + .unwrap_err() + .code, + SnapshotErrorCode::IntegrityError + ); + let child = pages + .dag() + .payloads() + .iter() + .find(|page| page.id != pages.dag().root()) + .unwrap(); + let txn = second.begin().await.unwrap(); + repository.inner.barrier(&txn).await.unwrap(); + txn.execute_unprepared("ALTER TABLE mst2_metadata_prepare_page DISABLE TRIGGER mst2_metadata_generation_mapping_guard").await.unwrap(); + txn.execute_raw(statement("UPDATE mst2_metadata_prepare_page SET expected_size=expected_size+1 WHERE prepare_id=$1 AND page_id=$2", [intent.prepare_id().into(), child.id.to_vec().into()])).await.unwrap(); + txn.execute_unprepared("ALTER TABLE mst2_metadata_prepare_page ENABLE TRIGGER mst2_metadata_generation_mapping_guard").await.unwrap(); + txn.commit().await.unwrap(); + assert_eq!( + repository + .observe_installed_dag(&intent) + .await + .unwrap_err() + .code, + SnapshotErrorCode::IntegrityError + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_retention_node").await, + 0 + ); +} + +#[tokio::test] +async fn generation_compact_receipt_checks_actual_seal_without_loading_the_dag() { + let (first, second, _schema) = fixture().await; + let repository = PostgresMetadataGenerationRepository::new(first.clone()) + .await + .unwrap(); + let pages = prepared("/"); + let intent = repository.begin_intent("compact", &pages).await.unwrap(); + repository + .install_pages(&intent, pages.dag().payloads()) + .await + .unwrap(); + let receipt = repository.finalize(&intent).await.unwrap(); + let source = receipt.legacy.identity.tagged_root_tree_oid.clone(); + let txn = second.begin().await.unwrap(); + repository + .verify_receipt_in_txn(&txn, &receipt, &source, "/") + .await + .unwrap(); + txn.rollback().await.unwrap(); + let txn = second.begin().await.unwrap(); + repository.inner.barrier(&txn).await.unwrap(); + txn.execute_unprepared( + "ALTER TABLE mst2_metadata_prepare DISABLE TRIGGER mst2_metadata_generation_seal_guard", + ) + .await + .unwrap(); + txn.execute_raw(statement( + "UPDATE mst2_metadata_prepare SET storage_seal=$1 WHERE prepare_id=$2", + [vec![0u8; 32].into(), intent.prepare_id().into()], + )) + .await + .unwrap(); + txn.execute_unprepared( + "ALTER TABLE mst2_metadata_prepare ENABLE TRIGGER mst2_metadata_generation_seal_guard", + ) + .await + .unwrap(); + txn.commit().await.unwrap(); + let txn = second.begin().await.unwrap(); + assert_eq!( + repository + .verify_receipt_in_txn(&txn, &receipt, &source, "/") + .await + .unwrap_err() + .code, + SnapshotErrorCode::IntegrityError + ); + txn.rollback().await.unwrap(); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_payload").await, + 2 + ); + let txn = second.begin().await.unwrap(); + repository.inner.barrier(&txn).await.unwrap(); + txn.execute_unprepared( + "ALTER TABLE mst2_metadata_prepare DISABLE TRIGGER mst2_metadata_generation_seal_guard", + ) + .await + .unwrap(); + txn.execute_raw(statement( + "UPDATE mst2_metadata_prepare SET storage_seal=$1, + canonical_bindings=set_byte(canonical_bindings,60,get_byte(canonical_bindings,60)#1) WHERE prepare_id=$2", + [intent.storage_seal.to_vec().into(),intent.prepare_id().into()], + )).await.unwrap(); + txn.execute_unprepared( + "ALTER TABLE mst2_metadata_prepare ENABLE TRIGGER mst2_metadata_generation_seal_guard", + ) + .await + .unwrap(); + txn.commit().await.unwrap(); + let txn = second.begin().await.unwrap(); + assert_eq!( + repository + .verify_receipt_in_txn(&txn, &receipt, &source, "/") + .await + .unwrap_err() + .code, + SnapshotErrorCode::IntegrityError + ); + txn.rollback().await.unwrap(); +} + +#[tokio::test] +async fn generation_db_guards_preserve_watermark_payload_scope_and_committed_mappings() { + let (first, second, _schema) = fixture().await; + let repository = PostgresMetadataGenerationRepository::new(first.clone()) + .await + .unwrap(); + let pages = prepared("/"); + let intent = repository.begin_intent("guards", &pages).await.unwrap(); + repository + .install_pages(&intent, pages.dag().payloads()) + .await + .unwrap(); + repository.finalize(&intent).await.unwrap(); + for sql in [ + "DELETE FROM mst2_metadata_lifetime", + "UPDATE mst2_metadata_lifetime SET generation=2", + "UPDATE mst2_metadata_lifetime SET state='REMOVED'", + "DELETE FROM mst2_metadata_payload", + "UPDATE mst2_metadata_payload SET generation=1", + "DELETE FROM mst2_metadata_storage_scope", + "UPDATE mst2_metadata_storage_scope SET storage_uuid=storage_uuid", + "DELETE FROM mst2_metadata_prepare_page", + "UPDATE mst2_metadata_prepare_page SET generation=1", + "UPDATE mst2_metadata_prepare SET storage_seal=NULL,canonical_bindings=NULL,bindings_digest=NULL,primary_scope=NULL", + ] { + assert!( + second.execute_unprepared(sql).await.is_err(), + "guard must reject {sql}" + ); + } + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_lifetime WHERE generation=1 AND state='LIVE'" + ) + .await, + 2 + ); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_prepare_page WHERE generation=1" + ) + .await, + 2 + ); + assert_eq!( + repository.finalize(&intent).await.unwrap().intent(), + &intent + ); + assert_eq!(scalar(&first, "SELECT count(*) FROM pg_indexes WHERE schemaname=current_schema() AND indexname='idx_mst2_metadata_prepare_page_lifetime'").await, 1); +} + +#[tokio::test] +async fn generation_finalize_row_lock_fences_a_late_mapping_insert_on_another_connection() { + let (first, second, _schema) = fixture().await; + let repository = PostgresMetadataGenerationRepository::new(first.clone()) + .await + .unwrap(); + let pages = prepared("/"); + let intent = repository + .begin_intent("late-mapping", &pages) + .await + .unwrap(); + repository + .install_pages(&intent, pages.dag().payloads()) + .await + .unwrap(); + let observation = repository.observe_installed_dag(&intent).await.unwrap(); + let held = repository.inner.transaction().await.unwrap(); + repository.inner.barrier(&held).await.unwrap(); + lock_prepare(&held, intent.prepare_id()).await.unwrap(); + let contender = second.begin().await.unwrap(); + let pid = contender + .query_one_raw(statement("SELECT pg_backend_pid() AS pid", [])) + .await + .unwrap() + .unwrap() + .try_get::("", "pid") + .unwrap(); + let prepare_id = intent.prepare_id().to_owned(); + let root = pages.dag().root(); + let size = pages.install_plan().unwrap().pages[&root] as i32; + let blocked = tokio::spawn(async move { + let result = contender.execute_raw(statement( + "INSERT INTO mst2_metadata_prepare_page(prepare_id,page_id,generation,expected_size) VALUES($1,$2,1,$3)", + [prepare_id.into(), root.to_vec().into(), size.into()], + )).await; + contender.rollback().await.unwrap(); + result + }); + let deadline = tokio::time::Instant::now() + Duration::from_secs(4); + loop { + let waiting = held + .query_one_raw(statement( + "SELECT EXISTS(SELECT 1 FROM pg_locks WHERE pid=$1 AND NOT granted) AS waiting", + [pid.into()], + )) + .await + .unwrap() + .unwrap() + .try_get::("", "waiting") + .unwrap(); + if waiting { + break; + } + assert!( + tokio::time::Instant::now() < deadline, + "late mapping writer must wait for the prepare row lock" + ); + tokio::task::yield_now().await; + } + repository + .finalize_in_txn(&held, &intent, &observation) + .await + .unwrap(); + held.commit().await.unwrap(); + let error = blocked.await.unwrap().unwrap_err(); + assert!( + error + .to_string() + .contains("metadata generation mappings are immutable"), + "{error}" + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_prepare_page").await, + 2 + ); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_prepare WHERE state='COMMITTED'" + ) + .await, + 1 + ); + repository.finalize(&intent).await.unwrap(); +} + +#[tokio::test] +async fn generation_wrong_primary_never_upgrades_intent_or_reports_false_absence() { + let (first, same_primary, _schema) = fixture().await; + let (other, _other_connection, _other_schema) = fixture().await; + let repository = PostgresMetadataGenerationRepository::new(first.clone()) + .await + .unwrap(); + let wrong = PostgresMetadataGenerationRepository::new(other.clone()) + .await + .unwrap(); + let pages = prepared("/"); + let intent = repository + .begin_intent("captured-primary", &pages) + .await + .unwrap(); + repository + .install_pages(&intent, pages.dag().payloads()) + .await + .unwrap(); + let receipt = repository.finalize(&intent).await.unwrap(); + assert_eq!( + rejected( + wrong + .install_pages(&intent, pages.dag().payloads()) + .await + .unwrap_err() + ), + SnapshotErrorCode::IntegrityError + ); + assert_eq!( + wrong.observe_installed_dag(&intent).await.unwrap_err().code, + SnapshotErrorCode::IntegrityError + ); + assert!(matches!( + repository + .inspect_prepare( + &other, + "captured-primary", + intent.manifest_digest(), + MetadataCommitPhase::Finalize + ) + .await, + Err(MetadataInstallError::CommitUncertain { + phase: MetadataCommitPhase::Finalize, + .. + }) + )); + assert_eq!( + repository + .inspect_prepare( + &same_primary, + "captured-primary", + intent.manifest_digest(), + MetadataCommitPhase::Finalize + ) + .await + .unwrap(), + GenerationPrepareObservation::Committed(Box::new(receipt)) + ); + assert_eq!( + scalar(&other, "SELECT count(*) FROM mst2_metadata_payload").await, + 0 + ); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_retention_root WHERE root_kind='prepare'" + ) + .await, + 2 + ); +} diff --git a/src/jupiter/storage/native_metadata_generations.rs b/src/jupiter/storage/native_metadata_generations.rs new file mode 100644 index 00000000..fd1323cd --- /dev/null +++ b/src/jupiter/storage/native_metadata_generations.rs @@ -0,0 +1,866 @@ +//! Generation-bound preparation repository, isolated from HTTP lease authority. + +use sha2::{Digest, Sha256}; + +use super::*; + +const MAX_BINDINGS_BYTES: usize = 12 + 48 * 4096; +const MAX_PRIMARY_SCOPE_BYTES: usize = 16384; + +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct GenerationPrepareIntent { + legacy: MetadataPrepareIntent, + metadata_root: [u8; 32], + root_generation: i64, + bindings_digest: [u8; 32], + primary_scope: Vec, + storage_seal: [u8; 32], +} + +impl GenerationPrepareIntent { + pub fn prepare_id(&self) -> &str { + self.legacy.prepare_id() + } + + pub fn manifest_digest(&self) -> [u8; 32] { + self.legacy.manifest_digest() + } + + pub fn operation_id(&self) -> &str { + self.legacy.operation_id() + } +} + +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct GenerationMetadataReceipt { + legacy: PreparedMetadataReceipt, + intent: GenerationPrepareIntent, +} + +impl GenerationMetadataReceipt { + pub fn intent(&self) -> &GenerationPrepareIntent { + &self.intent + } + + pub fn metadata_root(&self) -> [u8; 32] { + self.intent.metadata_root + } + + pub fn payload_bytes(&self) -> u64 { + self.legacy.payload_bytes() + } +} + +#[derive(Debug)] +pub struct GenerationDagObservation { + intent: GenerationPrepareIntent, + dag: ValidatedMetadataDag, +} + +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum GenerationPrepareObservation { + Absent, + Preparing(GenerationPrepareIntent), + Committed(Box), +} + +#[derive(Debug, Clone, PartialEq, Eq)] +struct GenerationBindings(BTreeMap<[u8; 32], (i64, u64)>); + +impl GenerationBindings { + fn encode(&self) -> Result, SnapshotError> { + if self.0.is_empty() || self.0.len() > MetadataDagLimits::default().nodes { + return Err(integrity("invalid metadata generation binding count")); + } + let mut bytes = Vec::with_capacity(12 + 48 * self.0.len()); + bytes.extend_from_slice(b"MST2GEN1"); + bytes.extend_from_slice(&(self.0.len() as u32).to_be_bytes()); + for (page, (generation, size)) in &self.0 { + if *generation <= 0 || !(HEADER_LEN as u64..=PAGE_MAX_BYTES as u64).contains(size) { + return Err(integrity("invalid metadata lifetime generation or size")); + } + bytes.extend_from_slice(page); + bytes.extend_from_slice(&generation.to_be_bytes()); + bytes.extend_from_slice(&size.to_be_bytes()); + } + Ok(bytes) + } + + fn decode(bytes: &[u8], plan: &MetadataInstallPlan) -> Result { + if bytes.len() < 60 || bytes.len() > MAX_BINDINGS_BYTES || &bytes[..8] != b"MST2GEN1" { + return Err(integrity("invalid metadata generation binding encoding")); + } + let count = u32::from_be_bytes(bytes[8..12].try_into().map_err(internal)?) as usize; + if count != plan.pages.len() || bytes.len() != 12 + count * 48 { + return Err(integrity( + "metadata generation binding coverage differs from plan", + )); + } + let mut bindings = BTreeMap::new(); + for entry in bytes[12..].chunks_exact(48) { + let page: [u8; 32] = entry[..32].try_into().map_err(internal)?; + let generation = i64::from_be_bytes(entry[32..40].try_into().map_err(internal)?); + let size = u64::from_be_bytes(entry[40..48].try_into().map_err(internal)?); + if generation <= 0 + || plan.pages.get(&page) != Some(&size) + || bindings.insert(page, (generation, size)).is_some() + { + return Err(integrity("invalid fixed metadata lifetime binding")); + } + } + let result = Self(bindings); + if result.encode()? != bytes { + return Err(integrity("noncanonical metadata generation bindings")); + } + Ok(result) + } +} + +struct FixedPlan { + stored: StoredPlan, + bindings: GenerationBindings, + intent: GenerationPrepareIntent, +} + +#[derive(Clone)] +pub struct PostgresMetadataGenerationRepository { + inner: PostgresMetadataInstallRepository, +} + +impl PostgresMetadataGenerationRepository { + pub async fn new(connection: DatabaseConnection) -> Result { + Ok(Self { + inner: PostgresMetadataInstallRepository::new(connection).await?, + }) + } + + pub async fn begin_intent( + &self, + operation_id: &str, + prepared: &PreparedNativeMetadataRetention, + ) -> Result { + validate_operation_id(operation_id)?; + let plan = prepared.install_plan()?; + let digest = plan.digest()?; + let primary_scope = self.primary_scope()?; + let txn = self.inner.transaction().await?; + let result = async { + self.inner.barrier(&txn).await?; + if let Some(fixed) = self.load_fixed_plan(&txn, operation_id, &digest).await? { + return Ok(fixed.intent); + } + let bindings = allocate_lifetimes(&txn, &plan).await?; + let canonical_bindings = bindings.encode()?; + let bindings_digest = Sha256::digest(&canonical_bindings).into(); + let prepare_id = uuid::Uuid::new_v4().to_string(); + let intent = GenerationPrepareIntent { + legacy: MetadataPrepareIntent { prepare_id: prepare_id.clone(), operation_id: operation_id.into(), manifest_digest: digest }, + metadata_root: plan.root, + root_generation: bindings.0.get(&plan.root).ok_or_else(|| integrity("root lifetime is missing"))?.0, + bindings_digest, + storage_seal: seal(&prepare_id, &digest, &plan.root, &bindings_digest, &primary_scope)?, + primary_scope, + }; + let identity = &plan.identity; + txn.execute_raw(statement( + "INSERT INTO mst2_metadata_prepare(prepare_id,operation_id,manifest_digest,canonical_plan, + source_domain,tagged_root_tree_oid,scope,schema_version,metadata_codec,materialization_policy, + fs_semantics,access_projection,verification_revision,projection_revision,metadata_root, + node_count,edge_count,total_bytes,state,canonical_bindings,bindings_digest,primary_scope,storage_seal) + VALUES($1,$2,$3,$4,$5,$6,$7,$8,$9,$10,$11,$12,$13,$14,$15,$16,$17,$18,'PREPARING',$19,$20,$21,$22)", + [prepare_id.clone().into(),operation_id.into(),digest.to_vec().into(),plan.encode()?.into(), + identity.source_domain.clone().into(),identity.tagged_root_tree_oid.clone().into(),identity.scope.clone().into(), + (identity.schema_version as i16).into(),(identity.metadata_codec as i16).into(), + (identity.materialization_policy as i16).into(),(identity.fs_semantics as i16).into(), + (identity.access_projection as i16).into(),identity.verification_revision.into(), + (identity.projection_revision as i16).into(),plan.root.to_vec().into(),(plan.pages.len() as i32).into(), + (plan.edges.len() as i32).into(),(plan.total_bytes as i64).into(),canonical_bindings.into(), + intent.bindings_digest.to_vec().into(),intent.primary_scope.clone().into(),intent.storage_seal.to_vec().into()], + )).await.map_err(internal)?; + let pages: Vec<_> = bindings.0.iter().map(|(page,(generation,size))| + json!({"page_id":hex::encode(page),"generation":generation,"size":size})).collect(); + txn.execute_raw(statement( + "INSERT INTO mst2_metadata_prepare_page(prepare_id,page_id,generation,expected_size) + SELECT $1,decode(p.page_id,'hex'),p.generation,p.size + FROM jsonb_to_recordset($2::jsonb) AS p(page_id text,generation bigint,size integer)", + [prepare_id.clone().into(),serde_json::to_string(&pages).map_err(internal)?.into()], + )).await.map_err(internal)?; + // Reuse only already LIVE graph rows. RESERVED pages without a graph + // are protected by the durable mappings, without inventing a DAG. + txn.execute_raw(statement( + "INSERT INTO mst2_retention_root(node_id,root_key,root_kind) + SELECT l.node_id,'prepare:'||p.prepare_id,'prepare' + FROM mst2_metadata_prepare_page p JOIN mst2_metadata_lifetime l + ON l.page_id=p.page_id AND l.generation=p.generation + JOIN mst2_retention_node n ON n.node_id=l.node_id AND n.state='LIVE' + WHERE p.prepare_id=$1 ON CONFLICT(node_id,root_key) DO NOTHING", + [prepare_id.into()], + )).await.map_err(internal)?; + Ok(intent) + }.await; + commit( + txn, + result, + operation_id, + digest, + MetadataCommitPhase::Intent, + ) + .await + } + + pub async fn install_page( + &self, + intent: &GenerationPrepareIntent, + payload: &MetadataPagePayload, + ) -> Result<(), MetadataInstallError> { + self.install_pages(intent, std::slice::from_ref(payload)) + .await + } + + pub async fn install_pages( + &self, + intent: &GenerationPrepareIntent, + payloads: &[MetadataPagePayload], + ) -> Result<(), MetadataInstallError> { + if payloads.is_empty() || payloads.len() > 64 { + return Err(SnapshotError::new( + SnapshotErrorCode::LimitExceeded, + "metadata installation batch must contain 1..=64 pages", + ) + .into()); + } + let mut ids = BTreeSet::new(); + for payload in payloads { + validate_payload(payload)?; + if !ids.insert(payload.id) { + return Err(integrity("duplicate page in metadata installation batch").into()); + } + } + let txn = self.inner.transaction().await?; + let result = async { + self.inner.barrier(&txn).await?; + let fixed = self.require_fixed_plan(&txn, intent).await?; + if fixed.stored.record.state == "COMMITTED" { + check_generation_payloads(&txn, &fixed).await?; + } + let mut pages = Vec::with_capacity(payloads.len()); + for page in payloads { + let &(generation,size) = fixed.bindings.0.get(&page.id) + .ok_or_else(|| integrity("metadata payload is not a member of its fixed generation installation"))?; + if size != page.size { + return Err(integrity("metadata payload size differs from its fixed lifetime")); + } + pages.push(json!({"page_id":hex::encode(page.id),"generation":generation,"size":size,"payload":hex::encode(&page.bytes)})); + } + let encoded = serde_json::to_string(&pages).map_err(internal)?; + txn.execute_raw(statement( + "INSERT INTO mst2_metadata_payload(page_id,generation,metadata_codec,byte_size,payload) + SELECT decode(p.page_id,'hex'),p.generation,$1,p.size,decode(p.payload,'hex') + FROM jsonb_to_recordset($2::jsonb) AS p(page_id text,generation bigint,size integer,payload text) + ON CONFLICT(page_id) DO NOTHING", + [fixed.stored.record.metadata_codec.into(),encoded.clone().into()], + )).await.map_err(internal)?; + let bad=txn.query_one_raw(statement( + "SELECT p.page_id FROM jsonb_to_recordset($2::jsonb) AS p(page_id text,generation bigint,size integer,payload text) + LEFT JOIN mst2_metadata_payload b ON b.page_id=decode(p.page_id,'hex') + WHERE b.page_id IS NULL OR b.generation IS DISTINCT FROM p.generation OR b.metadata_codec<>$1 + OR b.byte_size<>p.size OR b.payload<>decode(p.payload,'hex') LIMIT 1", + [fixed.stored.record.metadata_codec.into(),encoded.into()], + )).await.map_err(internal)?; + if bad.is_some() { + return Err(integrity("immutable payload conflicts with its exact metadata lifetime")); + } + Ok(()) + }.await; + commit( + txn, + result, + &intent.legacy.operation_id, + intent.manifest_digest(), + MetadataCommitPhase::Payload, + ) + .await + } + + /// Read, hash and parse the actual fixed-generation bytes outside the lock. + pub async fn observe_installed_dag( + &self, + intent: &GenerationPrepareIntent, + ) -> Result { + self.inner + .verify_primary_connection(&self.inner.connection) + .await?; + let fixed = self + .require_fixed_plan(&self.inner.connection, intent) + .await?; + let dag = load_fixed_dag(&self.inner.connection, &fixed).await?; + Ok(GenerationDagObservation { + intent: intent.clone(), + dag, + }) + } + + pub async fn finalize( + &self, + intent: &GenerationPrepareIntent, + ) -> Result { + let observation = self.observe_installed_dag(intent).await?; + self.finalize_observation(intent, &observation).await + } + + pub async fn finalize_observation( + &self, + intent: &GenerationPrepareIntent, + observation: &GenerationDagObservation, + ) -> Result { + if &observation.intent != intent { + return Err( + integrity("metadata observation belongs to another fixed installation").into(), + ); + } + let txn = self.inner.transaction().await?; + let result = self.finalize_in_txn(&txn, intent, observation).await; + commit( + txn, + result, + &intent.legacy.operation_id, + intent.manifest_digest(), + MetadataCommitPhase::Finalize, + ) + .await + } + + // Only a successful outer commit or the recovery barrier exposes a receipt. + async fn finalize_in_txn( + &self, + txn: &DatabaseTransaction, + intent: &GenerationPrepareIntent, + observation: &GenerationDagObservation, + ) -> Result { + if &observation.intent != intent { + return Err(integrity( + "metadata observation belongs to another fixed installation", + )); + } + self.inner.barrier(txn).await?; + // Mapping INSERT guards take this same row lock. No late insertion + // may cross coverage validation and the COMMITTED transition. + lock_prepare(txn, intent.prepare_id()).await?; + let fixed = self.require_fixed_plan(txn, intent).await?; + check_generation_payloads(txn, &fixed).await?; + let legacy = self + .inner + .finalize_stored_plan_in_txn(txn, &intent.legacy, &observation.dag, fixed.stored) + .await?; + txn.execute_raw(statement( + "UPDATE mst2_metadata_lifetime l SET state='LIVE' + FROM mst2_metadata_prepare_page p WHERE p.prepare_id=$1 AND p.page_id=l.page_id + AND p.generation=l.generation AND l.state='RESERVED'", + [intent.prepare_id().into()], + )) + .await + .map_err(internal)?; + Ok(GenerationMetadataReceipt { + legacy, + intent: intent.clone(), + }) + } + + /// Compact proof for a future atomic handoff. No DAG/page scan under the + /// mono writer lock: hash <=196620 binding bytes and check the root lifetime. + pub(crate) async fn verify_receipt_in_txn( + &self, + txn: &DatabaseTransaction, + receipt: &GenerationMetadataReceipt, + tagged_root_tree_oid: &str, + scope: &str, + ) -> Result<(), SnapshotError> { + self.require_scope(&receipt.intent)?; + self.inner + .verify_receipt_in_txn(txn, &receipt.legacy, tagged_root_tree_oid, scope) + .await?; + let row=txn.query_one_raw(statement( + "SELECT p.bindings_digest,p.primary_scope,p.storage_seal, + CASE WHEN octet_length(p.canonical_bindings)<=$2 THEN sha256(p.canonical_bindings) END AS actual_bindings_digest, + l.generation,l.state,l.metadata_codec,l.expected_size,m.generation AS root_generation, + b.generation AS payload_generation,b.byte_size + FROM mst2_metadata_prepare p LEFT JOIN mst2_metadata_prepare_page m + ON m.prepare_id=p.prepare_id AND m.page_id=p.metadata_root + LEFT JOIN mst2_metadata_lifetime l ON l.page_id=m.page_id AND l.generation=m.generation + LEFT JOIN mst2_metadata_payload b ON b.page_id=m.page_id + WHERE p.prepare_id=$1", + [receipt.intent.prepare_id().into(),(MAX_BINDINGS_BYTES as i32).into()], + )).await.map_err(internal)?.ok_or_else(|| unavailable("metadata generation receipt is missing"))?; + for (column, expected) in [ + ("bindings_digest", receipt.intent.bindings_digest.as_slice()), + ( + "actual_bindings_digest", + receipt.intent.bindings_digest.as_slice(), + ), + ("primary_scope", receipt.intent.primary_scope.as_slice()), + ("storage_seal", receipt.intent.storage_seal.as_slice()), + ] { + if row + .try_get::>>("", column) + .map_err(internal)? + .as_deref() + != Some(expected) + { + return Err(integrity( + "metadata generation receipt seal differs from durable storage", + )); + } + } + let generation = row + .try_get::>("", "generation") + .map_err(internal)?; + if generation != Some(receipt.intent.root_generation) + || row + .try_get::>("", "root_generation") + .map_err(internal)? + != generation + || row + .try_get::>("", "payload_generation") + .map_err(internal)? + != generation + || row + .try_get::>("", "state") + .map_err(internal)? + .as_deref() + != Some("LIVE") + || row + .try_get::>("", "metadata_codec") + .map_err(internal)? + != Some(receipt.legacy.identity.metadata_codec as i16) + || row + .try_get::>("", "expected_size") + .map_err(internal)? + != Some(receipt.legacy.root_payload_bytes as i32) + || row + .try_get::>("", "byte_size") + .map_err(internal)? + != Some(receipt.legacy.root_payload_bytes as i32) + { + return Err(unavailable( + "metadata generation receipt root lifetime is unavailable", + )); + } + Ok(()) + } + + pub async fn inspect_prepare( + &self, + fresh_primary: &DatabaseConnection, + operation_id: &str, + digest: [u8; 32], + phase: MetadataCommitPhase, + ) -> Result { + validate_operation_id(operation_id)?; + let txn = fresh_primary + .begin_with_config(Some(IsolationLevel::ReadCommitted), None) + .await + .map_err(|_| uncertain(operation_id, digest, phase))?; + if self.inner.barrier(&txn).await.is_err() { + let _ = txn.rollback().await; + return Err(uncertain(operation_id, digest, phase)); + } + let result = async { + match self.load_fixed_plan(&txn, operation_id, &digest).await? { + None => Ok(GenerationPrepareObservation::Absent), + Some(fixed) if fixed.stored.record.state == "COMMITTED" => { + check_generation_payloads(&txn, &fixed).await?; + load_fixed_dag(&txn, &fixed).await?; + verify_graph(&txn, &fixed.stored).await?; + Ok(GenerationPrepareObservation::Committed(Box::new( + GenerationMetadataReceipt { + legacy: fixed.stored.receipt()?, + intent: fixed.intent, + }, + ))) + } + Some(fixed) => Ok(GenerationPrepareObservation::Preparing(fixed.intent)), + } + } + .await; + txn.rollback() + .await + .map_err(|_| uncertain(operation_id, digest, phase))?; + result.map_err(|error: SnapshotError| { + if error.code == SnapshotErrorCode::Internal { + uncertain(operation_id, digest, phase) + } else { + MetadataInstallError::Rejected(error) + } + }) + } + + async fn require_fixed_plan( + &self, + connection: &C, + intent: &GenerationPrepareIntent, + ) -> Result { + self.require_scope(intent)?; + let fixed = self + .load_fixed_plan( + connection, + &intent.legacy.operation_id, + &intent.manifest_digest(), + ) + .await? + .ok_or_else(|| unavailable("fixed metadata preparation is missing"))?; + if &fixed.intent != intent { + return Err(integrity( + "metadata intent differs from its fixed generation seal", + )); + } + Ok(fixed) + } + + async fn load_fixed_plan( + &self, + connection: &C, + operation_id: &str, + digest: &[u8; 32], + ) -> Result, SnapshotError> { + let Some(stored) = load_plan(connection, operation_id, digest).await? else { + return Ok(None); + }; + let canonical = stored + .record + .canonical_bindings + .as_deref() + .ok_or_else(|| unavailable("legacy preparation has no generation seal"))?; + let bindings = GenerationBindings::decode(canonical, &stored.plan)?; + let bindings_digest: [u8; 32] = Sha256::digest(canonical).into(); + let primary_scope = stored + .record + .primary_scope + .clone() + .ok_or_else(|| integrity("fixed preparation primary scope is missing"))?; + let storage_seal = seal( + &stored.record.prepare_id, + digest, + &stored.plan.root, + &bindings_digest, + &primary_scope, + )?; + if stored.record.bindings_digest.as_deref() != Some(bindings_digest.as_slice()) + || stored.record.storage_seal.as_deref() != Some(storage_seal.as_slice()) + { + return Err(integrity("metadata generation seal is corrupt")); + } + let intent = GenerationPrepareIntent { + legacy: stored.intent()?, + metadata_root: stored.plan.root, + root_generation: bindings + .0 + .get(&stored.plan.root) + .ok_or_else(|| integrity("root lifetime is missing"))? + .0, + bindings_digest, + primary_scope, + storage_seal, + }; + self.require_scope(&intent)?; + let mut actual = BTreeMap::new(); + for row in &stored.prepare_pages { + let page: [u8; 32] = row.page_id.as_slice().try_into().map_err(internal)?; + let generation = row + .generation + .ok_or_else(|| integrity("fixed metadata page has no generation"))?; + actual.insert(page, (generation, row.expected_size as u64)); + } + if actual != bindings.0 { + return Err(integrity( + "metadata mappings differ from complete fixed generation bindings", + )); + } + check_lifetimes(connection, &stored).await?; + Ok(Some(FixedPlan { + stored, + bindings, + intent, + })) + } + + fn primary_scope(&self) -> Result, SnapshotError> { + let scope = &self.inner.storage_scope; + let bytes = serde_json::to_vec(&( + &scope.storage_uuid, + &scope.database, + scope.database_oid, + &scope.schema, + scope.schema_oid, + &scope.server_address, + scope.server_port, + )) + .map_err(internal)?; + if bytes.is_empty() || bytes.len() > MAX_PRIMARY_SCOPE_BYTES { + return Err(integrity("invalid captured metadata primary scope size")); + } + Ok(bytes) + } + + fn require_scope(&self, intent: &GenerationPrepareIntent) -> Result<(), SnapshotError> { + if intent.primary_scope != self.primary_scope()? { + return Err(integrity( + "metadata seal belongs to another captured primary storage scope", + )); + } + Ok(()) + } +} + +fn seal( + prepare_id: &str, + manifest: &[u8; 32], + root: &[u8; 32], + bindings: &[u8; 32], + scope: &[u8], +) -> Result<[u8; 32], SnapshotError> { + if scope.is_empty() || scope.len() > MAX_PRIMARY_SCOPE_BYTES { + return Err(integrity("invalid metadata generation seal primary scope")); + } + let id = uuid::Uuid::parse_str(prepare_id).map_err(internal)?; + let mut hash = Sha256::new(); + hash.update(b"MST2-METADATA-STORAGE-SEAL-1\0"); + hash.update(id.as_bytes()); + hash.update(manifest); + hash.update(root); + hash.update(bindings); + hash.update((scope.len() as u32).to_be_bytes()); + hash.update(scope); + Ok(hash.finalize().into()) +} + +async fn allocate_lifetimes( + txn: &DatabaseTransaction, + plan: &MetadataInstallPlan, +) -> Result { + let pages: Vec<_> = plan + .pages + .iter() + .map(|(page, size)| json!({"page_id":hex::encode(page),"size":size})) + .collect(); + let encoded = serde_json::to_string(&pages).map_err(internal)?; + txn.execute_raw(statement( + "INSERT INTO mst2_metadata_lifetime(page_id,node_id,generation,state,metadata_codec,expected_size) + SELECT decode(p.page_id,'hex'),'page:sha256:'||p.page_id,1,'RESERVED',$1,p.size + FROM jsonb_to_recordset($2::jsonb) AS p(page_id text,size integer) + ON CONFLICT(page_id) DO NOTHING", + [(plan.identity.metadata_codec as i16).into(),encoded.clone().into()], + )).await.map_err(internal)?; + let rows=txn.query_all_raw(statement( + "SELECT decode(p.page_id,'hex') AS page_id,p.size,l.generation,l.state,l.metadata_codec,l.expected_size, + n.state AS graph_state,n.kind AS graph_kind,n.bytes AS graph_bytes, + b.page_id IS NOT NULL AS payload_present,b.generation AS payload_generation, + b.metadata_codec AS payload_codec,b.byte_size AS payload_size, + EXISTS(SELECT 1 FROM mst2_retention_gc_op g WHERE g.node_id=l.node_id + AND g.operation='REMOVE' AND g.state IN ('PENDING','APPLIED')) AS tombstone + FROM jsonb_to_recordset($1::jsonb) AS p(page_id text,size integer) + JOIN mst2_metadata_lifetime l ON l.page_id=decode(p.page_id,'hex') + LEFT JOIN mst2_retention_node n ON n.node_id=l.node_id + LEFT JOIN mst2_metadata_payload b ON b.page_id=l.page_id ORDER BY l.page_id", + [encoded.into()], + )).await.map_err(internal)?; + let mut bindings = BTreeMap::new(); + for row in rows { + let state: String = row.try_get("", "state").map_err(internal)?; + let generation: i64 = row.try_get("", "generation").map_err(internal)?; + let size: i32 = row.try_get("", "size").map_err(internal)?; + if state == "LIVE" + && (!row + .try_get::("", "payload_present") + .map_err(internal)? + || row + .try_get::>("", "graph_state") + .map_err(internal)? + .is_none()) + { + return Err(unavailable( + "LIVE metadata lifetime has lost its payload or graph", + )); + } + if !["RESERVED", "LIVE"].contains(&state.as_str()) + || row.try_get::("", "tombstone").map_err(internal)? + || row + .try_get::>("", "graph_state") + .map_err(internal)? + .is_some_and(|state| state != "LIVE") + { + return Err(unavailable("metadata lifetime is deleting or removed")); + } + if generation <= 0 + || row.try_get::("", "metadata_codec").map_err(internal)? + != plan.identity.metadata_codec as i16 + || row.try_get::("", "expected_size").map_err(internal)? != size + || row + .try_get::>("", "graph_kind") + .map_err(internal)? + .is_some_and(|kind| kind != "page") + || row + .try_get::>("", "graph_bytes") + .map_err(internal)? + .is_some_and(|bytes| bytes != size as i64) + { + return Err(integrity( + "metadata lifetime profile conflicts with fixed plan", + )); + } + if row + .try_get::("", "payload_present") + .map_err(internal)? + && (row + .try_get::>("", "payload_generation") + .map_err(internal)? + != Some(generation) + || row + .try_get::>("", "payload_codec") + .map_err(internal)? + != Some(plan.identity.metadata_codec as i16) + || row + .try_get::>("", "payload_size") + .map_err(internal)? + != Some(size)) + { + return Err(integrity( + "existing metadata payload has another or unbound lifetime", + )); + } + let page: Vec = row.try_get("", "page_id").map_err(internal)?; + bindings.insert( + page.as_slice().try_into().map_err(internal)?, + (generation, size as u64), + ); + } + if bindings.len() != plan.pages.len() { + return Err(integrity("incomplete metadata lifetime allocation")); + } + Ok(GenerationBindings(bindings)) +} + +async fn check_lifetimes( + connection: &C, + stored: &StoredPlan, +) -> Result<(), SnapshotError> { + let row=connection.query_one_raw(statement( + "SELECT p.page_id,l.generation,l.state,l.metadata_codec,l.expected_size,p.generation AS expected_generation, + n.state AS graph_state,n.kind AS graph_kind,n.bytes AS graph_bytes, + EXISTS(SELECT 1 FROM mst2_retention_gc_op g WHERE g.node_id=l.node_id + AND g.operation='REMOVE' AND g.state IN ('PENDING','APPLIED')) AS tombstone + FROM mst2_metadata_prepare_page p LEFT JOIN mst2_metadata_lifetime l ON l.page_id=p.page_id + LEFT JOIN mst2_retention_node n ON n.node_id=l.node_id WHERE p.prepare_id=$1 + AND (l.page_id IS NULL OR l.generation IS DISTINCT FROM p.generation OR l.metadata_codec<>$2 + OR l.expected_size<>p.expected_size OR l.state NOT IN ('RESERVED','LIVE') + OR ($3='COMMITTED' AND l.state<>'LIVE') + OR (l.state='LIVE' AND n.node_id IS NULL) + OR n.state<>'LIVE' OR n.kind<>'page' OR n.bytes<>p.expected_size + OR EXISTS(SELECT 1 FROM mst2_retention_gc_op g WHERE g.node_id=l.node_id + AND g.operation='REMOVE' AND g.state IN ('PENDING','APPLIED'))) LIMIT 1", + [stored.record.prepare_id.clone().into(),stored.record.metadata_codec.into(),stored.record.state.clone().into()], + )).await.map_err(internal)?; + if let Some(row) = row { + if row + .try_get::>("", "state") + .map_err(internal)? + .is_some_and(|state| !["RESERVED", "LIVE"].contains(&state.as_str())) + || row + .try_get::>("", "graph_state") + .map_err(internal)? + .is_some_and(|state| state != "LIVE") + || row.try_get::("", "tombstone").map_err(internal)? + { + return Err(unavailable( + "fixed metadata lifetime is no longer installable", + )); + } + return Err(integrity( + "fixed metadata lifetime registry conflicts with its bindings", + )); + } + Ok(()) +} + +async fn check_generation_payloads( + connection: &C, + fixed: &FixedPlan, +) -> Result<(), SnapshotError> { + let row=connection.query_one_raw(statement( + "SELECT p.page_id FROM mst2_metadata_prepare_page p LEFT JOIN mst2_metadata_payload b ON b.page_id=p.page_id + WHERE p.prepare_id=$1 AND (b.page_id IS NULL OR b.generation IS DISTINCT FROM p.generation + OR b.metadata_codec<>$2 OR b.byte_size<>p.expected_size OR octet_length(b.payload)<>p.expected_size) LIMIT 1", + [fixed.intent.prepare_id().into(),fixed.stored.record.metadata_codec.into()], + )).await.map_err(internal)?; + if row.is_some() { + return Err(unavailable( + "exact metadata generation payload coverage is incomplete", + )); + } + Ok(()) +} + +async fn load_fixed_dag( + connection: &C, + fixed: &FixedPlan, +) -> Result { + let rows = connection + .query_all_raw(statement( + "SELECT b.page_id,b.generation,b.metadata_codec,b.byte_size,b.payload + FROM mst2_metadata_prepare_page p JOIN mst2_metadata_payload b + ON b.page_id=p.page_id AND b.generation=p.generation WHERE p.prepare_id=$1", + [fixed.intent.prepare_id().into()], + )) + .await + .map_err(internal)?; + let mut pages = Vec::with_capacity(rows.len()); + for row in rows { + let id: Vec = row.try_get("", "page_id").map_err(internal)?; + let id: [u8; 32] = id.as_slice().try_into().map_err(internal)?; + let generation: i64 = row.try_get("", "generation").map_err(internal)?; + let size: i32 = row.try_get("", "byte_size").map_err(internal)?; + if fixed.bindings.0.get(&id) != Some(&(generation, size as u64)) + || row.try_get::("", "metadata_codec").map_err(internal)? + != fixed.stored.record.metadata_codec + { + return Err(integrity( + "installed bytes differ from their fixed lifetime binding", + )); + } + let payload = MetadataPagePayload { + id, + size: size as u64, + bytes: row.try_get("", "payload").map_err(internal)?, + }; + validate_payload(&payload)?; + pages.push(payload); + } + if pages.len() != fixed.bindings.0.len() { + return Err(unavailable( + "exact metadata generation payload coverage is incomplete", + )); + } + ValidatedMetadataDag::validate( + MetadataDagCandidate { + metadata_codec: fixed.stored.plan.identity.metadata_codec, + root: fixed.stored.plan.root, + pages, + edges: fixed.stored.plan.edges.iter().copied().collect(), + }, + MetadataDagLimits::default(), + ) +} + +async fn lock_prepare(txn: &DatabaseTransaction, prepare_id: &str) -> Result<(), SnapshotError> { + txn.query_one_raw(statement( + "SELECT prepare_id FROM mst2_metadata_prepare WHERE prepare_id=$1 FOR UPDATE", + [prepare_id.into()], + )) + .await + .map_err(internal)? + .ok_or_else(|| unavailable("fixed metadata prepare is missing"))?; + Ok(()) +} + +#[cfg(test)] +#[path = "native_metadata_generation_tests.rs"] +mod tests; diff --git a/src/jupiter/storage/native_metadata_install.rs b/src/jupiter/storage/native_metadata_install.rs index 659989c4..d4be4cad 100644 --- a/src/jupiter/storage/native_metadata_install.rs +++ b/src/jupiter/storage/native_metadata_install.rs @@ -101,6 +101,7 @@ pub enum MetadataPrepareObservation { struct StoredPlan { record: mst2_metadata_prepare::Model, plan: MetadataInstallPlan, + prepare_pages: Vec, } impl StoredPlan { @@ -511,6 +512,17 @@ impl PostgresMetadataInstallRepository { ) -> Result { self.barrier(txn).await?; let stored = require_plan(txn, intent).await?; + self.finalize_stored_plan_in_txn(txn, intent, dag, stored) + .await + } + + async fn finalize_stored_plan_in_txn( + &self, + txn: &DatabaseTransaction, + intent: &MetadataPrepareIntent, + dag: &ValidatedMetadataDag, + stored: StoredPlan, + ) -> Result { let expected: BTreeSet<_> = dag .payloads() .iter() @@ -768,7 +780,7 @@ async fn load_plan( .await .map_err(internal)?; let mut coverage = BTreeSet::new(); - for row in rows { + for row in &rows { let page: [u8; 32] = row .page_id .as_slice() @@ -781,7 +793,11 @@ async fn load_plan( "stored metadata preparation coverage differs from its canonical plan", )); } - Ok(Some(StoredPlan { record, plan })) + Ok(Some(StoredPlan { + record, + plan, + prepare_pages: rows, + })) } async fn require_plan( @@ -1034,6 +1050,11 @@ fn unavailable(message: &str) -> SnapshotError { SnapshotError::new(SnapshotErrorCode::ObjectUnavailable, message) } +// Additive repository only. HTTP sessions continue using the existing installer +// until session anchors and lease adoption bind the generation seal. +#[path = "native_metadata_generations.rs"] +pub mod generations; + #[cfg(test)] #[path = "native_metadata_install_tests.rs"] mod tests; From a37837d6819a9bcc0eb0fce5cf251d065f8232d2 Mon Sep 17 00:00:00 2001 From: Xiaoyang Han Date: Wed, 7 Oct 2026 19:04:25 +0800 Subject: [PATCH 06/50] fix(mst2): parse fixed generation records as array chunks Fix native all-target/all-feature Clippy chunks_exact_to_as_chunks. Keep the prior exact length/count guard, canonical re-encoding, per-entry parsing and every generation regression unchanged; no lint suppression. Native gate run 37610549234 confirmed the single failing lint. --- src/jupiter/storage/native_metadata_generations.rs | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/src/jupiter/storage/native_metadata_generations.rs b/src/jupiter/storage/native_metadata_generations.rs index fd1323cd..f7957de9 100644 --- a/src/jupiter/storage/native_metadata_generations.rs +++ b/src/jupiter/storage/native_metadata_generations.rs @@ -97,7 +97,7 @@ impl GenerationBindings { )); } let mut bindings = BTreeMap::new(); - for entry in bytes[12..].chunks_exact(48) { + for entry in bytes[12..].as_chunks::<48>().0 { let page: [u8; 32] = entry[..32].try_into().map_err(internal)?; let generation = i64::from_be_bytes(entry[32..40].try_into().map_err(internal)?); let size = u64::from_be_bytes(entry[40..48].try_into().map_err(internal)?); From ccfb6999ed0fffd25cc5418e76c802ce239e12d5 Mon Sep 17 00:00:00 2001 From: Xiaoyang Han Date: Wed, 7 Oct 2026 19:28:38 +0800 Subject: [PATCH 07/50] feat(init): allow a fixed bootstrap commit timestamp (#58) Add explicit service init --yes --commit-time UNIX_SECONDS to reproduce initial root and materialized path parents for independent backends. Preserve current-time defaults and existing initialization persistence. Under the original advisory lock, reject fixed-input root commit/tree mismatches before object/ref changes. Frozen five-path source independently reviewed; nightly formatting and locked offline metadata passed. Native/PG/network push not run locally. --- src/commands/service/init.rs | 21 +- .../service/init_commit_time_tests.rs | 316 ++++++++++++++++++ src/context/mod.rs | 12 +- src/jupiter/service/mono_service.rs | 78 ++++- src/jupiter/utils/converter.rs | 46 ++- 5 files changed, 458 insertions(+), 15 deletions(-) create mode 100644 src/commands/service/init_commit_time_tests.rs diff --git a/src/commands/service/init.rs b/src/commands/service/init.rs index 4078d84b..69a7a712 100644 --- a/src/commands/service/init.rs +++ b/src/commands/service/init.rs @@ -1,6 +1,9 @@ use clap::{Arg, ArgAction, ArgMatches, Command}; -use crate::{common::errors::MegaResult, config::Config, context::bootstrap_monorepo}; +use crate::{ + common::errors::MegaResult, config::Config, context::bootstrap_monorepo_with_commit_time, + jupiter::utils::converter::BootstrapCommitTime, +}; pub fn cli() -> Command { Command::new("init") @@ -12,11 +15,19 @@ pub fn cli() -> Command { .required(true) .help("Confirm creation of the configured initial Monorepo graph"), ) + .arg( + Arg::new("commit-time") + .long("commit-time") + .value_name("UNIX_SECONDS") + .value_parser(clap::value_parser!(BootstrapCommitTime)) + .help("Set the initial commit's author and committer time (0..=4294967295)"), + ) } -pub(crate) async fn exec(config: Config, _args: &ArgMatches) -> MegaResult { +pub(crate) async fn exec(config: Config, args: &ArgMatches) -> MegaResult { let object_format = config.monorepo.object_format.as_str(); - bootstrap_monorepo(config).await?; + let commit_time = args.get_one::("commit-time").copied(); + bootstrap_monorepo_with_commit_time(config, commit_time).await?; tracing::info!( object_format, "Monorepo bootstrap completed; exiting without starting Git listeners" @@ -24,6 +35,10 @@ pub(crate) async fn exec(config: Config, _args: &ArgMatches) -> MegaResult { Ok(()) } +#[cfg(test)] +#[path = "init_commit_time_tests.rs"] +mod commit_time_tests; + #[cfg(test)] mod tests { use super::cli; diff --git a/src/commands/service/init_commit_time_tests.rs b/src/commands/service/init_commit_time_tests.rs new file mode 100644 index 00000000..74b2157e --- /dev/null +++ b/src/commands/service/init_commit_time_tests.rs @@ -0,0 +1,316 @@ +use std::{collections::BTreeMap, sync::Arc}; + +use git_internal::{ + hash::{HashKind, get_hash_kind, set_hash_kind_for_test}, + internal::object::{ObjectTrait, commit::Commit}, +}; +use sea_orm::{ActiveModelTrait, EntityTrait, IntoActiveModel, QueryOrder, Set}; +use serde_json::{Value, json}; + +use super::{cli, exec}; +use crate::{ + callisto::{mega_blob, mega_commit, mega_tree}, + ceres::pack::materialize::{lock_materialize_tests, materialize_path_refs}, + config::{Config, MonoConfig, MonoObjectFormat, testing::isolated_config}, + jupiter::{ + storage::{Storage, base_storage::StorageConnector, init::database_connection}, + tests::{TestSchemaGuard, test_db_config}, + utils::converter::{BootstrapCommitTime, FromMegaModel, MegaModelConverter}, + }, +}; + +#[test] +fn commit_time_cli_is_explicit_bounded_unsigned_unix_seconds() { + let default = cli().try_get_matches_from(["init", "--yes"]).unwrap(); + assert!( + default + .get_one::("commit-time") + .is_none() + ); + for value in ["0", "1790000000", "4294967295"] { + let args = cli() + .try_get_matches_from(["init", "--yes", "--commit-time", value]) + .unwrap(); + assert_eq!( + args.get_one::("commit-time"), + Some(&value.parse().unwrap()) + ); + } + for value in [ + "", + "-1", + "+1", + "1.0", + "1e3", + " 1", + "1 ", + "4294967296", + "18446744073709551616", + ] { + assert!( + cli() + .try_get_matches_from(["init", "--yes", &format!("--commit-time={value}")]) + .is_err(), + "invalid commit time accepted: {value:?}" + ); + assert!(value.parse::().is_err()); + } + assert!( + cli() + .try_get_matches_from(["init", "--commit-time", "1"]) + .is_err() + ); + assert!( + cli() + .try_get_matches_from(["init", "--yes", "--commit-time", "1", "--commit-time", "2"]) + .is_err() + ); +} + +#[test] +fn explicit_commit_time_reproduces_each_hash_format_and_restores_scope() { + let _scope = set_hash_kind_for_test(HashKind::Sha1); + for (format, kind) in [ + (MonoObjectFormat::Sha1, HashKind::Sha1), + (MonoObjectFormat::Sha256, HashKind::Sha256), + (MonoObjectFormat::Blake3, HashKind::Blake3), + ] { + let config = MonoConfig { + object_format: format, + ..Default::default() + }; + for seconds in ["0", "1790000000", "4294967295"] { + let time = seconds.parse().unwrap(); + let first = MegaModelConverter::init_with_commit_time(&config, Some(time)).unwrap(); + let second = MegaModelConverter::init_with_commit_time(&config, Some(time)).unwrap(); + assert_eq!(first.commit.id.kind(), kind); + assert_eq!(first.root_tree.id.kind(), kind); + assert_eq!(first.commit.id, second.commit.id); + assert_eq!( + first.commit.to_data().unwrap(), + second.commit.to_data().unwrap() + ); + assert_eq!( + first.root_tree.to_data().unwrap(), + second.root_tree.to_data().unwrap() + ); + assert_eq!(first.commit.author.timestamp.to_string(), seconds); + assert_eq!(first.commit.committer.timestamp.to_string(), seconds); + assert_eq!(get_hash_kind(), HashKind::Sha1); + } + } +} + +struct Fixture { + config: Config, + storage: Storage, + schema: TestSchemaGuard, + _directory: tempfile::TempDir, +} + +async fn fixture() -> Fixture { + let directory = tempfile::tempdir().unwrap(); + let (database, schema) = test_db_config(directory.path()).await; + let mut config = isolated_config(directory.path().join("config")); + config.database = database; + config.monorepo.root_dirs = vec!["bench".to_string()]; + config.redis.url = "redis://127.0.0.1:1".to_string(); + let args = cli() + .try_get_matches_from(["init", "--yes", "--commit-time", "1790000000"]) + .unwrap(); + exec(config.clone(), &args).await.unwrap(); + let connection = Arc::new(database_connection(&config.database).await.unwrap()); + let object_store = + crate::jupiter::storage::object_storage::build_object_storage(&config.object_storage) + .await + .unwrap(); + let storage = Storage::new_with_connection(Arc::new(config.clone()), connection, object_store) + .await + .unwrap(); + Fixture { + config, + storage, + schema, + _directory: directory, + } +} + +async fn graph_identity(storage: &Storage) -> Value { + let mono = storage.mono_storage(); + let db = mono.get_connection(); + let commits = mega_commit::Entity::find() + .order_by_asc(mega_commit::Column::CommitId) + .all(db) + .await + .unwrap(); + let trees = mega_tree::Entity::find().all(db).await.unwrap(); + let blobs = mega_blob::Entity::find().all(db).await.unwrap(); + let mut raw_blobs = BTreeMap::new(); + for blob in blobs { + raw_blobs.insert( + blob.blob_id.clone(), + storage + .git_service + .get_object_as_bytes(&blob.blob_id) + .await + .unwrap(), + ); + } + let refs = mono.get_all_refs("/", false).await.unwrap(); + let path_refs = mono.get_all_refs("/bench", false).await.unwrap(); + json!({ + "commits": commits.into_iter().map(|commit| json!({ + "id": commit.commit_id, "tree": commit.tree, "parents": commit.parents_id, + "author": commit.author, "committer": commit.committer, "message": commit.content, + })).collect::>(), + "trees": trees.into_iter().map(|tree| (tree.tree_id, tree.sub_trees)).collect::>(), + "blobs": raw_blobs, + "root_refs": refs.into_iter().map(|r| (r.ref_name, (r.ref_commit_hash, r.ref_tree_hash))).collect::>(), + "path_refs": path_refs.into_iter().map(|r| (r.ref_name, (r.ref_commit_hash, r.ref_tree_hash))).collect::>(), + }) +} + +#[tokio::test] +async fn cli_bootstrap_two_real_schemas_share_seed_root_tree_and_path_parent() { + let _lock = lock_materialize_tests().await; + let first = fixture().await; + let second = fixture().await; + assert_ne!(first.schema.schema(), second.schema.schema()); + let first_root = first + .storage + .mono_storage() + .get_main_ref("/") + .await + .unwrap() + .unwrap(); + let second_root = second + .storage + .mono_storage() + .get_main_ref("/") + .await + .unwrap() + .unwrap(); + assert_eq!(first_root.ref_commit_hash, second_root.ref_commit_hash); + assert_eq!(first_root.ref_tree_hash, second_root.ref_tree_hash); + let mut seeds = Vec::new(); + for fixture in [&first, &second] { + let refs = materialize_path_refs(&fixture.storage, "/bench") + .await + .unwrap(); + assert_eq!(refs.len(), 1); + let parent = fixture + .storage + .mono_storage() + .get_commit_by_hash(&refs[0].ref_commit_hash) + .await + .unwrap() + .unwrap(); + let root = fixture + .storage + .mono_storage() + .get_commit_by_hash(&first_root.ref_commit_hash) + .await + .unwrap() + .unwrap(); + assert_eq!(parent.author, root.author); + assert_eq!(parent.committer, root.committer); + assert_eq!(parent.parents_id, json!([])); + let parent = Commit::from_mega_model(parent); + let seed = Commit::new_with_kind( + HashKind::Sha1, + parent.author.clone(), + parent.committer.clone(), + parent.tree_id, + vec![parent.id], + "\npaired seed", + ) + .unwrap(); + assert_eq!(seed.parent_commit_ids, vec![parent.id]); + seeds.push(seed.to_data().unwrap()); + } + assert_eq!(seeds[0], seeds[1]); + assert_eq!( + graph_identity(&first.storage).await, + graph_identity(&second.storage).await + ); +} + +#[tokio::test] +async fn cli_bootstrap_existing_root_replays_exactly_and_rejects_changed_inputs_without_writes() { + let fixture = fixture().await; + let before = graph_identity(&fixture.storage).await; + let args = cli() + .try_get_matches_from(["init", "--yes", "--commit-time", "1790000000"]) + .unwrap(); + exec(fixture.config.clone(), &args).await.unwrap(); + assert_eq!(graph_identity(&fixture.storage).await, before); + let changed_time = cli() + .try_get_matches_from(["init", "--yes", "--commit-time", "1790000001"]) + .unwrap(); + let error = exec(fixture.config.clone(), &changed_time) + .await + .unwrap_err(); + assert!(error.to_string().contains("does not match")); + assert_eq!(graph_identity(&fixture.storage).await, before); + let mut changed_config = fixture.config.clone(); + changed_config + .monorepo + .root_dirs + .push("different".to_string()); + let error = exec(changed_config, &args).await.unwrap_err(); + assert!(error.to_string().contains("does not match")); + assert_eq!(graph_identity(&fixture.storage).await, before); + let default_args = cli().try_get_matches_from(["init", "--yes"]).unwrap(); + exec(fixture.config.clone(), &default_args).await.unwrap(); + assert_eq!(graph_identity(&fixture.storage).await, before); +} + +#[tokio::test] +async fn cli_bootstrap_checks_stored_root_objects_even_when_ref_ids_match() { + let fixture = fixture().await; + let mono = fixture.storage.mono_storage(); + let root = mono.get_main_ref("/").await.unwrap().unwrap(); + let args = cli() + .try_get_matches_from(["init", "--yes", "--commit-time", "1790000000"]) + .unwrap(); + let commit = mono + .get_commit_by_hash(&root.ref_commit_hash) + .await + .unwrap() + .unwrap(); + let original_message = commit.content.clone(); + let mut changed = commit.into_active_model(); + changed.content = Set(Some("\nchanged stored commit".to_string())); + let changed = changed.update(mono.get_connection()).await.unwrap(); + let corrupt_commit = graph_identity(&fixture.storage).await; + assert!( + exec(fixture.config.clone(), &args) + .await + .unwrap_err() + .to_string() + .contains("does not match") + ); + assert_eq!(graph_identity(&fixture.storage).await, corrupt_commit); + let mut restored = changed.into_active_model(); + restored.content = Set(original_message); + restored.update(mono.get_connection()).await.unwrap(); + let tree = mono + .get_tree_by_hash(&root.ref_tree_hash) + .await + .unwrap() + .unwrap(); + let mut bytes = tree.sub_trees.clone(); + bytes.push(b'x'); + let mut changed_tree = tree.into_active_model(); + changed_tree.sub_trees = Set(bytes); + changed_tree.update(mono.get_connection()).await.unwrap(); + let corrupt_tree = graph_identity(&fixture.storage).await; + assert!( + exec(fixture.config.clone(), &args) + .await + .unwrap_err() + .to_string() + .contains("does not match") + ); + assert_eq!(graph_identity(&fixture.storage).await, corrupt_tree); +} diff --git a/src/context/mod.rs b/src/context/mod.rs index a4c05d22..560cf6fa 100644 --- a/src/context/mod.rs +++ b/src/context/mod.rs @@ -31,6 +31,7 @@ use crate::{ mono_storage::MonoStorage, vault_storage::VaultStorage, }, + utils::converter::BootstrapCommitTime, }, }; @@ -75,6 +76,13 @@ pub struct AppContext { /// Redis, notification workers, and Git listeners are deliberately /// outside this one-shot path. pub(crate) async fn bootstrap_monorepo(config: crate::config::Config) -> Result<(), MegaError> { + bootstrap_monorepo_with_commit_time(config, None).await +} + +pub(crate) async fn bootstrap_monorepo_with_commit_time( + config: crate::config::Config, + commit_time: Option, +) -> Result<(), MegaError> { config.validate()?; config.monorepo.object_hash_kind()?; let config = Arc::new(config); @@ -111,7 +119,9 @@ pub(crate) async fn bootstrap_monorepo(config: crate::config::Config) -> Result< }, }; - mono_service.bootstrap_monorepo(&config.monorepo).await + mono_service + .bootstrap_monorepo_with_commit_time(&config.monorepo, commit_time) + .await } impl AppContext { diff --git a/src/jupiter/service/mono_service.rs b/src/jupiter/service/mono_service.rs index 8574a734..a94668a9 100644 --- a/src/jupiter/service/mono_service.rs +++ b/src/jupiter/service/mono_service.rs @@ -3,7 +3,7 @@ use std::sync::Arc; use futures::{StreamExt, stream}; use git_internal::internal::{ metadata::{EntryMeta, MetaAttached}, - object::blob::Blob, + object::{ObjectTrait, blob::Blob}, pack::entry::Entry, }; use sea_orm::{ActiveModelTrait, ConnectionTrait, IntoActiveModel, TransactionTrait}; @@ -19,7 +19,9 @@ use crate::{ base_storage::{BaseStorage, StorageConnector}, mono_storage::MonoStorage, }, - utils::converter::{IntoMegaModel, MegaModelConverter, MegaObjectModel, process_entry}, + utils::converter::{ + BootstrapCommitTime, IntoMegaModel, MegaModelConverter, MegaObjectModel, process_entry, + }, }, }; @@ -84,7 +86,7 @@ impl MonoService { pub async fn init_monorepo(&self, mono_config: &MonoConfig) -> Result<(), MegaError> { mono_config.ensure_normal_service_object_format()?; - self.initialize_monorepo(mono_config).await + self.initialize_monorepo(mono_config, None).await } /// Initializes an empty Monorepo for a controlled, one-shot bootstrap. @@ -97,12 +99,69 @@ impl MonoService { mono_config: &MonoConfig, ) -> Result<(), MegaError> { mono_config.object_hash_kind()?; - self.initialize_monorepo(mono_config).await + self.initialize_monorepo(mono_config, None).await + } + + pub(crate) async fn bootstrap_monorepo_with_commit_time( + &self, + mono_config: &MonoConfig, + commit_time: Option, + ) -> Result<(), MegaError> { + mono_config.object_hash_kind()?; + self.initialize_monorepo(mono_config, commit_time).await } - async fn initialize_monorepo(&self, mono_config: &MonoConfig) -> Result<(), MegaError> { + async fn ensure_existing_root_matches_expected( + &self, + txn: &sea_orm::DatabaseTransaction, + root_ref: &crate::callisto::mega_refs::Model, + expected: &MegaModelConverter, + ) -> Result<(), MegaError> { + let mismatch = || { + MegaError::Other( + "existing Monorepo root does not match the configured initial graph and commit time; refusing to modify it" + .to_string(), + ) + }; + if root_ref.ref_commit_hash != expected.commit.id.to_string() + || root_ref.ref_tree_hash != expected.root_tree.id.to_string() + { + return Err(mismatch()); + } + let commits = self + .mono_storage + .get_commits_by_hashes_fallible(txn, std::slice::from_ref(&root_ref.ref_commit_hash)) + .await?; + let commit = commits.first().ok_or_else(mismatch)?; + let tree = self + .mono_storage + .get_tree_by_hash_in_txn(&root_ref.ref_tree_hash, txn) + .await? + .ok_or_else(mismatch)?; + if commit.tree != expected.commit.tree_id.to_string() + || commit.parents_id != serde_json::json!([]) + || commit.author.as_deref().map(str::as_bytes) + != Some(expected.commit.author.to_data()?.as_slice()) + || commit.committer.as_deref().map(str::as_bytes) + != Some(expected.commit.committer.to_data()?.as_slice()) + || commit.content.as_deref() != Some(expected.commit.message.as_str()) + || tree.sub_trees != expected.root_tree.to_data()? + { + return Err(mismatch()); + } + Ok(()) + } + + async fn initialize_monorepo( + &self, + mono_config: &MonoConfig, + commit_time: Option, + ) -> Result<(), MegaError> { let txn = self.mono_storage.get_connection().begin().await?; acquire_monorepo_initialization_lock(&txn).await?; + let expected = commit_time + .map(|time| MegaModelConverter::init_with_commit_time(mono_config, Some(time))) + .transpose()?; if let Some(root_ref) = self.mono_storage.get_main_ref_in_txn("/", &txn).await? { ensure_existing_root_ref_matches_config( @@ -110,11 +169,18 @@ impl MonoService { &root_ref.ref_commit_hash, &root_ref.ref_tree_hash, )?; + if let Some(expected) = &expected { + self.ensure_existing_root_matches_expected(&txn, &root_ref, expected) + .await?; + } txn.commit().await?; tracing::info!("Monorepo Directory Already Inited, skip init process!"); return Ok(()); } - let converter = MegaModelConverter::init(mono_config)?; + let converter = match expected { + Some(expected) => expected, + None => MegaModelConverter::init(mono_config)?, + }; let commit = converter .commit .into_mega_model(EntryMeta::default()) diff --git a/src/jupiter/utils/converter.rs b/src/jupiter/utils/converter.rs index 40f3502f..697816c4 100644 --- a/src/jupiter/utils/converter.rs +++ b/src/jupiter/utils/converter.rs @@ -1,4 +1,4 @@ -use std::{cell::RefCell, collections::HashMap}; +use std::{cell::RefCell, collections::HashMap, str::FromStr}; use git_internal::{ hash::{HashKind, ObjectHash, get_hash_kind, set_hash_kind}, @@ -663,6 +663,21 @@ pub struct MegaModelConverter { pub refs: mega_refs::ActiveModel, } +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub(crate) struct BootstrapCommitTime(u32); + +impl FromStr for BootstrapCommitTime { + type Err = String; + + fn from_str(value: &str) -> Result { + let invalid = || "commit time must be Unix seconds in 0..=4294967295".to_string(); + if value.is_empty() || !value.bytes().all(|byte| byte.is_ascii_digit()) { + return Err(invalid()); + } + value.parse::().map(Self).map_err(|_| invalid()) + } +} + struct InitializationHashKindGuard { previous: HashKind, } @@ -725,11 +740,32 @@ impl MegaModelConverter { } pub fn init(mono_config: &MonoConfig) -> Result { + Self::init_with_commit_time(mono_config, None) + } + + pub(crate) fn init_with_commit_time( + mono_config: &MonoConfig, + commit_time: Option, + ) -> Result { let hash_kind = mono_config.object_hash_kind()?; - Ok(with_initialization_hash_kind(hash_kind, || { + with_initialization_hash_kind(hash_kind, || { let (tree_maps, blob_maps, root_tree) = init_trees(mono_config); - let commit = Commit::from_tree_id(root_tree.id, vec![], "\nInit Mega Directory"); + let commit = match commit_time { + Some(BootstrapCommitTime(seconds)) => Commit::new_with_kind( + hash_kind, + Signature::from_data( + format!("author mega {seconds} +0800").into_bytes(), + )?, + Signature::from_data( + format!("committer mega {seconds} +0800").into_bytes(), + )?, + root_tree.id, + vec![], + "\nInit Mega Directory", + )?, + None => Commit::from_tree_id(root_tree.id, vec![], "\nInit Mega Directory"), + }; let mega_ref = mega_refs::Model { id: generate_id(), @@ -753,8 +789,8 @@ impl MegaModelConverter { refs: mega_ref.into(), }; converter.traverse_from_root(); - converter - })) + Ok(converter) + }) } } From c00668434759df4f1bd981c9bb945c1bbaf15f94 Mon Sep 17 00:00:00 2001 From: Xiaoyang Han Date: Wed, 7 Oct 2026 19:32:45 +0800 Subject: [PATCH 08/50] feat(mst2): preserve metadata lifetime history and explicit terminal receipts (#59) Add composite immutable lifetime history, a separate current watermark, persisted domain-bound preparation seals, explicit abort/coverage retirement and same-primary terminal replay. Preserve existing v3 serving, G1 NULL-domain receipts, original guards and bootstrap C1 blobs. No payload deletion, qualified collector, generation replacement or production adoption is enabled. Twelve frozen code/test paths independently source-reviewed; native validation and performance remain unmeasured. --- src/api/router/snapshot_content_tests.rs | 3 + .../snapshot_generation_history_fixture.rs | 31 + .../snapshot_generation_upgrade_tests.rs | 1 + src/callisto/mod.rs | 1 + src/callisto/mst2_metadata_current.rs | 13 + src/callisto/mst2_metadata_lifetime.rs | 1 + src/callisto/mst2_metadata_prepare.rs | 3 + ...0300_add_mst2_metadata_lifetime_history.rs | 67 ++ src/jupiter/migration/mod.rs | 10 +- .../storage/native_metadata_generations.rs | 123 +++- .../storage/native_metadata_history.rs | 357 ++++++++++ .../storage/native_metadata_history_tests.rs | 634 ++++++++++++++++++ 12 files changed, 1225 insertions(+), 19 deletions(-) create mode 100644 src/api/router/snapshot_generation_history_fixture.rs create mode 100644 src/callisto/mst2_metadata_current.rs create mode 100644 src/jupiter/migration/m20261007_000300_add_mst2_metadata_lifetime_history.rs create mode 100644 src/jupiter/storage/native_metadata_history.rs create mode 100644 src/jupiter/storage/native_metadata_history_tests.rs diff --git a/src/api/router/snapshot_content_tests.rs b/src/api/router/snapshot_content_tests.rs index 3a7c8390..5c85ffbd 100644 --- a/src/api/router/snapshot_content_tests.rs +++ b/src/api/router/snapshot_content_tests.rs @@ -85,6 +85,9 @@ mod durable_sessions; #[path = "snapshot_generation_upgrade_tests.rs"] mod generation_upgrade; +#[path = "snapshot_generation_history_fixture.rs"] +mod generation_history_fixture; + #[derive(Default)] struct ReadCounts { whole: AtomicUsize, diff --git a/src/api/router/snapshot_generation_history_fixture.rs b/src/api/router/snapshot_generation_history_fixture.rs new file mode 100644 index 00000000..9a694424 --- /dev/null +++ b/src/api/router/snapshot_generation_history_fixture.rs @@ -0,0 +1,31 @@ +use sea_orm::{ConnectionTrait, DatabaseConnection, DbBackend, Statement}; + +pub(super) async fn restore_empty_g1_schema(db: &DatabaseConnection) { + let count = db + .query_one_raw(Statement::from_string( + DbBackend::Postgres, + "SELECT count(*)::bigint AS n FROM mst2_metadata_current", + )) + .await + .unwrap() + .unwrap() + .try_get::("", "n") + .unwrap(); + assert_eq!( + count, 0, + "upgrade fixture has no generation-bound incarnation to erase" + ); + db.execute_unprepared( + "DROP TABLE mst2_metadata_current; + DROP FUNCTION mst2_metadata_current_guard(); + DROP INDEX idx_mst2_metadata_lifetime_node_generation; + ALTER TABLE mst2_metadata_lifetime DROP CONSTRAINT mst2_metadata_lifetime_pkey, + ADD PRIMARY KEY(page_id),ADD UNIQUE(node_id); + ALTER TABLE mst2_metadata_prepare DROP CONSTRAINT mst2_metadata_prepare_state_check, + DROP CONSTRAINT mst2_metadata_prepare_terminal_check, + ADD CONSTRAINT mst2_metadata_prepare_state_check CHECK (state IN ('PREPARING','COMMITTED')), + ADD CONSTRAINT mst2_metadata_prepare_check CHECK ((state='COMMITTED')=(committed_at IS NOT NULL)), + DROP COLUMN graph_domain,DROP COLUMN aborted_at,DROP COLUMN coverage_retired_at; + DELETE FROM seaql_migrations WHERE version='m20261007_000300_add_mst2_metadata_lifetime_history'" + ).await.unwrap(); +} diff --git a/src/api/router/snapshot_generation_upgrade_tests.rs b/src/api/router/snapshot_generation_upgrade_tests.rs index c2afda08..9f14f96d 100644 --- a/src/api/router/snapshot_generation_upgrade_tests.rs +++ b/src/api/router/snapshot_generation_upgrade_tests.rs @@ -42,6 +42,7 @@ async fn mst2_generation_additive_upgrade_preserves_legacy_v3_sid_lease_and_new_ ); let pages = scalar(db, "SELECT count(*) FROM mst2_metadata_payload").await; let original = success_json(fixture.send("GET", "descriptor", Body::empty()).await).await; + super::generation_history_fixture::restore_empty_g1_schema(db).await; // The unchanged v3 installer created these durable rows. Remove only the // empty additive schema in this isolated fixture to reproduce the previous // deployed schema; the SID, lease, CAS bytes and all original guards survive. diff --git a/src/callisto/mod.rs b/src/callisto/mod.rs index 8eb43be5..8bdddc74 100644 --- a/src/callisto/mod.rs +++ b/src/callisto/mod.rs @@ -66,6 +66,7 @@ pub mod mega_view_root_chain_scan; pub mod mega_webhook; pub mod mega_webhook_delivery; pub mod mega_webhook_event_type; +pub mod mst2_metadata_current; pub mod mst2_metadata_lifetime; pub mod mst2_metadata_payload; pub mod mst2_metadata_prepare; diff --git a/src/callisto/mst2_metadata_current.rs b/src/callisto/mst2_metadata_current.rs new file mode 100644 index 00000000..0c759a1c --- /dev/null +++ b/src/callisto/mst2_metadata_current.rs @@ -0,0 +1,13 @@ +use sea_orm::entity::prelude::*; + +#[derive(Clone, Debug, PartialEq, DeriveEntityModel, Eq)] +#[sea_orm(table_name = "mst2_metadata_current")] +pub struct Model { + #[sea_orm(primary_key, auto_increment = false)] + pub page_id: Vec, + pub generation: i64, +} + +#[derive(Copy, Clone, Debug, EnumIter, DeriveRelation)] +pub enum Relation {} +impl ActiveModelBehavior for ActiveModel {} diff --git a/src/callisto/mst2_metadata_lifetime.rs b/src/callisto/mst2_metadata_lifetime.rs index 9807c27a..d9bc99da 100644 --- a/src/callisto/mst2_metadata_lifetime.rs +++ b/src/callisto/mst2_metadata_lifetime.rs @@ -6,6 +6,7 @@ pub struct Model { #[sea_orm(primary_key, auto_increment = false)] pub page_id: Vec, pub node_id: String, + #[sea_orm(primary_key, auto_increment = false)] pub generation: i64, pub state: String, pub metadata_codec: i16, diff --git a/src/callisto/mst2_metadata_prepare.rs b/src/callisto/mst2_metadata_prepare.rs index 98b017a8..c586cf0d 100644 --- a/src/callisto/mst2_metadata_prepare.rs +++ b/src/callisto/mst2_metadata_prepare.rs @@ -12,6 +12,7 @@ pub struct Model { pub bindings_digest: Option>, pub primary_scope: Option>, pub storage_seal: Option>, + pub graph_domain: Option, pub source_domain: String, pub tagged_root_tree_oid: String, pub scope: String, @@ -29,6 +30,8 @@ pub struct Model { pub state: String, pub created_at: DateTimeWithTimeZone, pub committed_at: Option, + pub aborted_at: Option, + pub coverage_retired_at: Option, } #[derive(Copy, Clone, Debug, EnumIter, DeriveRelation)] diff --git a/src/jupiter/migration/m20261007_000300_add_mst2_metadata_lifetime_history.rs b/src/jupiter/migration/m20261007_000300_add_mst2_metadata_lifetime_history.rs new file mode 100644 index 00000000..51dcee8f --- /dev/null +++ b/src/jupiter/migration/m20261007_000300_add_mst2_metadata_lifetime_history.rs @@ -0,0 +1,67 @@ +//! Preserve historical incarnations and explicit preparation terminal states. + +use sea_orm_migration::prelude::*; + +#[derive(DeriveMigrationName)] +pub struct Migration; + +#[async_trait::async_trait] +impl MigrationTrait for Migration { + async fn up(&self, manager: &SchemaManager) -> Result<(), DbErr> { + manager.get_connection().execute_unprepared( + "ALTER TABLE mst2_metadata_lifetime + DROP CONSTRAINT mst2_metadata_lifetime_pkey, + DROP CONSTRAINT mst2_metadata_lifetime_node_id_key, + ADD PRIMARY KEY(page_id,generation); + CREATE TABLE mst2_metadata_current ( + page_id bytea PRIMARY KEY CHECK (octet_length(page_id)=32), + generation bigint NOT NULL CHECK (generation>0), + FOREIGN KEY(page_id,generation) REFERENCES mst2_metadata_lifetime(page_id,generation) + ); + INSERT INTO mst2_metadata_current(page_id,generation) + SELECT page_id,generation FROM mst2_metadata_lifetime; + CREATE FUNCTION mst2_metadata_current_guard() RETURNS trigger LANGUAGE plpgsql AS $$ + BEGIN RAISE EXCEPTION 'metadata current watermark requires generation-fenced collection'; END $$; + CREATE TRIGGER mst2_metadata_current_guard BEFORE UPDATE OR DELETE + ON mst2_metadata_current FOR EACH ROW EXECUTE FUNCTION mst2_metadata_current_guard(); + ALTER TABLE mst2_metadata_prepare + ADD COLUMN graph_domain text CHECK (graph_domain IN ('generic-v1','qualified-v1')), + ADD COLUMN aborted_at timestamptz, + ADD COLUMN coverage_retired_at timestamptz, + DROP CONSTRAINT mst2_metadata_prepare_state_check, + DROP CONSTRAINT mst2_metadata_prepare_check, + ADD CONSTRAINT mst2_metadata_prepare_state_check CHECK (state IN ('PREPARING','COMMITTED','ABORTED')), + ADD CONSTRAINT mst2_metadata_prepare_terminal_check CHECK ( + (state='COMMITTED')=(committed_at IS NOT NULL) + AND (state='ABORTED')=(aborted_at IS NOT NULL) + AND (coverage_retired_at IS NULL OR state='COMMITTED') + AND (state<>'ABORTED' OR storage_seal IS NOT NULL) + AND (graph_domain IS NULL OR storage_seal IS NOT NULL) + ); + CREATE OR REPLACE FUNCTION mst2_metadata_generation_seal_guard() RETURNS trigger LANGUAGE plpgsql AS $$ + BEGIN + IF NEW.canonical_bindings IS DISTINCT FROM OLD.canonical_bindings + OR NEW.bindings_digest IS DISTINCT FROM OLD.bindings_digest + OR NEW.primary_scope IS DISTINCT FROM OLD.primary_scope + OR NEW.storage_seal IS DISTINCT FROM OLD.storage_seal + OR NEW.graph_domain IS DISTINCT FROM OLD.graph_domain THEN + RAISE EXCEPTION 'metadata generation seal cannot be rebound'; + END IF; + IF OLD.state IN ('COMMITTED','ABORTED') AND NEW.state IS DISTINCT FROM OLD.state THEN + RAISE EXCEPTION 'metadata preparation terminal state cannot be revived'; + END IF; + IF OLD.committed_at IS NOT NULL AND NEW.committed_at IS DISTINCT FROM OLD.committed_at + OR OLD.aborted_at IS NOT NULL AND NEW.aborted_at IS DISTINCT FROM OLD.aborted_at + OR OLD.coverage_retired_at IS NOT NULL AND NEW.coverage_retired_at IS DISTINCT FROM OLD.coverage_retired_at THEN + RAISE EXCEPTION 'metadata preparation terminal receipt cannot be rewritten'; + END IF; + RETURN NEW; + END $$; + CREATE INDEX idx_mst2_metadata_lifetime_node_generation ON mst2_metadata_lifetime(node_id,generation)" + ).await.map(|_| ()) + } + + async fn down(&self, _manager: &SchemaManager) -> Result<(), DbErr> { + Ok(()) + } +} diff --git a/src/jupiter/migration/mod.rs b/src/jupiter/migration/mod.rs index 6ceda285..4a302ee3 100644 --- a/src/jupiter/migration/mod.rs +++ b/src/jupiter/migration/mod.rs @@ -147,6 +147,7 @@ mod m20261005_000300_add_mst2_metadata_install; mod m20261006_000100_add_view_tables; mod m20261007_000100_add_mst2_snapshot_sessions; mod m20261007_000200_add_mst2_metadata_generations; +mod m20261007_000300_add_mst2_metadata_lifetime_history; mod runner; pub use m20260905_000100_add_push_queue::ensure_queue_control_seed; pub use runner::apply_migrations; @@ -282,6 +283,7 @@ impl MigratorTrait for Migrator { Box::new(m20261006_000100_add_view_tables::Migration), Box::new(m20261007_000100_add_mst2_snapshot_sessions::Migration), Box::new(m20261007_000200_add_mst2_metadata_generations::Migration), + Box::new(m20261007_000300_add_mst2_metadata_lifetime_history::Migration), ] } } @@ -1192,7 +1194,7 @@ mod tests { async fn import_repo_alias_rows_canonicalized() { let names = migration_names(); assert_eq!( - &names[names.len() - 10..names.len() - 1], + &names[names.len() - 11..names.len() - 2], &[ "m20260923_000200_canonicalize_import_repo_paths".to_string(), "m20260925_000100_media_paging".to_string(), @@ -1207,9 +1209,13 @@ mod tests { "native retention, metadata installation and view tables follow media paging" ); assert_eq!( - names.last().unwrap(), + &names[names.len() - 2], "m20261007_000200_add_mst2_metadata_generations" ); + assert_eq!( + names.last().unwrap(), + "m20261007_000300_add_mst2_metadata_lifetime_history" + ); let db = alias_db().await; insert_repo(&db, 1, "/third-party//a").await; diff --git a/src/jupiter/storage/native_metadata_generations.rs b/src/jupiter/storage/native_metadata_generations.rs index f7957de9..29189469 100644 --- a/src/jupiter/storage/native_metadata_generations.rs +++ b/src/jupiter/storage/native_metadata_generations.rs @@ -6,6 +6,33 @@ use super::*; const MAX_BINDINGS_BYTES: usize = 12 + 48 * 4096; const MAX_PRIMARY_SCOPE_BYTES: usize = 16384; +const GENERIC_GRAPH_DOMAIN: &str = "generic-v1"; + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +enum GraphDomain { + LegacyGeneric, + Generic, + Qualified, +} + +impl GraphDomain { + fn from_stored(domain: Option<&str>) -> Result { + match domain { + None => Ok(Self::LegacyGeneric), + Some(GENERIC_GRAPH_DOMAIN) => Ok(Self::Generic), + Some("qualified-v1") => Ok(Self::Qualified), + _ => Err(integrity("unknown stored metadata graph domain")), + } + } + + fn stored(self) -> Option<&'static str> { + match self { + Self::LegacyGeneric => None, + Self::Generic => Some(GENERIC_GRAPH_DOMAIN), + Self::Qualified => Some("qualified-v1"), + } + } +} #[derive(Debug, Clone, PartialEq, Eq)] pub struct GenerationPrepareIntent { @@ -13,8 +40,9 @@ pub struct GenerationPrepareIntent { metadata_root: [u8; 32], root_generation: i64, bindings_digest: [u8; 32], - primary_scope: Vec, + primary_scope: Box<[u8]>, storage_seal: [u8; 32], + graph_domain: GraphDomain, } impl GenerationPrepareIntent { @@ -125,12 +153,14 @@ struct FixedPlan { #[derive(Clone)] pub struct PostgresMetadataGenerationRepository { inner: PostgresMetadataInstallRepository, + graph_domain: &'static str, } impl PostgresMetadataGenerationRepository { pub async fn new(connection: DatabaseConnection) -> Result { Ok(Self { inner: PostgresMetadataInstallRepository::new(connection).await?, + graph_domain: GENERIC_GRAPH_DOMAIN, }) } @@ -149,7 +179,7 @@ impl PostgresMetadataGenerationRepository { if let Some(fixed) = self.load_fixed_plan(&txn, operation_id, &digest).await? { return Ok(fixed.intent); } - let bindings = allocate_lifetimes(&txn, &plan).await?; + let bindings = allocate_lifetimes(&txn, &plan, self.graph_domain).await?; let canonical_bindings = bindings.encode()?; let bindings_digest = Sha256::digest(&canonical_bindings).into(); let prepare_id = uuid::Uuid::new_v4().to_string(); @@ -158,16 +188,17 @@ impl PostgresMetadataGenerationRepository { metadata_root: plan.root, root_generation: bindings.0.get(&plan.root).ok_or_else(|| integrity("root lifetime is missing"))?.0, bindings_digest, - storage_seal: seal(&prepare_id, &digest, &plan.root, &bindings_digest, &primary_scope)?, - primary_scope, + storage_seal: seal(&prepare_id, &digest, &plan.root, &bindings_digest, &primary_scope,Some(self.graph_domain))?, + primary_scope:primary_scope.into_boxed_slice(), + graph_domain:GraphDomain::from_stored(Some(self.graph_domain))?, }; let identity = &plan.identity; txn.execute_raw(statement( "INSERT INTO mst2_metadata_prepare(prepare_id,operation_id,manifest_digest,canonical_plan, source_domain,tagged_root_tree_oid,scope,schema_version,metadata_codec,materialization_policy, fs_semantics,access_projection,verification_revision,projection_revision,metadata_root, - node_count,edge_count,total_bytes,state,canonical_bindings,bindings_digest,primary_scope,storage_seal) - VALUES($1,$2,$3,$4,$5,$6,$7,$8,$9,$10,$11,$12,$13,$14,$15,$16,$17,$18,'PREPARING',$19,$20,$21,$22)", + node_count,edge_count,total_bytes,state,canonical_bindings,bindings_digest,primary_scope,storage_seal,graph_domain) + VALUES($1,$2,$3,$4,$5,$6,$7,$8,$9,$10,$11,$12,$13,$14,$15,$16,$17,$18,'PREPARING',$19,$20,$21,$22,$23)", [prepare_id.clone().into(),operation_id.into(),digest.to_vec().into(),plan.encode()?.into(), identity.source_domain.clone().into(),identity.tagged_root_tree_oid.clone().into(),identity.scope.clone().into(), (identity.schema_version as i16).into(),(identity.metadata_codec as i16).into(), @@ -175,7 +206,7 @@ impl PostgresMetadataGenerationRepository { (identity.access_projection as i16).into(),identity.verification_revision.into(), (identity.projection_revision as i16).into(),plan.root.to_vec().into(),(plan.pages.len() as i32).into(), (plan.edges.len() as i32).into(),(plan.total_bytes as i64).into(),canonical_bindings.into(), - intent.bindings_digest.to_vec().into(),intent.primary_scope.clone().into(),intent.storage_seal.to_vec().into()], + intent.bindings_digest.to_vec().into(),intent.primary_scope.to_vec().into(),intent.storage_seal.to_vec().into(),self.graph_domain.into()], )).await.map_err(internal)?; let pages: Vec<_> = bindings.0.iter().map(|(page,(generation,size))| json!({"page_id":hex::encode(page),"generation":generation,"size":size})).collect(); @@ -380,24 +411,38 @@ impl PostgresMetadataGenerationRepository { .verify_receipt_in_txn(txn, &receipt.legacy, tagged_root_tree_oid, scope) .await?; let row=txn.query_one_raw(statement( - "SELECT p.bindings_digest,p.primary_scope,p.storage_seal, + "SELECT p.bindings_digest,p.primary_scope,p.storage_seal,p.graph_domain,p.coverage_retired_at IS NOT NULL AS retired, CASE WHEN octet_length(p.canonical_bindings)<=$2 THEN sha256(p.canonical_bindings) END AS actual_bindings_digest, l.generation,l.state,l.metadata_codec,l.expected_size,m.generation AS root_generation, b.generation AS payload_generation,b.byte_size FROM mst2_metadata_prepare p LEFT JOIN mst2_metadata_prepare_page m ON m.prepare_id=p.prepare_id AND m.page_id=p.metadata_root LEFT JOIN mst2_metadata_lifetime l ON l.page_id=m.page_id AND l.generation=m.generation + LEFT JOIN mst2_metadata_current c ON c.page_id=l.page_id AND c.generation=l.generation LEFT JOIN mst2_metadata_payload b ON b.page_id=m.page_id - WHERE p.prepare_id=$1", + WHERE p.prepare_id=$1 AND c.page_id IS NOT NULL", [receipt.intent.prepare_id().into(),(MAX_BINDINGS_BYTES as i32).into()], )).await.map_err(internal)?.ok_or_else(|| unavailable("metadata generation receipt is missing"))?; + let domain = row + .try_get::>("", "graph_domain") + .map_err(internal)?; + if GraphDomain::from_stored(domain.as_deref())? != receipt.intent.graph_domain { + return Err(integrity( + "metadata generation receipt graph domain differs from storage", + )); + } + if row.try_get::("", "retired").map_err(internal)? { + return Err(unavailable( + "metadata generation receipt coverage was retired", + )); + } for (column, expected) in [ ("bindings_digest", receipt.intent.bindings_digest.as_slice()), ( "actual_bindings_digest", receipt.intent.bindings_digest.as_slice(), ), - ("primary_scope", receipt.intent.primary_scope.as_slice()), + ("primary_scope", receipt.intent.primary_scope.as_ref()), ("storage_seal", receipt.intent.storage_seal.as_slice()), ] { if row @@ -525,6 +570,9 @@ impl PostgresMetadataGenerationRepository { let Some(stored) = load_plan(connection, operation_id, digest).await? else { return Ok(None); }; + if stored.record.coverage_retired_at.is_some() { + return Err(unavailable("metadata preparation coverage was retired")); + } let canonical = stored .record .canonical_bindings @@ -543,6 +591,7 @@ impl PostgresMetadataGenerationRepository { &stored.plan.root, &bindings_digest, &primary_scope, + stored.record.graph_domain.as_deref(), )?; if stored.record.bindings_digest.as_deref() != Some(bindings_digest.as_slice()) || stored.record.storage_seal.as_deref() != Some(storage_seal.as_slice()) @@ -558,8 +607,9 @@ impl PostgresMetadataGenerationRepository { .ok_or_else(|| integrity("root lifetime is missing"))? .0, bindings_digest, - primary_scope, + primary_scope: primary_scope.into_boxed_slice(), storage_seal, + graph_domain: GraphDomain::from_stored(stored.record.graph_domain.as_deref())?, }; self.require_scope(&intent)?; let mut actual = BTreeMap::new(); @@ -602,11 +652,16 @@ impl PostgresMetadataGenerationRepository { } fn require_scope(&self, intent: &GenerationPrepareIntent) -> Result<(), SnapshotError> { - if intent.primary_scope != self.primary_scope()? { + if intent.primary_scope.as_ref() != self.primary_scope()?.as_slice() { return Err(integrity( "metadata seal belongs to another captured primary storage scope", )); } + if intent.graph_domain.stored().unwrap_or(GENERIC_GRAPH_DOMAIN) != self.graph_domain { + return Err(integrity( + "metadata seal belongs to another immutable graph domain", + )); + } Ok(()) } } @@ -617,13 +672,24 @@ fn seal( root: &[u8; 32], bindings: &[u8; 32], scope: &[u8], + graph_domain: Option<&str>, ) -> Result<[u8; 32], SnapshotError> { if scope.is_empty() || scope.len() > MAX_PRIMARY_SCOPE_BYTES { return Err(integrity("invalid metadata generation seal primary scope")); } let id = uuid::Uuid::parse_str(prepare_id).map_err(internal)?; let mut hash = Sha256::new(); - hash.update(b"MST2-METADATA-STORAGE-SEAL-1\0"); + match graph_domain { + None => hash.update(b"MST2-METADATA-STORAGE-SEAL-1\0"), + Some(domain) => { + if ![GENERIC_GRAPH_DOMAIN, "qualified-v1"].contains(&domain) { + return Err(integrity("unknown metadata graph domain")); + } + hash.update(b"MST2-METADATA-STORAGE-SEAL-2\0"); + hash.update((domain.len() as u32).to_be_bytes()); + hash.update(domain.as_bytes()); + } + } hash.update(id.as_bytes()); hash.update(manifest); hash.update(root); @@ -636,6 +702,7 @@ fn seal( async fn allocate_lifetimes( txn: &DatabaseTransaction, plan: &MetadataInstallPlan, + graph_domain: &str, ) -> Result { let pages: Vec<_> = plan .pages @@ -647,9 +714,26 @@ async fn allocate_lifetimes( "INSERT INTO mst2_metadata_lifetime(page_id,node_id,generation,state,metadata_codec,expected_size) SELECT decode(p.page_id,'hex'),'page:sha256:'||p.page_id,1,'RESERVED',$1,p.size FROM jsonb_to_recordset($2::jsonb) AS p(page_id text,size integer) - ON CONFLICT(page_id) DO NOTHING", + ON CONFLICT(page_id,generation) DO NOTHING", [(plan.identity.metadata_codec as i16).into(),encoded.clone().into()], )).await.map_err(internal)?; + txn.execute_raw(statement( + "INSERT INTO mst2_metadata_current(page_id,generation) + SELECT decode(p.page_id,'hex'),1 FROM jsonb_to_recordset($1::jsonb) AS p(page_id text,size integer) + ON CONFLICT(page_id) DO NOTHING", + [encoded.clone().into()], + )).await.map_err(internal)?; + if txn.query_one_raw(statement( + "SELECT p.page_id FROM jsonb_to_recordset($1::jsonb) AS p(page_id text,size integer) + JOIN mst2_metadata_prepare_page m ON m.page_id=decode(p.page_id,'hex') + JOIN mst2_metadata_current c ON c.page_id=m.page_id AND c.generation=m.generation + JOIN mst2_metadata_prepare q ON q.prepare_id=m.prepare_id + WHERE coalesce(q.graph_domain,'generic-v1')<>$2 + AND (q.state='PREPARING' OR (q.state='COMMITTED' AND q.coverage_retired_at IS NULL)) LIMIT 1", + [encoded.clone().into(),graph_domain.into()], + )).await.map_err(internal)?.is_some() { + return Err(unavailable("metadata lifetime is covered by another graph domain")); + } let rows=txn.query_all_raw(statement( "SELECT decode(p.page_id,'hex') AS page_id,p.size,l.generation,l.state,l.metadata_codec,l.expected_size, n.state AS graph_state,n.kind AS graph_kind,n.bytes AS graph_bytes, @@ -658,7 +742,8 @@ async fn allocate_lifetimes( EXISTS(SELECT 1 FROM mst2_retention_gc_op g WHERE g.node_id=l.node_id AND g.operation='REMOVE' AND g.state IN ('PENDING','APPLIED')) AS tombstone FROM jsonb_to_recordset($1::jsonb) AS p(page_id text,size integer) - JOIN mst2_metadata_lifetime l ON l.page_id=decode(p.page_id,'hex') + JOIN mst2_metadata_current c ON c.page_id=decode(p.page_id,'hex') + JOIN mst2_metadata_lifetime l ON l.page_id=c.page_id AND l.generation=c.generation LEFT JOIN mst2_retention_node n ON n.node_id=l.node_id LEFT JOIN mst2_metadata_payload b ON b.page_id=l.page_id ORDER BY l.page_id", [encoded.into()], @@ -748,9 +833,10 @@ async fn check_lifetimes( n.state AS graph_state,n.kind AS graph_kind,n.bytes AS graph_bytes, EXISTS(SELECT 1 FROM mst2_retention_gc_op g WHERE g.node_id=l.node_id AND g.operation='REMOVE' AND g.state IN ('PENDING','APPLIED')) AS tombstone - FROM mst2_metadata_prepare_page p LEFT JOIN mst2_metadata_lifetime l ON l.page_id=p.page_id + FROM mst2_metadata_prepare_page p LEFT JOIN mst2_metadata_lifetime l ON l.page_id=p.page_id AND l.generation=p.generation + LEFT JOIN mst2_metadata_current c ON c.page_id=p.page_id AND c.generation=p.generation LEFT JOIN mst2_retention_node n ON n.node_id=l.node_id WHERE p.prepare_id=$1 - AND (l.page_id IS NULL OR l.generation IS DISTINCT FROM p.generation OR l.metadata_codec<>$2 + AND (l.page_id IS NULL OR c.page_id IS NULL OR l.generation IS DISTINCT FROM p.generation OR l.metadata_codec<>$2 OR l.expected_size<>p.expected_size OR l.state NOT IN ('RESERVED','LIVE') OR ($3='COMMITTED' AND l.state<>'LIVE') OR (l.state='LIVE' AND n.node_id IS NULL) @@ -864,3 +950,6 @@ async fn lock_prepare(txn: &DatabaseTransaction, prepare_id: &str) -> Result<(), #[cfg(test)] #[path = "native_metadata_generation_tests.rs"] mod tests; + +#[path = "native_metadata_history.rs"] +pub mod history; diff --git a/src/jupiter/storage/native_metadata_history.rs b/src/jupiter/storage/native_metadata_history.rs new file mode 100644 index 00000000..e093a4cc --- /dev/null +++ b/src/jupiter/storage/native_metadata_history.rs @@ -0,0 +1,357 @@ +//! Explicit preparation termination. Payload deletion remains forbidden. + +use super::*; + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum MetadataTerminalAction { + Abort, + RetireCoverage, +} + +#[derive(Debug, thiserror::Error)] +pub enum MetadataTerminalError { + #[error(transparent)] + Rejected(#[from] SnapshotError), + #[error("metadata {action:?} commit outcome is unknown for operation {operation_id}")] + CommitUncertain { + operation_id: String, + manifest_digest: [u8; 32], + action: MetadataTerminalAction, + }, +} + +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct MetadataTerminalReceipt { + intent: GenerationPrepareIntent, + action: MetadataTerminalAction, + terminal_at: sea_orm::prelude::DateTimeWithTimeZone, +} + +impl MetadataTerminalReceipt { + pub fn intent(&self) -> &GenerationPrepareIntent { + &self.intent + } + + pub fn action(&self) -> MetadataTerminalAction { + self.action + } +} + +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum MetadataTerminalObservation { + Active, + Terminated(Box), +} + +impl PostgresMetadataGenerationRepository { + pub async fn abort( + &self, + intent: &GenerationPrepareIntent, + ) -> Result { + self.terminate(intent, MetadataTerminalAction::Abort, None) + .await + } + + pub async fn retire_prepare_coverage( + &self, + receipt: &GenerationMetadataReceipt, + ) -> Result { + self.terminate( + &receipt.intent, + MetadataTerminalAction::RetireCoverage, + Some(receipt), + ) + .await + } + + async fn terminate( + &self, + intent: &GenerationPrepareIntent, + action: MetadataTerminalAction, + receipt: Option<&GenerationMetadataReceipt>, + ) -> Result { + self.require_scope(intent)?; + let txn = self.inner.transaction().await?; + let result = self.terminate_in_txn(&txn, intent, action, receipt).await; + match result { + Ok(receipt) => { + txn.commit().await.map_err(|_| uncertain(intent, action))?; + Ok(receipt) + } + Err(error) => { + txn.rollback().await.map_err(internal)?; + Err(error.into()) + } + } + } + + async fn terminate_in_txn( + &self, + txn: &DatabaseTransaction, + intent: &GenerationPrepareIntent, + action: MetadataTerminalAction, + receipt: Option<&GenerationMetadataReceipt>, + ) -> Result { + self.inner.barrier(txn).await?; + lock_prepare(txn, intent.prepare_id()).await?; + let record = self.require_terminal_record(txn, intent).await?; + if let Some(terminal) = terminal_receipt(&record, intent, action)? { + verify_no_prepare_coverage(txn, intent).await?; + return Ok(terminal); + } + let expected = match action { + MetadataTerminalAction::Abort => "PREPARING", + MetadataTerminalAction::RetireCoverage => "COMMITTED", + }; + if record.state != expected { + return Err(SnapshotError::new( + SnapshotErrorCode::Conflict, + "metadata prepare is in another terminal state", + )); + } + let fixed = self.require_fixed_plan(txn, intent).await?; + if action == MetadataTerminalAction::RetireCoverage { + let receipt = receipt + .ok_or_else(|| integrity("coverage retirement requires a definitive receipt"))?; + if receipt.legacy != fixed.stored.receipt()? || receipt.intent != fixed.intent { + return Err(integrity( + "coverage retirement receipt differs from its fixed preparation", + )); + } + check_generation_payloads(txn, &fixed).await?; + verify_graph(txn, &fixed.stored).await?; + } + let roots = mst2_retention_root::Entity::find() + .filter( + mst2_retention_root::Column::RootKey.eq(format!("prepare:{}", intent.prepare_id())), + ) + .limit((MetadataDagLimits::default().nodes + 1) as u64) + .all(txn) + .await + .map_err(internal)?; + let expected_nodes: BTreeSet<_> = fixed.bindings.0.keys().map(node_id).collect(); + if roots + .iter() + .any(|root| root.root_kind != "prepare" || !expected_nodes.contains(&root.node_id)) + { + return Err(integrity( + "terminal prepare coverage differs from its fixed lifetime mappings", + )); + } + if action == MetadataTerminalAction::RetireCoverage + && roots + .iter() + .map(|root| root.node_id.clone()) + .collect::>() + != expected_nodes + { + return Err(unavailable( + "coverage retirement cannot recover missing or transferred prepare roots", + )); + } + txn.execute_raw(statement( + "DELETE FROM mst2_retention_root r USING mst2_metadata_prepare_page p + WHERE p.prepare_id=$1 AND r.root_key='prepare:'||p.prepare_id AND r.root_kind='prepare' + AND r.node_id='page:sha256:'||encode(p.page_id,'hex')", + [intent.prepare_id().into()], + )) + .await + .map_err(internal)?; + let sql=match action { + MetadataTerminalAction::Abort=> + "UPDATE mst2_metadata_prepare SET state='ABORTED',aborted_at=clock_timestamp() + WHERE prepare_id=$1 AND state='PREPARING' AND storage_seal=$2 RETURNING *", + MetadataTerminalAction::RetireCoverage=> + "UPDATE mst2_metadata_prepare SET coverage_retired_at=clock_timestamp() + WHERE prepare_id=$1 AND state='COMMITTED' AND coverage_retired_at IS NULL AND storage_seal=$2 RETURNING *", + }; + let row = txn + .query_one_raw(statement( + sql, + [ + intent.prepare_id().into(), + intent.storage_seal.to_vec().into(), + ], + )) + .await + .map_err(internal)? + .ok_or_else(|| integrity("metadata terminal CAS changed during its barrier"))?; + let column = match action { + MetadataTerminalAction::Abort => "aborted_at", + MetadataTerminalAction::RetireCoverage => "coverage_retired_at", + }; + let terminal_at = row.try_get("", column).map_err(internal)?; + Ok(MetadataTerminalReceipt { + intent: intent.clone(), + action, + terminal_at, + }) + } + + /// A fresh same-primary barrier resolves a lost terminal commit response. + /// An old receipt is replayed without consulting or changing a newer lifetime. + pub async fn inspect_terminal( + &self, + fresh_primary: &DatabaseConnection, + intent: &GenerationPrepareIntent, + action: MetadataTerminalAction, + ) -> Result { + self.require_scope(intent)?; + let txn = fresh_primary + .begin_with_config(Some(IsolationLevel::ReadCommitted), None) + .await + .map_err(|_| uncertain(intent, action))?; + if self.inner.barrier(&txn).await.is_err() { + let _ = txn.rollback().await; + return Err(uncertain(intent, action)); + } + let result = async { + let record = self.require_terminal_record(&txn, intent).await?; + match terminal_receipt(&record, intent, action)? { + Some(receipt) => { + verify_no_prepare_coverage(&txn, intent).await?; + Ok(MetadataTerminalObservation::Terminated(Box::new(receipt))) + } + None => Ok(MetadataTerminalObservation::Active), + } + } + .await; + txn.rollback() + .await + .map_err(|_| uncertain(intent, action))?; + result.map_err(|error: SnapshotError| { + if error.code == SnapshotErrorCode::Internal { + uncertain(intent, action) + } else { + MetadataTerminalError::Rejected(error) + } + }) + } + + async fn require_terminal_record( + &self, + connection: &C, + intent: &GenerationPrepareIntent, + ) -> Result { + self.require_scope(intent)?; + let record = mst2_metadata_prepare::Entity::find() + .filter(mst2_metadata_prepare::Column::OperationId.eq(intent.operation_id())) + .one(connection) + .await + .map_err(internal)? + .ok_or_else(|| unavailable("sealed metadata preparation is missing"))?; + let canonical = record + .canonical_bindings + .as_deref() + .ok_or_else(|| unavailable("legacy preparation has no generation seal"))?; + let plan = MetadataInstallPlan::decode(&record.canonical_plan, &intent.manifest_digest())?; + let bindings = GenerationBindings::decode(canonical, &plan)?; + let actual_digest: [u8; 32] = Sha256::digest(canonical).into(); + let domain = GraphDomain::from_stored(record.graph_domain.as_deref())?; + if record.prepare_id != intent.prepare_id() + || record.manifest_digest.as_slice() != intent.manifest_digest() + || record.metadata_root.as_slice() != intent.metadata_root + || record.bindings_digest.as_deref() != Some(intent.bindings_digest.as_slice()) + || actual_digest != intent.bindings_digest + || record.primary_scope.as_deref() != Some(intent.primary_scope.as_ref()) + || record.storage_seal.as_deref() != Some(intent.storage_seal.as_slice()) + || domain != intent.graph_domain + || bindings.0.get(&intent.metadata_root).map(|pair| pair.0) + != Some(intent.root_generation) + || seal( + intent.prepare_id(), + &intent.manifest_digest(), + &intent.metadata_root, + &intent.bindings_digest, + &intent.primary_scope, + domain.stored(), + )? != intent.storage_seal + { + return Err(integrity( + "terminal metadata action differs from its immutable generation seal", + )); + } + let pages = mst2_metadata_prepare_page::Entity::find() + .filter(mst2_metadata_prepare_page::Column::PrepareId.eq(intent.prepare_id())) + .limit((MetadataDagLimits::default().nodes + 1) as u64) + .all(connection) + .await + .map_err(internal)?; + let mut actual = BTreeMap::new(); + for page in pages { + let id: [u8; 32] = page.page_id.as_slice().try_into().map_err(internal)?; + actual.insert( + id, + ( + page.generation + .ok_or_else(|| integrity("terminal preparation has an unbound mapping"))?, + page.expected_size as u64, + ), + ); + } + if actual != bindings.0 { + return Err(integrity( + "terminal metadata mappings differ from the fixed seal", + )); + } + Ok(record) + } +} + +fn terminal_receipt( + record: &mst2_metadata_prepare::Model, + intent: &GenerationPrepareIntent, + action: MetadataTerminalAction, +) -> Result, SnapshotError> { + let terminal_at = match action { + MetadataTerminalAction::Abort if record.state == "ABORTED" => Some( + record + .aborted_at + .ok_or_else(|| integrity("aborted metadata preparation has no terminal receipt"))?, + ), + MetadataTerminalAction::RetireCoverage if record.state == "COMMITTED" => { + record.coverage_retired_at + } + _ => None, + }; + Ok(terminal_at.map(|terminal_at| MetadataTerminalReceipt { + intent: intent.clone(), + action, + terminal_at, + })) +} + +async fn verify_no_prepare_coverage( + connection: &C, + intent: &GenerationPrepareIntent, +) -> Result<(), SnapshotError> { + if connection + .query_one_raw(statement( + "SELECT node_id FROM mst2_retention_root WHERE root_key='prepare:'||$1 LIMIT 1", + [intent.prepare_id().into()], + )) + .await + .map_err(internal)? + .is_some() + { + return Err(integrity( + "terminal metadata preparation still has prepare coverage", + )); + } + Ok(()) +} + +fn uncertain( + intent: &GenerationPrepareIntent, + action: MetadataTerminalAction, +) -> MetadataTerminalError { + MetadataTerminalError::CommitUncertain { + operation_id: intent.operation_id().into(), + manifest_digest: intent.manifest_digest(), + action, + } +} + +#[cfg(test)] +#[path = "native_metadata_history_tests.rs"] +mod tests; diff --git a/src/jupiter/storage/native_metadata_history_tests.rs b/src/jupiter/storage/native_metadata_history_tests.rs new file mode 100644 index 00000000..250d99a6 --- /dev/null +++ b/src/jupiter/storage/native_metadata_history_tests.rs @@ -0,0 +1,634 @@ +use sea_orm::{Database, PaginatorTrait}; +use sea_orm_migration::MigratorTrait; + +use super::*; +use crate::{ + callisto::{mst2_metadata_current, mst2_metadata_lifetime}, + jupiter::{ + migration::Migrator, + tests::{TestSchemaGuard, test_db_config}, + }, +}; + +async fn fixture() -> (DatabaseConnection, DatabaseConnection, TestSchemaGuard) { + let temp = tempfile::tempdir().unwrap(); + let (config, schema) = test_db_config(temp.path()).await; + let first = Database::connect(config.db_url.clone()).await.unwrap(); + Migrator::up(&first, None).await.unwrap(); + let second = Database::connect(config.db_url).await.unwrap(); + (first, second, schema) +} + +fn prepared() -> PreparedNativeMetadataRetention { + let entries = [mst2_codec::metapage::Entry::file( + mst2_codec::metapage::EntryKind::Regular, + b"file", + 3, + [42; 32], + )]; + let child = Page::build(&entries).unwrap(); + let root_entries = [mst2_codec::metapage::Entry::dir(b"one", page_id(&child))]; + let root = Page::build(&root_entries).unwrap(); + let mut builder = crate::ceres::snapshot::retention_dag::MetadataDagBuilder::new( + MetadataDagLimits::default(), + ); + builder.add_directory(&child, &entries).unwrap(); + builder.add_directory(&root, &root_entries).unwrap(); + PreparedNativeMetadataRetention::test_installation( + std::sync::Arc::new(builder.finish(page_id(&root)).unwrap()), + "/", + ) +} + +async fn scalar(db: &DatabaseConnection, sql: &str) -> i64 { + db.query_one_raw(statement(sql, [])) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap() +} + +fn rejected(error: MetadataTerminalError) -> SnapshotErrorCode { + match error { + MetadataTerminalError::Rejected(error) => error.code, + _ => panic!("expected definite terminal rejection"), + } +} + +#[tokio::test] +async fn history_keeps_old_composite_fk_bindings_while_current_is_an_independent_watermark() { + let (first, second, _schema) = fixture().await; + let repository = PostgresMetadataGenerationRepository::new(first.clone()) + .await + .unwrap(); + let pages = prepared(); + let intent = repository.begin_intent("history", &pages).await.unwrap(); + repository + .install_pages(&intent, pages.dag().payloads()) + .await + .unwrap(); + let page = &pages.dag().payloads()[0]; + second.execute_raw(statement( + "INSERT INTO mst2_metadata_lifetime(page_id,node_id,generation,state,metadata_codec,expected_size) + VALUES($1,$2,2,'RESERVED',1,$3)", + [page.id.to_vec().into(),node_id(&page.id).into(),(page.size as i32).into()], + )).await.unwrap(); + assert!( + mst2_metadata_lifetime::Entity::find_by_id((page.id.to_vec(), 1)) + .one(&second) + .await + .unwrap() + .is_some() + ); + assert!( + mst2_metadata_lifetime::Entity::find_by_id((page.id.to_vec(), 2)) + .one(&second) + .await + .unwrap() + .is_some() + ); + assert_eq!( + mst2_metadata_current::Entity::find_by_id(page.id.to_vec()) + .one(&second) + .await + .unwrap() + .unwrap() + .generation, + 1 + ); + assert!( + second + .execute_raw(statement( + "UPDATE mst2_metadata_current SET generation=2 WHERE page_id=$1", + [page.id.to_vec().into()] + )) + .await + .is_err() + ); + assert_eq!(scalar(&first,"SELECT count(*) FROM mst2_metadata_prepare_page p JOIN mst2_metadata_lifetime l ON l.page_id=p.page_id AND l.generation=p.generation WHERE p.generation=1").await,2); + assert_eq!(scalar(&first,"SELECT count(*) FROM mst2_metadata_payload b JOIN mst2_metadata_lifetime l ON l.page_id=b.page_id AND l.generation=b.generation WHERE b.generation=1").await,2); + repository.finalize(&intent).await.unwrap(); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_lifetime").await, + 3 + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_current").await, + 2 + ); +} + +#[tokio::test] +async fn history_abort_fences_a_late_installer_waiting_on_the_real_retention_barrier() { + let (first, second, _schema) = fixture().await; + let repository = PostgresMetadataGenerationRepository::new(first.clone()) + .await + .unwrap(); + let other = PostgresMetadataGenerationRepository::new(second.clone()) + .await + .unwrap(); + let pages = prepared(); + let intent = repository + .begin_intent("abort-late-install", &pages) + .await + .unwrap(); + repository + .install_page(&intent, &pages.dag().payloads()[0]) + .await + .unwrap(); + let held = repository.inner.transaction().await.unwrap(); + repository.inner.barrier(&held).await.unwrap(); + let late_intent = intent.clone(); + let late_pages = pages.dag().payloads().to_vec(); + let late = tokio::spawn(async move { other.install_pages(&late_intent, &late_pages).await }); + let deadline = tokio::time::Instant::now() + Duration::from_secs(4); + loop { + let waiting=held.query_one_raw(statement( + "SELECT EXISTS(SELECT 1 FROM pg_locks WHERE locktype='advisory' AND NOT granted + AND classid=$1::integer::oid AND objid=hashtext(current_schema())::oid + AND database=(SELECT oid FROM pg_database WHERE datname=current_database())) AS waiting", + [RETENTION_LOCK_KEY.into()], + )).await.unwrap().unwrap().try_get::("","waiting").unwrap(); + if waiting { + break; + } + assert!( + tokio::time::Instant::now() < deadline, + "late installer must wait on the captured schema barrier" + ); + tokio::task::yield_now().await; + } + let terminal = repository + .terminate_in_txn(&held, &intent, MetadataTerminalAction::Abort, None) + .await + .unwrap(); + held.commit().await.unwrap(); + assert!(matches!( + late.await.unwrap(), + Err(MetadataInstallError::Rejected(SnapshotError { + code: SnapshotErrorCode::IntegrityError, + .. + })) + )); + assert_eq!(repository.abort(&intent).await.unwrap(), terminal); + assert_eq!( + repository + .inspect_terminal(&second, &intent, MetadataTerminalAction::Abort) + .await + .unwrap(), + MetadataTerminalObservation::Terminated(Box::new(terminal)) + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_payload").await, + 1 + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_prepare_page").await, + 2 + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_retention_node").await, + 0 + ); + assert_eq!(scalar(&first,"SELECT count(*) FROM mst2_metadata_prepare WHERE state='ABORTED' AND aborted_at IS NOT NULL").await,1); +} + +#[tokio::test] +async fn history_abort_after_observation_rejects_old_finalize_and_preserves_other_prepares() { + let (first, second, _schema) = fixture().await; + let repository = PostgresMetadataGenerationRepository::new(first.clone()) + .await + .unwrap(); + let other = PostgresMetadataGenerationRepository::new(second.clone()) + .await + .unwrap(); + let pages = prepared(); + let owner = repository + .begin_intent("committed-owner", &pages) + .await + .unwrap(); + repository + .install_pages(&owner, pages.dag().payloads()) + .await + .unwrap(); + let owner_receipt = repository.finalize(&owner).await.unwrap(); + let abandoned = other + .begin_intent("abandoned-shared", &pages) + .await + .unwrap(); + let observation = other.observe_installed_dag(&abandoned).await.unwrap(); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_retention_root").await, + 4 + ); + other.abort(&abandoned).await.unwrap(); + assert!(matches!( + other.finalize_observation(&abandoned, &observation).await, + Err(MetadataInstallError::Rejected(_)) + )); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_retention_root").await, + 2 + ); + assert_eq!(repository.finalize(&owner).await.unwrap(), owner_receipt); + assert_eq!( + rejected(repository.abort(&owner).await.unwrap_err()), + SnapshotErrorCode::Conflict + ); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_lifetime WHERE state='LIVE' AND generation=1" + ) + .await, + 2 + ); + assert_eq!( + scalar(&second, "SELECT count(*) FROM mst2_metadata_payload").await, + 2 + ); +} + +#[tokio::test] +async fn history_terminal_outer_rollback_keeps_preparing_protection_and_durable_recovery_active() { + let (first, second, _schema) = fixture().await; + let repository = PostgresMetadataGenerationRepository::new(first.clone()) + .await + .unwrap(); + let pages = prepared(); + let intent = repository + .begin_intent("abort-rollback", &pages) + .await + .unwrap(); + let held = repository.inner.transaction().await.unwrap(); + repository + .terminate_in_txn(&held, &intent, MetadataTerminalAction::Abort, None) + .await + .unwrap(); + held.rollback().await.unwrap(); + assert_eq!( + repository + .inspect_terminal(&second, &intent, MetadataTerminalAction::Abort) + .await + .unwrap(), + MetadataTerminalObservation::Active + ); + assert_eq!(scalar(&first,"SELECT count(*) FROM mst2_metadata_prepare WHERE state='PREPARING' AND aborted_at IS NULL").await,1); + repository + .install_pages(&intent, pages.dag().payloads()) + .await + .unwrap(); + repository.finalize(&intent).await.unwrap(); +} + +#[tokio::test] +async fn history_retirement_replays_original_receipt_and_never_recreates_lost_prepare_coverage() { + let (first, second, _schema) = fixture().await; + let repository = PostgresMetadataGenerationRepository::new(first.clone()) + .await + .unwrap(); + let pages = prepared(); + let intent = repository.begin_intent("retire", &pages).await.unwrap(); + repository + .install_pages(&intent, pages.dag().payloads()) + .await + .unwrap(); + let receipt = repository.finalize(&intent).await.unwrap(); + let terminal = repository.retire_prepare_coverage(&receipt).await.unwrap(); + assert_eq!( + repository.retire_prepare_coverage(&receipt).await.unwrap(), + terminal + ); + assert_eq!( + repository + .inspect_terminal(&second, &intent, MetadataTerminalAction::RetireCoverage) + .await + .unwrap(), + MetadataTerminalObservation::Terminated(Box::new(terminal)) + ); + assert!(matches!( + repository + .install_pages(&intent, pages.dag().payloads()) + .await, + Err(MetadataInstallError::Rejected(SnapshotError { + code: SnapshotErrorCode::ObjectUnavailable, + .. + })) + )); + assert!(repository.finalize(&intent).await.is_err()); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_retention_root").await, + 0 + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_payload").await, + 2 + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_prepare_page").await, + 2 + ); + assert_eq!(scalar(&first,"SELECT count(*) FROM mst2_metadata_prepare WHERE state='COMMITTED' AND coverage_retired_at IS NOT NULL").await,1); + assert!( + second + .execute_unprepared("UPDATE mst2_metadata_prepare SET coverage_retired_at=NULL") + .await + .is_err() + ); +} + +#[tokio::test] +async fn history_committed_missing_payload_rejects_a_late_installer_without_repairing_corruption() { + let (first, second, _schema) = fixture().await; + let repository = PostgresMetadataGenerationRepository::new(first.clone()) + .await + .unwrap(); + let pages = prepared(); + let intent = repository + .begin_intent("committed-missing", &pages) + .await + .unwrap(); + repository + .install_page(&intent, &pages.dag().payloads()[0]) + .await + .unwrap(); + // Persist a corrupt COMMITTED fixture without deleting bytes or disabling + // any guard. A late installer must refuse to fill its missing page. + let txn = second.begin().await.unwrap(); + repository.inner.barrier(&txn).await.unwrap(); + PostgresRetentionRepository::retain_group_in_txn( + &txn, + pages.dag().nodes(), + pages.dag().edges(), + &[RetentionRoot::Prepare(intent.prepare_id().into())], + ) + .await + .unwrap(); + txn.execute_unprepared("UPDATE mst2_metadata_lifetime SET state='LIVE'") + .await + .unwrap(); + txn.execute_raw(statement("UPDATE mst2_metadata_prepare SET state='COMMITTED',committed_at=clock_timestamp() WHERE prepare_id=$1",[intent.prepare_id().into()])).await.unwrap(); + txn.commit().await.unwrap(); + assert!(matches!( + repository + .install_pages(&intent, pages.dag().payloads()) + .await, + Err(MetadataInstallError::Rejected(SnapshotError { + code: SnapshotErrorCode::ObjectUnavailable, + .. + })) + )); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_payload").await, + 1 + ); + assert_eq!( + scalar(&second, "SELECT count(*) FROM mst2_retention_root").await, + 2 + ); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_prepare WHERE state='COMMITTED'" + ) + .await, + 1 + ); +} + +#[tokio::test] +async fn history_graph_domain_is_durable_sealed_and_cannot_be_changed_by_a_new_repository() { + let (first, second, _schema) = fixture().await; + let repository = PostgresMetadataGenerationRepository::new(first.clone()) + .await + .unwrap(); + let mut qualified = PostgresMetadataGenerationRepository::new(second.clone()) + .await + .unwrap(); + qualified.graph_domain = "qualified-v1"; + let pages = prepared(); + let intent = repository + .begin_intent("generic-domain", &pages) + .await + .unwrap(); + assert_eq!(intent.graph_domain, GraphDomain::Generic); + assert!(matches!( + qualified.begin_intent("generic-domain", &pages).await, + Err(MetadataInstallError::Rejected(SnapshotError { + code: SnapshotErrorCode::IntegrityError, + .. + })) + )); + assert!(matches!( + qualified.begin_intent("cross-domain-page", &pages).await, + Err(MetadataInstallError::Rejected(SnapshotError { + code: SnapshotErrorCode::ObjectUnavailable, + .. + })) + )); + assert!(matches!( + qualified + .install_pages(&intent, pages.dag().payloads()) + .await, + Err(MetadataInstallError::Rejected(SnapshotError { + code: SnapshotErrorCode::IntegrityError, + .. + })) + )); + assert!( + second + .execute_unprepared("UPDATE mst2_metadata_prepare SET graph_domain='qualified-v1'") + .await + .is_err() + ); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_prepare WHERE graph_domain='generic-v1'" + ) + .await, + 1 + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_payload").await, + 0 + ); + assert_eq!( + scalar(&second, "SELECT count(*) FROM mst2_metadata_current").await, + 2 + ); +} + +#[tokio::test] +async fn history_terminal_wrong_primary_is_unknown_and_current_delete_payload_update_stay_forbidden() + { + let (first, second, _schema) = fixture().await; + let (other, _other_connection, _other_schema) = fixture().await; + let repository = PostgresMetadataGenerationRepository::new(first.clone()) + .await + .unwrap(); + let pages = prepared(); + let intent = repository + .begin_intent("terminal-primary", &pages) + .await + .unwrap(); + repository + .install_page(&intent, &pages.dag().payloads()[0]) + .await + .unwrap(); + assert!(matches!( + repository + .inspect_terminal(&other, &intent, MetadataTerminalAction::Abort) + .await, + Err(MetadataTerminalError::CommitUncertain { .. }) + )); + assert_eq!( + repository + .inspect_terminal(&second, &intent, MetadataTerminalAction::Abort) + .await + .unwrap(), + MetadataTerminalObservation::Active + ); + for sql in [ + "DELETE FROM mst2_metadata_current", + "DELETE FROM mst2_metadata_lifetime", + "DELETE FROM mst2_metadata_payload", + "UPDATE mst2_metadata_payload SET payload=payload", + "DELETE FROM mst2_metadata_storage_scope", + ] { + assert!(second.execute_unprepared(sql).await.is_err(), "{sql}"); + } + repository.abort(&intent).await.unwrap(); + assert!( + second + .execute_unprepared( + "UPDATE mst2_metadata_prepare SET state='PREPARING',aborted_at=NULL" + ) + .await + .is_err() + ); + assert_eq!( + mst2_metadata_current::Entity::find() + .count(&first) + .await + .unwrap(), + 2 + ); +} + +#[tokio::test] +async fn history_migration_preserves_preexisting_g1_null_domain_seal_and_composite_mappings() { + let temp = tempfile::tempdir().unwrap(); + let (config, _schema) = test_db_config(temp.path()).await; + let first = Database::connect(config.db_url.clone()).await.unwrap(); + let second = Database::connect(config.db_url).await.unwrap(); + let at = Migrator::migrations() + .iter() + .position(|migration| { + migration.name() == "m20261007_000300_add_mst2_metadata_lifetime_history" + }) + .unwrap(); + Migrator::up(&first, Some(at.try_into().unwrap())) + .await + .unwrap(); + let inner = PostgresMetadataInstallRepository::new(first.clone()) + .await + .unwrap(); + let pages = prepared(); + let plan = pages.install_plan().unwrap(); + let manifest = plan.digest().unwrap(); + let bindings = GenerationBindings( + plan.pages + .iter() + .map(|(page, size)| (*page, (1, *size))) + .collect(), + ); + let canonical = bindings.encode().unwrap(); + let digest: [u8; 32] = Sha256::digest(&canonical).into(); + let scope = serde_json::to_vec(&( + &inner.storage_scope.storage_uuid, + &inner.storage_scope.database, + inner.storage_scope.database_oid, + &inner.storage_scope.schema, + inner.storage_scope.schema_oid, + &inner.storage_scope.server_address, + inner.storage_scope.server_port, + )) + .unwrap(); + let prepare_id = uuid::Uuid::new_v4().to_string(); + let old_seal = seal(&prepare_id, &manifest, &plan.root, &digest, &scope, None).unwrap(); + let txn = first.begin().await.unwrap(); + inner.barrier(&txn).await.unwrap(); + for page in pages.dag().payloads() { + txn.execute_raw(statement("INSERT INTO mst2_metadata_lifetime(page_id,node_id,generation,state,metadata_codec,expected_size) VALUES($1,$2,1,'RESERVED',1,$3)", + [page.id.to_vec().into(),node_id(&page.id).into(),(page.size as i32).into()])).await.unwrap(); + txn.execute_raw(statement("INSERT INTO mst2_metadata_payload(page_id,generation,metadata_codec,byte_size,payload) VALUES($1,1,1,$2,$3)", + [page.id.to_vec().into(),(page.size as i32).into(),page.bytes.clone().into()])).await.unwrap(); + } + let identity = &plan.identity; + txn.execute_raw(statement( + "INSERT INTO mst2_metadata_prepare(prepare_id,operation_id,manifest_digest,canonical_plan, + source_domain,tagged_root_tree_oid,scope,schema_version,metadata_codec,materialization_policy, + fs_semantics,access_projection,verification_revision,projection_revision,metadata_root, + node_count,edge_count,total_bytes,state,canonical_bindings,bindings_digest,primary_scope,storage_seal) + VALUES($1,'g1-before-history',$2,$3,$4,$5,$6,$7,$8,$9,$10,$11,$12,$13,$14,$15,$16,$17,'PREPARING',$18,$19,$20,$21)", + [prepare_id.clone().into(),manifest.to_vec().into(),plan.encode().unwrap().into(),identity.source_domain.clone().into(), + identity.tagged_root_tree_oid.clone().into(),identity.scope.clone().into(),(identity.schema_version as i16).into(), + (identity.metadata_codec as i16).into(),(identity.materialization_policy as i16).into(),(identity.fs_semantics as i16).into(), + (identity.access_projection as i16).into(),identity.verification_revision.into(),(identity.projection_revision as i16).into(), + plan.root.to_vec().into(),(plan.pages.len() as i32).into(),(plan.edges.len() as i32).into(),(plan.total_bytes as i64).into(), + canonical.into(),digest.to_vec().into(),scope.into(),old_seal.to_vec().into()], + )).await.unwrap(); + for (page, size) in &plan.pages { + txn.execute_raw(statement("INSERT INTO mst2_metadata_prepare_page(prepare_id,page_id,generation,expected_size) VALUES($1,$2,1,$3)", + [prepare_id.clone().into(),page.to_vec().into(),(*size as i32).into()])).await.unwrap(); + } + PostgresRetentionRepository::retain_group_in_txn( + &txn, + pages.dag().nodes(), + pages.dag().edges(), + &[RetentionRoot::Prepare(prepare_id.clone())], + ) + .await + .unwrap(); + txn.execute_unprepared("UPDATE mst2_metadata_lifetime SET state='LIVE'") + .await + .unwrap(); + txn.execute_raw(statement("UPDATE mst2_metadata_prepare SET state='COMMITTED',committed_at=clock_timestamp() WHERE prepare_id=$1",[prepare_id.clone().into()])).await.unwrap(); + txn.commit().await.unwrap(); + Migrator::up(&first, Some(1)).await.unwrap(); + let repository = PostgresMetadataGenerationRepository::new(second.clone()) + .await + .unwrap(); + let restored = repository + .begin_intent("g1-before-history", &pages) + .await + .unwrap(); + assert_eq!(restored.graph_domain, GraphDomain::LegacyGeneric); + assert_eq!(restored.storage_seal, old_seal); + assert_eq!(restored.prepare_id(), prepare_id); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_current WHERE generation=1" + ) + .await, + 2 + ); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_prepare WHERE graph_domain IS NULL" + ) + .await, + 1 + ); + let receipt = repository.finalize(&restored).await.unwrap(); + repository.retire_prepare_coverage(&receipt).await.unwrap(); + assert_eq!(scalar(&first,"SELECT count(*) FROM mst2_metadata_prepare_page p JOIN mst2_metadata_lifetime l ON l.page_id=p.page_id AND l.generation=p.generation").await,2); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_payload").await, + 2 + ); +} From 4ec499d83e5c908c94bd723b4e9cab95f1369980 Mon Sep 17 00:00:00 2001 From: Xiaoyang Han Date: Wed, 7 Oct 2026 20:31:20 +0800 Subject: [PATCH 09/50] fix(mst2): timestamp retained roots during native batch acquisition Fix the two native batch root inserts to supply PostgreSQL now() for the existing required created_at column. Preserve graph locks, identities, ON CONFLICT semantics and all HTTP/generation assertions. Keep the reproducible bootstrap fixture consistent with the validated default import directory. Native run 37615218353 exposed 29 missing-timestamp failures and three invalid-fixture failures; source review, nightly formatting and locked offline metadata passed, native retest pending. --- src/commands/service/init_commit_time_tests.rs | 2 +- src/jupiter/storage/mst2_retention.rs | 4 ++-- src/jupiter/storage/native_metadata_generations.rs | 4 ++-- 3 files changed, 5 insertions(+), 5 deletions(-) diff --git a/src/commands/service/init_commit_time_tests.rs b/src/commands/service/init_commit_time_tests.rs index 74b2157e..543fca98 100644 --- a/src/commands/service/init_commit_time_tests.rs +++ b/src/commands/service/init_commit_time_tests.rs @@ -113,7 +113,7 @@ async fn fixture() -> Fixture { let (database, schema) = test_db_config(directory.path()).await; let mut config = isolated_config(directory.path().join("config")); config.database = database; - config.monorepo.root_dirs = vec!["bench".to_string()]; + config.monorepo.root_dirs = vec!["bench".to_string(), "third-party".to_string()]; config.redis.url = "redis://127.0.0.1:1".to_string(); let args = cli() .try_get_matches_from(["init", "--yes", "--commit-time", "1790000000"]) diff --git a/src/jupiter/storage/mst2_retention.rs b/src/jupiter/storage/mst2_retention.rs index b336b7a1..3931fe86 100644 --- a/src/jupiter/storage/mst2_retention.rs +++ b/src/jupiter/storage/mst2_retention.rs @@ -167,8 +167,8 @@ impl PostgresRetentionRepository { return Err(integrity("retention root is bound to another DAG or kind")); } savepoint.execute_raw(statement( - "INSERT INTO mst2_retention_root(node_id,root_key,root_kind) - SELECT $1,p.key,p.kind FROM jsonb_to_recordset($2::jsonb) AS p(key text,kind text) + "INSERT INTO mst2_retention_root(node_id,root_key,root_kind,created_at) + SELECT $1,p.key,p.kind,now() FROM jsonb_to_recordset($2::jsonb) AS p(key text,kind text) ON CONFLICT(node_id,root_key) DO NOTHING",[node_id.into(),encoded.into()], )).await.map_err(internal)?; Ok(()) diff --git a/src/jupiter/storage/native_metadata_generations.rs b/src/jupiter/storage/native_metadata_generations.rs index 29189469..9ea9f766 100644 --- a/src/jupiter/storage/native_metadata_generations.rs +++ b/src/jupiter/storage/native_metadata_generations.rs @@ -219,8 +219,8 @@ impl PostgresMetadataGenerationRepository { // Reuse only already LIVE graph rows. RESERVED pages without a graph // are protected by the durable mappings, without inventing a DAG. txn.execute_raw(statement( - "INSERT INTO mst2_retention_root(node_id,root_key,root_kind) - SELECT l.node_id,'prepare:'||p.prepare_id,'prepare' + "INSERT INTO mst2_retention_root(node_id,root_key,root_kind,created_at) + SELECT l.node_id,'prepare:'||p.prepare_id,'prepare',now() FROM mst2_metadata_prepare_page p JOIN mst2_metadata_lifetime l ON l.page_id=p.page_id AND l.generation=p.generation JOIN mst2_retention_node n ON n.node_id=l.node_id AND n.state='LIVE' From 7afd1be2b125f77d7c6f4cfed15ff8f84ed6c44f Mon Sep 17 00:00:00 2001 From: Xiaoyang Han Date: Wed, 7 Oct 2026 20:45:27 +0800 Subject: [PATCH 10/50] feat(mst2): collect exact qualified metadata lifetimes atomically (#60) Add permanent lifetime/domain fences, a generation-qualified graph, bounded discovery and persisted exact GC operations. Physical payload removal, edges/counters, immutable history and APPLIED receipt commit together; fresh preparation requires exact prior removal proof and protects against ABA. Preserve current v3 serving, upstream updates and native timestamp/bootstrap fixes. Nineteen real PostgreSQL regressions added; final frozen source reviewed and formatting/offline metadata passed, native validation and performance pending. Qualified production sessions, collector and generic-to-qualified adoption remain closed. --- src/api/router/snapshot_content_tests.rs | 3 + .../snapshot_generation_history_fixture.rs | 1 + .../snapshot_generation_qualified_fixture.rs | 56 + src/callisto/mst2_metadata_lifetime.rs | 1 + ...7_000400_add_mst2_qualified_metadata_gc.rs | 22 + ...m20261007_000400_qualified_metadata_gc.sql | 639 +++++++++ src/jupiter/migration/mod.rs | 12 +- .../storage/native_metadata_generations.rs | 54 +- .../storage/native_metadata_history.rs | 179 ++- .../storage/native_metadata_qualified.rs | 936 ++++++++++++ .../native_metadata_qualified_tests.rs | 1261 +++++++++++++++++ 11 files changed, 3123 insertions(+), 41 deletions(-) create mode 100644 src/api/router/snapshot_generation_qualified_fixture.rs create mode 100644 src/jupiter/migration/m20261007_000400_add_mst2_qualified_metadata_gc.rs create mode 100644 src/jupiter/migration/m20261007_000400_qualified_metadata_gc.sql create mode 100644 src/jupiter/storage/native_metadata_qualified.rs create mode 100644 src/jupiter/storage/native_metadata_qualified_tests.rs diff --git a/src/api/router/snapshot_content_tests.rs b/src/api/router/snapshot_content_tests.rs index 5c85ffbd..79bc47ef 100644 --- a/src/api/router/snapshot_content_tests.rs +++ b/src/api/router/snapshot_content_tests.rs @@ -88,6 +88,9 @@ mod generation_upgrade; #[path = "snapshot_generation_history_fixture.rs"] mod generation_history_fixture; +#[path = "snapshot_generation_qualified_fixture.rs"] +mod generation_qualified_fixture; + #[derive(Default)] struct ReadCounts { whole: AtomicUsize, diff --git a/src/api/router/snapshot_generation_history_fixture.rs b/src/api/router/snapshot_generation_history_fixture.rs index 9a694424..0f6e7005 100644 --- a/src/api/router/snapshot_generation_history_fixture.rs +++ b/src/api/router/snapshot_generation_history_fixture.rs @@ -1,6 +1,7 @@ use sea_orm::{ConnectionTrait, DatabaseConnection, DbBackend, Statement}; pub(super) async fn restore_empty_g1_schema(db: &DatabaseConnection) { + super::generation_qualified_fixture::restore_empty_history_schema(db).await; let count = db .query_one_raw(Statement::from_string( DbBackend::Postgres, diff --git a/src/api/router/snapshot_generation_qualified_fixture.rs b/src/api/router/snapshot_generation_qualified_fixture.rs new file mode 100644 index 00000000..e7020864 --- /dev/null +++ b/src/api/router/snapshot_generation_qualified_fixture.rs @@ -0,0 +1,56 @@ +use sea_orm::{ConnectionTrait, DatabaseConnection, DbBackend, Statement}; + +pub(super) async fn restore_empty_history_schema(db: &DatabaseConnection) { + let n:i64=db.query_one_raw(Statement::from_string(DbBackend::Postgres, + "SELECT (SELECT count(*) FROM mst2_metadata_current)+(SELECT count(*) FROM mst2_metadata_graph_node) + +(SELECT count(*) FROM mst2_metadata_graph_root)+(SELECT count(*) FROM mst2_metadata_graph_edge) + +(SELECT count(*) FROM mst2_metadata_gc_op) AS n")) + .await.unwrap().unwrap().try_get("","n").unwrap(); + assert_eq!( + n, 0, + "upgrade fixture has no qualified or current incarnation to erase" + ); + db.execute_unprepared( + "DO $$ DECLARE t text; tr text; BEGIN + FOREACH t IN ARRAY ARRAY['mst2_metadata_lifetime','mst2_metadata_current','mst2_metadata_payload', + 'mst2_metadata_prepare','mst2_metadata_prepare_page','mst2_retention_node','mst2_retention_edge', + 'mst2_retention_root','mst2_retention_gc_op','mst2_snapshot_context','mst2_snapshot_lease'] LOOP + FOREACH tr IN ARRAY ARRAY['mst2_metadata_statement_barrier','mst2_metadata_generic_domain_guard', + 'mst2_metadata_session_domain_guard','mst2_metadata_qualified_mapping_guard', + 'mst2_metadata_lifetime_insert_guard','mst2_metadata_lifetime_removed', + 'mst2_metadata_current_insert_guard','mst2_metadata_current_protected', + 'mst2_metadata_payload_fenced','mst2_metadata_payload_removed'] LOOP + EXECUTE format('DROP TRIGGER IF EXISTS %I ON %I',tr,t); + END LOOP; + END LOOP; + END $$; + DROP TABLE mst2_metadata_graph_root,mst2_metadata_graph_edge,mst2_metadata_graph_node,mst2_metadata_gc_op; + ALTER TABLE mst2_metadata_lifetime DROP COLUMN graph_domain; + ALTER TABLE mst2_metadata_prepare_page DROP CONSTRAINT mst2_metadata_prepare_page_prepare_id_page_id_generation_key; + ALTER TABLE mst2_metadata_prepare DROP CONSTRAINT mst2_metadata_prepare_prepare_id_storage_seal_key; + DROP INDEX idx_mst2_snapshot_context_metadata_root; + DO $$ DECLARE function_row record; BEGIN + FOR function_row IN SELECT proc.oid::regprocedure::text AS signature FROM pg_proc proc JOIN pg_namespace n ON n.oid=proc.pronamespace + WHERE n.nspname=current_schema() AND proc.proname LIKE 'mst2_metadata_%' + AND proc.proname NOT IN ('mst2_metadata_lifetime_guard','mst2_metadata_current_guard', + 'mst2_metadata_generation_seal_guard','mst2_metadata_generation_mapping_guard','mst2_metadata_payload_immutable') LOOP + EXECUTE 'DROP FUNCTION '||function_row.signature; + END LOOP; + END $$; + CREATE OR REPLACE FUNCTION mst2_metadata_current_guard() RETURNS trigger LANGUAGE plpgsql AS $$ + BEGIN RAISE EXCEPTION 'metadata current watermark requires generation-fenced collection'; END $$; + CREATE OR REPLACE FUNCTION mst2_metadata_lifetime_guard() RETURNS trigger LANGUAGE plpgsql AS $$ + BEGIN + IF TG_OP='DELETE' THEN RAISE EXCEPTION 'metadata lifetime watermark cannot be deleted'; END IF; + IF NEW.page_id IS DISTINCT FROM OLD.page_id OR NEW.node_id IS DISTINCT FROM OLD.node_id + OR NEW.generation IS DISTINCT FROM OLD.generation OR NEW.metadata_codec IS DISTINCT FROM OLD.metadata_codec + OR NEW.expected_size IS DISTINCT FROM OLD.expected_size THEN RAISE EXCEPTION 'metadata lifetime identity is immutable'; END IF; + IF NEW.state IS DISTINCT FROM OLD.state AND NOT (OLD.state='RESERVED' AND NEW.state='LIVE') THEN + RAISE EXCEPTION 'metadata lifetime transition requires generation collector'; END IF; + RETURN NEW; + END $$; + CREATE TRIGGER mst2_metadata_payload_immutable BEFORE UPDATE OR DELETE ON mst2_metadata_payload + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_payload_immutable(); + DELETE FROM seaql_migrations WHERE version='m20261007_000400_add_mst2_qualified_metadata_gc'" + ).await.unwrap(); +} diff --git a/src/callisto/mst2_metadata_lifetime.rs b/src/callisto/mst2_metadata_lifetime.rs index d9bc99da..ba0fa4bc 100644 --- a/src/callisto/mst2_metadata_lifetime.rs +++ b/src/callisto/mst2_metadata_lifetime.rs @@ -11,6 +11,7 @@ pub struct Model { pub state: String, pub metadata_codec: i16, pub expected_size: i32, + pub graph_domain: String, } #[derive(Copy, Clone, Debug, EnumIter, DeriveRelation)] diff --git a/src/jupiter/migration/m20261007_000400_add_mst2_qualified_metadata_gc.rs b/src/jupiter/migration/m20261007_000400_add_mst2_qualified_metadata_gc.rs new file mode 100644 index 00000000..0faacb16 --- /dev/null +++ b/src/jupiter/migration/m20261007_000400_add_mst2_qualified_metadata_gc.rs @@ -0,0 +1,22 @@ +//! Exact-incarnation metadata graph and atomic, replayable payload collection. + +use mst2_codec::metapage::{HEADER_LEN, PAGE_MAX_BYTES}; +use sea_orm_migration::prelude::*; + +#[derive(DeriveMigrationName)] +pub struct Migration; + +#[async_trait::async_trait] +impl MigrationTrait for Migration { + async fn up(&self, manager: &SchemaManager) -> Result<(), DbErr> { + let sql = include_str!("m20261007_000400_qualified_metadata_gc.sql") + .replace("$HEADER_LEN$", &HEADER_LEN.to_string()) + .replace("$PAGE_MAX_BYTES$", &PAGE_MAX_BYTES.to_string()); + manager.get_connection().execute_unprepared(&sql).await?; + Ok(()) + } + + async fn down(&self, _manager: &SchemaManager) -> Result<(), DbErr> { + Ok(()) + } +} diff --git a/src/jupiter/migration/m20261007_000400_qualified_metadata_gc.sql b/src/jupiter/migration/m20261007_000400_qualified_metadata_gc.sql new file mode 100644 index 00000000..a2a84cf0 --- /dev/null +++ b/src/jupiter/migration/m20261007_000400_qualified_metadata_gc.sql @@ -0,0 +1,639 @@ +LOCK TABLE mst2_metadata_lifetime,mst2_metadata_current,mst2_metadata_payload, + mst2_metadata_prepare,mst2_metadata_prepare_page,mst2_retention_node, + mst2_retention_edge,mst2_retention_root,mst2_retention_gc_op, + mst2_snapshot_context,mst2_snapshot_lease IN ACCESS EXCLUSIVE MODE; + +ALTER TABLE mst2_metadata_lifetime ADD COLUMN graph_domain text NOT NULL DEFAULT 'generic-v1' + CHECK (graph_domain IN ('generic-v1','qualified-v1')); +ALTER TABLE mst2_metadata_prepare_page ADD UNIQUE(prepare_id,page_id,generation); +ALTER TABLE mst2_metadata_prepare ADD UNIQUE(prepare_id,storage_seal); +CREATE INDEX idx_mst2_snapshot_context_metadata_root ON mst2_snapshot_context(metadata_root,snapshot_id); +CREATE INDEX idx_mst2_metadata_lifetime_qualified_scan ON mst2_metadata_lifetime(page_id,generation) + WHERE graph_domain='qualified-v1' AND state IN ('RESERVED','LIVE'); + +CREATE TABLE mst2_metadata_graph_node ( + page_id bytea NOT NULL CHECK (octet_length(page_id)=32), + generation bigint NOT NULL CHECK (generation>0), + state text NOT NULL CHECK (state IN ('LIVE','DELETING')), + metadata_codec smallint NOT NULL CHECK (metadata_codec=1), + bytes bigint NOT NULL CHECK (bytes BETWEEN $HEADER_LEN$ AND $PAGE_MAX_BYTES$), + incoming_refs bigint NOT NULL DEFAULT 0 CHECK (incoming_refs>=0), + PRIMARY KEY(page_id,generation), + FOREIGN KEY(page_id,generation) REFERENCES mst2_metadata_lifetime(page_id,generation) +); +CREATE INDEX idx_mst2_metadata_graph_node_gc ON mst2_metadata_graph_node(state,incoming_refs,page_id,generation); +CREATE TABLE mst2_metadata_graph_edge ( + parent_page bytea NOT NULL, + parent_generation bigint NOT NULL, + child_page bytea NOT NULL, + child_generation bigint NOT NULL, + PRIMARY KEY(parent_page,parent_generation,child_page,child_generation), + FOREIGN KEY(parent_page,parent_generation) REFERENCES mst2_metadata_graph_node(page_id,generation), + FOREIGN KEY(child_page,child_generation) REFERENCES mst2_metadata_graph_node(page_id,generation), + CHECK (parent_page<>child_page OR parent_generation<>child_generation) +); +CREATE INDEX idx_mst2_metadata_graph_edge_child ON mst2_metadata_graph_edge(child_page,child_generation,parent_page,parent_generation); +CREATE TABLE mst2_metadata_graph_root ( + prepare_id text NOT NULL, + storage_seal bytea NOT NULL CHECK (octet_length(storage_seal)=32), + page_id bytea NOT NULL, + generation bigint NOT NULL, + PRIMARY KEY(prepare_id,page_id,generation), + FOREIGN KEY(prepare_id,storage_seal) REFERENCES mst2_metadata_prepare(prepare_id,storage_seal), + FOREIGN KEY(prepare_id,page_id,generation) REFERENCES mst2_metadata_prepare_page(prepare_id,page_id,generation), + FOREIGN KEY(page_id,generation) REFERENCES mst2_metadata_graph_node(page_id,generation) +); +CREATE INDEX idx_mst2_metadata_graph_root_page ON mst2_metadata_graph_root(page_id,generation,prepare_id); +CREATE TABLE mst2_metadata_gc_op ( + operation_id uuid PRIMARY KEY, + page_id bytea NOT NULL CHECK (octet_length(page_id)=32), + generation bigint NOT NULL CHECK (generation>0), + primary_scope bytea NOT NULL CHECK (octet_length(primary_scope) BETWEEN 1 AND 16384), + graph_domain text NOT NULL CHECK (graph_domain='qualified-v1'), + metadata_codec smallint NOT NULL CHECK (metadata_codec=1), + expected_size integer NOT NULL CHECK (expected_size BETWEEN $HEADER_LEN$ AND $PAGE_MAX_BYTES$), + graph_present boolean NOT NULL, + had_payload boolean NOT NULL, + payload_delete_xid bigint, + state text NOT NULL CHECK (state IN ('PENDING','APPLIED')), + created_at timestamptz NOT NULL DEFAULT clock_timestamp(), + completed_at timestamptz, + UNIQUE(page_id,generation), + FOREIGN KEY(page_id,generation) REFERENCES mst2_metadata_lifetime(page_id,generation), + CHECK ((state='APPLIED')=(completed_at IS NOT NULL)) +); +CREATE INDEX idx_mst2_metadata_gc_op_pending ON mst2_metadata_gc_op(created_at,operation_id) WHERE state='PENDING'; + +CREATE FUNCTION mst2_metadata_dml_barrier() RETURNS trigger LANGUAGE plpgsql VOLATILE AS $$ +BEGIN + IF pg_is_in_recovery() OR current_setting('transaction_isolation')<>'read committed' THEN + RAISE EXCEPTION 'metadata mutation requires primary READ COMMITTED'; + END IF; + PERFORM set_config('lock_timeout','5000ms',true); + PERFORM pg_advisory_xact_lock(1296717362,hashtext(current_schema())); + RETURN NULL; +END $$; +DO $$ DECLARE t text; BEGIN + FOREACH t IN ARRAY ARRAY['mst2_metadata_lifetime','mst2_metadata_current','mst2_metadata_payload', + 'mst2_metadata_prepare','mst2_metadata_prepare_page','mst2_metadata_graph_node', + 'mst2_metadata_graph_edge','mst2_metadata_graph_root','mst2_metadata_gc_op', + 'mst2_retention_node','mst2_retention_edge','mst2_retention_root','mst2_retention_gc_op', + 'mst2_snapshot_context','mst2_snapshot_lease'] LOOP + EXECUTE format('CREATE TRIGGER mst2_metadata_statement_barrier BEFORE INSERT OR UPDATE OR DELETE ON %I FOR EACH STATEMENT EXECUTE FUNCTION mst2_metadata_dml_barrier()',t); + END LOOP; +END $$; + +CREATE FUNCTION mst2_metadata_scope_matches(s bytea) RETURNS boolean LANGUAGE plpgsql VOLATILE AS $$ +DECLARE actual jsonb; +BEGIN + IF pg_is_in_recovery() OR current_setting('transaction_isolation')<>'read committed' THEN RETURN false; END IF; + SELECT jsonb_build_array(x.storage_uuid,current_database(),d.oid::bigint,current_schema(),n.oid::bigint, + inet_server_addr()::text,inet_server_port()) INTO actual + FROM mst2_metadata_storage_scope x JOIN pg_database d ON d.datname=current_database() + JOIN pg_namespace n ON n.nspname=current_schema() WHERE x.singleton=1; + RETURN actual IS NOT NULL AND convert_from(s,'UTF8')::jsonb=actual; +EXCEPTION WHEN OTHERS THEN RETURN false; +END $$; + +CREATE FUNCTION mst2_metadata_assert_uncovered(p bytea,g bigint) RETURNS void LANGUAGE plpgsql VOLATILE AS $$ +DECLARE nid text:='page:sha256:'||encode(p,'hex'); +BEGIN + IF EXISTS(SELECT 1 FROM mst2_metadata_graph_root WHERE page_id=p AND generation=g) + OR EXISTS(SELECT 1 FROM mst2_metadata_graph_edge WHERE child_page=p AND child_generation=g) + OR EXISTS(SELECT 1 FROM mst2_retention_node WHERE node_id=nid) + OR EXISTS(SELECT 1 FROM mst2_retention_edge WHERE parent_id=nid OR child_id=nid) + OR EXISTS(SELECT 1 FROM mst2_retention_root WHERE node_id=nid) + OR EXISTS(SELECT 1 FROM mst2_retention_gc_op WHERE node_id=nid) + OR EXISTS(SELECT 1 FROM mst2_metadata_prepare_page m JOIN mst2_metadata_prepare q USING(prepare_id) + WHERE m.page_id=p AND (q.state='PREPARING' OR (q.state='COMMITTED' AND q.coverage_retired_at IS NULL))) + OR EXISTS(SELECT 1 FROM mst2_snapshot_context WHERE metadata_root=p) + OR EXISTS(SELECT 1 FROM mst2_metadata_prepare_page m JOIN mst2_snapshot_context s USING(prepare_id) + WHERE m.page_id=p) THEN + RAISE EXCEPTION 'metadata lifetime has durable or ambiguous coverage'; + END IF; +END $$; + +CREATE FUNCTION mst2_metadata_gc_proof(p bytea,g bigint,stage text,graph_present boolean,payload_present boolean) +RETURNS void LANGUAGE plpgsql VOLATILE AS $$ +DECLARE l mst2_metadata_lifetime%ROWTYPE; n mst2_metadata_graph_node%ROWTYPE; b mst2_metadata_payload%ROWTYPE; +BEGIN + SELECT h.* INTO l FROM mst2_metadata_current c JOIN mst2_metadata_lifetime h USING(page_id,generation) + WHERE c.page_id=p AND c.generation=g FOR UPDATE OF c,h; + IF NOT FOUND OR l.graph_domain<>'qualified-v1' + OR (stage='CLAIM' AND l.state NOT IN ('RESERVED','LIVE')) + OR (stage='PENDING' AND l.state<>'DELETING') + OR (stage='REMOVED' AND l.state<>'REMOVED') + OR stage NOT IN ('CLAIM','PENDING','REMOVED') THEN + RAISE EXCEPTION 'metadata GC is not the exact current qualified lifetime'; + END IF; + PERFORM mst2_metadata_assert_uncovered(p,g); + SELECT * INTO n FROM mst2_metadata_graph_node WHERE page_id=p AND generation=g FOR UPDATE; + IF FOUND IS DISTINCT FROM graph_present THEN RAISE EXCEPTION 'metadata GC graph presence changed'; END IF; + IF graph_present AND (n.metadata_codec<>l.metadata_codec OR n.bytes<>l.expected_size OR n.incoming_refs<>0 + OR (stage='CLAIM' AND n.state<>'LIVE') OR (stage='PENDING' AND n.state<>'DELETING')) THEN + RAISE EXCEPTION 'metadata GC graph profile or counter changed'; + END IF; + IF graph_present AND (SELECT count(*) FROM (SELECT 1 FROM mst2_metadata_graph_edge + WHERE parent_page=p AND parent_generation=g LIMIT 16385) bounded)>16384 THEN + RAISE EXCEPTION 'metadata GC outgoing edge limit exceeded'; + END IF; + SELECT * INTO b FROM mst2_metadata_payload WHERE page_id=p FOR UPDATE; + IF FOUND IS DISTINCT FROM payload_present THEN RAISE EXCEPTION 'metadata GC payload presence changed'; END IF; + IF payload_present AND (b.generation IS DISTINCT FROM g OR b.metadata_codec<>l.metadata_codec OR b.byte_size<>l.expected_size) THEN + RAISE EXCEPTION 'metadata GC payload profile or generation changed'; + END IF; + IF stage='CLAIM' AND l.state='LIVE' AND NOT (graph_present AND payload_present) THEN + RAISE EXCEPTION 'LIVE metadata corruption is not collectable'; + END IF; + IF stage='CLAIM' AND l.state='RESERVED' AND (graph_present OR NOT EXISTS( + SELECT 1 FROM mst2_metadata_prepare_page m JOIN mst2_metadata_prepare q USING(prepare_id) + WHERE m.page_id=p AND m.generation=g AND q.graph_domain='qualified-v1' + AND q.state='ABORTED' AND q.storage_seal IS NOT NULL AND m.expected_size=l.expected_size) + OR EXISTS(SELECT 1 FROM mst2_metadata_prepare_page m JOIN mst2_metadata_prepare q USING(prepare_id) + WHERE m.page_id=p AND m.generation=g AND q.state<>'ABORTED')) THEN + RAISE EXCEPTION 'RESERVED metadata needs an explicitly aborted qualified origin'; + END IF; +END $$; + +CREATE FUNCTION mst2_metadata_gc_op_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE AS $$ +DECLARE l mst2_metadata_lifetime%ROWTYPE; +BEGIN + IF TG_OP='DELETE' THEN RAISE EXCEPTION 'metadata GC evidence cannot be deleted'; END IF; + IF NOT mst2_metadata_scope_matches(NEW.primary_scope) THEN RAISE EXCEPTION 'metadata GC wrong primary scope'; END IF; + IF TG_OP='INSERT' THEN + IF NEW.state<>'PENDING' OR NEW.completed_at IS NOT NULL OR NEW.payload_delete_xid IS NOT NULL THEN RAISE EXCEPTION 'metadata GC must start PENDING'; END IF; + PERFORM mst2_metadata_gc_proof(NEW.page_id,NEW.generation,'CLAIM',NEW.graph_present,NEW.had_payload); + SELECT * INTO l FROM mst2_metadata_lifetime WHERE page_id=NEW.page_id AND generation=NEW.generation; + IF NEW.graph_domain<>l.graph_domain OR NEW.metadata_codec<>l.metadata_codec OR NEW.expected_size<>l.expected_size THEN + RAISE EXCEPTION 'metadata GC immutable profile conflict'; + END IF; + NEW.created_at:=clock_timestamp(); + ELSE + IF ROW(NEW.operation_id,NEW.page_id,NEW.generation,NEW.primary_scope,NEW.graph_domain, + NEW.metadata_codec,NEW.expected_size,NEW.graph_present,NEW.had_payload,NEW.created_at) + IS DISTINCT FROM ROW(OLD.operation_id,OLD.page_id,OLD.generation,OLD.primary_scope,OLD.graph_domain, + OLD.metadata_codec,OLD.expected_size,OLD.graph_present,OLD.had_payload,OLD.created_at) THEN + RAISE EXCEPTION 'metadata GC identity cannot change'; + END IF; + IF OLD.state='APPLIED' AND (NEW.state<>OLD.state OR NEW.completed_at IS DISTINCT FROM OLD.completed_at + OR NEW.payload_delete_xid IS DISTINCT FROM OLD.payload_delete_xid) THEN + RAISE EXCEPTION 'metadata GC receipt cannot change'; + END IF; + IF OLD.state='PENDING' AND NEW.state='APPLIED' THEN + IF OLD.had_payload AND OLD.payload_delete_xid IS DISTINCT FROM txid_current() THEN + RAISE EXCEPTION 'metadata GC has no same-transaction payload delete proof'; + END IF; + PERFORM mst2_metadata_gc_proof(NEW.page_id,NEW.generation,'REMOVED',false,false); + NEW.completed_at:=clock_timestamp(); + ELSIF NEW.state IS DISTINCT FROM OLD.state OR NEW.completed_at IS DISTINCT FROM OLD.completed_at THEN + RAISE EXCEPTION 'metadata GC invalid receipt transition'; + END IF; + IF NEW.payload_delete_xid IS DISTINCT FROM OLD.payload_delete_xid THEN + IF OLD.state<>'PENDING' OR NEW.state<>'PENDING' OR NOT OLD.had_payload THEN + RAISE EXCEPTION 'invalid metadata delete transaction proof'; + END IF; + PERFORM mst2_metadata_gc_proof(OLD.page_id,OLD.generation,'PENDING',OLD.graph_present,true); + NEW.payload_delete_xid:=txid_current(); + END IF; + END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_metadata_gc_op_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_metadata_gc_op + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_gc_op_guard(); + +CREATE OR REPLACE FUNCTION mst2_metadata_lifetime_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE AS $$ +DECLARE op mst2_metadata_gc_op%ROWTYPE; +BEGIN + IF TG_OP='DELETE' THEN RAISE EXCEPTION 'metadata lifetime watermark cannot be deleted'; END IF; + IF ROW(NEW.page_id,NEW.node_id,NEW.generation,NEW.metadata_codec,NEW.expected_size,NEW.graph_domain) + IS DISTINCT FROM ROW(OLD.page_id,OLD.node_id,OLD.generation,OLD.metadata_codec,OLD.expected_size,OLD.graph_domain) THEN + RAISE EXCEPTION 'metadata lifetime identity is immutable'; + END IF; + IF NEW.state=OLD.state THEN RETURN NEW; END IF; + IF OLD.graph_domain='generic-v1' THEN + IF OLD.state='RESERVED' AND NEW.state='LIVE' THEN RETURN NEW; END IF; + RAISE EXCEPTION 'generic lifetime transition requires original collector'; + END IF; + IF OLD.state='RESERVED' AND NEW.state='LIVE' THEN + IF NOT EXISTS(SELECT 1 FROM mst2_metadata_graph_node n JOIN mst2_metadata_payload b USING(page_id,generation) + WHERE n.page_id=OLD.page_id AND n.generation=OLD.generation AND n.state='LIVE' + AND n.metadata_codec=OLD.metadata_codec AND n.bytes=OLD.expected_size + AND b.metadata_codec=OLD.metadata_codec AND b.byte_size=OLD.expected_size) + OR NOT EXISTS(SELECT 1 FROM mst2_metadata_graph_root r WHERE r.page_id=OLD.page_id AND r.generation=OLD.generation) THEN + RAISE EXCEPTION 'qualified LIVE transition needs its graph payload and prepare root'; + END IF; + RETURN NEW; + END IF; + SELECT * INTO op FROM mst2_metadata_gc_op WHERE page_id=OLD.page_id AND generation=OLD.generation AND state='PENDING'; + IF NOT FOUND OR NOT mst2_metadata_scope_matches(op.primary_scope) THEN RAISE EXCEPTION 'metadata transition has no exact GC op'; END IF; + IF OLD.state IN ('RESERVED','LIVE') AND NEW.state='DELETING' THEN + PERFORM mst2_metadata_assert_uncovered(OLD.page_id,OLD.generation); + RETURN NEW; + END IF; + IF OLD.state='DELETING' AND NEW.state='REMOVED' THEN + IF op.had_payload AND op.payload_delete_xid IS DISTINCT FROM txid_current() THEN + RAISE EXCEPTION 'metadata removal has no same-transaction payload delete proof'; + END IF; + PERFORM mst2_metadata_gc_proof(OLD.page_id,OLD.generation,'PENDING',false,false); + RETURN NEW; + END IF; + RAISE EXCEPTION 'metadata lifetime invalid state transition'; +END $$; + +CREATE FUNCTION mst2_metadata_lifetime_removed() RETURNS trigger LANGUAGE plpgsql VOLATILE AS $$ +BEGIN + IF OLD.state='DELETING' AND NEW.state='REMOVED' THEN + UPDATE mst2_metadata_gc_op SET state='APPLIED' + WHERE page_id=NEW.page_id AND generation=NEW.generation AND state='PENDING'; + IF NOT FOUND THEN RAISE EXCEPTION 'metadata removal lost its GC operation'; END IF; + END IF; + RETURN NULL; +END $$; +CREATE TRIGGER mst2_metadata_lifetime_removed AFTER UPDATE ON mst2_metadata_lifetime + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_lifetime_removed(); + +CREATE OR REPLACE FUNCTION mst2_metadata_current_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE AS $$ +DECLARE op mst2_metadata_gc_op%ROWTYPE; l mst2_metadata_lifetime%ROWTYPE; +BEGIN + IF TG_OP='DELETE' THEN RAISE EXCEPTION 'metadata current watermark cannot be deleted'; END IF; + IF TG_OP='INSERT' THEN + IF NEW.generation<>1 OR EXISTS(SELECT 1 FROM mst2_metadata_lifetime WHERE page_id=NEW.page_id AND generation<>1) THEN + RAISE EXCEPTION 'initial metadata current cannot reset history'; + END IF; + SELECT * INTO l FROM mst2_metadata_lifetime WHERE page_id=NEW.page_id AND generation=NEW.generation; + IF l.graph_domain='qualified-v1' THEN + PERFORM mst2_metadata_assert_uncovered(NEW.page_id,NEW.generation); + IF EXISTS(SELECT 1 FROM mst2_metadata_payload WHERE page_id=NEW.page_id) + OR EXISTS(SELECT 1 FROM mst2_metadata_graph_node WHERE page_id=NEW.page_id) THEN + RAISE EXCEPTION 'initial qualified current cannot adopt old bytes or graph'; + END IF; + END IF; + RETURN NEW; + END IF; + IF NEW.page_id<>OLD.page_id OR OLD.generation=9223372036854775807 OR NEW.generation<>OLD.generation+1 THEN + RAISE EXCEPTION 'metadata current requires exact next-generation CAS'; + END IF; + SELECT * INTO op FROM mst2_metadata_gc_op WHERE page_id=OLD.page_id AND generation=OLD.generation AND state='APPLIED'; + IF NOT FOUND OR NOT mst2_metadata_scope_matches(op.primary_scope) THEN RAISE EXCEPTION 'fresh metadata requires exact APPLIED proof'; END IF; + PERFORM mst2_metadata_gc_proof(OLD.page_id,OLD.generation,'REMOVED',false,false); + SELECT * INTO l FROM mst2_metadata_lifetime WHERE page_id=NEW.page_id AND generation=NEW.generation; + IF NOT FOUND OR l.state<>'RESERVED' OR l.graph_domain<>'qualified-v1' + OR l.metadata_codec<>op.metadata_codec OR l.expected_size<>op.expected_size THEN + RAISE EXCEPTION 'fresh metadata incarnation profile mismatch'; + END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_metadata_current_insert_guard BEFORE INSERT ON mst2_metadata_current + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_current_guard(); + +CREATE FUNCTION mst2_metadata_current_protected() RETURNS trigger LANGUAGE plpgsql VOLATILE AS $$ +BEGIN + IF EXISTS(SELECT 1 FROM mst2_metadata_lifetime WHERE page_id=NEW.page_id AND generation=NEW.generation AND graph_domain='qualified-v1') + AND NOT EXISTS(SELECT 1 FROM mst2_metadata_prepare_page m JOIN mst2_metadata_prepare q USING(prepare_id) + JOIN mst2_metadata_current c ON c.page_id=m.page_id AND c.generation=m.generation + WHERE m.page_id=NEW.page_id AND m.generation=NEW.generation AND q.graph_domain='qualified-v1' + AND q.storage_seal IS NOT NULL AND (q.state='PREPARING' OR (q.state='COMMITTED' AND q.coverage_retired_at IS NULL))) THEN + RAISE EXCEPTION 'qualified current change must commit with fresh prepare protection'; + END IF; + RETURN NULL; +END $$; +CREATE CONSTRAINT TRIGGER mst2_metadata_current_protected AFTER INSERT OR UPDATE ON mst2_metadata_current + DEFERRABLE INITIALLY DEFERRED FOR EACH ROW EXECUTE FUNCTION mst2_metadata_current_protected(); + +CREATE FUNCTION mst2_metadata_lifetime_insert_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE AS $$ +DECLARE c mst2_metadata_current%ROWTYPE; old_owner text; op mst2_metadata_gc_op%ROWTYPE; +BEGIN + SELECT * INTO c FROM mst2_metadata_current WHERE page_id=NEW.page_id; + IF FOUND THEN + SELECT graph_domain INTO old_owner FROM mst2_metadata_lifetime WHERE page_id=c.page_id AND generation=c.generation; + IF old_owner IS DISTINCT FROM NEW.graph_domain THEN RAISE EXCEPTION 'metadata incarnation domain cannot be adopted'; END IF; + IF NEW.graph_domain='qualified-v1' AND NEW.generation<>c.generation THEN + SELECT * INTO op FROM mst2_metadata_gc_op WHERE page_id=c.page_id AND generation=c.generation AND state='APPLIED'; + IF NOT FOUND OR NOT mst2_metadata_scope_matches(op.primary_scope) OR c.generation=9223372036854775807 + OR NEW.generation<>c.generation+1 OR NEW.state<>'RESERVED' + OR NEW.metadata_codec<>op.metadata_codec OR NEW.expected_size<>op.expected_size THEN + RAISE EXCEPTION 'fresh historical incarnation needs exact removal proof'; + END IF; + PERFORM mst2_metadata_gc_proof(c.page_id,c.generation,'REMOVED',false,false); + END IF; + ELSIF NEW.graph_domain='qualified-v1' AND (NEW.generation<>1 OR NEW.state<>'RESERVED' + OR EXISTS(SELECT 1 FROM mst2_metadata_lifetime WHERE page_id=NEW.page_id AND graph_domain<>NEW.graph_domain)) THEN + RAISE EXCEPTION 'initial qualified incarnation cannot adopt old history'; + END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_metadata_lifetime_insert_guard BEFORE INSERT ON mst2_metadata_lifetime + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_lifetime_insert_guard(); + +CREATE FUNCTION mst2_metadata_qualified_mapping_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE AS $$ +DECLARE owner text; domain text; st text; +BEGIN + IF TG_OP<>'INSERT' THEN RETURN NEW; END IF; + SELECT l.graph_domain INTO owner FROM mst2_metadata_current c JOIN mst2_metadata_lifetime l USING(page_id,generation) + WHERE c.page_id=NEW.page_id; + SELECT graph_domain INTO domain FROM mst2_metadata_prepare WHERE prepare_id=NEW.prepare_id; + IF owner='qualified-v1' OR domain='qualified-v1' THEN + SELECT l.state INTO st FROM mst2_metadata_current c JOIN mst2_metadata_lifetime l USING(page_id,generation) + WHERE c.page_id=NEW.page_id AND c.generation=NEW.generation AND l.graph_domain='qualified-v1'; + IF owner IS DISTINCT FROM domain OR st IS NULL OR st NOT IN ('RESERVED','LIVE') + OR EXISTS(SELECT 1 FROM mst2_metadata_gc_op WHERE page_id=NEW.page_id AND generation=NEW.generation) + OR NOT EXISTS(SELECT 1 FROM mst2_metadata_prepare q WHERE q.prepare_id=NEW.prepare_id + AND q.state='PREPARING' AND q.storage_seal IS NOT NULL AND mst2_metadata_scope_matches(q.primary_scope)) THEN + RAISE EXCEPTION 'qualified mapping cannot cross domain/current/state fence'; + END IF; + END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_metadata_qualified_mapping_guard BEFORE INSERT ON mst2_metadata_prepare_page + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_qualified_mapping_guard(); + +CREATE FUNCTION mst2_metadata_generic_domain_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE AS $$ +DECLARE record jsonb; ident text; key text; +BEGIN + FOREACH key IN ARRAY ARRAY['node_id','parent_id','child_id'] LOOP + IF TG_OP<>'INSERT' THEN + record:=to_jsonb(OLD); ident:=record->>key; + IF EXISTS(SELECT 1 FROM mst2_metadata_current c JOIN mst2_metadata_lifetime l USING(page_id,generation) + WHERE l.graph_domain='qualified-v1' AND l.node_id=ident) THEN RAISE EXCEPTION 'generic graph cannot touch qualified incarnation'; END IF; + END IF; + IF TG_OP<>'DELETE' THEN + record:=to_jsonb(NEW); ident:=record->>key; + IF EXISTS(SELECT 1 FROM mst2_metadata_current c JOIN mst2_metadata_lifetime l USING(page_id,generation) + WHERE l.graph_domain='qualified-v1' AND l.node_id=ident) THEN RAISE EXCEPTION 'generic graph cannot adopt qualified incarnation'; END IF; + END IF; + END LOOP; + IF TG_OP='DELETE' THEN RETURN OLD; END IF; + RETURN NEW; +END $$; +DO $$ DECLARE t text; BEGIN + FOREACH t IN ARRAY ARRAY['mst2_retention_node','mst2_retention_edge','mst2_retention_root','mst2_retention_gc_op'] LOOP + EXECUTE format('CREATE TRIGGER mst2_metadata_generic_domain_guard BEFORE INSERT OR UPDATE OR DELETE ON %I FOR EACH ROW EXECUTE FUNCTION mst2_metadata_generic_domain_guard()',t); + END LOOP; +END $$; + +CREATE FUNCTION mst2_metadata_session_domain_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE AS $$ +DECLARE sid text; pid text; +BEGIN + IF TG_OP='DELETE' THEN RETURN OLD; END IF; + IF TG_TABLE_NAME='mst2_snapshot_context' THEN pid:=NEW.prepare_id; + ELSE sid:=NEW.snapshot_id; SELECT prepare_id INTO pid FROM mst2_snapshot_context WHERE snapshot_id=sid; END IF; + IF EXISTS(SELECT 1 FROM mst2_metadata_prepare WHERE prepare_id=pid AND graph_domain='qualified-v1') THEN + RAISE EXCEPTION 'qualified production session adoption is closed'; + END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_metadata_session_domain_guard BEFORE INSERT OR UPDATE ON mst2_snapshot_context + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_session_domain_guard(); +CREATE TRIGGER mst2_metadata_session_domain_guard BEFORE INSERT OR UPDATE ON mst2_snapshot_lease + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_session_domain_guard(); + +CREATE FUNCTION mst2_metadata_graph_node_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE AS $$ +DECLARE l mst2_metadata_lifetime%ROWTYPE; op mst2_metadata_gc_op%ROWTYPE; +BEGIN + IF TG_OP='DELETE' THEN + SELECT * INTO op FROM mst2_metadata_gc_op WHERE page_id=OLD.page_id AND generation=OLD.generation AND state='PENDING'; + IF NOT FOUND OR NOT mst2_metadata_scope_matches(op.primary_scope) OR OLD.state<>'DELETING' THEN + RAISE EXCEPTION 'qualified node deletion needs exact pending operation'; + END IF; + IF op.had_payload AND op.payload_delete_xid IS DISTINCT FROM txid_current() THEN RAISE EXCEPTION 'node deletion has no payload delete proof'; END IF; + PERFORM mst2_metadata_gc_proof(OLD.page_id,OLD.generation,'PENDING',true,false); + IF EXISTS(SELECT 1 FROM mst2_metadata_graph_edge WHERE parent_page=OLD.page_id AND parent_generation=OLD.generation) THEN + RAISE EXCEPTION 'qualified node still has outgoing edges'; + END IF; + RETURN OLD; + END IF; + SELECT l0.* INTO l FROM mst2_metadata_current c JOIN mst2_metadata_lifetime l0 USING(page_id,generation) + WHERE c.page_id=NEW.page_id AND c.generation=NEW.generation; + IF NOT FOUND OR l.graph_domain<>'qualified-v1' OR l.metadata_codec<>NEW.metadata_codec OR l.expected_size<>NEW.bytes THEN + RAISE EXCEPTION 'qualified node does not match exact lifetime'; + END IF; + IF NEW.incoming_refs<>(SELECT count(*) FROM mst2_metadata_graph_edge WHERE child_page=NEW.page_id AND child_generation=NEW.generation) THEN + RAISE EXCEPTION 'qualified node counter differs from actual unique edges'; + END IF; + IF TG_OP='INSERT' THEN + IF NEW.state<>'LIVE' OR l.state NOT IN ('RESERVED','LIVE') OR NOT EXISTS( + SELECT 1 FROM mst2_metadata_prepare_page m JOIN mst2_metadata_prepare q USING(prepare_id) + WHERE m.page_id=NEW.page_id AND m.generation=NEW.generation AND q.graph_domain='qualified-v1' AND q.state='PREPARING') THEN + RAISE EXCEPTION 'qualified node creation needs active fixed preparation'; + END IF; + ELSE + IF ROW(NEW.page_id,NEW.generation,NEW.metadata_codec,NEW.bytes) IS DISTINCT FROM ROW(OLD.page_id,OLD.generation,OLD.metadata_codec,OLD.bytes) THEN + RAISE EXCEPTION 'qualified node identity is immutable'; + END IF; + IF NEW.state<>OLD.state AND NOT (OLD.state='LIVE' AND NEW.state='DELETING' AND EXISTS( + SELECT 1 FROM mst2_metadata_gc_op WHERE page_id=OLD.page_id AND generation=OLD.generation AND state='PENDING')) THEN + RAISE EXCEPTION 'qualified node invalid state transition'; + END IF; + END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_metadata_graph_node_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_metadata_graph_node + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_graph_node_guard(); + +CREATE FUNCTION mst2_metadata_graph_root_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE AS $$ +BEGIN + IF TG_OP='UPDATE' THEN RAISE EXCEPTION 'qualified roots cannot be retargeted'; END IF; + IF TG_OP='DELETE' THEN RETURN OLD; END IF; + IF NOT EXISTS(SELECT 1 FROM mst2_metadata_prepare q JOIN mst2_metadata_current c ON c.page_id=NEW.page_id AND c.generation=NEW.generation + JOIN mst2_metadata_lifetime l USING(page_id,generation) JOIN mst2_metadata_graph_node n USING(page_id,generation) + WHERE q.prepare_id=NEW.prepare_id AND q.storage_seal=NEW.storage_seal AND q.state='PREPARING' + AND q.graph_domain='qualified-v1' AND l.graph_domain='qualified-v1' AND l.state IN ('RESERVED','LIVE') + AND n.state='LIVE' AND mst2_metadata_scope_matches(q.primary_scope)) THEN + RAISE EXCEPTION 'qualified root requires its active exact sealed preparation'; + END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_metadata_graph_root_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_metadata_graph_root + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_graph_root_guard(); + +CREATE FUNCTION mst2_metadata_graph_edge_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE AS $$ +DECLARE op mst2_metadata_gc_op%ROWTYPE; +BEGIN + IF TG_OP='UPDATE' THEN RAISE EXCEPTION 'qualified edge identity is immutable'; END IF; + IF TG_OP='DELETE' THEN + SELECT * INTO op FROM mst2_metadata_gc_op WHERE page_id=OLD.parent_page AND generation=OLD.parent_generation AND state='PENDING'; + IF NOT FOUND OR NOT mst2_metadata_scope_matches(op.primary_scope) THEN RAISE EXCEPTION 'edge deletion has no exact operation'; END IF; + IF op.had_payload AND op.payload_delete_xid IS DISTINCT FROM txid_current() THEN RAISE EXCEPTION 'edge deletion has no payload delete proof'; END IF; + IF EXISTS(SELECT 1 FROM mst2_metadata_payload WHERE page_id=OLD.parent_page) + OR NOT EXISTS(SELECT 1 FROM mst2_metadata_graph_node WHERE page_id=OLD.parent_page + AND generation=OLD.parent_generation AND state='DELETING' AND incoming_refs=0) THEN + RAISE EXCEPTION 'edge deletion is outside its atomic byte-removal phase'; + END IF; + RETURN OLD; + END IF; + IF (SELECT count(*) FROM mst2_metadata_graph_node n JOIN mst2_metadata_current c USING(page_id,generation) + JOIN mst2_metadata_lifetime l USING(page_id,generation) + WHERE ((n.page_id=NEW.parent_page AND n.generation=NEW.parent_generation) + OR (n.page_id=NEW.child_page AND n.generation=NEW.child_generation)) + AND n.state='LIVE' AND l.state IN ('RESERVED','LIVE') AND l.graph_domain='qualified-v1')<>2 + OR NOT EXISTS(SELECT 1 FROM mst2_metadata_prepare_page a JOIN mst2_metadata_prepare_page b USING(prepare_id) + JOIN mst2_metadata_prepare q USING(prepare_id) WHERE a.page_id=NEW.parent_page AND a.generation=NEW.parent_generation + AND b.page_id=NEW.child_page AND b.generation=NEW.child_generation AND q.state='PREPARING' AND q.graph_domain='qualified-v1') THEN + RAISE EXCEPTION 'edge creation requires active exact qualified endpoints'; + END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_metadata_graph_edge_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_metadata_graph_edge + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_graph_edge_guard(); + +CREATE FUNCTION mst2_metadata_edges_added() RETURNS trigger LANGUAGE plpgsql VOLATILE AS $$ +DECLARE reachable text[]; links jsonb; adjacency jsonb; indegrees bigint[]; + queue integer[]:=ARRAY[]::integer[]; head integer:=1; parent_index integer; child_index integer; i integer; +BEGIN + IF (SELECT count(*) FROM added_edges)>16384 THEN RAISE EXCEPTION 'qualified edge batch exceeds limit'; END IF; + IF NOT EXISTS(SELECT 1 FROM added_edges) THEN RETURN NULL; END IF; + IF EXISTS(SELECT 1 FROM mst2_metadata_graph_node n JOIN ( + SELECT child_page,child_generation,count(*) AS delta FROM added_edges GROUP BY child_page,child_generation + ) d ON d.child_page=n.page_id AND d.child_generation=n.generation + WHERE n.incoming_refs+d.delta<>(SELECT count(*) FROM mst2_metadata_graph_edge x WHERE x.child_page=n.page_id AND x.child_generation=n.generation)) THEN + RAISE EXCEPTION 'qualified insertion found preexisting counter drift'; + END IF; + UPDATE mst2_metadata_graph_node n SET incoming_refs=n.incoming_refs+d.delta FROM ( + SELECT child_page,child_generation,count(*) AS delta FROM added_edges GROUP BY child_page,child_generation + ) d WHERE n.page_id=d.child_page AND n.generation=d.child_generation; + -- One bounded closure per statement; topological accounting below is local. + WITH RECURSIVE walk(page_id,generation) AS ( + SELECT page_id,generation FROM ( + SELECT parent_page AS page_id,parent_generation AS generation FROM added_edges + UNION SELECT child_page,child_generation FROM added_edges + ) starts UNION + SELECT x.child_page,x.child_generation FROM walk w JOIN mst2_metadata_graph_edge x + ON x.parent_page=w.page_id AND x.parent_generation=w.generation + ) SELECT array_agg(encode(page_id,'hex')||':'||generation ORDER BY page_id,generation) + INTO reachable FROM (SELECT * FROM walk LIMIT 4097) bounded; + IF cardinality(reachable)>4096 THEN RAISE EXCEPTION 'qualified graph node audit overflow'; END IF; + WITH nodes AS ( + SELECT ordinality::integer AS i,decode(split_part(key,':',1),'hex') AS page_id, + split_part(key,':',2)::bigint AS generation FROM unnest(reachable) WITH ORDINALITY n(key,ordinality) + ) SELECT coalesce(jsonb_agg(jsonb_build_array(p.i,c.i)),'[]'::jsonb) INTO links FROM ( + SELECT e.* FROM nodes p JOIN mst2_metadata_graph_edge e + ON e.parent_page=p.page_id AND e.parent_generation=p.generation LIMIT 16385 + ) e JOIN nodes p ON p.page_id=e.parent_page AND p.generation=e.parent_generation + JOIN nodes c ON c.page_id=e.child_page AND c.generation=e.child_generation; + IF jsonb_array_length(links)>16384 THEN RAISE EXCEPTION 'qualified graph edge audit overflow'; END IF; + SELECT coalesce(jsonb_object_agg(parent,children),'{}'::jsonb) INTO adjacency FROM ( + SELECT item->>0 AS parent,jsonb_agg((item->>1)::integer) AS children + FROM jsonb_array_elements(links) x(item) GROUP BY item->>0 + ) grouped; + SELECT array_agg(coalesce(counts.refs,0) ORDER BY n.i) INTO indegrees + FROM generate_series(1,cardinality(reachable)) n(i) LEFT JOIN ( + SELECT (item->>1)::integer AS child,count(*) AS refs FROM jsonb_array_elements(links) x(item) GROUP BY item->>1 + ) counts ON counts.child=n.i; + FOR i IN 1..cardinality(reachable) LOOP + IF indegrees[i]=0 THEN queue:=array_append(queue,i); END IF; + END LOOP; + WHILE head<=cardinality(queue) LOOP + parent_index:=queue[head]; head:=head+1; + FOR child_index IN SELECT value::text::integer FROM jsonb_array_elements(coalesce(adjacency->parent_index::text,'[]'::jsonb)) LOOP + indegrees[child_index]:=indegrees[child_index]-1; + IF indegrees[child_index]=0 THEN queue:=array_append(queue,child_index); END IF; + END LOOP; + END LOOP; + IF cardinality(queue)<>cardinality(reachable) THEN RAISE EXCEPTION 'qualified graph contains a cycle'; END IF; + RETURN NULL; +END $$; +CREATE TRIGGER mst2_metadata_edges_added AFTER INSERT ON mst2_metadata_graph_edge + REFERENCING NEW TABLE AS added_edges FOR EACH STATEMENT EXECUTE FUNCTION mst2_metadata_edges_added(); +CREATE FUNCTION mst2_metadata_edges_removed() RETURNS trigger LANGUAGE plpgsql VOLATILE AS $$ +BEGIN + IF EXISTS(SELECT 1 FROM mst2_metadata_graph_node n JOIN ( + SELECT child_page,child_generation,count(*) AS delta FROM removed_edges GROUP BY child_page,child_generation + ) d ON d.child_page=n.page_id AND d.child_generation=n.generation + WHERE n.incoming_refs(SELECT count(*) FROM mst2_metadata_graph_edge x WHERE x.child_page=n.page_id AND x.child_generation=n.generation)) THEN + RAISE EXCEPTION 'qualified subtraction found counter drift'; + END IF; + UPDATE mst2_metadata_graph_node n SET incoming_refs=n.incoming_refs-d.delta FROM ( + SELECT child_page,child_generation,count(*) AS delta FROM removed_edges GROUP BY child_page,child_generation + ) d WHERE n.page_id=d.child_page AND n.generation=d.child_generation; + RETURN NULL; +END $$; +CREATE TRIGGER mst2_metadata_edges_removed AFTER DELETE ON mst2_metadata_graph_edge + REFERENCING OLD TABLE AS removed_edges FOR EACH STATEMENT EXECUTE FUNCTION mst2_metadata_edges_removed(); + +CREATE FUNCTION mst2_metadata_gc_claimed() RETURNS trigger LANGUAGE plpgsql VOLATILE AS $$ +BEGIN + UPDATE mst2_metadata_graph_node SET state='DELETING' WHERE page_id=NEW.page_id AND generation=NEW.generation; + UPDATE mst2_metadata_lifetime SET state='DELETING' WHERE page_id=NEW.page_id AND generation=NEW.generation; + RETURN NULL; +END $$; +CREATE TRIGGER mst2_metadata_gc_claimed AFTER INSERT ON mst2_metadata_gc_op + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_gc_claimed(); + +CREATE FUNCTION mst2_metadata_gc_finish(id uuid) RETURNS void LANGUAGE plpgsql VOLATILE AS $$ +DECLARE op mst2_metadata_gc_op%ROWTYPE; +BEGIN + IF pg_is_in_recovery() OR current_setting('transaction_isolation')<>'read committed' THEN RAISE EXCEPTION 'GC finish needs primary READ COMMITTED'; END IF; + PERFORM set_config('lock_timeout','5000ms',true); + PERFORM pg_advisory_xact_lock(1296717362,hashtext(current_schema())); + SELECT * INTO op FROM mst2_metadata_gc_op WHERE operation_id=id FOR UPDATE; + IF NOT FOUND OR NOT mst2_metadata_scope_matches(op.primary_scope) THEN RAISE EXCEPTION 'GC finish wrong exact operation scope'; END IF; + IF op.state='APPLIED' THEN RETURN; END IF; + IF op.had_payload AND op.payload_delete_xid IS DISTINCT FROM txid_current() THEN + RAISE EXCEPTION 'GC finish has no same-transaction payload delete proof'; + END IF; + PERFORM mst2_metadata_gc_proof(op.page_id,op.generation,'PENDING',op.graph_present,false); + DELETE FROM mst2_metadata_graph_edge WHERE parent_page=op.page_id AND parent_generation=op.generation; + DELETE FROM mst2_metadata_graph_node WHERE page_id=op.page_id AND generation=op.generation; + UPDATE mst2_metadata_lifetime SET state='REMOVED' WHERE page_id=op.page_id AND generation=op.generation AND state='DELETING'; + IF NOT FOUND THEN RAISE EXCEPTION 'GC finish lost exact lifetime'; END IF; +END $$; + +CREATE FUNCTION mst2_metadata_payload_fenced() RETURNS trigger LANGUAGE plpgsql VOLATILE AS $$ +DECLARE l mst2_metadata_lifetime%ROWTYPE; op mst2_metadata_gc_op%ROWTYPE; selector text; +BEGIN + IF TG_OP='UPDATE' THEN RAISE EXCEPTION 'MST2 metadata payload UPDATE is forbidden'; END IF; + IF TG_OP='INSERT' THEN + SELECT h.* INTO l FROM mst2_metadata_current c JOIN mst2_metadata_lifetime h USING(page_id,generation) WHERE c.page_id=NEW.page_id; + IF FOUND AND l.graph_domain='qualified-v1' THEN + IF NEW.generation IS DISTINCT FROM l.generation OR l.state NOT IN ('RESERVED','LIVE') + OR NEW.metadata_codec<>l.metadata_codec OR NEW.byte_size<>l.expected_size + OR NOT EXISTS(SELECT 1 FROM mst2_metadata_prepare_page m JOIN mst2_metadata_prepare q USING(prepare_id) + WHERE m.page_id=NEW.page_id AND m.generation=NEW.generation AND q.state='PREPARING' + AND q.graph_domain='qualified-v1' AND q.storage_seal IS NOT NULL AND mst2_metadata_scope_matches(q.primary_scope)) THEN + RAISE EXCEPTION 'qualified payload INSERT crossed its exact active generation'; + END IF; + END IF; + RETURN NEW; + END IF; + selector:=current_setting('mega2.metadata_gc_operation',true); + IF OLD.generation IS NULL OR selector IS NULL OR selector='' THEN RAISE EXCEPTION 'payload DELETE has no qualified operation selector'; END IF; + SELECT * INTO op FROM mst2_metadata_gc_op WHERE operation_id=selector::uuid AND page_id=OLD.page_id AND generation=OLD.generation FOR UPDATE; + IF NOT FOUND OR op.state<>'PENDING' OR NOT op.had_payload OR NOT mst2_metadata_scope_matches(op.primary_scope) THEN + RAISE EXCEPTION 'payload DELETE requires exact persistent pending proof'; + END IF; + PERFORM mst2_metadata_gc_proof(OLD.page_id,OLD.generation,'PENDING',op.graph_present,true); + UPDATE mst2_metadata_gc_op SET payload_delete_xid=txid_current() WHERE operation_id=op.operation_id; + RETURN OLD; +END $$; +DROP TRIGGER mst2_metadata_payload_immutable ON mst2_metadata_payload; +CREATE TRIGGER mst2_metadata_payload_fenced BEFORE INSERT OR UPDATE OR DELETE ON mst2_metadata_payload + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_payload_fenced(); +CREATE FUNCTION mst2_metadata_payload_removed() RETURNS trigger LANGUAGE plpgsql VOLATILE AS $$ +DECLARE op uuid; +BEGIN + SELECT operation_id INTO op FROM mst2_metadata_gc_op WHERE page_id=OLD.page_id AND generation=OLD.generation AND state='PENDING'; + IF NOT FOUND THEN RAISE EXCEPTION 'deleted payload lost exact pending operation'; END IF; + PERFORM mst2_metadata_gc_finish(op); + RETURN NULL; +END $$; +CREATE TRIGGER mst2_metadata_payload_removed AFTER DELETE ON mst2_metadata_payload + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_payload_removed(); + +CREATE FUNCTION mst2_metadata_gc_apply(id uuid) RETURNS void LANGUAGE plpgsql VOLATILE AS $$ +DECLARE op mst2_metadata_gc_op%ROWTYPE; +BEGIN + IF pg_is_in_recovery() OR current_setting('transaction_isolation')<>'read committed' THEN RAISE EXCEPTION 'GC apply needs primary READ COMMITTED'; END IF; + PERFORM set_config('lock_timeout','5000ms',true); + PERFORM pg_advisory_xact_lock(1296717362,hashtext(current_schema())); + SELECT * INTO op FROM mst2_metadata_gc_op WHERE operation_id=id FOR UPDATE; + IF NOT FOUND OR NOT mst2_metadata_scope_matches(op.primary_scope) THEN RAISE EXCEPTION 'GC apply missing exact scope'; END IF; + IF op.state='APPLIED' THEN RETURN; END IF; + PERFORM mst2_metadata_gc_proof(op.page_id,op.generation,'PENDING',op.graph_present,op.had_payload); + IF op.had_payload THEN + PERFORM set_config('mega2.metadata_gc_operation',op.operation_id::text,true); + DELETE FROM mst2_metadata_payload WHERE page_id=op.page_id AND generation=op.generation; + IF NOT FOUND THEN RAISE EXCEPTION 'GC apply lost captured payload'; END IF; + ELSE + PERFORM mst2_metadata_gc_finish(id); + END IF; +END $$; diff --git a/src/jupiter/migration/mod.rs b/src/jupiter/migration/mod.rs index 4a302ee3..384af69a 100644 --- a/src/jupiter/migration/mod.rs +++ b/src/jupiter/migration/mod.rs @@ -148,6 +148,7 @@ mod m20261006_000100_add_view_tables; mod m20261007_000100_add_mst2_snapshot_sessions; mod m20261007_000200_add_mst2_metadata_generations; mod m20261007_000300_add_mst2_metadata_lifetime_history; +mod m20261007_000400_add_mst2_qualified_metadata_gc; mod runner; pub use m20260905_000100_add_push_queue::ensure_queue_control_seed; pub use runner::apply_migrations; @@ -284,6 +285,7 @@ impl MigratorTrait for Migrator { Box::new(m20261007_000100_add_mst2_snapshot_sessions::Migration), Box::new(m20261007_000200_add_mst2_metadata_generations::Migration), Box::new(m20261007_000300_add_mst2_metadata_lifetime_history::Migration), + Box::new(m20261007_000400_add_mst2_qualified_metadata_gc::Migration), ] } } @@ -1194,7 +1196,7 @@ mod tests { async fn import_repo_alias_rows_canonicalized() { let names = migration_names(); assert_eq!( - &names[names.len() - 11..names.len() - 2], + &names[names.len() - 12..names.len() - 3], &[ "m20260923_000200_canonicalize_import_repo_paths".to_string(), "m20260925_000100_media_paging".to_string(), @@ -1209,13 +1211,17 @@ mod tests { "native retention, metadata installation and view tables follow media paging" ); assert_eq!( - &names[names.len() - 2], + &names[names.len() - 3], "m20261007_000200_add_mst2_metadata_generations" ); assert_eq!( - names.last().unwrap(), + &names[names.len() - 2], "m20261007_000300_add_mst2_metadata_lifetime_history" ); + assert_eq!( + names.last().unwrap(), + "m20261007_000400_add_mst2_qualified_metadata_gc" + ); let db = alias_db().await; insert_repo(&db, 1, "/third-party//a").await; diff --git a/src/jupiter/storage/native_metadata_generations.rs b/src/jupiter/storage/native_metadata_generations.rs index 9ea9f766..a2cad697 100644 --- a/src/jupiter/storage/native_metadata_generations.rs +++ b/src/jupiter/storage/native_metadata_generations.rs @@ -168,6 +168,15 @@ impl PostgresMetadataGenerationRepository { &self, operation_id: &str, prepared: &PreparedNativeMetadataRetention, + ) -> Result { + self.begin_with_reopens(operation_id, prepared, &[]).await + } + + async fn begin_with_reopens( + &self, + operation_id: &str, + prepared: &PreparedNativeMetadataRetention, + reopens: &[qualified::MetadataGcReceipt], ) -> Result { validate_operation_id(operation_id)?; let plan = prepared.install_plan()?; @@ -177,8 +186,12 @@ impl PostgresMetadataGenerationRepository { let result = async { self.inner.barrier(&txn).await?; if let Some(fixed) = self.load_fixed_plan(&txn, operation_id, &digest).await? { + qualified::verify_reopen_replay(&txn, &fixed, reopens).await?; return Ok(fixed.intent); } + if !reopens.is_empty() { + qualified::reopen_in_txn(&txn, &plan, reopens).await?; + } let bindings = allocate_lifetimes(&txn, &plan, self.graph_domain).await?; let canonical_bindings = bindings.encode()?; let bindings_digest = Sha256::digest(&canonical_bindings).into(); @@ -218,6 +231,9 @@ impl PostgresMetadataGenerationRepository { )).await.map_err(internal)?; // Reuse only already LIVE graph rows. RESERVED pages without a graph // are protected by the durable mappings, without inventing a DAG. + if self.graph_domain == "qualified-v1" { + qualified::retain_existing_roots(&txn, &intent).await?; + } else { txn.execute_raw(statement( "INSERT INTO mst2_retention_root(node_id,root_key,root_kind,created_at) SELECT l.node_id,'prepare:'||p.prepare_id,'prepare',now() @@ -227,6 +243,7 @@ impl PostgresMetadataGenerationRepository { WHERE p.prepare_id=$1 ON CONFLICT(node_id,root_key) DO NOTHING", [prepare_id.into()], )).await.map_err(internal)?; + } Ok(intent) }.await; commit( @@ -288,6 +305,7 @@ impl PostgresMetadataGenerationRepository { "INSERT INTO mst2_metadata_payload(page_id,generation,metadata_codec,byte_size,payload) SELECT decode(p.page_id,'hex'),p.generation,$1,p.size,decode(p.payload,'hex') FROM jsonb_to_recordset($2::jsonb) AS p(page_id text,generation bigint,size integer,payload text) + WHERE NOT EXISTS(SELECT 1 FROM mst2_metadata_payload b WHERE b.page_id=decode(p.page_id,'hex')) ON CONFLICT(page_id) DO NOTHING", [fixed.stored.record.metadata_codec.into(),encoded.clone().into()], )).await.map_err(internal)?; @@ -379,10 +397,13 @@ impl PostgresMetadataGenerationRepository { lock_prepare(txn, intent.prepare_id()).await?; let fixed = self.require_fixed_plan(txn, intent).await?; check_generation_payloads(txn, &fixed).await?; - let legacy = self - .inner - .finalize_stored_plan_in_txn(txn, &intent.legacy, &observation.dag, fixed.stored) - .await?; + let legacy = if intent.graph_domain == GraphDomain::Qualified { + qualified::finalize_graph(txn, &fixed, &observation.dag).await? + } else { + self.inner + .finalize_stored_plan_in_txn(txn, &intent.legacy, &observation.dag, fixed.stored) + .await? + }; txn.execute_raw(statement( "UPDATE mst2_metadata_lifetime l SET state='LIVE' FROM mst2_metadata_prepare_page p WHERE p.prepare_id=$1 AND p.page_id=l.page_id @@ -407,6 +428,9 @@ impl PostgresMetadataGenerationRepository { scope: &str, ) -> Result<(), SnapshotError> { self.require_scope(&receipt.intent)?; + if receipt.intent.graph_domain == GraphDomain::Qualified { + return Err(unavailable("qualified production handoff is not admitted")); + } self.inner .verify_receipt_in_txn(txn, &receipt.legacy, tagged_root_tree_oid, scope) .await?; @@ -515,7 +539,7 @@ impl PostgresMetadataGenerationRepository { Some(fixed) if fixed.stored.record.state == "COMMITTED" => { check_generation_payloads(&txn, &fixed).await?; load_fixed_dag(&txn, &fixed).await?; - verify_graph(&txn, &fixed.stored).await?; + verify_fixed_graph(&txn, &fixed).await?; Ok(GenerationPrepareObservation::Committed(Box::new( GenerationMetadataReceipt { legacy: fixed.stored.receipt()?, @@ -704,6 +728,9 @@ async fn allocate_lifetimes( plan: &MetadataInstallPlan, graph_domain: &str, ) -> Result { + if graph_domain == "qualified-v1" { + return qualified::allocate_lifetimes(txn, plan).await; + } let pages: Vec<_> = plan .pages .iter() @@ -828,6 +855,9 @@ async fn check_lifetimes( connection: &C, stored: &StoredPlan, ) -> Result<(), SnapshotError> { + if stored.record.graph_domain.as_deref() == Some("qualified-v1") { + return qualified::check_lifetimes(connection, stored).await; + } let row=connection.query_one_raw(statement( "SELECT p.page_id,l.generation,l.state,l.metadata_codec,l.expected_size,p.generation AS expected_generation, n.state AS graph_state,n.kind AS graph_kind,n.bytes AS graph_bytes, @@ -953,3 +983,17 @@ mod tests; #[path = "native_metadata_history.rs"] pub mod history; + +#[path = "native_metadata_qualified.rs"] +pub mod qualified; + +async fn verify_fixed_graph( + connection: &C, + fixed: &FixedPlan, +) -> Result<(), SnapshotError> { + if fixed.intent.graph_domain == GraphDomain::Qualified { + qualified::verify_graph(connection, fixed).await + } else { + verify_graph(connection, &fixed.stored).await + } +} diff --git a/src/jupiter/storage/native_metadata_history.rs b/src/jupiter/storage/native_metadata_history.rs index e093a4cc..6ea7f821 100644 --- a/src/jupiter/storage/native_metadata_history.rs +++ b/src/jupiter/storage/native_metadata_history.rs @@ -39,11 +39,77 @@ impl MetadataTerminalReceipt { #[derive(Debug, Clone, PartialEq, Eq)] pub enum MetadataTerminalObservation { + /// The stored state matches this action; mutation still requires its full CAS proof. Active, Terminated(Box), } impl PostgresMetadataGenerationRepository { + /// Recover an immutable sealed intent after restart, including retired/aborted history. + /// Current lifetimes and payloads are deliberately not consulted or revived. + pub async fn capture_terminal_intent( + &self, + operation_id: &str, + manifest_digest: [u8; 32], + ) -> Result { + validate_operation_id(operation_id)?; + let txn = self.inner.transaction().await?; + self.inner.barrier(&txn).await?; + let result = async { + let record = mst2_metadata_prepare::Entity::find() + .filter(mst2_metadata_prepare::Column::OperationId.eq(operation_id)) + .one(&txn) + .await + .map_err(internal)? + .ok_or_else(|| unavailable("terminal preparation history is missing"))?; + if record.manifest_digest.as_slice() != manifest_digest { + return Err(integrity( + "terminal recovery manifest differs from durable history", + )); + } + let plan = MetadataInstallPlan::decode(&record.canonical_plan, &manifest_digest)?; + let canonical = record.canonical_bindings.as_deref().ok_or_else(|| { + unavailable("legacy unsealed preparation has no fixed terminal capability") + })?; + let bindings = GenerationBindings::decode(canonical, &plan)?; + let digest: [u8; 32] = Sha256::digest(canonical).into(); + let primary = record + .primary_scope + .as_deref() + .ok_or_else(|| integrity("terminal history has no primary scope"))?; + let domain = GraphDomain::from_stored(record.graph_domain.as_deref())?; + let intent = GenerationPrepareIntent { + legacy: MetadataPrepareIntent { + prepare_id: record.prepare_id.clone(), + operation_id: operation_id.into(), + manifest_digest, + }, + metadata_root: plan.root, + root_generation: bindings + .0 + .get(&plan.root) + .ok_or_else(|| integrity("terminal root binding is missing"))? + .0, + bindings_digest: digest, + primary_scope: primary.into(), + graph_domain: domain, + storage_seal: seal( + &record.prepare_id, + &manifest_digest, + &plan.root, + &digest, + primary, + domain.stored(), + )?, + }; + self.require_terminal_record(&txn, &intent).await?; + Ok(intent) + } + .await; + txn.rollback().await.map_err(internal)?; + result + } + pub async fn abort( &self, intent: &GenerationPrepareIntent, @@ -85,7 +151,7 @@ impl PostgresMetadataGenerationRepository { } } - async fn terminate_in_txn( + pub(super) async fn terminate_in_txn( &self, txn: &DatabaseTransaction, intent: &GenerationPrepareIntent, @@ -119,44 +185,63 @@ impl PostgresMetadataGenerationRepository { )); } check_generation_payloads(txn, &fixed).await?; - verify_graph(txn, &fixed.stored).await?; + verify_fixed_graph(txn, &fixed).await?; } - let roots = mst2_retention_root::Entity::find() - .filter( - mst2_retention_root::Column::RootKey.eq(format!("prepare:{}", intent.prepare_id())), + if intent.graph_domain == GraphDomain::Qualified { + qualified::verify_roots( + txn, + &fixed, + action == MetadataTerminalAction::RetireCoverage, ) - .limit((MetadataDagLimits::default().nodes + 1) as u64) - .all(txn) + .await?; + txn.execute_raw(statement( + "DELETE FROM mst2_metadata_graph_root WHERE prepare_id=$1 AND storage_seal=$2", + [ + intent.prepare_id().into(), + intent.storage_seal.to_vec().into(), + ], + )) .await .map_err(internal)?; - let expected_nodes: BTreeSet<_> = fixed.bindings.0.keys().map(node_id).collect(); - if roots - .iter() - .any(|root| root.root_kind != "prepare" || !expected_nodes.contains(&root.node_id)) - { - return Err(integrity( - "terminal prepare coverage differs from its fixed lifetime mappings", - )); - } - if action == MetadataTerminalAction::RetireCoverage - && roots + } else { + let roots = mst2_retention_root::Entity::find() + .filter( + mst2_retention_root::Column::RootKey + .eq(format!("prepare:{}", intent.prepare_id())), + ) + .limit((MetadataDagLimits::default().nodes + 1) as u64) + .all(txn) + .await + .map_err(internal)?; + let expected_nodes: BTreeSet<_> = fixed.bindings.0.keys().map(node_id).collect(); + if roots .iter() - .map(|root| root.node_id.clone()) - .collect::>() - != expected_nodes - { - return Err(unavailable( - "coverage retirement cannot recover missing or transferred prepare roots", - )); - } - txn.execute_raw(statement( - "DELETE FROM mst2_retention_root r USING mst2_metadata_prepare_page p + .any(|root| root.root_kind != "prepare" || !expected_nodes.contains(&root.node_id)) + { + return Err(integrity( + "terminal prepare coverage differs from its fixed lifetime mappings", + )); + } + if action == MetadataTerminalAction::RetireCoverage + && roots + .iter() + .map(|root| root.node_id.clone()) + .collect::>() + != expected_nodes + { + return Err(unavailable( + "coverage retirement cannot recover missing or transferred prepare roots", + )); + } + txn.execute_raw(statement( + "DELETE FROM mst2_retention_root r USING mst2_metadata_prepare_page p WHERE p.prepare_id=$1 AND r.root_key='prepare:'||p.prepare_id AND r.root_kind='prepare' AND r.node_id='page:sha256:'||encode(p.page_id,'hex')", - [intent.prepare_id().into()], - )) - .await - .map_err(internal)?; + [intent.prepare_id().into()], + )) + .await + .map_err(internal)?; + } let sql=match action { MetadataTerminalAction::Abort=> "UPDATE mst2_metadata_prepare SET state='ABORTED',aborted_at=clock_timestamp() @@ -212,7 +297,19 @@ impl PostgresMetadataGenerationRepository { verify_no_prepare_coverage(&txn, intent).await?; Ok(MetadataTerminalObservation::Terminated(Box::new(receipt))) } - None => Ok(MetadataTerminalObservation::Active), + None => { + let expected = match action { + MetadataTerminalAction::Abort => "PREPARING", + MetadataTerminalAction::RetireCoverage => "COMMITTED", + }; + if record.state != expected { + return Err(SnapshotError::new( + SnapshotErrorCode::Conflict, + "metadata prepare is in another terminal state", + )); + } + Ok(MetadataTerminalObservation::Active) + } } } .await; @@ -325,6 +422,22 @@ async fn verify_no_prepare_coverage( connection: &C, intent: &GenerationPrepareIntent, ) -> Result<(), SnapshotError> { + if intent.graph_domain == GraphDomain::Qualified { + if connection + .query_one_raw(statement( + "SELECT prepare_id FROM mst2_metadata_graph_root WHERE prepare_id=$1 LIMIT 1", + [intent.prepare_id().into()], + )) + .await + .map_err(internal)? + .is_some() + { + return Err(integrity( + "terminal qualified preparation still has coverage", + )); + } + return Ok(()); + } if connection .query_one_raw(statement( "SELECT node_id FROM mst2_retention_root WHERE root_key='prepare:'||$1 LIMIT 1", diff --git a/src/jupiter/storage/native_metadata_qualified.rs b/src/jupiter/storage/native_metadata_qualified.rs new file mode 100644 index 00000000..5a29c9c6 --- /dev/null +++ b/src/jupiter/storage/native_metadata_qualified.rs @@ -0,0 +1,936 @@ +//! Qualified metadata incarnations. Runtime serving adoption is intentionally closed. + +use std::ops::Deref; + +use super::*; + +pub struct PostgresQualifiedMetadataRepository { + inner: PostgresMetadataGenerationRepository, +} + +impl Deref for PostgresQualifiedMetadataRepository { + type Target = PostgresMetadataGenerationRepository; + + fn deref(&self) -> &Self::Target { + &self.inner + } +} + +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct MetadataLifetime { + page_id: [u8; 32], + generation: i64, + metadata_codec: i16, + expected_size: i32, + primary_scope: Box<[u8]>, +} + +impl MetadataLifetime { + pub fn page_id(&self) -> [u8; 32] { + self.page_id + } + + pub fn generation(&self) -> i64 { + self.generation + } +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum MetadataGcPhase { + Claim, + Apply, +} + +#[derive(Debug, thiserror::Error)] +pub enum MetadataGcError { + #[error(transparent)] + Rejected(#[from] SnapshotError), + #[error("metadata GC {phase:?} commit outcome is unknown for {operation_id}")] + CommitUncertain { + operation_id: String, + page_id: [u8; 32], + generation: i64, + phase: MetadataGcPhase, + }, +} + +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct MetadataGcClaim { + operation_id: String, + lifetime: MetadataLifetime, + graph_present: bool, + had_payload: bool, +} + +impl MetadataGcClaim { + pub fn operation_id(&self) -> &str { + &self.operation_id + } + + pub fn lifetime(&self) -> &MetadataLifetime { + &self.lifetime + } +} + +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct MetadataGcReceipt { + claim: MetadataGcClaim, + created_at: sea_orm::prelude::DateTimeWithTimeZone, + completed_at: sea_orm::prelude::DateTimeWithTimeZone, +} + +impl MetadataGcReceipt { + pub fn claim(&self) -> &MetadataGcClaim { + &self.claim + } +} + +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum MetadataGcObservation { + Absent, + Pending(Box), + Applied(Box), +} + +struct GcRecord { + claim: MetadataGcClaim, + state: String, + created_at: sea_orm::prelude::DateTimeWithTimeZone, + completed_at: Option, +} + +impl GcRecord { + fn receipt(self) -> Result { + if self.state != "APPLIED" { + return Err(unavailable("metadata GC operation is not APPLIED")); + } + Ok(MetadataGcReceipt { + claim: self.claim, + created_at: self.created_at, + completed_at: self + .completed_at + .ok_or_else(|| integrity("APPLIED metadata GC receipt has no timestamp"))?, + }) + } +} + +impl PostgresQualifiedMetadataRepository { + pub async fn new(connection: DatabaseConnection) -> Result { + Ok(Self { + inner: PostgresMetadataGenerationRepository { + inner: PostgresMetadataInstallRepository::new(connection).await?, + graph_domain: "qualified-v1", + }, + }) + } + + pub async fn lifetime(&self, page_id: [u8; 32]) -> Result { + self.inner + .inner + .verify_primary_connection(&self.inner.inner.connection) + .await?; + let row = self + .inner + .inner + .connection + .query_one_raw(statement( + "SELECT l.generation,l.metadata_codec,l.expected_size FROM mst2_metadata_current c + JOIN mst2_metadata_lifetime l USING(page_id,generation) + WHERE c.page_id=$1 AND l.graph_domain='qualified-v1'", + [page_id.to_vec().into()], + )) + .await + .map_err(internal)? + .ok_or_else(|| unavailable("qualified lifetime is missing"))?; + Ok(MetadataLifetime { + page_id, + generation: row.try_get("", "generation").map_err(internal)?, + metadata_codec: row.try_get("", "metadata_codec").map_err(internal)?, + expected_size: row.try_get("", "expected_size").map_err(internal)?, + primary_scope: self.inner.primary_scope()?.into_boxed_slice(), + }) + } + + /// Read-only recovery selector; claim still requires this tuple to be current. + pub async fn historical_lifetime( + &self, + page_id: [u8; 32], + generation: i64, + ) -> Result { + if generation <= 0 { + return Err(integrity("historical generation must be positive")); + } + self.inner + .inner + .verify_primary_connection(&self.inner.inner.connection) + .await?; + let row = self + .inner + .inner + .connection + .query_one_raw(statement( + "SELECT metadata_codec,expected_size FROM mst2_metadata_lifetime + WHERE page_id=$1 AND generation=$2 AND graph_domain='qualified-v1'", + [page_id.to_vec().into(), generation.into()], + )) + .await + .map_err(internal)? + .ok_or_else(|| unavailable("qualified historical incarnation is missing"))?; + Ok(MetadataLifetime { + page_id, + generation, + metadata_codec: row.try_get("", "metadata_codec").map_err(internal)?, + expected_size: row.try_get("", "expected_size").map_err(internal)?, + primary_scope: self.inner.primary_scope()?.into_boxed_slice(), + }) + } + + fn require_lifetime_scope(&self, lifetime: &MetadataLifetime) -> Result<(), SnapshotError> { + if lifetime.primary_scope.as_ref() != self.inner.primary_scope()?.as_slice() { + return Err(integrity( + "metadata GC belongs to another captured primary scope", + )); + } + Ok(()) + } + + /// Bounded indexed discovery only. Every returned tuple still needs a full claim proof. + pub async fn scan_current_lifetimes( + &self, + after: Option<([u8; 32], i64)>, + limit: u16, + ) -> Result, SnapshotError> { + if !(1..=256).contains(&limit) { + return Err(integrity("metadata scan limit must be 1..=256")); + } + self.inner + .inner + .verify_primary_connection(&self.inner.inner.connection) + .await?; + let (page, generation) = after.map(|(p, g)| (p.to_vec(), g)).unwrap_or_default(); + let rows=self.inner.inner.connection.query_all_raw(statement( + "SELECT l.page_id,l.generation,l.metadata_codec,l.expected_size FROM mst2_metadata_lifetime l + JOIN mst2_metadata_current c USING(page_id,generation) + WHERE l.graph_domain='qualified-v1' AND l.state IN ('RESERVED','LIVE') AND (l.page_id,l.generation)>($1,$2) + ORDER BY l.page_id,l.generation LIMIT $3", + [page.into(),generation.into(),i64::from(limit).into()], + )).await.map_err(internal)?; + let scope = self.inner.primary_scope()?.into_boxed_slice(); + rows.into_iter() + .map(|row| { + let page: Vec = row.try_get("", "page_id").map_err(internal)?; + Ok(MetadataLifetime { + page_id: page.as_slice().try_into().map_err(internal)?, + generation: row.try_get("", "generation").map_err(internal)?, + metadata_codec: row.try_get("", "metadata_codec").map_err(internal)?, + expected_size: row.try_get("", "expected_size").map_err(internal)?, + primary_scope: scope.clone(), + }) + }) + .collect() + } + + pub async fn claim( + &self, + operation_id: &str, + lifetime: &MetadataLifetime, + ) -> Result { + validate_gc_operation(operation_id)?; + self.require_lifetime_scope(lifetime)?; + let txn = self.inner.inner.transaction().await?; + let result = self.claim_in_txn(&txn, operation_id, lifetime).await; + finish_gc_transaction(txn, result, operation_id, lifetime, MetadataGcPhase::Claim).await + } + + async fn claim_in_txn( + &self, + txn: &DatabaseTransaction, + operation_id: &str, + lifetime: &MetadataLifetime, + ) -> Result { + self.require_lifetime_scope(lifetime)?; + self.inner.inner.barrier(txn).await?; + if let Some(record) = load_gc(txn, operation_id).await? { + if &record.claim.lifetime != lifetime { + return Err(integrity( + "GC operation cannot be retargeted to another lifetime", + )); + } + return Ok(record.claim); + } + let row=txn.query_one_raw(statement( + "SELECT EXISTS(SELECT 1 FROM mst2_metadata_graph_node n WHERE n.page_id=l.page_id AND n.generation=l.generation) AS graph_present, + EXISTS(SELECT 1 FROM mst2_metadata_payload b WHERE b.page_id=l.page_id) AS had_payload + FROM mst2_metadata_current c JOIN mst2_metadata_lifetime l USING(page_id,generation) + WHERE l.page_id=$1 AND l.generation=$2 AND l.graph_domain='qualified-v1' + AND l.metadata_codec=$3 AND l.expected_size=$4 FOR UPDATE OF c,l", + [lifetime.page_id.to_vec().into(),lifetime.generation.into(),lifetime.metadata_codec.into(),lifetime.expected_size.into()], + )).await.map_err(internal)?.ok_or_else(|| unavailable("GC candidate no longer names its captured current incarnation"))?; + let graph_present: bool = row.try_get("", "graph_present").map_err(internal)?; + let had_payload: bool = row.try_get("", "had_payload").map_err(internal)?; + txn.execute_raw(statement( + "INSERT INTO mst2_metadata_gc_op(operation_id,page_id,generation,primary_scope,graph_domain, + metadata_codec,expected_size,graph_present,had_payload,state) + VALUES($1::uuid,$2,$3,$4,'qualified-v1',$5,$6,$7,$8,'PENDING')", + [operation_id.into(),lifetime.page_id.to_vec().into(),lifetime.generation.into(), + lifetime.primary_scope.to_vec().into(),lifetime.metadata_codec.into(),lifetime.expected_size.into(), + graph_present.into(),had_payload.into()], + )).await.map_err(internal)?; + Ok(load_gc(txn, operation_id) + .await? + .ok_or_else(|| integrity("claimed GC operation disappeared"))? + .claim) + } + + pub async fn apply( + &self, + claim: &MetadataGcClaim, + ) -> Result { + self.require_lifetime_scope(&claim.lifetime)?; + let txn = self.inner.inner.transaction().await?; + let result = self.apply_in_txn(&txn, claim).await; + finish_gc_transaction( + txn, + result, + &claim.operation_id, + &claim.lifetime, + MetadataGcPhase::Apply, + ) + .await + } + + async fn apply_in_txn( + &self, + txn: &DatabaseTransaction, + claim: &MetadataGcClaim, + ) -> Result { + self.require_lifetime_scope(&claim.lifetime)?; + self.inner.inner.barrier(txn).await?; + let record = load_gc(txn, &claim.operation_id) + .await? + .ok_or_else(|| unavailable("metadata GC operation is missing"))?; + if &record.claim != claim { + return Err(integrity( + "metadata GC claim differs from immutable operation", + )); + } + if record.state == "APPLIED" { + return record.receipt(); + } + txn.query_one_raw(statement( + "SELECT mst2_metadata_gc_apply($1::uuid)", + [claim.operation_id.clone().into()], + )) + .await + .map_err(internal)?; + load_gc(txn, &claim.operation_id) + .await? + .ok_or_else(|| integrity("applied metadata GC operation disappeared"))? + .receipt() + } + + /// Absence is definitive only after a fresh same-primary completion barrier. + pub async fn inspect_gc( + &self, + fresh_primary: &DatabaseConnection, + operation_id: &str, + lifetime: &MetadataLifetime, + phase: MetadataGcPhase, + ) -> Result { + validate_gc_operation(operation_id)?; + self.require_lifetime_scope(lifetime)?; + let txn = fresh_primary + .begin_with_config(Some(IsolationLevel::ReadCommitted), None) + .await + .map_err(|_| gc_uncertain(operation_id, lifetime, phase))?; + if self.inner.inner.barrier(&txn).await.is_err() { + let _ = txn.rollback().await; + return Err(gc_uncertain(operation_id, lifetime, phase)); + } + let result = async { + let Some(record) = load_gc(&txn, operation_id).await? else { + return Ok(MetadataGcObservation::Absent); + }; + if &record.claim.lifetime != lifetime { + return Err(integrity( + "GC inspection differs from its captured lifetime", + )); + } + match record.state.as_str() { + "PENDING" => Ok(MetadataGcObservation::Pending(Box::new(record.claim))), + "APPLIED" => Ok(MetadataGcObservation::Applied(Box::new(record.receipt()?))), + _ => Err(integrity("unknown metadata GC operation state")), + } + } + .await; + txn.rollback() + .await + .map_err(|_| gc_uncertain(operation_id, lifetime, phase))?; + result.map_err(|error: SnapshotError| { + if error.code == SnapshotErrorCode::Internal { + gc_uncertain(operation_id, lifetime, phase) + } else { + MetadataGcError::Rejected(error) + } + }) + } + + pub async fn pending_gc(&self, limit: u16) -> Result, SnapshotError> { + if !(1..=256).contains(&limit) { + return Err(integrity("metadata pending GC limit must be 1..=256")); + } + self.inner + .inner + .verify_primary_connection(&self.inner.inner.connection) + .await?; + let rows=self.inner.inner.connection.query_all_raw(statement( + "SELECT operation_id::text,page_id,generation,primary_scope,metadata_codec,expected_size, + graph_present,had_payload,state,created_at,completed_at FROM mst2_metadata_gc_op WHERE state='PENDING' + ORDER BY created_at,operation_id LIMIT $1", + [i64::from(limit).into()], + )).await.map_err(internal)?; + let mut claims = Vec::with_capacity(rows.len()); + for row in rows { + let record = gc_record(row)?; + self.require_lifetime_scope(&record.claim.lifetime)?; + claims.push(record.claim); + } + Ok(claims) + } + + pub async fn begin_fresh_intent( + &self, + operation_id: &str, + prepared: &PreparedNativeMetadataRetention, + applied: &[MetadataGcReceipt], + ) -> Result { + if applied.is_empty() || applied.len() > MetadataDagLimits::default().nodes { + return Err(integrity("fresh begin requires bounded exact APPLIED receipts").into()); + } + for receipt in applied { + self.require_lifetime_scope(&receipt.claim.lifetime)?; + } + self.inner + .begin_with_reopens(operation_id, prepared, applied) + .await + } +} + +fn validate_gc_operation(operation: &str) -> Result<(), SnapshotError> { + let uuid = uuid::Uuid::parse_str(operation).map_err(internal)?; + if uuid.get_version_num() != 4 || uuid.to_string() != operation { + return Err(integrity("metadata GC operation must be canonical UUID v4")); + } + Ok(()) +} + +fn gc_uncertain( + operation: &str, + lifetime: &MetadataLifetime, + phase: MetadataGcPhase, +) -> MetadataGcError { + MetadataGcError::CommitUncertain { + operation_id: operation.into(), + page_id: lifetime.page_id, + generation: lifetime.generation, + phase, + } +} + +async fn finish_gc_transaction( + txn: DatabaseTransaction, + result: Result, + operation: &str, + lifetime: &MetadataLifetime, + phase: MetadataGcPhase, +) -> Result { + match result { + Ok(value) => { + txn.commit() + .await + .map_err(|_| gc_uncertain(operation, lifetime, phase))?; + Ok(value) + } + Err(error) => { + txn.rollback().await.map_err(internal)?; + Err(error.into()) + } + } +} + +async fn load_gc( + connection: &C, + operation: &str, +) -> Result, SnapshotError> { + let Some(row)=connection.query_one_raw(statement( + "SELECT operation_id::text,page_id,generation,primary_scope,metadata_codec,expected_size, + graph_present,had_payload,state,created_at,completed_at FROM mst2_metadata_gc_op WHERE operation_id=$1::uuid", + [operation.into()], + )).await.map_err(internal)? else { return Ok(None); }; + Ok(Some(gc_record(row)?)) +} + +fn gc_record(row: QueryResult) -> Result { + let page: Vec = row.try_get("", "page_id").map_err(internal)?; + Ok(GcRecord { + claim: MetadataGcClaim { + operation_id: row.try_get("", "operation_id").map_err(internal)?, + lifetime: MetadataLifetime { + page_id: page.as_slice().try_into().map_err(internal)?, + generation: row.try_get("", "generation").map_err(internal)?, + metadata_codec: row.try_get("", "metadata_codec").map_err(internal)?, + expected_size: row.try_get("", "expected_size").map_err(internal)?, + primary_scope: row + .try_get::>("", "primary_scope") + .map_err(internal)? + .into_boxed_slice(), + }, + graph_present: row.try_get("", "graph_present").map_err(internal)?, + had_payload: row.try_get("", "had_payload").map_err(internal)?, + }, + state: row.try_get("", "state").map_err(internal)?, + created_at: row.try_get("", "created_at").map_err(internal)?, + completed_at: row.try_get("", "completed_at").map_err(internal)?, + }) +} + +async fn validate_applied_receipts( + connection: &C, + receipts: &[MetadataGcReceipt], +) -> Result<(), SnapshotError> { + if receipts.is_empty() { + return Ok(()); + } + let operations: Vec<_> = receipts + .iter() + .map(|r| r.claim.operation_id.as_str()) + .collect(); + let rows=connection.query_all_raw(statement( + "SELECT operation_id::text,page_id,generation,primary_scope,metadata_codec,expected_size, + graph_present,had_payload,state,created_at,completed_at FROM mst2_metadata_gc_op + WHERE operation_id IN (SELECT jsonb_array_elements_text($1::jsonb)::uuid) LIMIT 4097", + [serde_json::to_string(&operations).map_err(internal)?.into()], + )).await.map_err(internal)?; + let mut actual = BTreeMap::new(); + for row in rows { + let record = gc_record(row)?; + actual.insert(record.claim.operation_id.clone(), record.receipt()?); + } + if actual.len() != receipts.len() + || receipts + .iter() + .any(|r| actual.get(&r.claim.operation_id) != Some(r)) + { + return Err(integrity( + "fresh proof differs from exact immutable APPLIED operation and timestamps", + )); + } + Ok(()) +} + +pub(super) async fn reopen_in_txn( + txn: &DatabaseTransaction, + plan: &MetadataInstallPlan, + receipts: &[MetadataGcReceipt], +) -> Result<(), SnapshotError> { + let mut pages = BTreeSet::new(); + validate_applied_receipts(txn, receipts).await?; + for receipt in receipts { + let lifetime = &receipt.claim.lifetime; + if !pages.insert(lifetime.page_id) + || plan.pages.get(&lifetime.page_id).copied() != Some(lifetime.expected_size as u64) + || plan.identity.metadata_codec as i16 != lifetime.metadata_codec + { + return Err(integrity( + "fresh receipts are not unique exact members of the new plan", + )); + } + lifetime + .generation + .checked_add(1) + .ok_or_else(|| unavailable("metadata generation watermark exhausted"))?; + } + let operations: Vec<_> = receipts + .iter() + .map(|r| r.claim.operation_id.as_str()) + .collect(); + let encoded = serde_json::to_string(&operations).map_err(internal)?; + txn.execute_raw(statement( + "INSERT INTO mst2_metadata_lifetime(page_id,node_id,generation,state,metadata_codec,expected_size,graph_domain) + SELECT o.page_id,'page:sha256:'||encode(o.page_id,'hex'),o.generation+1,'RESERVED',o.metadata_codec,o.expected_size,'qualified-v1' + FROM mst2_metadata_gc_op o WHERE operation_id IN (SELECT jsonb_array_elements_text($1::jsonb)::uuid) + ORDER BY o.page_id ON CONFLICT(page_id,generation) DO NOTHING", + [encoded.clone().into()], + )).await.map_err(internal)?; + let changed=txn.execute_raw(statement( + "UPDATE mst2_metadata_current c SET generation=o.generation+1 FROM mst2_metadata_gc_op o + WHERE o.operation_id IN (SELECT jsonb_array_elements_text($1::jsonb)::uuid) + AND c.page_id=o.page_id AND c.generation=o.generation", + [encoded.into()], + )).await.map_err(internal)?; + if changed.rows_affected() != receipts.len() as u64 { + return Err(unavailable( + "fresh receipts do not name all current removed incarnations", + )); + } + Ok(()) +} + +pub(super) async fn verify_reopen_replay( + connection: &C, + fixed: &FixedPlan, + receipts: &[MetadataGcReceipt], +) -> Result<(), SnapshotError> { + validate_applied_receipts(connection, receipts).await?; + let mut pages = BTreeSet::new(); + for receipt in receipts { + let old = &receipt.claim.lifetime; + if !pages.insert(old.page_id) + || old.primary_scope != fixed.intent.primary_scope + || old.generation.checked_add(1) != fixed.bindings.0.get(&old.page_id).map(|b| b.0) + || Some(old.expected_size as u64) != fixed.bindings.0.get(&old.page_id).map(|b| b.1) + { + return Err(integrity( + "fresh replay receipts differ from the fixed new incarnation bindings", + )); + } + } + Ok(()) +} + +pub(super) async fn allocate_lifetimes( + txn: &DatabaseTransaction, + plan: &MetadataInstallPlan, +) -> Result { + let pages: Vec<_> = plan + .pages + .iter() + .map(|(id, size)| json!({"id":hex::encode(id),"size":size})) + .collect(); + let encoded = serde_json::to_string(&pages).map_err(internal)?; + if txn.query_one_raw(statement( + "SELECT l.page_id FROM jsonb_to_recordset($1::jsonb) p(id text,size integer) + JOIN mst2_metadata_current c ON c.page_id=decode(p.id,'hex') + JOIN mst2_metadata_lifetime l USING(page_id,generation) WHERE l.graph_domain<>'qualified-v1' LIMIT 1", + [encoded.clone().into()], + )).await.map_err(internal)?.is_some() { + return Err(unavailable("metadata incarnation permanently belongs to another graph domain")); + } + txn.execute_raw(statement( + "INSERT INTO mst2_metadata_lifetime(page_id,node_id,generation,state,metadata_codec,expected_size,graph_domain) + SELECT decode(p.id,'hex'),'page:sha256:'||p.id,1,'RESERVED',$1,p.size,'qualified-v1' + FROM jsonb_to_recordset($2::jsonb) p(id text,size integer) + WHERE NOT EXISTS(SELECT 1 FROM mst2_metadata_current c WHERE c.page_id=decode(p.id,'hex')) + ON CONFLICT(page_id,generation) DO NOTHING", + [(plan.identity.metadata_codec as i16).into(),encoded.clone().into()], + )).await.map_err(internal)?; + txn.execute_raw(statement( + "INSERT INTO mst2_metadata_current(page_id,generation) SELECT decode(p.id,'hex'),1 + FROM jsonb_to_recordset($1::jsonb) p(id text,size integer) + WHERE NOT EXISTS(SELECT 1 FROM mst2_metadata_current c WHERE c.page_id=decode(p.id,'hex'))", + [encoded.clone().into()], + )).await.map_err(internal)?; + let rows=txn.query_all_raw(statement( + "SELECT l.page_id,l.generation,l.state,l.graph_domain,l.metadata_codec,l.expected_size,p.size, + n.state AS graph_state,n.metadata_codec AS graph_codec,n.bytes AS graph_bytes, + b.page_id IS NOT NULL AS payload_present,b.generation AS payload_generation,b.metadata_codec AS payload_codec,b.byte_size AS payload_size, + EXISTS(SELECT 1 FROM mst2_metadata_gc_op o WHERE o.page_id=l.page_id AND o.generation=l.generation) AS tombstone, + EXISTS(SELECT 1 FROM mst2_retention_node x WHERE x.node_id=l.node_id) + OR EXISTS(SELECT 1 FROM mst2_retention_gc_op x WHERE x.node_id=l.node_id) AS generic_graph + FROM jsonb_to_recordset($1::jsonb) p(id text,size integer) + JOIN mst2_metadata_current c ON c.page_id=decode(p.id,'hex') + JOIN mst2_metadata_lifetime l USING(page_id,generation) + LEFT JOIN mst2_metadata_graph_node n USING(page_id,generation) + LEFT JOIN mst2_metadata_payload b ON b.page_id=l.page_id ORDER BY l.page_id", + [encoded.into()], + )).await.map_err(internal)?; + let mut bindings = BTreeMap::new(); + for row in rows { + let state: String = row.try_get("", "state").map_err(internal)?; + let generation: i64 = row.try_get("", "generation").map_err(internal)?; + let size: i32 = row.try_get("", "size").map_err(internal)?; + let graph: Option = row.try_get("", "graph_state").map_err(internal)?; + let payload: bool = row.try_get("", "payload_present").map_err(internal)?; + if !["RESERVED", "LIVE"].contains(&state.as_str()) + || row.try_get::("", "tombstone").map_err(internal)? + || row.try_get::("", "generic_graph").map_err(internal)? + || graph.as_deref().is_some_and(|s| s != "LIVE") + { + return Err(unavailable( + "qualified incarnation is removed, deleting, or has generic evidence", + )); + } + if state == "LIVE" && (!payload || graph.is_none()) { + return Err(unavailable( + "LIVE qualified incarnation lost bytes or graph", + )); + } + if row + .try_get::("", "graph_domain") + .map_err(internal)? + != "qualified-v1" + || row.try_get::("", "metadata_codec").map_err(internal)? + != plan.identity.metadata_codec as i16 + || row.try_get::("", "expected_size").map_err(internal)? != size + || (graph.is_some() + && (row + .try_get::>("", "graph_codec") + .map_err(internal)? + != Some(plan.identity.metadata_codec as i16) + || row + .try_get::>("", "graph_bytes") + .map_err(internal)? + != Some(i64::from(size)))) + || (payload + && (row + .try_get::>("", "payload_generation") + .map_err(internal)? + != Some(generation) + || row + .try_get::>("", "payload_codec") + .map_err(internal)? + != Some(plan.identity.metadata_codec as i16) + || row + .try_get::>("", "payload_size") + .map_err(internal)? + != Some(size))) + { + return Err(integrity( + "qualified incarnation immutable profile mismatch", + )); + } + let page: Vec = row.try_get("", "page_id").map_err(internal)?; + bindings.insert( + page.as_slice().try_into().map_err(internal)?, + (generation, size as u64), + ); + } + if bindings.len() != plan.pages.len() { + return Err(integrity("qualified lifetime allocation is incomplete")); + } + Ok(GenerationBindings(bindings)) +} + +pub(super) async fn retain_existing_roots( + txn: &DatabaseTransaction, + intent: &GenerationPrepareIntent, +) -> Result<(), SnapshotError> { + txn.execute_raw(statement( + "INSERT INTO mst2_metadata_graph_root(prepare_id,storage_seal,page_id,generation) + SELECT q.prepare_id,q.storage_seal,m.page_id,m.generation FROM mst2_metadata_prepare q + JOIN mst2_metadata_prepare_page m USING(prepare_id) + JOIN mst2_metadata_graph_node n USING(page_id,generation) WHERE q.prepare_id=$1 AND n.state='LIVE' + ON CONFLICT(prepare_id,page_id,generation) DO NOTHING", + [intent.prepare_id().into()], + )).await.map_err(internal)?; + Ok(()) +} + +pub(super) async fn check_lifetimes( + connection: &C, + stored: &StoredPlan, +) -> Result<(), SnapshotError> { + if connection.query_one_raw(statement( + "SELECT m.page_id FROM mst2_metadata_prepare_page m + LEFT JOIN mst2_metadata_current c ON c.page_id=m.page_id AND c.generation=m.generation + LEFT JOIN mst2_metadata_lifetime l ON l.page_id=m.page_id AND l.generation=m.generation + LEFT JOIN mst2_metadata_graph_node n ON n.page_id=m.page_id AND n.generation=m.generation + WHERE m.prepare_id=$1 AND (c.page_id IS NULL OR l.page_id IS NULL OR l.graph_domain<>'qualified-v1' + OR l.state NOT IN ('RESERVED','LIVE') OR l.metadata_codec<>$2 OR l.expected_size<>m.expected_size + OR ($3='COMMITTED' AND l.state<>'LIVE') OR (l.state='LIVE' AND n.page_id IS NULL) + OR n.state<>'LIVE' OR n.metadata_codec<>$2 OR n.bytes<>m.expected_size + OR EXISTS(SELECT 1 FROM mst2_metadata_gc_op o WHERE o.page_id=m.page_id AND o.generation=m.generation) + OR EXISTS(SELECT 1 FROM mst2_retention_node x WHERE x.node_id=l.node_id) + OR EXISTS(SELECT 1 FROM mst2_retention_gc_op x WHERE x.node_id=l.node_id)) LIMIT 1", + [stored.record.prepare_id.clone().into(),stored.record.metadata_codec.into(),stored.record.state.clone().into()], + )).await.map_err(internal)?.is_some() { return Err(unavailable("fixed qualified incarnation is no longer installable")); } + Ok(()) +} + +pub(super) async fn finalize_graph( + txn: &DatabaseTransaction, + fixed: &FixedPlan, + dag: &ValidatedMetadataDag, +) -> Result { + let actual_pages: BTreeSet<_> = dag.payloads().iter().map(|p| (p.id, p.size)).collect(); + let actual_edges: BTreeSet<_> = dag + .edges() + .iter() + .map(|e| (e.parent.clone(), e.child.clone())) + .collect(); + if dag.root() != fixed.stored.plan.root + || actual_pages + != fixed + .stored + .plan + .pages + .iter() + .map(|(p, s)| (*p, *s)) + .collect() + || actual_edges + != fixed + .stored + .plan + .edges + .iter() + .map(|(p, c)| (node_id(p), node_id(c))) + .collect() + { + return Err(integrity( + "qualified DAG observation differs from immutable plan", + )); + } + if fixed.stored.record.state == "COMMITTED" { + verify_graph(txn, fixed).await?; + return fixed.stored.receipt(); + } + txn.execute_raw(statement( + "INSERT INTO mst2_metadata_graph_node(page_id,generation,state,metadata_codec,bytes) + SELECT m.page_id,m.generation,'LIVE',$2,m.expected_size FROM mst2_metadata_prepare_page m + WHERE m.prepare_id=$1 AND NOT EXISTS(SELECT 1 FROM mst2_metadata_graph_node n WHERE n.page_id=m.page_id AND n.generation=m.generation)", + [fixed.intent.prepare_id().into(),fixed.stored.record.metadata_codec.into()], + )).await.map_err(internal)?; + let edges:Vec<_>=fixed.stored.plan.edges.iter().map(|(p,c)|json!({"parent":hex::encode(p),"pg":fixed.bindings.0[p].0,"child":hex::encode(c),"cg":fixed.bindings.0[c].0})).collect(); + txn.execute_raw(statement( + "INSERT INTO mst2_metadata_graph_edge(parent_page,parent_generation,child_page,child_generation) + SELECT decode(e.parent,'hex'),e.pg,decode(e.child,'hex'),e.cg + FROM jsonb_to_recordset($1::jsonb) e(parent text,pg bigint,child text,cg bigint) + ON CONFLICT(parent_page,parent_generation,child_page,child_generation) DO NOTHING", + [serde_json::to_string(&edges).map_err(internal)?.into()], + )).await.map_err(internal)?; + retain_existing_roots(txn, &fixed.intent).await?; + verify_graph(txn, fixed).await?; + let changed = txn + .execute_raw(statement( + "UPDATE mst2_metadata_prepare SET state='COMMITTED',committed_at=clock_timestamp() + WHERE prepare_id=$1 AND state='PREPARING' AND storage_seal=$2", + [ + fixed.intent.prepare_id().into(), + fixed.intent.storage_seal.to_vec().into(), + ], + )) + .await + .map_err(internal)?; + if changed.rows_affected() != 1 { + return Err(integrity("qualified finalize lost its prepare CAS")); + } + fixed.stored.receipt() +} + +pub(super) async fn verify_graph( + connection: &C, + fixed: &FixedPlan, +) -> Result<(), SnapshotError> { + let rows=connection.query_all_raw(statement( + "SELECT n.page_id,n.generation,n.state,n.metadata_codec,n.bytes,n.incoming_refs, + (SELECT count(*) FROM mst2_metadata_graph_edge e WHERE e.child_page=n.page_id AND e.child_generation=n.generation) AS actual_refs + FROM mst2_metadata_prepare_page m JOIN mst2_metadata_graph_node n USING(page_id,generation) + WHERE m.prepare_id=$1 ORDER BY n.page_id LIMIT 4097", + [fixed.intent.prepare_id().into()], + )).await.map_err(internal)?; + if rows.len() != fixed.bindings.0.len() { + return Err(unavailable("qualified graph coverage is incomplete")); + } + for row in rows { + let page: Vec = row.try_get("", "page_id").map_err(internal)?; + let page: [u8; 32] = page.as_slice().try_into().map_err(internal)?; + let &(generation, size) = fixed + .bindings + .0 + .get(&page) + .ok_or_else(|| integrity("qualified graph has an extra node"))?; + if row.try_get::("", "generation").map_err(internal)? != generation + || row.try_get::("", "state").map_err(internal)? != "LIVE" + || row.try_get::("", "metadata_codec").map_err(internal)? + != fixed.stored.record.metadata_codec + || row.try_get::("", "bytes").map_err(internal)? != size as i64 + || row.try_get::("", "incoming_refs").map_err(internal)? + != row.try_get::("", "actual_refs").map_err(internal)? + { + return Err(integrity("qualified graph profile or counter drift")); + } + } + let rows=connection.query_all_raw(statement( + "SELECT e.parent_page,e.parent_generation,e.child_page,e.child_generation FROM mst2_metadata_prepare_page m + JOIN mst2_metadata_graph_edge e ON e.parent_page=m.page_id AND e.parent_generation=m.generation + WHERE m.prepare_id=$1 LIMIT 16385", + [fixed.intent.prepare_id().into()], + )).await.map_err(internal)?; + let mut actual = BTreeSet::new(); + for row in rows { + let parent: Vec = row.try_get("", "parent_page").map_err(internal)?; + let child: Vec = row.try_get("", "child_page").map_err(internal)?; + actual.insert(( + parent, + row.try_get::("", "parent_generation") + .map_err(internal)?, + child, + row.try_get::("", "child_generation") + .map_err(internal)?, + )); + } + let expected: BTreeSet<_> = fixed + .stored + .plan + .edges + .iter() + .map(|(p, c)| { + ( + p.to_vec(), + fixed.bindings.0[p].0, + c.to_vec(), + fixed.bindings.0[c].0, + ) + }) + .collect(); + if actual != expected { + return Err(integrity( + "qualified graph edges differ from immutable plan", + )); + } + verify_roots(connection, fixed, true).await +} + +pub(super) async fn verify_roots( + connection: &C, + fixed: &FixedPlan, + complete: bool, +) -> Result<(), SnapshotError> { + let rows=connection.query_all_raw(statement( + "SELECT page_id,generation,storage_seal FROM mst2_metadata_graph_root WHERE prepare_id=$1 LIMIT 4097", + [fixed.intent.prepare_id().into()], + )).await.map_err(internal)?; + let mut actual = BTreeSet::new(); + for row in rows { + let page: Vec = row.try_get("", "page_id").map_err(internal)?; + let page: [u8; 32] = page.as_slice().try_into().map_err(internal)?; + let generation: i64 = row.try_get("", "generation").map_err(internal)?; + let seal: Vec = row.try_get("", "storage_seal").map_err(internal)?; + if seal.as_slice() != fixed.intent.storage_seal + || fixed.bindings.0.get(&page).map(|b| b.0) != Some(generation) + { + return Err(integrity( + "qualified prepare root differs from its immutable seal", + )); + } + actual.insert((page, generation)); + } + if complete + && actual + != fixed + .bindings + .0 + .iter() + .map(|(p, (g, _))| (*p, *g)) + .collect() + { + return Err(unavailable( + "qualified prepare lost or transferred original roots", + )); + } + Ok(()) +} + +#[cfg(test)] +#[path = "native_metadata_qualified_tests.rs"] +mod tests; diff --git a/src/jupiter/storage/native_metadata_qualified_tests.rs b/src/jupiter/storage/native_metadata_qualified_tests.rs new file mode 100644 index 00000000..a7b73493 --- /dev/null +++ b/src/jupiter/storage/native_metadata_qualified_tests.rs @@ -0,0 +1,1261 @@ +use std::sync::Arc; + +use mst2_codec::metapage::{Entry, EntryKind}; +use sea_orm::Database; +use sea_orm_migration::MigratorTrait; + +use super::{ + super::history::{MetadataTerminalAction, MetadataTerminalError, MetadataTerminalObservation}, + *, +}; +use crate::{ + ceres::snapshot::retention_dag::MetadataDagBuilder, + jupiter::{ + migration::Migrator, + tests::{TestSchemaGuard, test_db_config}, + }, +}; + +fn prepared() -> PreparedNativeMetadataRetention { + let entries = [Entry::file(EntryKind::Regular, b"file", 3, [42; 32])]; + let child = Page::build(&entries).unwrap(); + let roots = [ + Entry::dir(b"one", page_id(&child)), + Entry::dir(b"two", page_id(&child)), + ]; + let root = Page::build(&roots).unwrap(); + let mut builder = MetadataDagBuilder::new(MetadataDagLimits::default()); + builder.add_directory(&child, &entries).unwrap(); + builder.add_directory(&root, &roots).unwrap(); + PreparedNativeMetadataRetention::test_installation( + Arc::new(builder.finish(page_id(&root)).unwrap()), + "/", + ) +} + +async fn fixture() -> (DatabaseConnection, DatabaseConnection, TestSchemaGuard) { + let temp = tempfile::tempdir().unwrap(); + let (config, schema) = test_db_config(temp.path()).await; + let first = Database::connect(config.db_url.clone()).await.unwrap(); + Migrator::up(&first, None).await.unwrap(); + let second = Database::connect(config.db_url).await.unwrap(); + (first, second, schema) +} + +async fn scalar(db: &C, sql: &str) -> i64 { + db.query_one_raw(statement(sql, [])) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap() +} + +async fn installed( + db: &DatabaseConnection, + operation: &str, +) -> ( + PostgresQualifiedMetadataRepository, + PreparedNativeMetadataRetention, + GenerationMetadataReceipt, +) { + let repo = PostgresQualifiedMetadataRepository::new(db.clone()) + .await + .unwrap(); + let pages = prepared(); + let intent = repo.begin_intent(operation, &pages).await.unwrap(); + repo.install_pages(&intent, pages.dag().payloads()) + .await + .unwrap(); + let receipt = repo.finalize(&intent).await.unwrap(); + (repo, pages, receipt) +} + +async fn claim_root( + repo: &PostgresQualifiedMetadataRepository, + receipt: &GenerationMetadataReceipt, +) -> MetadataGcClaim { + let lifetime = repo.lifetime(receipt.metadata_root()).await.unwrap(); + repo.claim(&uuid::Uuid::new_v4().to_string(), &lifetime) + .await + .unwrap() +} + +async fn wait_for_waiter(txn: &DatabaseTransaction) { + let deadline = tokio::time::Instant::now() + Duration::from_secs(4); + loop { + let waiting:bool=txn.query_one_raw(statement( + "SELECT EXISTS(SELECT 1 FROM pg_locks WHERE locktype='advisory' AND NOT granted + AND classid=$1::integer::oid AND objid=hashtext(current_schema())::oid + AND database=(SELECT oid FROM pg_database WHERE datname=current_database())) AS waiting", + [RETENTION_LOCK_KEY.into()], + )).await.unwrap().unwrap().try_get("","waiting").unwrap(); + if waiting { + break; + } + assert!( + tokio::time::Instant::now() < deadline, + "real independent connection must wait on retention barrier" + ); + tokio::task::yield_now().await; + } +} + +#[tokio::test] +async fn qualified_graph_exact_fks_unique_edges_and_permanent_domains() { + let (first, second, _schema) = fixture().await; + let (repo, pages, receipt) = installed(&first, "qualified-graph").await; + repo.install_pages(receipt.intent(), pages.dag().payloads()) + .await + .unwrap(); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_graph_node").await, + 2 + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_graph_edge").await, + 1 + ); + assert_eq!( + scalar( + &first, + "SELECT sum(incoming_refs)::bigint FROM mst2_metadata_graph_node" + ) + .await, + 1 + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_retention_node").await, + 0 + ); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_lifetime WHERE graph_domain='qualified-v1'" + ) + .await, + 2 + ); + repo.retire_prepare_coverage(&receipt).await.unwrap(); + let generic = PostgresMetadataGenerationRepository::new(second.clone()) + .await + .unwrap(); + assert!( + generic + .begin_intent("cannot-adopt-retired-qualified", &pages) + .await + .is_err() + ); + for sql in [ + "UPDATE mst2_metadata_lifetime SET graph_domain='generic-v1'", + "UPDATE mst2_metadata_graph_node SET incoming_refs=incoming_refs+1", + "UPDATE mst2_metadata_graph_edge SET child_generation=child_generation+1", + "DELETE FROM mst2_metadata_current", + "UPDATE mst2_metadata_payload SET payload=payload", + "DELETE FROM mst2_metadata_storage_scope", + ] { + assert!(second.execute_unprepared(sql).await.is_err(), "{sql}"); + } + let claim = claim_root(&repo, &receipt).await; + let applied = repo.apply(&claim).await.unwrap(); + assert_eq!(applied.claim(), &claim); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_gc_op WHERE state='APPLIED'" + ) + .await, + 1 + ); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_prepare_page WHERE generation=1" + ) + .await, + 2 + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_lifetime").await, + 2 + ); +} + +#[tokio::test] +async fn qualified_shared_child_and_other_prepare_are_real_collection_vetoes() { + let (first, second, _schema) = fixture().await; + let (repo, pages, receipt) = installed(&first, "shared-owner-one").await; + let other = PostgresQualifiedMetadataRepository::new(second) + .await + .unwrap(); + let intent = other + .begin_intent("shared-owner-two", &pages) + .await + .unwrap(); + let shared = other.finalize(&intent).await.unwrap(); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_graph_root").await, + 4 + ); + repo.retire_prepare_coverage(&receipt).await.unwrap(); + let root = repo.lifetime(receipt.metadata_root()).await.unwrap(); + assert!( + repo.claim(&uuid::Uuid::new_v4().to_string(), &root) + .await + .is_err() + ); + other.retire_prepare_coverage(&shared).await.unwrap(); + let child = pages + .dag() + .payloads() + .iter() + .find(|p| p.id != receipt.metadata_root()) + .unwrap(); + let child = repo.lifetime(child.id).await.unwrap(); + assert!( + repo.claim(&uuid::Uuid::new_v4().to_string(), &child) + .await + .is_err() + ); + let root_claim = repo + .claim(&uuid::Uuid::new_v4().to_string(), &root) + .await + .unwrap(); + repo.apply(&root_claim).await.unwrap(); + assert_eq!( + scalar(&first, "SELECT incoming_refs FROM mst2_metadata_graph_node").await, + 0 + ); + let child_claim = repo + .claim(&uuid::Uuid::new_v4().to_string(), &child) + .await + .unwrap(); + repo.apply(&child_claim).await.unwrap(); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_payload").await, + 0 + ); +} + +#[tokio::test] +async fn qualified_pending_preparation_and_lost_roots_fail_closed() { + let (first, second, _schema) = fixture().await; + let (repo, pages, receipt) = installed(&first, "lost-roots").await; + second + .execute_raw(statement( + "DELETE FROM mst2_metadata_graph_root WHERE prepare_id=$1", + [receipt.intent().prepare_id().into()], + )) + .await + .unwrap(); + assert!(repo.retire_prepare_coverage(&receipt).await.is_err()); + let lifetime = repo.lifetime(receipt.metadata_root()).await.unwrap(); + assert!( + repo.claim(&uuid::Uuid::new_v4().to_string(), &lifetime) + .await + .is_err() + ); + let preparing = repo.begin_intent("pending-owner", &pages).await.unwrap(); + assert!( + repo.claim(&uuid::Uuid::new_v4().to_string(), &lifetime) + .await + .is_err() + ); + repo.abort(&preparing).await.unwrap(); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_payload").await, + 2 + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_gc_op").await, + 0 + ); +} + +#[tokio::test] +async fn qualified_explicit_aborted_orphans_with_and_without_payload_collect() { + for partial in [false, true] { + let (first, second, _schema) = fixture().await; + let repo = PostgresQualifiedMetadataRepository::new(first.clone()) + .await + .unwrap(); + let pages = prepared(); + let intent = repo.begin_intent("orphan", &pages).await.unwrap(); + if partial { + repo.install_pages(&intent, pages.dag().payloads()) + .await + .unwrap(); + } + let lifetime = repo.lifetime(pages.dag().root()).await.unwrap(); + assert!( + repo.claim(&uuid::Uuid::new_v4().to_string(), &lifetime) + .await + .is_err() + ); + repo.abort(&intent).await.unwrap(); + let claim = repo + .claim(&uuid::Uuid::new_v4().to_string(), &lifetime) + .await + .unwrap(); + assert!(!claim.graph_present); + assert_eq!(claim.had_payload, partial); + repo.apply(&claim).await.unwrap(); + assert_eq!( + repo.inspect_gc( + &second, + claim.operation_id(), + &lifetime, + MetadataGcPhase::Apply + ) + .await + .unwrap(), + MetadataGcObservation::Applied(Box::new(repo.apply(&claim).await.unwrap())) + ); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_lifetime WHERE state='REMOVED'" + ) + .await, + 1 + ); + } +} + +#[tokio::test] +async fn qualified_abort_and_collection_fence_a_late_installer_at_actual_pg_lock() { + let (first, second, _schema) = fixture().await; + let repo = PostgresQualifiedMetadataRepository::new(first.clone()) + .await + .unwrap(); + let pages = prepared(); + let intent = repo.begin_intent("late-installer", &pages).await.unwrap(); + repo.install_pages(&intent, pages.dag().payloads()) + .await + .unwrap(); + let lifetime = repo.lifetime(pages.dag().root()).await.unwrap(); + let held = first.begin().await.unwrap(); + repo.inner.inner.barrier(&held).await.unwrap(); + let late_intent = intent.clone(); + let late_pages = pages.dag().payloads().to_vec(); + let late = tokio::spawn(async move { + let late_repo = PostgresQualifiedMetadataRepository::new(second) + .await + .unwrap(); + late_repo.install_pages(&late_intent, &late_pages).await + }); + wait_for_waiter(&held).await; + repo.terminate_in_txn(&held, &intent, MetadataTerminalAction::Abort, None) + .await + .unwrap(); + let claim = repo + .claim_in_txn(&held, &uuid::Uuid::new_v4().to_string(), &lifetime) + .await + .unwrap(); + repo.apply_in_txn(&held, &claim).await.unwrap(); + held.commit().await.unwrap(); + assert!(late.await.unwrap().is_err()); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_payload").await, + 1 + ); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_gc_op WHERE state='APPLIED'" + ) + .await, + 1 + ); +} + +#[tokio::test] +async fn qualified_statement_barrier_raw_writer_waits_before_row_lock_and_rechecks_domain() { + for generic in [true, false] { + let (first, second, _schema) = fixture().await; + let (repo, _pages, receipt) = installed(&first, "raw-writer").await; + repo.retire_prepare_coverage(&receipt).await.unwrap(); + let root = repo.lifetime(receipt.metadata_root()).await.unwrap(); + let held = first.begin().await.unwrap(); + repo.inner.inner.barrier(&held).await.unwrap(); + let sql = if generic { + format!( + "INSERT INTO mst2_retention_node(node_id,kind,bytes,state) VALUES('page:sha256:{}','page',{},'LIVE')", + hex::encode(root.page_id), + root.expected_size + ) + } else { + format!( + "UPDATE mst2_metadata_graph_node SET state='LIVE' WHERE page_id=decode('{}','hex') AND generation={}", + hex::encode(root.page_id), + root.generation + ) + }; + let writer = tokio::spawn(async move { second.execute_unprepared(&sql).await }); + wait_for_waiter(&held).await; + held.query_one_raw(statement("SELECT page_id FROM mst2_metadata_graph_node WHERE page_id=$1 AND generation=$2 FOR UPDATE NOWAIT", + [root.page_id.to_vec().into(),root.generation.into()])).await.unwrap().unwrap(); + let claim = repo + .claim_in_txn(&held, &uuid::Uuid::new_v4().to_string(), &root) + .await + .unwrap(); + held.commit().await.unwrap(); + assert!(writer.await.unwrap().is_err()); + repo.apply(&claim).await.unwrap(); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_retention_node").await, + 0 + ); + } +} + +#[tokio::test] +async fn qualified_claim_outer_rollback_and_lost_commit_response_use_fresh_barrier() { + let (first, second, _schema) = fixture().await; + let (repo, _pages, receipt) = installed(&first, "claim-recovery").await; + repo.retire_prepare_coverage(&receipt).await.unwrap(); + let lifetime = repo.lifetime(receipt.metadata_root()).await.unwrap(); + let id = uuid::Uuid::new_v4().to_string(); + let held = first.begin().await.unwrap(); + repo.claim_in_txn(&held, &id, &lifetime).await.unwrap(); + held.rollback().await.unwrap(); + assert_eq!( + repo.inspect_gc(&second, &id, &lifetime, MetadataGcPhase::Claim) + .await + .unwrap(), + MetadataGcObservation::Absent + ); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_graph_node WHERE state='LIVE'" + ) + .await, + 2 + ); + let held = first.begin().await.unwrap(); + let claim = repo.claim_in_txn(&held, &id, &lifetime).await.unwrap(); + held.commit().await.unwrap(); // The caller deliberately discards the commit response. + assert_eq!( + repo.inspect_gc(&second, &id, &lifetime, MetadataGcPhase::Claim) + .await + .unwrap(), + MetadataGcObservation::Pending(Box::new(claim.clone())) + ); + assert_eq!(repo.pending_gc(1).await.unwrap(), vec![claim.clone()]); + repo.apply(&claim).await.unwrap(); +} + +#[tokio::test] +async fn qualified_apply_faults_rollback_bytes_graph_counters_and_receipt_with_guards_enabled() { + for (table, event, predicate) in [ + ("mst2_metadata_payload", "AFTER DELETE", "true"), + ("mst2_metadata_graph_edge", "AFTER DELETE", "true"), + ( + "mst2_metadata_gc_op", + "BEFORE UPDATE", + "NEW.state='APPLIED'", + ), + ] { + let (first, second, _schema) = fixture().await; + let (repo, _pages, receipt) = installed(&first, "apply-fault").await; + repo.retire_prepare_coverage(&receipt).await.unwrap(); + let claim = claim_root(&repo, &receipt).await; + let granularity = if table == "mst2_metadata_graph_edge" { + "STATEMENT" + } else { + "ROW" + }; + let sql = format!( + "CREATE FUNCTION test_gc_fault() RETURNS trigger LANGUAGE plpgsql AS $$ BEGIN IF {predicate} THEN RAISE EXCEPTION 'injected atomic GC failure'; END IF; RETURN NEW; END $$; CREATE TRIGGER zzz_test_gc_fault {event} ON {table} FOR EACH {granularity} EXECUTE FUNCTION test_gc_fault()" + ); + second.execute_unprepared(&sql).await.unwrap(); + assert!(repo.apply(&claim).await.is_err()); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_payload").await, + 2 + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_graph_edge").await, + 1 + ); + assert_eq!( + scalar( + &first, + "SELECT sum(incoming_refs)::bigint FROM mst2_metadata_graph_node" + ) + .await, + 1 + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_graph_node").await, + 2 + ); + assert_eq!(scalar(&first,"SELECT count(*) FROM mst2_metadata_gc_op WHERE state='PENDING' AND payload_delete_xid IS NULL").await,1); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_lifetime WHERE state='DELETING'" + ) + .await, + 1 + ); + assert_eq!(scalar(&first,"SELECT count(*) FROM pg_trigger WHERE tgname IN ('mst2_metadata_payload_fenced','mst2_metadata_payload_removed','mst2_metadata_graph_node_guard','mst2_metadata_gc_op_guard') AND tgenabled='O' AND tgrelid IN (SELECT oid FROM pg_class WHERE relnamespace=(SELECT oid FROM pg_namespace WHERE nspname=current_schema()))").await,4); + second + .execute_unprepared(&format!("DROP TRIGGER zzz_test_gc_fault ON {table}")) + .await + .unwrap(); + repo.apply(&claim).await.unwrap(); + } +} + +#[tokio::test] +async fn qualified_direct_selected_delete_completes_all_bookkeeping_and_guc_is_not_authority() { + let (first, second, _schema) = fixture().await; + let (repo, _pages, receipt) = installed(&first, "direct-delete").await; + repo.retire_prepare_coverage(&receipt).await.unwrap(); + let claim = claim_root(&repo, &receipt).await; + for sql in [ + "UPDATE mst2_metadata_gc_op SET had_payload=false", + "UPDATE mst2_metadata_gc_op SET graph_present=false", + "UPDATE mst2_metadata_gc_op SET generation=generation+1", + ] { + assert!(second.execute_unprepared(sql).await.is_err(), "{sql}"); + } + assert!( + second + .query_one_raw(statement( + "SELECT mst2_metadata_gc_finish($1::uuid)", + [claim.operation_id.clone().into()] + )) + .await + .is_err() + ); + second + .execute_unprepared("UPDATE mst2_metadata_gc_op SET payload_delete_xid=1") + .await + .unwrap(); + let marked = second.begin().await.unwrap(); + marked + .execute_raw(statement( + "UPDATE mst2_metadata_gc_op SET payload_delete_xid=42 WHERE operation_id=$1::uuid", + [claim.operation_id.clone().into()], + )) + .await + .unwrap(); + let same_xid:bool=marked.query_one_raw(statement( + "SELECT payload_delete_xid=txid_current() AS same_xid FROM mst2_metadata_gc_op WHERE operation_id=$1::uuid", + [claim.operation_id.clone().into()], + )).await.unwrap().unwrap().try_get("","same_xid").unwrap(); + assert!( + same_xid, + "database must replace a caller marker with the actual transaction identity" + ); + let error = marked + .query_one_raw(statement( + "SELECT mst2_metadata_gc_finish($1::uuid)", + [claim.operation_id.clone().into()], + )) + .await + .unwrap_err(); + assert!( + error.to_string().contains("payload presence changed"), + "{error}" + ); + marked.rollback().await.unwrap(); + assert!( + second + .query_one_raw(statement( + "SELECT mst2_metadata_gc_finish($1::uuid)", + [claim.operation_id.clone().into()] + )) + .await + .is_err() + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_payload").await, + 2 + ); + assert!( + second + .execute_unprepared("DELETE FROM mst2_metadata_payload") + .await + .is_err() + ); + let txn = second.begin().await.unwrap(); + txn.execute_raw(statement( + "SELECT set_config('mega2.metadata_gc_operation',$1,true)", + [uuid::Uuid::new_v4().to_string().into()], + )) + .await + .unwrap(); + assert!( + txn.execute_raw(statement( + "DELETE FROM mst2_metadata_payload WHERE page_id=$1 AND generation=$2", + [ + claim.lifetime.page_id.to_vec().into(), + claim.lifetime.generation.into() + ] + )) + .await + .is_err() + ); + txn.rollback().await.unwrap(); + let txn = second.begin().await.unwrap(); + txn.execute_raw(statement( + "SELECT set_config('mega2.metadata_gc_operation',$1,true)", + [claim.operation_id.clone().into()], + )) + .await + .unwrap(); + txn.execute_raw(statement( + "DELETE FROM mst2_metadata_payload WHERE page_id=$1 AND generation=$2", + [ + claim.lifetime.page_id.to_vec().into(), + claim.lifetime.generation.into(), + ], + )) + .await + .unwrap(); + assert_eq!( + scalar( + &txn, + "SELECT count(*) FROM mst2_metadata_gc_op WHERE state='APPLIED'" + ) + .await, + 1 + ); + assert_eq!( + scalar(&txn, "SELECT count(*) FROM mst2_metadata_graph_edge").await, + 0 + ); + assert_eq!( + scalar(&txn, "SELECT incoming_refs FROM mst2_metadata_graph_node").await, + 0 + ); + txn.commit().await.unwrap(); + repo.apply(&claim).await.unwrap(); + assert!( + second + .execute_unprepared("DELETE FROM mst2_metadata_gc_op") + .await + .is_err() + ); +} + +#[tokio::test] +async fn qualified_fresh_aba_replays_old_receipts_without_deleting_g2_or_reviving_old_intents() { + let (first, second, _schema) = fixture().await; + let (repo, pages, receipt) = installed(&first, "aba-old").await; + let observation = repo.observe_installed_dag(receipt.intent()).await.unwrap(); + let terminal = repo.retire_prepare_coverage(&receipt).await.unwrap(); + let claim = claim_root(&repo, &receipt).await; + second.execute_unprepared("CREATE TABLE test_payload_deletes(n integer NOT NULL); CREATE FUNCTION test_record_delete() RETURNS trigger LANGUAGE plpgsql AS $$ BEGIN INSERT INTO test_payload_deletes VALUES(1); RETURN NULL; END $$; CREATE TRIGGER zzz_test_record_delete AFTER DELETE ON mst2_metadata_payload FOR EACH ROW EXECUTE FUNCTION test_record_delete()").await.unwrap(); + let held = first.begin().await.unwrap(); + let applied = repo.apply_in_txn(&held, &claim).await.unwrap(); + held.commit().await.unwrap(); // Deliberately lose delivery; inspect recovers the exact durable receipt. + assert_eq!( + repo.inspect_gc( + &second, + claim.operation_id(), + claim.lifetime(), + MetadataGcPhase::Apply + ) + .await + .unwrap(), + MetadataGcObservation::Applied(Box::new(applied.clone())) + ); + assert!( + repo.begin_intent("ordinary-cannot-reopen", &pages) + .await + .is_err() + ); + let fresh = repo + .begin_fresh_intent("aba-fresh", &pages, std::slice::from_ref(&applied)) + .await + .unwrap(); + assert_eq!( + repo.lifetime(receipt.metadata_root()) + .await + .unwrap() + .generation(), + 2 + ); + repo.install_pages(&fresh, pages.dag().payloads()) + .await + .unwrap(); + repo.finalize(&fresh).await.unwrap(); + assert!( + repo.install_pages(receipt.intent(), pages.dag().payloads()) + .await + .is_err() + ); + assert!( + repo.finalize_observation(receipt.intent(), &observation) + .await + .is_err() + ); + let restarted = PostgresQualifiedMetadataRepository::new(second.clone()) + .await + .unwrap(); + let old = restarted + .historical_lifetime(claim.lifetime.page_id, 1) + .await + .unwrap(); + let recovered = restarted + .inspect_gc(&second, claim.operation_id(), &old, MetadataGcPhase::Apply) + .await + .unwrap(); + let MetadataGcObservation::Applied(recovered) = recovered else { + panic!("restart must recover historical APPLIED receipt"); + }; + assert_eq!(restarted.apply(recovered.claim()).await.unwrap(), applied); + assert!( + restarted + .claim(&uuid::Uuid::new_v4().to_string(), &old) + .await + .is_err() + ); + let terminal_intent = restarted + .capture_terminal_intent( + receipt.intent().operation_id(), + receipt.intent().manifest_digest(), + ) + .await + .unwrap(); + assert_eq!( + restarted + .inspect_terminal( + &second, + &terminal_intent, + MetadataTerminalAction::RetireCoverage + ) + .await + .unwrap(), + MetadataTerminalObservation::Terminated(Box::new(terminal.clone())) + ); + assert_eq!( + repo.retire_prepare_coverage(&receipt).await.unwrap(), + terminal + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM test_payload_deletes").await, + 1 + ); + let stale = second.begin().await.unwrap(); + stale + .execute_raw(statement( + "SELECT set_config('mega2.metadata_gc_operation',$1,true)", + [claim.operation_id.clone().into()], + )) + .await + .unwrap(); + assert!( + stale + .execute_raw(statement( + "DELETE FROM mst2_metadata_payload WHERE page_id=$1 AND generation=2", + [claim.lifetime.page_id.to_vec().into()] + )) + .await + .is_err() + ); + stale.rollback().await.unwrap(); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_payload").await, + 2 + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_lifetime").await, + 3 + ); + assert_eq!( + repo.begin_fresh_intent("aba-fresh", &pages, std::slice::from_ref(&applied)) + .await + .unwrap(), + fresh + ); + let mut forged = applied.clone(); + forged.created_at += chrono::Duration::seconds(1); + assert!( + repo.begin_fresh_intent("aba-fresh", &pages, &[forged]) + .await + .is_err() + ); + assert!( + repo.begin_fresh_intent("stale-fresh-proof", &pages, &[applied]) + .await + .is_err() + ); +} + +#[tokio::test] +async fn qualified_multi_page_fresh_failure_rolls_back_all_watermarks_and_new_history() { + let (first, second, _schema) = fixture().await; + let (repo, pages, receipt) = installed(&first, "multi-old").await; + repo.retire_prepare_coverage(&receipt).await.unwrap(); + let root = claim_root(&repo, &receipt).await; + let root = repo.apply(&root).await.unwrap(); + let child_id = pages + .dag() + .payloads() + .iter() + .find(|p| p.id != receipt.metadata_root()) + .unwrap() + .id; + let child = repo.lifetime(child_id).await.unwrap(); + let child = repo + .claim(&uuid::Uuid::new_v4().to_string(), &child) + .await + .unwrap(); + let child = repo.apply(&child).await.unwrap(); + second.execute_unprepared("CREATE FUNCTION test_fresh_fault() RETURNS trigger LANGUAGE plpgsql AS $$ BEGIN IF NEW.generation=2 AND NEW.page_id=(SELECT page_id FROM mst2_metadata_current ORDER BY page_id DESC LIMIT 1) THEN RAISE EXCEPTION 'injected multi-page fresh CAS failure'; END IF; RETURN NEW; END $$; CREATE TRIGGER zzz_test_fresh_fault BEFORE UPDATE ON mst2_metadata_current FOR EACH ROW EXECUTE FUNCTION test_fresh_fault()").await.unwrap(); + assert!( + repo.begin_fresh_intent("multi-fresh", &pages, &[root.clone(), child.clone()]) + .await + .is_err() + ); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_current WHERE generation=1" + ) + .await, + 2 + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_lifetime").await, + 2 + ); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_prepare WHERE operation_id='multi-fresh'" + ) + .await, + 0 + ); + second + .execute_unprepared("DROP TRIGGER zzz_test_fresh_fault ON mst2_metadata_current") + .await + .unwrap(); + let intent = repo + .begin_fresh_intent("multi-fresh", &pages, &[root, child]) + .await + .unwrap(); + repo.install_pages(&intent, pages.dag().payloads()) + .await + .unwrap(); + repo.finalize(&intent).await.unwrap(); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_current WHERE generation=2" + ) + .await, + 2 + ); +} + +#[tokio::test] +async fn qualified_wrong_primary_and_repeatable_read_never_prove_absence_or_authorize_delete() { + let (first, second, _schema) = fixture().await; + let (other, _other_second, _other_schema) = fixture().await; + let (repo, _pages, receipt) = installed(&first, "scope").await; + repo.retire_prepare_coverage(&receipt).await.unwrap(); + let claim = claim_root(&repo, &receipt).await; + assert!(matches!( + repo.inspect_gc( + &other, + claim.operation_id(), + claim.lifetime(), + MetadataGcPhase::Apply + ) + .await, + Err(MetadataGcError::CommitUncertain { .. }) + )); + let txn = second + .begin_with_config(Some(IsolationLevel::RepeatableRead), None) + .await + .unwrap(); + assert!( + txn.execute_unprepared("DELETE FROM mst2_metadata_payload") + .await + .is_err() + ); + txn.rollback().await.unwrap(); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_payload").await, + 2 + ); + repo.apply(&claim).await.unwrap(); +} + +#[tokio::test] +async fn qualified_null_bytes_and_retired_generic_incarnations_cannot_be_adopted() { + let (first, second, _schema) = fixture().await; + let pages = prepared(); + let payload = &pages.dag().payloads()[0]; + first.execute_raw(statement("INSERT INTO mst2_metadata_payload(page_id,metadata_codec,byte_size,payload) VALUES($1,1,$2,$3)", + [payload.id.to_vec().into(),(payload.size as i32).into(),payload.bytes.clone().into()])).await.unwrap(); + let qualified = PostgresQualifiedMetadataRepository::new(second.clone()) + .await + .unwrap(); + assert!( + qualified + .begin_intent("cannot-adopt-null", &pages) + .await + .is_err() + ); + assert!( + second + .execute_unprepared("DELETE FROM mst2_metadata_payload") + .await + .is_err() + ); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_payload WHERE generation IS NULL" + ) + .await, + 1 + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_current").await, + 0 + ); + let (first, second, _second_schema) = fixture().await; + let generic = PostgresMetadataGenerationRepository::new(first.clone()) + .await + .unwrap(); + let intent = generic + .begin_intent("generic-history", &pages) + .await + .unwrap(); + generic + .install_pages(&intent, pages.dag().payloads()) + .await + .unwrap(); + let receipt = generic.finalize(&intent).await.unwrap(); + generic.retire_prepare_coverage(&receipt).await.unwrap(); + let qualified = PostgresQualifiedMetadataRepository::new(second) + .await + .unwrap(); + assert!( + qualified + .begin_intent("cannot-adopt-retired-generic", &pages) + .await + .is_err() + ); + assert!(qualified.lifetime(payload.id).await.is_err()); +} + +#[tokio::test] +async fn qualified_committed_partial_payload_is_rejected_without_repair() { + let (first, second, _schema) = fixture().await; + let repo = PostgresQualifiedMetadataRepository::new(first.clone()) + .await + .unwrap(); + let pages = prepared(); + let intent = repo + .begin_intent("corrupt-partial-committed", &pages) + .await + .unwrap(); + repo.install_page(&intent, &pages.dag().payloads()[0]) + .await + .unwrap(); + second.execute_raw(statement("UPDATE mst2_metadata_prepare SET state='COMMITTED',committed_at=clock_timestamp() WHERE prepare_id=$1", + [intent.prepare_id().into()])).await.unwrap(); + assert!( + repo.install_pages(&intent, pages.dag().payloads()) + .await + .is_err() + ); + let lifetime = repo.lifetime(pages.dag().root()).await.unwrap(); + assert!( + repo.claim(&uuid::Uuid::new_v4().to_string(), &lifetime) + .await + .is_err() + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_payload").await, + 1 + ); +} + +#[tokio::test] +async fn qualified_cross_action_terminal_inspection_conflicts_and_production_adoption_stays_closed() +{ + let (first, second, _schema) = fixture().await; + let (repo, pages, receipt) = installed(&first, "cross-action-committed").await; + assert!(matches!( + repo.inspect_terminal(&second, receipt.intent(), MetadataTerminalAction::Abort) + .await, + Err(MetadataTerminalError::Rejected(SnapshotError { + code: SnapshotErrorCode::Conflict, + .. + })) + )); + let pending = repo + .begin_intent("cross-action-aborted", &pages) + .await + .unwrap(); + repo.abort(&pending).await.unwrap(); + assert!(matches!( + repo.inspect_terminal(&second, &pending, MetadataTerminalAction::RetireCoverage) + .await, + Err(MetadataTerminalError::Rejected(SnapshotError { + code: SnapshotErrorCode::Conflict, + .. + })) + )); + assert!(repo.abort(receipt.intent()).await.is_err()); + let sql="INSERT INTO mst2_snapshot_context(snapshot_id,canonical_descriptor,instance_id,commit_oid,root_tree_oid, + metadata_root,prepare_id,publication_sequence,writer_epoch,state) VALUES('blocked',$1,'i','c','r',$2,$3,0,1,'READY')"; + assert!( + second + .execute_raw(statement( + sql, + [ + vec![1_u8].into(), + receipt.metadata_root().to_vec().into(), + receipt.intent().prepare_id().into() + ] + )) + .await + .is_err() + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_snapshot_context").await, + 0 + ); + assert_eq!( + repo.inspect_terminal( + &second, + receipt.intent(), + MetadataTerminalAction::RetireCoverage + ) + .await + .unwrap(), + MetadataTerminalObservation::Active + ); +} + +#[tokio::test] +async fn qualified_deferred_current_guard_denies_unprotected_raw_fresh_commit() { + let (first, second, _schema) = fixture().await; + let (repo, _pages, receipt) = installed(&first, "raw-fresh").await; + repo.retire_prepare_coverage(&receipt).await.unwrap(); + let claim = claim_root(&repo, &receipt).await; + repo.apply(&claim).await.unwrap(); + let txn = second.begin().await.unwrap(); + txn.execute_raw(statement("INSERT INTO mst2_metadata_lifetime(page_id,node_id,generation,state,metadata_codec,expected_size,graph_domain) VALUES($1,$2,2,'RESERVED',1,$3,'qualified-v1')", + [claim.lifetime.page_id.to_vec().into(),node_id(&claim.lifetime.page_id).into(),claim.lifetime.expected_size.into()])).await.unwrap(); + txn.execute_raw(statement( + "UPDATE mst2_metadata_current SET generation=2 WHERE page_id=$1", + [claim.lifetime.page_id.to_vec().into()], + )) + .await + .unwrap(); + assert!(txn.commit().await.is_err()); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_current WHERE generation=2" + ) + .await, + 0 + ); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_lifetime WHERE generation=2" + ) + .await, + 0 + ); +} + +#[tokio::test] +async fn qualified_statement_dag_audit_accepts_wide_and_chain_graphs_and_rejects_a_cycle() { + for wide in [true, false] { + let (first, second, _schema) = fixture().await; + let repo = PostgresQualifiedMetadataRepository::new(first.clone()) + .await + .unwrap(); + let mut builder = MetadataDagBuilder::new(MetadataDagLimits::default()); + let mut root_entries = Vec::new(); + let mut last = [0_u8; 32]; + for i in 0..64_u8 { + let entries = if wide || i == 0 { + vec![Entry::file(EntryKind::Regular, b"file", 3, [i; 32])] + } else { + vec![Entry::dir(b"child", last)] + }; + let page = Page::build(&entries).unwrap(); + builder.add_directory(&page, &entries).unwrap(); + last = page_id(&page); + root_entries.push(Entry::dir(format!("d{i:03}").as_bytes(), last)); + } + let root = if wide { + let root = Page::build(&root_entries).unwrap(); + builder.add_directory(&root, &root_entries).unwrap(); + page_id(&root) + } else { + last + }; + let pages = PreparedNativeMetadataRetention::test_installation( + Arc::new(builder.finish(root).unwrap()), + "/", + ); + let intent = repo.begin_intent("statement-dag", &pages).await.unwrap(); + repo.install_pages(&intent, &pages.dag().payloads()[..64]) + .await + .unwrap(); + if wide { + repo.install_page(&intent, &pages.dag().payloads()[64]) + .await + .unwrap(); + } + let receipt = repo.finalize(&intent).await.unwrap(); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_graph_node").await, + if wide { 65 } else { 64 } + ); + assert_eq!( + scalar( + &first, + "SELECT sum(incoming_refs)::bigint FROM mst2_metadata_graph_node" + ) + .await, + if wide { 64 } else { 63 } + ); + let pending = repo.begin_intent("cycle-attempt", &pages).await.unwrap(); + let leaf = pages + .dag() + .payloads() + .iter() + .find(|p| { + !pages + .dag() + .edges() + .iter() + .any(|e| e.parent == node_id(&p.id)) + }) + .unwrap() + .id; + let error=second.execute_raw(statement( + "INSERT INTO mst2_metadata_graph_edge(parent_page,parent_generation,child_page,child_generation) VALUES($1,1,$2,1)", + [leaf.to_vec().into(),root.to_vec().into()], + )).await.unwrap_err(); + assert!(error.to_string().contains("cycle"), "{error}"); + assert_eq!( + scalar( + &first, + "SELECT sum(incoming_refs)::bigint FROM mst2_metadata_graph_node" + ) + .await, + if wide { 64 } else { 63 } + ); + repo.abort(&pending).await.unwrap(); + repo.retire_prepare_coverage(&receipt).await.unwrap(); + } +} + +#[tokio::test] +async fn qualified_statement_dag_overflow_rolls_back_without_disabling_production_guards() { + let (first, second, _schema) = fixture().await; + let repo = PostgresQualifiedMetadataRepository::new(first.clone()) + .await + .unwrap(); + let pages = prepared(); + let intent = repo.begin_intent("overflow-proof", &pages).await.unwrap(); + assert_eq!(MetadataDagLimits::default().nodes, 4096); + assert_eq!(MetadataDagLimits::default().edges, 16384); + let txn = second.begin().await.unwrap(); + txn.execute_raw(statement( + "INSERT INTO mst2_metadata_lifetime(page_id,node_id,generation,state,metadata_codec,expected_size,graph_domain) + SELECT sha256(convert_to('bound-node-'||i,'UTF8')),'page:sha256:'||encode(sha256(convert_to('bound-node-'||i,'UTF8')),'hex'), + 1,'RESERVED',1,$1,'qualified-v1' FROM generate_series(1,4097) x(i)", + [(pages.dag().payloads()[0].size as i32).into()], + )).await.unwrap(); + txn.execute_unprepared("INSERT INTO mst2_metadata_current(page_id,generation) SELECT l.page_id,l.generation FROM mst2_metadata_lifetime l WHERE NOT EXISTS(SELECT 1 FROM mst2_metadata_current c WHERE c.page_id=l.page_id)").await.unwrap(); + txn.execute_raw(statement( + "INSERT INTO mst2_metadata_prepare_page(prepare_id,page_id,generation,expected_size) + SELECT $1,l.page_id,l.generation,l.expected_size FROM mst2_metadata_lifetime l + WHERE NOT EXISTS(SELECT 1 FROM mst2_metadata_prepare_page m WHERE m.prepare_id=$1 AND m.page_id=l.page_id)", + [intent.prepare_id().into()], + )).await.unwrap(); + txn.execute_raw(statement( + "INSERT INTO mst2_metadata_graph_node(page_id,generation,state,metadata_codec,bytes) + SELECT page_id,generation,'LIVE',1,expected_size FROM mst2_metadata_prepare_page WHERE prepare_id=$1", + [intent.prepare_id().into()], + )).await.unwrap(); + let error=txn.execute_unprepared( + "INSERT INTO mst2_metadata_graph_edge(parent_page,parent_generation,child_page,child_generation) + SELECT p.page_id,1,c.page_id,1 FROM (SELECT page_id FROM mst2_metadata_graph_node ORDER BY page_id LIMIT 1) p + CROSS JOIN mst2_metadata_graph_node c WHERE c.page_id<>p.page_id" + ).await.unwrap_err(); + assert!(error.to_string().contains("overflow"), "{error}"); + txn.rollback().await.unwrap(); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_graph_node").await, + 0 + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_current").await, + 2 + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_lifetime").await, + 2 + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_prepare_page").await, + 2 + ); + let captured = repo.scan_current_lifetimes(None, 1).await.unwrap(); + assert_eq!(captured.len(), 1); + let rest = repo + .scan_current_lifetimes(Some((captured[0].page_id(), captured[0].generation())), 1) + .await + .unwrap(); + assert_eq!(rest.len(), 1); + assert_ne!(rest[0].page_id(), captured[0].page_id()); +} + +#[tokio::test] +async fn qualified_aborted_intent_is_recovered_from_history_after_repository_restart() { + let (first, second, _schema) = fixture().await; + let repo = PostgresQualifiedMetadataRepository::new(first) + .await + .unwrap(); + let pages = prepared(); + let intent = repo.begin_intent("restart-aborted", &pages).await.unwrap(); + let terminal = repo.abort(&intent).await.unwrap(); + drop(repo); + let restarted = PostgresQualifiedMetadataRepository::new(second.clone()) + .await + .unwrap(); + let recovered = restarted + .capture_terminal_intent("restart-aborted", intent.manifest_digest()) + .await + .unwrap(); + assert_eq!( + restarted + .inspect_terminal(&second, &recovered, MetadataTerminalAction::Abort) + .await + .unwrap(), + MetadataTerminalObservation::Terminated(Box::new(terminal)) + ); + assert!( + restarted + .install_pages(&recovered, pages.dag().payloads()) + .await + .is_err() + ); + assert_eq!( + scalar(&second, "SELECT count(*) FROM mst2_metadata_payload").await, + 0 + ); +} From e044c064a13fd853450b3c752e2b6d2aa0998335 Mon Sep 17 00:00:00 2001 From: Xiaoyang Han Date: Wed, 7 Oct 2026 21:17:20 +0800 Subject: [PATCH 11/50] fix(mst2): observe renewal locks without exhausting the test pool (#61) Observe pg_locks through the transaction already holding the retention lock, and clear its statistics snapshot before each poll. Keep all original expiry, root-release and byte-count assertions. Stream repository test diagnostics before a possible CI cancellation. Native regression validation is pending. --- .github/workflows/repository-gates.yml | 2 +- src/api/router/snapshot_session_tests.rs | 12 +++++++++--- 2 files changed, 10 insertions(+), 4 deletions(-) diff --git a/.github/workflows/repository-gates.yml b/.github/workflows/repository-gates.yml index b656c665..313fc5a5 100644 --- a/.github/workflows/repository-gates.yml +++ b/.github/workflows/repository-gates.yml @@ -72,7 +72,7 @@ jobs: set -euo pipefail source .env.test export no_proxy="$NO_PROXY" - cargo test --all + cargo test --all -- --nocapture - name: Stop isolated test dependencies if: always() diff --git a/src/api/router/snapshot_session_tests.rs b/src/api/router/snapshot_session_tests.rs index 43a9c440..187499a4 100644 --- a/src/api/router/snapshot_session_tests.rs +++ b/src/api/router/snapshot_session_tests.rs @@ -1,4 +1,4 @@ -use sea_orm::{DatabaseConnection, DbBackend, Statement, TransactionTrait}; +use sea_orm::{ConnectionTrait, DatabaseConnection, DbBackend, Statement, TransactionTrait}; use tokio::sync::Barrier; use super::*; @@ -22,7 +22,7 @@ fn resolve_request(scope: &str) -> Request { .unwrap() } -async fn scalar(db: &DatabaseConnection, sql: &str) -> i64 { +async fn scalar(db: &C, sql: &str) -> i64 { db.query_one_raw(Statement::from_string(DbBackend::Postgres, sql.to_owned())) .await .unwrap() @@ -575,8 +575,14 @@ async fn mst2_durable_http_renew_waits_for_lock_before_checking_database_deadlin }; tokio::time::timeout(Duration::from_secs(30), async { loop { + held.execute_raw(Statement::from_string( + DbBackend::Postgres, + "SELECT pg_stat_clear_snapshot()".to_owned(), + )) + .await + .unwrap(); let blocked = scalar( - db, + &held, "SELECT count(*) FROM pg_locks l JOIN pg_stat_activity a ON a.pid=l.pid WHERE l.locktype='advisory' AND NOT l.granted AND a.datname=current_database() AND a.application_name=current_schema()", From 78d1e1accdca590a5e67ae9e5111f69b1a2b152e Mon Sep 17 00:00:00 2001 From: Xiaoyang Han Date: Wed, 7 Oct 2026 22:21:29 +0800 Subject: [PATCH 12/50] fix(mst2): keep lookup metadata-only and restore native regression coverage (#63) Use one fixed-path metadata walk and current verified size/digest facts for lookup without whole-content reads. Preserve the original zero-body-read assertions and HEAD behavior. Correct three observed native fixture failures without changing production publication or payload guards. Native execution of this combined source remains pending. --- src/api/router/snapshot_content.rs | 17 +- .../router/snapshot_lookup_metadata_tests.rs | 271 ++++++++++++++++++ src/api/router/snapshot_router.rs | 26 +- src/api/router/snapshot_session_tests.rs | 3 + src/contract/git_protocol/mod.rs | 1 + .../service/native_publication_push_tests.rs | 39 ++- .../storage/native_metadata_install_tests.rs | 71 ++++- 7 files changed, 392 insertions(+), 36 deletions(-) create mode 100644 src/api/router/snapshot_lookup_metadata_tests.rs diff --git a/src/api/router/snapshot_content.rs b/src/api/router/snapshot_content.rs index d1dd2a15..6e419a61 100644 --- a/src/api/router/snapshot_content.rs +++ b/src/api/router/snapshot_content.rs @@ -34,11 +34,11 @@ struct ResolvedFile { raw: Vec, } -struct ResolvedFileMetadata { +pub(super) struct ResolvedFileMetadata { fs_kind: FsKind, oid: String, - digest: [u8; 32], - size: u64, + pub(super) digest: [u8; 32], + pub(super) size: u64, } #[allow(clippy::result_large_err)] @@ -93,6 +93,17 @@ async fn resolve_file_metadata( + handler: &T, + fs_kind: FsKind, + oid: String, + path: &str, + expected_digest: Option<&str>, +) -> Result { let storage = handler.get_context(); let mut verified = storage .mono_storage() diff --git a/src/api/router/snapshot_lookup_metadata_tests.rs b/src/api/router/snapshot_lookup_metadata_tests.rs new file mode 100644 index 00000000..81c11cdb --- /dev/null +++ b/src/api/router/snapshot_lookup_metadata_tests.rs @@ -0,0 +1,271 @@ +use super::*; + +fn lookup_body(paths: &[&str]) -> Body { + Body::from(json!({"paths": paths}).to_string()) +} + +#[tokio::test] +async fn mst2_durable_http_lookup_uses_verified_metadata_after_service_rebuild_without_body_reads() +{ + let fixture = Fixture::new_with_pg_config(true).await; + let state = rebuilt(&fixture).await; + assert!(state.storage.native_snapshot_sessions.get().is_none()); + assert!(!Arc::ptr_eq( + &state.storage.native_projection_cache, + &fixture.state.storage.native_projection_cache + )); + let paths = [ + "/file", + "/alias", + "/executable", + "/empty", + "/link", + "/nested/file", + "/directory", + "/nested", + "/missing", + "/file/child", + "/link/child", + ]; + let value = success_json( + app(&state) + .oneshot(fixture.request("POST", "lookup", lookup_body(&paths))) + .await + .unwrap(), + ) + .await; + assert_eq!(value["snapshot_id"], fixture.snapshot); + let results = value["results"].as_array().unwrap(); + assert_eq!(results.len(), paths.len()); + for (result, path) in results.iter().zip(paths) { + assert_eq!(result["path"], path); + } + for (index, kind, size, content_digest) in [ + (0, "regular", fixture.raw.len(), fixture.digest_string()), + (1, "regular", fixture.raw.len(), fixture.digest_string()), + (2, "executable", fixture.raw.len(), fixture.digest_string()), + (3, "regular", 0, format!("sha256:{}", hex_of(&digest(&[])))), + ( + 4, + "symlink", + 4, + format!("sha256:{}", hex_of(&digest(b"file"))), + ), + (5, "regular", fixture.raw.len(), fixture.digest_string()), + ] { + assert_eq!(results[index]["status"], "found"); + assert_eq!( + results[index]["node"], + json!({ + "fs_kind": kind, + "name": paths[index].rsplit('/').next().unwrap(), + "size": size.to_string(), + "content_digest": content_digest, + }) + ); + } + let proofs = value["proof_pages"].as_array().unwrap(); + assert!(!proofs.is_empty()); + let mut proof_digests = std::collections::HashSet::new(); + for proof in proofs { + let bytes = STANDARD + .decode(proof["data_base64"].as_str().unwrap()) + .unwrap(); + mst2_codec::metapage::Page::decode(&bytes).unwrap(); + let digest = format!("sha256:{}", hex_of(&mst2_codec::metapage::page_id(&bytes))); + assert_eq!(proof["digest"], digest); + assert!(proof_digests.insert(digest)); + } + for index in [6, 7] { + let result = &results[index]; + assert_eq!(result["status"], "found"); + assert_eq!(result["node"]["fs_kind"], "directory"); + assert_eq!( + result["node"]["name"], + paths[index].rsplit('/').next().unwrap() + ); + assert_eq!(result["node"]["node_class"], "native_tree"); + assert_eq!(result["node"]["lifecycle"], "mutable"); + assert!(proof_digests.contains(result["node"]["directory_root"].as_str().unwrap())); + } + for (index, status) in [ + (8, "absent"), + (9, "not_directory"), + (10, "symlink_traversal"), + ] { + assert_eq!( + results[index], + json!({"path": paths[index], "status": status}) + ); + } + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn mst2_durable_http_lookup_missing_and_noncurrent_facts_never_fall_back_to_body_or_cache() { + let fixture = Fixture::new_with_pg_config(true).await; + fixture.map("/file").await; + let original = fixture.fact().await; + fixture.counts.reset(); + fixture.delete_fact().await; + error( + fixture + .send("POST", "lookup", lookup_body(&["/file"])) + .await, + 503, + "METADATA_NOT_READY", + true, + ) + .await; + fixture.counts.assert(0, 0); + for (domain, kind, generation) in [ + ("git", "blob", 1), + ("other", "blob", MST2_VERIFICATION_VERSION), + ("git", "tree", MST2_VERIFICATION_VERSION), + ] { + let mut fact = original.clone(); + fact.storage_domain = domain.to_string(); + fact.object_kind = kind.to_string(); + fact.verification_version = generation; + fixture.replace_fact(fact).await; + error( + fixture + .send("POST", "lookup", lookup_body(&["/alias"])) + .await, + 503, + "METADATA_NOT_READY", + true, + ) + .await; + fixture.counts.assert(0, 0); + } + fixture.replace_fact(original).await; + success_json( + fixture + .send("POST", "lookup", lookup_body(&["/file"])) + .await, + ) + .await; + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn mst2_durable_http_lookup_invalid_verified_facts_fail_before_body_reads() { + let fixture = Fixture::new_with_pg_config(true).await; + let original = fixture.fact().await; + for case in 0..5 { + let mut fact = original.clone(); + match case { + 0 => fact.state = "PENDING".to_string(), + 1 => fact.verification_version = MST2_VERIFICATION_VERSION + 1, + 2 => fact.size = -1, + 3 => fact.size = 8_796_093_022_209, + 4 => { + fact.raw_sha256.pop().unwrap(); + } + _ => unreachable!(), + } + fixture.replace_fact(fact).await; + error( + fixture + .send("POST", "lookup", lookup_body(&["/file"])) + .await, + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + fixture.counts.assert(0, 0); + } + fixture.replace_fact(original).await; + success_json( + fixture + .send("POST", "lookup", lookup_body(&["/file"])) + .await, + ) + .await; + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn mst2_durable_http_lookup_symlink_facts_outside_profile_fail_without_body_reads() { + let fixture = Fixture::new_with_pg_config(true).await; + let mono = fixture.state.storage.mono_storage(); + let main = mono.get_main_ref("/").await.unwrap().unwrap(); + let handler = MonoApiService::from(&fixture.state); + let root = handler.get_tree_by_hash(&main.ref_tree_hash).await.unwrap(); + let MetadataWalkOutcome::FoundFile { oid, .. } = + resolve_abs_metadata(&handler, &root, "/project/link") + .await + .unwrap() + else { + panic!("fixture link must be a fixed file"); + }; + let original = mono + .get_verified_blobs(vec![oid.clone()]) + .await + .unwrap() + .remove(&oid) + .unwrap(); + for size in [0, 4096] { + let mut fact = original.clone().into_active_model(); + fact.size = Set(size); + fact.update(mono.get_connection()).await.unwrap(); + error( + fixture + .send("POST", "lookup", lookup_body(&["/link"])) + .await, + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + fixture.counts.assert(0, 0); + } + let mut restored = original.into_active_model(); + restored.size = Set(4); + restored.update(mono.get_connection()).await.unwrap(); + let result = success_json( + fixture + .send("POST", "lookup", lookup_body(&["/link"])) + .await, + ) + .await; + assert_eq!(result["results"][0]["node"]["size"], "4"); + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn mst2_durable_http_lookup_verified_fact_database_failure_is_typed_without_body_reads() { + let fixture = Fixture::new_with_pg_config(true).await; + let mono = fixture.state.storage.mono_storage(); + mono.get_connection() + .execute_unprepared( + "ALTER TABLE mst2_verified_object RENAME TO mst2_verified_object_unavailable", + ) + .await + .unwrap(); + error( + fixture + .send("POST", "lookup", lookup_body(&["/file"])) + .await, + 500, + "INTERNAL", + true, + ) + .await; + fixture.counts.assert(0, 0); + mono.get_connection() + .execute_unprepared( + "ALTER TABLE mst2_verified_object_unavailable RENAME TO mst2_verified_object", + ) + .await + .unwrap(); + success_json( + fixture + .send("POST", "lookup", lookup_body(&["/file"])) + .await, + ) + .await; + fixture.counts.assert(0, 0); +} diff --git a/src/api/router/snapshot_router.rs b/src/api/router/snapshot_router.rs index f2c3515f..40876942 100644 --- a/src/api/router/snapshot_router.rs +++ b/src/api/router/snapshot_router.rs @@ -28,8 +28,8 @@ use crate::{ descriptor::build as build_descriptor, error::{SnapshotError, SnapshotErrorCode}, pages::{ - WalkOutcome, base64_of, build_directory_page, build_directory_page_with_work, hex_of, - proof_pages, resolve_abs, + MetadataWalkOutcome, WalkOutcome, base64_of, build_directory_page, + build_directory_page_with_work, hex_of, proof_pages, resolve_abs, resolve_abs_metadata, }, projection_observation::{NativeResolveSource, ResolvedProjection}, runtime::{now_unix, runtime}, @@ -1299,11 +1299,11 @@ async fn lookup( validate_scope_relative_path(path).map_err(mst2_error_response)?; let abs_path = abs_view_path(&ctx.built.descriptor.scope, path); let mut entry = json!({"path": path}); - match resolve_abs(handler.as_ref(), &root_tree, &abs_path) + match resolve_abs_metadata(handler.as_ref(), &root_tree, &abs_path) .await .map_err(mst2_error_response)? { - WalkOutcome::FoundDir => { + MetadataWalkOutcome::FoundDir => { let built = build_directory_page(handler.as_ref(), &root_tree, &abs_path) .await .map_err(mst2_error_response)?; @@ -1320,25 +1320,23 @@ async fn lookup( entry["node"] = node; deepest_dirs.push(abs_path); } - WalkOutcome::FoundFile { - fs_kind, - size, - digest, - .. - } => { + MetadataWalkOutcome::FoundFile { fs_kind, oid } => { + let fact = + content::verified_file_metadata(handler.as_ref(), fs_kind, oid, path, None) + .await?; entry["status"] = json!("found"); let mut node = json!({"fs_kind": fs_kind.as_str()}); if let Some(name) = path.rsplit('/').next() { node["name"] = json!(name); } - node["size"] = json!(size.to_string()); - node["content_digest"] = json!(format!("sha256:{}", hex_of(&digest))); + node["size"] = json!(fact.size.to_string()); + node["content_digest"] = json!(format!("sha256:{}", hex_of(&fact.digest))); entry["node"] = node; } - WalkOutcome::Absent => { + MetadataWalkOutcome::Absent => { entry["status"] = json!("absent"); } - WalkOutcome::NotDirectory { symlink } => { + MetadataWalkOutcome::NotDirectory { symlink } => { entry["status"] = if symlink { json!("symlink_traversal") } else { diff --git a/src/api/router/snapshot_session_tests.rs b/src/api/router/snapshot_session_tests.rs index 187499a4..7cf495b4 100644 --- a/src/api/router/snapshot_session_tests.rs +++ b/src/api/router/snapshot_session_tests.rs @@ -1125,3 +1125,6 @@ async fn mst2_durable_http_warm_reads_renew_and_frame_delivery_reject_primary_sc success_json(fixture.send("GET", "descriptor", Body::empty()).await).await; fixture.counts.assert(0, 0); } + +#[path = "snapshot_lookup_metadata_tests.rs"] +mod metadata_lookup; diff --git a/src/contract/git_protocol/mod.rs b/src/contract/git_protocol/mod.rs index e3b4dc7d..6a87531e 100644 --- a/src/contract/git_protocol/mod.rs +++ b/src/contract/git_protocol/mod.rs @@ -954,6 +954,7 @@ mod tests { .unwrap(); let bytes = Arc::new(Mutex::new(Vec::new())); let subscriber = tracing_subscriber::fmt() + .with_ansi(false) .with_writer(Writer(bytes.clone())) .finish(); let guard = tracing::subscriber::set_default(subscriber); diff --git a/src/jupiter/service/native_publication_push_tests.rs b/src/jupiter/service/native_publication_push_tests.rs index 8f2f9f76..4fb89aa1 100644 --- a/src/jupiter/service/native_publication_push_tests.rs +++ b/src/jupiter/service/native_publication_push_tests.rs @@ -314,13 +314,22 @@ async fn http_resolve_captured_before_commit_never_mixes_new_sequence_with_old_d assert!(matches!(wh03_exec(&storage,id).await,ExecuteOutcome::Done{..})); tokio::time::timeout(Duration::from_secs(5),release.wait()).await.unwrap(); let ((status,delayed),delayed_observations)=tokio::time::timeout(Duration::from_secs(10),&mut reader).await.unwrap().unwrap(); - assert_eq!(status,200,"{delayed}");assert_eq!(delayed["descriptor"],v1["descriptor"]); - assert_eq!(delayed["publication_sequence"],v1["publication_sequence"]); - assert_eq!(delayed_observations.len(),1); - let delayed_identity=delayed_observations[0].test_identity(); - for field in ["root_commit_oid","root_tree_oid","native_certificate_receipt_id","native_writer_epoch","native_publication_sequence","snapshot_id","metadata_root"] { - assert_eq!(delayed_identity[field],first_identity[field],"{field} mixed with a later publication"); - } + assert_eq!(status,503,"{delayed}"); + assert_eq!(delayed["error"]["code"],"SNAPSHOT_NOT_READY"); + assert_eq!(delayed["error"]["retryable"],true); + assert!(delayed.get("descriptor").is_none()); + assert!(delayed.get("publication_sequence").is_none()); + assert!(delayed_observations.is_empty(),"rejected stale handoff emitted a success observation"); + use tower::ServiceExt; + let old_app=crate::api::router::snapshot_router::routers(state.clone()).with_state(state.clone()); + let old=old_app.oneshot(axum::http::Request::builder().method("GET") + .uri(format!("/snapshots/{}/descriptor",v1["descriptor"]["snapshot_id"].as_str().unwrap())) + .header("x-mega-snapshot-lease",v1["lease_id"].as_str().unwrap()) + .body(axum::body::Body::empty()).unwrap()).await.unwrap(); + assert_eq!(old.status(),200); + let old=axum::body::to_bytes(old.into_body(),1_048_576).await.unwrap(); + let old:serde_json::Value=serde_json::from_slice(&old).unwrap(); + assert_eq!(old["descriptor"],v1["descriptor"],"already protected old SID changed with the writer"); let ((status,v2),next_observations)=crate::ceres::snapshot::projection_observation::with_observations(http_resolve(state.clone(),"/")).await; assert_eq!(status,200,"{v2}");assert_eq!(v2["publication_sequence"],"2"); assert_ne!(v2["descriptor"]["snapshot_id"],v1["descriptor"]["snapshot_id"]); @@ -328,6 +337,14 @@ async fn http_resolve_captured_before_commit_never_mixes_new_sequence_with_old_d let next_identity=next_observations[0].test_identity(); assert_eq!(next_identity["native_publication_sequence"],v2["publication_sequence"]); assert_eq!(next_identity["snapshot_id"],v2["descriptor"]["snapshot_id"]); + assert_eq!(next_identity["metadata_root"],v2["descriptor"]["metadata_root"]); + assert_eq!(next_identity["namespace_view_id"],v2["descriptor"]["namespace_view_id"]); + assert_eq!(next_identity["instance_id"],v2["descriptor"]["instance_id"]); + assert_eq!(next_identity["native_writer_epoch"],v2["writer_epoch"]); + let native_v2=storage.mono_storage().read_native_publication_head(storage.config().mst2.instance_uuid.as_deref().unwrap()).await.unwrap(); + assert_eq!(next_identity["root_commit_oid"],git_internal::hash::ObjectHash::from_hex_for_kind(git_internal::hash::get_hash_kind(),&native_v2.root.commit).unwrap().to_tagged_string()); + assert_eq!(next_identity["root_tree_oid"],git_internal::hash::ObjectHash::from_hex_for_kind(git_internal::hash::get_hash_kind(),&native_v2.root.tree).unwrap().to_tagged_string()); + assert_eq!(next_identity["native_certificate_receipt_id"].as_u64(),Some(native_v2.token.certificate.unwrap() as u64)); assert_ne!(next_identity["native_certificate_receipt_id"],first_identity["native_certificate_receipt_id"]); let ((status,_),failed_observations)=crate::ceres::snapshot::projection_observation::with_observations(http_resolve(state,"/absent-observation-scope")).await; assert_eq!(status,404); @@ -338,8 +355,8 @@ async fn http_resolve_captured_before_commit_never_mixes_new_sequence_with_old_d let directory=writer.test_directory(_temp.path()); let records=std::fs::read_to_string(directory.join("records.jsonl")).unwrap(); let records:Vec=records.lines().map(|line|serde_json::from_str(line).unwrap()).collect(); - assert_eq!(records.len(),4,"failed resolve must not be a successful observation"); - for (record,observation) in [(&records[0],&first_observations[0]),(&records[2],&delayed_observations[0]),(&records[3],&next_observations[0])] { + assert_eq!(records.len(),3,"failed resolve must not be a successful observation"); + for (record,observation) in [(&records[0],&first_observations[0]),(&records[2],&next_observations[0])] { let typed=serde_json::to_value(observation.wire_record()).unwrap(); assert_eq!(record["payload"],typed,"writer must preserve the full validated operation tuple"); assert_eq!(record["payload"].as_object().unwrap().len(),38); @@ -347,8 +364,8 @@ async fn http_resolve_captured_before_commit_never_mixes_new_sequence_with_old_d assert_ne!(records[0]["payload"]["request_id"],records[2]["payload"]["request_id"]); let status:serde_json::Value=serde_json::from_slice(&std::fs::read(directory.join("status.json")).unwrap()).unwrap(); assert_eq!(status["closed"],true); - assert_eq!(status["accepted_records"],4); - assert_eq!(status["written_records"],4); + assert_eq!(status["accepted_records"],3); + assert_eq!(status["written_records"],3); assert_eq!(status["first_error_code"],0); } } diff --git a/src/jupiter/storage/native_metadata_install_tests.rs b/src/jupiter/storage/native_metadata_install_tests.rs index 4b47fe38..f7662117 100644 --- a/src/jupiter/storage/native_metadata_install_tests.rs +++ b/src/jupiter/storage/native_metadata_install_tests.rs @@ -458,24 +458,79 @@ async fn overwrite_installed_payload_for_test( bytes: Vec, ) { assert_eq!(bytes.len(), page.bytes.len()); + let guard_modes = payload_trigger_modes_for_test(connection).await; + assert!( + guard_modes + .iter() + .any(|(name, _)| name == "mst2_metadata_payload_fenced") + ); + assert!(guard_modes.iter().all(|(_, mode)| mode == "O")); + assert!( + connection + .execute_raw(statement( + "UPDATE mst2_metadata_payload SET payload=payload WHERE page_id=$1", + [page.id.to_vec().into()], + )) + .await + .is_err(), + "production payload UPDATE fence must still reject writes" + ); let txn = connection.begin().await.unwrap(); + txn.execute_unprepared("SELECT pg_advisory_xact_lock(1296717362,hashtext(current_schema()))") + .await + .unwrap(); txn.execute_unprepared( - "ALTER TABLE mst2_metadata_payload DISABLE TRIGGER mst2_metadata_payload_immutable", + "ALTER TABLE mst2_metadata_payload DISABLE TRIGGER mst2_metadata_payload_fenced", ) .await .unwrap(); - txn.execute_raw(statement( - "UPDATE mst2_metadata_payload SET payload=$2 WHERE page_id=$1", - [page.id.to_vec().into(), bytes.into()], - )) - .await - .unwrap(); + let result = txn + .execute_raw(statement( + "UPDATE mst2_metadata_payload SET payload=$2 WHERE page_id=$1", + [page.id.to_vec().into(), bytes.into()], + )) + .await + .unwrap(); + assert_eq!(result.rows_affected(), 1); txn.execute_unprepared( - "ALTER TABLE mst2_metadata_payload ENABLE TRIGGER mst2_metadata_payload_immutable", + "ALTER TABLE mst2_metadata_payload ENABLE TRIGGER mst2_metadata_payload_fenced", ) .await .unwrap(); txn.commit().await.unwrap(); + assert_eq!( + payload_trigger_modes_for_test(connection).await, + guard_modes + ); + assert!( + connection + .execute_raw(statement( + "UPDATE mst2_metadata_payload SET payload=payload WHERE page_id=$1", + [page.id.to_vec().into()], + )) + .await + .is_err(), + "corruption injection must restore the production UPDATE fence" + ); +} + +async fn payload_trigger_modes_for_test(connection: &DatabaseConnection) -> Vec<(String, String)> { + connection + .query_all_raw(statement( + "SELECT tgname,tgenabled::text AS mode FROM pg_trigger \ + WHERE tgrelid='mst2_metadata_payload'::regclass AND NOT tgisinternal ORDER BY tgname", + [], + )) + .await + .unwrap() + .into_iter() + .map(|row| { + ( + row.try_get("", "tgname").unwrap(), + row.try_get("", "mode").unwrap(), + ) + }) + .collect() } #[tokio::test] From 9101b8f59d4e59b7b4a95c968a8c96313e5ad15e Mon Sep 17 00:00:00 2001 From: Xiaoyang Han Date: Wed, 7 Oct 2026 22:29:22 +0800 Subject: [PATCH 13/50] perf(mst2): validate and seal fixed metadata installation once (#62) Mint an opaque capability from complete immutable preparation evidence and freeze its durable identity and membership. PREPARING batches use bounded exact-member checks while preserving current primary, scope, state, physical lifetime and payload validation. COMMITTED replay and finalize retain their complete oracle. Native validation and measured performance remain pending. --- src/api/router/snapshot_content_tests.rs | 3 + .../snapshot_generation_upgrade_tests.rs | 1 + .../snapshot_install_capability_fixture.rs | 33 + ...1007_000500_add_mst2_install_capability.rs | 21 + .../m20261007_000500_install_capability.sql | 138 ++++ src/jupiter/migration/mod.rs | 14 +- .../storage/native_metadata_install.rs | 8 + .../native_metadata_install_capability.rs | 464 +++++++++++ ...metadata_install_capability_fault_tests.rs | 217 +++++ ...ative_metadata_install_capability_tests.rs | 782 ++++++++++++++++++ .../storage/native_metadata_install_tests.rs | 3 + .../storage/native_snapshot_session.rs | 6 +- 12 files changed, 1685 insertions(+), 5 deletions(-) create mode 100644 src/api/router/snapshot_install_capability_fixture.rs create mode 100644 src/jupiter/migration/m20261007_000500_add_mst2_install_capability.rs create mode 100644 src/jupiter/migration/m20261007_000500_install_capability.sql create mode 100644 src/jupiter/storage/native_metadata_install_capability.rs create mode 100644 src/jupiter/storage/native_metadata_install_capability_fault_tests.rs create mode 100644 src/jupiter/storage/native_metadata_install_capability_tests.rs diff --git a/src/api/router/snapshot_content_tests.rs b/src/api/router/snapshot_content_tests.rs index 79bc47ef..5399ad42 100644 --- a/src/api/router/snapshot_content_tests.rs +++ b/src/api/router/snapshot_content_tests.rs @@ -91,6 +91,9 @@ mod generation_history_fixture; #[path = "snapshot_generation_qualified_fixture.rs"] mod generation_qualified_fixture; +#[path = "snapshot_install_capability_fixture.rs"] +mod install_capability_fixture; + #[derive(Default)] struct ReadCounts { whole: AtomicUsize, diff --git a/src/api/router/snapshot_generation_upgrade_tests.rs b/src/api/router/snapshot_generation_upgrade_tests.rs index 9f14f96d..08684866 100644 --- a/src/api/router/snapshot_generation_upgrade_tests.rs +++ b/src/api/router/snapshot_generation_upgrade_tests.rs @@ -42,6 +42,7 @@ async fn mst2_generation_additive_upgrade_preserves_legacy_v3_sid_lease_and_new_ ); let pages = scalar(db, "SELECT count(*) FROM mst2_metadata_payload").await; let original = success_json(fixture.send("GET", "descriptor", Body::empty()).await).await; + super::install_capability_fixture::restore_pre_capability_schema(db).await; super::generation_history_fixture::restore_empty_g1_schema(db).await; // The unchanged v3 installer created these durable rows. Remove only the // empty additive schema in this isolated fixture to reproduce the previous diff --git a/src/api/router/snapshot_install_capability_fixture.rs b/src/api/router/snapshot_install_capability_fixture.rs new file mode 100644 index 00000000..1806c7ab --- /dev/null +++ b/src/api/router/snapshot_install_capability_fixture.rs @@ -0,0 +1,33 @@ +use sea_orm::{ConnectionTrait, DatabaseConnection, DbBackend, Statement}; + +pub(super) async fn restore_pre_capability_schema(db: &DatabaseConnection) { + let bound:i64=db.query_one_raw(Statement::from_string(DbBackend::Postgres, + "SELECT (SELECT count(*) FROM mst2_metadata_prepare WHERE storage_seal IS NOT NULL + OR canonical_bindings IS NOT NULL OR bindings_digest IS NOT NULL OR primary_scope IS NOT NULL + OR graph_domain IS NOT NULL)+(SELECT count(*) FROM mst2_metadata_prepare_page WHERE generation IS NOT NULL) + +(SELECT count(*) FROM mst2_metadata_payload WHERE generation IS NOT NULL)")) + .await.unwrap().unwrap().try_get_by_index(0).unwrap(); + assert_eq!( + bound, 0, + "old-schema fixture only contains unbound legacy installations" + ); + // Reproduce the deployment before this additive migration. Preserve every + // original plan, member, byte, context, lease and protection root. + db.execute_unprepared( + "DROP TRIGGER mst2_00_install_capability_barrier ON mst2_metadata_prepare; + DROP TRIGGER mst2_00_install_capability_barrier ON mst2_metadata_prepare_page; + DROP TRIGGER mst2_install_capability_truncate_guard ON mst2_metadata_prepare; + DROP TRIGGER mst2_install_capability_truncate_guard ON mst2_metadata_prepare_page; + DROP TRIGGER mst2_install_capability_prepare_guard ON mst2_metadata_prepare; + DROP TRIGGER mst2_install_capability_mapping_guard ON mst2_metadata_prepare_page; + DROP TABLE mst2_metadata_install_seal; + DROP FUNCTION mst2_install_capability_barrier(); + DROP FUNCTION mst2_install_capability_truncate_guard(); + DROP FUNCTION mst2_install_capability_prepare_guard(); + DROP FUNCTION mst2_install_capability_mapping_guard(); + DROP FUNCTION mst2_install_capability_register(); + DELETE FROM seaql_migrations WHERE version='m20261007_000500_add_mst2_install_capability'", + ) + .await + .unwrap(); +} diff --git a/src/jupiter/migration/m20261007_000500_add_mst2_install_capability.rs b/src/jupiter/migration/m20261007_000500_add_mst2_install_capability.rs new file mode 100644 index 00000000..95ddaca5 --- /dev/null +++ b/src/jupiter/migration/m20261007_000500_add_mst2_install_capability.rs @@ -0,0 +1,21 @@ +//! Freeze a fully validated legacy plan without granting generation authority. + +use sea_orm_migration::prelude::*; + +#[derive(DeriveMigrationName)] +pub struct Migration; + +#[async_trait::async_trait] +impl MigrationTrait for Migration { + async fn up(&self, manager: &SchemaManager) -> Result<(), DbErr> { + manager + .get_connection() + .execute_unprepared(include_str!("m20261007_000500_install_capability.sql")) + .await?; + Ok(()) + } + + async fn down(&self, _manager: &SchemaManager) -> Result<(), DbErr> { + Ok(()) + } +} diff --git a/src/jupiter/migration/m20261007_000500_install_capability.sql b/src/jupiter/migration/m20261007_000500_install_capability.sql new file mode 100644 index 00000000..b8114adf --- /dev/null +++ b/src/jupiter/migration/m20261007_000500_install_capability.sql @@ -0,0 +1,138 @@ +LOCK TABLE mst2_metadata_prepare,mst2_metadata_prepare_page IN ACCESS EXCLUSIVE MODE; + +CREATE TABLE mst2_metadata_install_seal ( + prepare_id text PRIMARY KEY REFERENCES mst2_metadata_prepare(prepare_id), + operation_id text NOT NULL CHECK (octet_length(operation_id) BETWEEN 1 AND 255), + manifest_digest bytea NOT NULL CHECK (octet_length(manifest_digest)=32), + members_digest bytea NOT NULL CHECK (octet_length(members_digest)=32), + primary_scope bytea NOT NULL CHECK (octet_length(primary_scope) BETWEEN 1 AND 16384), + install_seal bytea NOT NULL CHECK (octet_length(install_seal)=32), + created_at timestamptz NOT NULL DEFAULT clock_timestamp() +); + +-- The actual regulated table, rather than caller search_path, selects the lock. +CREATE FUNCTION mst2_install_capability_barrier() RETURNS trigger LANGUAGE plpgsql VOLATILE AS $$ +BEGIN + IF pg_catalog.current_schema() IS DISTINCT FROM TG_TABLE_SCHEMA + OR pg_catalog.pg_is_in_recovery() + OR pg_catalog.current_setting('transaction_isolation')<>'read committed' THEN + RAISE EXCEPTION 'install capability mutation requires its actual primary schema and READ COMMITTED'; + END IF; + PERFORM pg_catalog.set_config('search_path',pg_catalog.quote_ident(TG_TABLE_SCHEMA)||',pg_catalog,pg_temp',true); + PERFORM pg_catalog.set_config('lock_timeout','5000ms',true); + PERFORM pg_catalog.pg_advisory_xact_lock(1296717362,pg_catalog.hashtext(TG_TABLE_SCHEMA)); + RETURN NULL; +END $$; + +CREATE FUNCTION mst2_install_capability_register() RETURNS trigger LANGUAGE plpgsql VOLATILE AS $$ +DECLARE p record; members bytea; actual jsonb; member_count bigint; member_bytes bigint; has_root boolean; +BEGIN + IF TG_OP<>'INSERT' THEN RAISE EXCEPTION 'install capability registration is immutable'; END IF; + IF pg_catalog.current_schema() IS DISTINCT FROM TG_TABLE_SCHEMA THEN + RAISE EXCEPTION 'install capability registration is outside its actual schema'; + END IF; + EXECUTE pg_catalog.format('SELECT * FROM %I.mst2_metadata_prepare WHERE prepare_id=$1',TG_TABLE_SCHEMA) + INTO p USING NEW.prepare_id; + IF p.prepare_id IS NULL OR p.operation_id IS DISTINCT FROM NEW.operation_id + OR p.manifest_digest IS DISTINCT FROM NEW.manifest_digest + OR pg_catalog.sha256(p.canonical_plan) IS DISTINCT FROM NEW.manifest_digest + OR p.state NOT IN ('PREPARING','COMMITTED') OR p.coverage_retired_at IS NOT NULL + OR p.canonical_bindings IS NOT NULL OR p.bindings_digest IS NOT NULL + OR p.primary_scope IS NOT NULL OR p.storage_seal IS NOT NULL OR p.graph_domain IS NOT NULL THEN + RAISE EXCEPTION 'install capability requires a fixed unbound legacy preparation'; + END IF; + EXECUTE pg_catalog.format( + 'SELECT count(*),sum(expected_size),bool_or(page_id=$2), + sha256(string_agg(page_id||int4send(expected_size)|| + CASE WHEN generation IS NULL THEN decode(''00'',''hex'') + ELSE decode(''01'',''hex'')||int8send(generation) END,''''::bytea ORDER BY page_id)) + FROM %I.mst2_metadata_prepare_page WHERE prepare_id=$1',TG_TABLE_SCHEMA) + INTO member_count,member_bytes,has_root,members USING NEW.prepare_id,p.metadata_root; + IF member_count<>p.node_count OR member_bytes<>p.total_bytes OR has_root IS DISTINCT FROM true + OR members IS DISTINCT FROM NEW.members_digest THEN + RAISE EXCEPTION 'install capability complete membership proof disagrees'; + END IF; + EXECUTE pg_catalog.format( + 'SELECT EXISTS(SELECT 1 FROM %I.mst2_metadata_prepare_page WHERE prepare_id=$1 AND generation IS NOT NULL)', + TG_TABLE_SCHEMA) INTO has_root USING NEW.prepare_id; + IF has_root THEN RAISE EXCEPTION 'legacy install capability cannot adopt generation bindings'; END IF; + EXECUTE pg_catalog.format( + 'SELECT jsonb_build_array(s.storage_uuid,current_database(),d.oid::bigint,$1,n.oid::bigint, + inet_server_addr()::text,inet_server_port()) + FROM %I.mst2_metadata_storage_scope s + JOIN pg_catalog.pg_database d ON d.datname=current_database() + JOIN pg_catalog.pg_namespace n ON n.nspname=$1 WHERE s.singleton=1',TG_TABLE_SCHEMA) + INTO actual USING TG_TABLE_SCHEMA; + IF actual IS NULL OR pg_catalog.convert_from(NEW.primary_scope,'UTF8')::jsonb IS DISTINCT FROM actual + OR NEW.install_seal IS DISTINCT FROM pg_catalog.sha256( + pg_catalog.convert_to('MST2-LEGACY-INSTALL-CAPABILITY-1','UTF8')||pg_catalog.decode('00','hex')|| + pg_catalog.uuid_send(NEW.prepare_id::uuid)|| + pg_catalog.int4send(pg_catalog.octet_length(NEW.operation_id))||pg_catalog.convert_to(NEW.operation_id,'UTF8')|| + NEW.manifest_digest||NEW.members_digest|| + pg_catalog.int4send(pg_catalog.octet_length(NEW.primary_scope))||NEW.primary_scope) THEN + RAISE EXCEPTION 'install capability seal or actual primary scope disagrees'; + END IF; + RETURN NEW; +END $$; + +CREATE FUNCTION mst2_install_capability_prepare_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE AS $$ +DECLARE registered boolean; +BEGIN + IF pg_catalog.current_schema() IS DISTINCT FROM TG_TABLE_SCHEMA THEN + RAISE EXCEPTION 'install capability preparation is outside its actual schema'; + END IF; + EXECUTE pg_catalog.format('SELECT EXISTS(SELECT 1 FROM %I.mst2_metadata_install_seal WHERE prepare_id=$1)',TG_TABLE_SCHEMA) + INTO registered USING OLD.prepare_id; + IF registered AND (TG_OP='DELETE' OR + (pg_catalog.to_jsonb(NEW)-ARRAY['state','committed_at','aborted_at','coverage_retired_at']) IS DISTINCT FROM + (pg_catalog.to_jsonb(OLD)-ARRAY['state','committed_at','aborted_at','coverage_retired_at'])) THEN + RAISE EXCEPTION 'registered metadata preparation identity is immutable'; + END IF; + IF TG_OP='DELETE' THEN RETURN OLD; END IF; + RETURN NEW; +END $$; + +CREATE FUNCTION mst2_install_capability_mapping_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE AS $$ +DECLARE old_id text; new_id text; registered boolean; +BEGIN + IF pg_catalog.current_schema() IS DISTINCT FROM TG_TABLE_SCHEMA THEN + RAISE EXCEPTION 'install capability membership is outside its actual schema'; + END IF; + IF TG_OP<>'INSERT' THEN old_id:=OLD.prepare_id; END IF; + IF TG_OP<>'DELETE' THEN new_id:=NEW.prepare_id; END IF; + EXECUTE pg_catalog.format( + 'SELECT EXISTS(SELECT 1 FROM %I.mst2_metadata_install_seal WHERE prepare_id=$1 OR prepare_id=$2)',TG_TABLE_SCHEMA) + INTO registered USING old_id,new_id; + IF registered THEN RAISE EXCEPTION 'registered metadata membership is immutable'; END IF; + IF TG_OP='DELETE' THEN RETURN OLD; END IF; + RETURN NEW; +END $$; + +CREATE FUNCTION mst2_install_capability_truncate_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE AS $$ +DECLARE registered boolean; +BEGIN + IF pg_catalog.current_schema() IS DISTINCT FROM TG_TABLE_SCHEMA + OR pg_catalog.pg_is_in_recovery() + OR pg_catalog.current_setting('transaction_isolation')<>'read committed' THEN + RAISE EXCEPTION 'install capability truncation requires its actual primary schema and READ COMMITTED'; + END IF; + PERFORM pg_catalog.set_config('lock_timeout','5000ms',true); + PERFORM pg_catalog.pg_advisory_xact_lock(1296717362,pg_catalog.hashtext(TG_TABLE_SCHEMA)); + EXECUTE pg_catalog.format('SELECT EXISTS(SELECT 1 FROM %I.mst2_metadata_install_seal)',TG_TABLE_SCHEMA) + INTO registered; + IF registered THEN RAISE EXCEPTION 'registered metadata evidence cannot be truncated'; END IF; + RETURN NULL; +END $$; + +DO $$ DECLARE t text; BEGIN + FOREACH t IN ARRAY ARRAY['mst2_metadata_prepare','mst2_metadata_prepare_page','mst2_metadata_install_seal'] LOOP + EXECUTE pg_catalog.format('CREATE TRIGGER mst2_00_install_capability_barrier BEFORE INSERT OR UPDATE OR DELETE ON %I FOR EACH STATEMENT EXECUTE FUNCTION mst2_install_capability_barrier()',t); + EXECUTE pg_catalog.format('CREATE TRIGGER mst2_install_capability_truncate_guard BEFORE TRUNCATE ON %I FOR EACH STATEMENT EXECUTE FUNCTION mst2_install_capability_truncate_guard()',t); + END LOOP; +END $$; +CREATE TRIGGER mst2_install_capability_register BEFORE INSERT OR UPDATE OR DELETE + ON mst2_metadata_install_seal FOR EACH ROW EXECUTE FUNCTION mst2_install_capability_register(); +CREATE TRIGGER mst2_install_capability_prepare_guard BEFORE UPDATE OR DELETE + ON mst2_metadata_prepare FOR EACH ROW EXECUTE FUNCTION mst2_install_capability_prepare_guard(); +CREATE TRIGGER mst2_install_capability_mapping_guard BEFORE INSERT OR UPDATE OR DELETE + ON mst2_metadata_prepare_page FOR EACH ROW EXECUTE FUNCTION mst2_install_capability_mapping_guard(); diff --git a/src/jupiter/migration/mod.rs b/src/jupiter/migration/mod.rs index 384af69a..fa611fce 100644 --- a/src/jupiter/migration/mod.rs +++ b/src/jupiter/migration/mod.rs @@ -149,6 +149,7 @@ mod m20261007_000100_add_mst2_snapshot_sessions; mod m20261007_000200_add_mst2_metadata_generations; mod m20261007_000300_add_mst2_metadata_lifetime_history; mod m20261007_000400_add_mst2_qualified_metadata_gc; +mod m20261007_000500_add_mst2_install_capability; mod runner; pub use m20260905_000100_add_push_queue::ensure_queue_control_seed; pub use runner::apply_migrations; @@ -286,6 +287,7 @@ impl MigratorTrait for Migrator { Box::new(m20261007_000200_add_mst2_metadata_generations::Migration), Box::new(m20261007_000300_add_mst2_metadata_lifetime_history::Migration), Box::new(m20261007_000400_add_mst2_qualified_metadata_gc::Migration), + Box::new(m20261007_000500_add_mst2_install_capability::Migration), ] } } @@ -1196,7 +1198,7 @@ mod tests { async fn import_repo_alias_rows_canonicalized() { let names = migration_names(); assert_eq!( - &names[names.len() - 12..names.len() - 3], + &names[names.len() - 13..names.len() - 4], &[ "m20260923_000200_canonicalize_import_repo_paths".to_string(), "m20260925_000100_media_paging".to_string(), @@ -1211,17 +1213,21 @@ mod tests { "native retention, metadata installation and view tables follow media paging" ); assert_eq!( - &names[names.len() - 3], + &names[names.len() - 4], "m20261007_000200_add_mst2_metadata_generations" ); assert_eq!( - &names[names.len() - 2], + &names[names.len() - 3], "m20261007_000300_add_mst2_metadata_lifetime_history" ); assert_eq!( - names.last().unwrap(), + &names[names.len() - 2], "m20261007_000400_add_mst2_qualified_metadata_gc" ); + assert_eq!( + names.last().unwrap(), + "m20261007_000500_add_mst2_install_capability" + ); let db = alias_db().await; insert_repo(&db, 1, "/third-party//a").await; diff --git a/src/jupiter/storage/native_metadata_install.rs b/src/jupiter/storage/native_metadata_install.rs index d4be4cad..735dcc54 100644 --- a/src/jupiter/storage/native_metadata_install.rs +++ b/src/jupiter/storage/native_metadata_install.rs @@ -1050,6 +1050,14 @@ fn unavailable(message: &str) -> SnapshotError { SnapshotError::new(SnapshotErrorCode::ObjectUnavailable, message) } +#[path = "native_metadata_install_capability.rs"] +mod capability; +pub(crate) use capability::ValidatedLegacyInstallCapability; + +#[cfg(test)] +#[path = "native_metadata_install_capability_tests.rs"] +mod capability_tests; + // Additive repository only. HTTP sessions continue using the existing installer // until session anchors and lease adoption bind the generation seal. #[path = "native_metadata_generations.rs"] diff --git a/src/jupiter/storage/native_metadata_install_capability.rs b/src/jupiter/storage/native_metadata_install_capability.rs new file mode 100644 index 00000000..348405e1 --- /dev/null +++ b/src/jupiter/storage/native_metadata_install_capability.rs @@ -0,0 +1,464 @@ +//! Reuse pure plan validation only after its durable membership is frozen. + +use sha2::{Digest, Sha256}; + +use super::{ + BTreeMap, BTreeSet, ConnectionTrait, DatabaseTransaction, MetadataCommitPhase, + MetadataInstallError, MetadataInstallIdentity, MetadataPagePayload, MetadataPrepareIntent, + PostgresMetadataInstallRepository, PrimaryStorageScope, QueryResult, SnapshotError, + SnapshotErrorCode, check_payload_coverage, commit, integrity, internal, json, + load_installed_dag, require_plan, statement, unavailable, validate_payload, verify_graph, +}; + +const SEAL_DOMAIN: &[u8] = b"MST2-LEGACY-INSTALL-CAPABILITY-1\0"; + +#[derive(Debug)] +pub(crate) struct ValidatedLegacyInstallCapability { + intent: MetadataPrepareIntent, + identity: MetadataInstallIdentity, + root: [u8; 32], + edge_count: usize, + total_bytes: u64, + members: BTreeMap<[u8; 32], (i32, Option)>, + members_digest: [u8; 32], + scope: PrimaryStorageScope, + primary_scope: Vec, + install_seal: [u8; 32], +} + +impl PostgresMetadataInstallRepository { + pub(crate) async fn mint_legacy_install_capability( + &self, + intent: &MetadataPrepareIntent, + ) -> Result { + let txn = self.transaction().await?; + let result = async { + self.capability_barrier(&txn).await?; + let stored = require_plan(&txn, intent).await?; + let record = &stored.record; + if record.storage_seal.is_some() + || record.canonical_bindings.is_some() + || record.bindings_digest.is_some() + || record.primary_scope.is_some() + || record.graph_domain.is_some() + || record.coverage_retired_at.is_some() + || record.aborted_at.is_some() + { + return Err(unavailable("install capability requires an active unbound legacy preparation")); + } + let mut members = BTreeMap::new(); + for member in &stored.prepare_pages { + if member.generation.is_some() { + return Err(integrity("legacy install capability cannot adopt generation bindings")); + } + let id = member.page_id.as_slice().try_into().map_err(internal)?; + members.insert(id, (member.expected_size, member.generation)); + } + if record.state == "COMMITTED" { + load_installed_dag(&txn, &stored).await?; + check_payload_coverage(&txn, &stored).await?; + verify_graph(&txn, &stored).await?; + } + let members_digest = member_digest(&members); + let primary_scope = scope_bytes(&self.storage_scope)?; + let install_seal = seal(intent, &members_digest, &primary_scope)?; + txn.execute_raw(statement( + "INSERT INTO mst2_metadata_install_seal(prepare_id,operation_id,manifest_digest,members_digest,primary_scope,install_seal) + VALUES($1,$2,$3,$4,$5,$6) ON CONFLICT(prepare_id) DO NOTHING", + [intent.prepare_id.clone().into(),intent.operation_id.clone().into(), + intent.manifest_digest.to_vec().into(),members_digest.to_vec().into(), + primary_scope.clone().into(),install_seal.to_vec().into()], + )).await.map_err(internal)?; + let capability = ValidatedLegacyInstallCapability { + intent: intent.clone(), identity: stored.plan.identity, root: stored.plan.root, + edge_count: stored.plan.edges.len(), total_bytes: stored.plan.total_bytes, + members,members_digest,scope:self.storage_scope.clone(),primary_scope,install_seal, + }; + read_registered_prepare(&txn, &capability).await?; + Ok(capability) + }.await; + commit( + txn, + result, + &intent.operation_id, + intent.manifest_digest, + MetadataCommitPhase::Intent, + ) + .await + } + + pub(crate) async fn install_pages_validated( + &self, + capability: &ValidatedLegacyInstallCapability, + payloads: &[MetadataPagePayload], + ) -> Result<(), MetadataInstallError> { + if capability.scope != self.storage_scope { + return Err( + integrity("install capability belongs to another captured primary scope").into(), + ); + } + if payloads.is_empty() || payloads.len() > 64 { + return Err(SnapshotError::new( + SnapshotErrorCode::LimitExceeded, + "metadata installation batch must contain 1..=64 pages", + ) + .into()); + } + let mut ids = BTreeSet::new(); + for payload in payloads { + validate_payload(payload)?; + if !ids.insert(payload.id) { + return Err(integrity("duplicate page in metadata installation batch").into()); + } + if capability + .members + .get(&payload.id) + .map(|member| member.0 as u64) + != Some(payload.size) + { + return Err(integrity( + "metadata payload is not a member of its validated installation", + ) + .into()); + } + } + let pages: Vec<_> = payloads + .iter() + .map(|p| { + json!({ + "page_id":hex::encode(p.id),"size":p.size,"payload":hex::encode(&p.bytes) + }) + }) + .collect(); + let encoded = serde_json::to_string(&pages).map_err(internal)?; + let txn = self.transaction().await?; + let result = async { + self.capability_barrier(&txn).await?; + let state = read_registered_prepare(&txn,capability).await?; + check_requested_members(&txn,capability,&encoded,payloads.len()).await?; + if state == "COMMITTED" { + // Committed replay retains the complete receipt oracle and never repairs bytes. + let stored = require_plan(&txn,&capability.intent).await?; + load_installed_dag(&txn,&stored).await?; + check_payload_coverage(&txn,&stored).await?; + verify_graph(&txn,&stored).await?; + } else { + txn.execute_raw(statement( + "INSERT INTO mst2_metadata_payload(page_id,metadata_codec,byte_size,payload) + SELECT decode(p.page_id,'hex'),$1,p.size,decode(p.payload,'hex') + FROM jsonb_to_recordset($2::jsonb) AS p(page_id text,size integer,payload text) + ON CONFLICT(page_id) DO NOTHING", + [(capability.identity.metadata_codec as i16).into(),encoded.clone().into()], + )).await.map_err(internal)?; + } + let bad = txn.query_one_raw(statement( + "SELECT p.page_id FROM jsonb_to_recordset($2::jsonb) AS p(page_id text,size integer,payload text) + LEFT JOIN mst2_metadata_payload b ON b.page_id=decode(p.page_id,'hex') + WHERE b.page_id IS NULL OR b.metadata_codec<>$1 OR b.byte_size<>p.size + OR b.payload<>decode(p.payload,'hex') LIMIT 1", + [(capability.identity.metadata_codec as i16).into(),encoded.into()], + )).await.map_err(internal)?; + if bad.is_some() { + return Err(integrity("immutable metadata payload identity conflicts with stored bytes")); + } + Ok(()) + }.await; + commit( + txn, + result, + &capability.intent.operation_id, + capability.intent.manifest_digest, + MetadataCommitPhase::Payload, + ) + .await + } + + pub(super) async fn capability_barrier( + &self, + txn: &DatabaseTransaction, + ) -> Result<(), SnapshotError> { + let schema = txn + .query_one_raw(statement( + "SELECT pg_catalog.current_schema() AS schema", + [], + )) + .await + .map_err(internal)? + .ok_or_else(|| internal("primary schema is missing"))? + .try_get::("", "schema") + .map_err(internal)?; + if schema != self.storage_scope.schema { + return Err(integrity( + "install capability transaction is outside its captured schema", + )); + } + txn.execute_raw(statement( + "SELECT pg_catalog.set_config('search_path',pg_catalog.quote_ident($1)||',pg_catalog,pg_temp',true)", + [schema.into()], + )).await.map_err(internal)?; + self.barrier(txn).await + } +} + +async fn read_registered_prepare( + txn: &DatabaseTransaction, + capability: &ValidatedLegacyInstallCapability, +) -> Result { + let row=txn.query_one_raw(statement( + "SELECT p.prepare_id,p.operation_id,p.manifest_digest,p.source_domain,p.tagged_root_tree_oid,p.scope, + p.schema_version,p.metadata_codec,p.materialization_policy,p.fs_semantics,p.access_projection, + p.verification_revision,p.projection_revision,p.metadata_root,p.node_count,p.edge_count,p.total_bytes, + p.state,p.committed_at IS NOT NULL AS committed,p.aborted_at IS NOT NULL AS aborted, + p.coverage_retired_at IS NOT NULL AS retired, + p.storage_seal IS NOT NULL OR p.canonical_bindings IS NOT NULL OR p.bindings_digest IS NOT NULL + OR p.primary_scope IS NOT NULL OR p.graph_domain IS NOT NULL AS generation_bound, + s.operation_id AS sealed_operation,s.manifest_digest AS sealed_manifest,s.members_digest, + s.primary_scope AS sealed_scope,s.install_seal + FROM mst2_metadata_prepare p JOIN mst2_metadata_install_seal s USING(prepare_id) + WHERE p.prepare_id=$1", + [capability.intent.prepare_id.clone().into()], + )).await.map_err(internal)?.ok_or_else(||unavailable("validated install registration is missing"))?; + let identity = &capability.identity; + for (column, expected) in [ + ("prepare_id", &capability.intent.prepare_id), + ("operation_id", &capability.intent.operation_id), + ("sealed_operation", &capability.intent.operation_id), + ("source_domain", &identity.source_domain), + ("tagged_root_tree_oid", &identity.tagged_root_tree_oid), + ("scope", &identity.scope), + ] { + if row.try_get::("", column).map_err(internal)? != *expected { + return Err(integrity( + "validated install identity differs from durable registration", + )); + } + } + for (column, expected) in [ + ("schema_version", identity.schema_version), + ("metadata_codec", identity.metadata_codec), + ("materialization_policy", identity.materialization_policy), + ("fs_semantics", identity.fs_semantics), + ("access_projection", identity.access_projection), + ("projection_revision", identity.projection_revision), + ] { + if row.try_get::("", column).map_err(internal)? != expected as i16 { + return Err(integrity( + "validated install profile differs from durable registration", + )); + } + } + for (column, expected) in [ + ( + "manifest_digest", + capability.intent.manifest_digest.as_slice(), + ), + ( + "sealed_manifest", + capability.intent.manifest_digest.as_slice(), + ), + ("metadata_root", capability.root.as_slice()), + ("members_digest", capability.members_digest.as_slice()), + ("sealed_scope", capability.primary_scope.as_slice()), + ("install_seal", capability.install_seal.as_slice()), + ] { + if row + .try_get::>("", column) + .map_err(internal)? + .as_slice() + != expected + { + return Err(integrity( + "validated install proof differs from durable registration", + )); + } + } + let state = row.try_get::("", "state").map_err(internal)?; + if row + .try_get::("", "verification_revision") + .map_err(internal)? + != identity.verification_revision + || row.try_get::("", "node_count").map_err(internal)? as usize + != capability.members.len() + || row.try_get::("", "edge_count").map_err(internal)? as usize != capability.edge_count + || row.try_get::("", "total_bytes").map_err(internal)? as u64 != capability.total_bytes + || !["PREPARING", "COMMITTED"].contains(&state.as_str()) + || row.try_get::("", "committed").map_err(internal)? != (state == "COMMITTED") + || row.try_get::("", "aborted").map_err(internal)? + || row.try_get::("", "retired").map_err(internal)? + || row + .try_get::("", "generation_bound") + .map_err(internal)? + { + return Err(unavailable( + "validated install preparation is no longer active with its fixed profile", + )); + } + Ok(state) +} + +async fn check_requested_members( + txn: &DatabaseTransaction, + capability: &ValidatedLegacyInstallCapability, + encoded: &str, + expected_count: usize, +) -> Result<(), SnapshotError> { + let rows=txn.query_all_raw(statement( + "SELECT decode(p.page_id,'hex') AS page_id,m.expected_size,m.generation AS member_generation, + b.page_id IS NOT NULL AS payload_present,b.generation AS payload_generation, + c.generation AS current_generation,l.generation AS lifetime_generation,l.graph_domain, + l.state AS lifetime_state,l.metadata_codec AS lifetime_codec,l.expected_size AS lifetime_size, + n.state AS graph_state,n.kind AS graph_kind,n.bytes AS graph_bytes, + EXISTS(SELECT 1 FROM mst2_retention_gc_op g WHERE g.node_id='page:sha256:'||p.page_id + AND g.operation='REMOVE' AND g.state IN ('PENDING','APPLIED')) AS tombstone + FROM jsonb_to_recordset($2::jsonb) AS p(page_id text,size integer,payload text) + LEFT JOIN mst2_metadata_prepare_page m ON m.prepare_id=$1 AND m.page_id=decode(p.page_id,'hex') + LEFT JOIN mst2_metadata_payload b ON b.page_id=m.page_id + LEFT JOIN mst2_metadata_current c ON c.page_id=m.page_id + LEFT JOIN mst2_metadata_lifetime l ON l.page_id=c.page_id AND l.generation=c.generation + LEFT JOIN mst2_retention_node n ON n.node_id='page:sha256:'||p.page_id", + [capability.intent.prepare_id.clone().into(),encoded.into()], + )).await.map_err(internal)?; + if rows.len() != expected_count { + return Err(integrity("requested install membership is incomplete")); + } + for row in rows { + let page: Vec = row.try_get("", "page_id").map_err(internal)?; + let page: [u8; 32] = page.as_slice().try_into().map_err(internal)?; + let size = row + .try_get::>("", "expected_size") + .map_err(internal)?; + let generation = row + .try_get::>("", "member_generation") + .map_err(internal)?; + if capability.members.get(&page).copied() != size.map(|size| (size, generation)) { + return Err(integrity( + "requested member differs from validated installation", + )); + } + check_physical_member( + &row, + capability.identity.metadata_codec as i16, + size.ok_or_else(|| integrity("requested member is missing"))?, + )?; + } + Ok(()) +} + +fn check_physical_member(row: &QueryResult, codec: i16, size: i32) -> Result<(), SnapshotError> { + let current = row + .try_get::>("", "current_generation") + .map_err(internal)?; + let lifetime = row + .try_get::>("", "lifetime_generation") + .map_err(internal)?; + let payload = row + .try_get::>("", "payload_generation") + .map_err(internal)?; + let graph = row + .try_get::>("", "graph_state") + .map_err(internal)?; + if row.try_get::("", "tombstone").map_err(internal)? + || graph.as_deref().is_some_and(|state| state != "LIVE") + || row + .try_get::>("", "graph_kind") + .map_err(internal)? + .is_some_and(|kind| kind != "page") + || row + .try_get::>("", "graph_bytes") + .map_err(internal)? + .is_some_and(|bytes| bytes != size as i64) + { + return Err(unavailable( + "requested generic metadata graph is unavailable", + )); + } + if payload.is_some() && (payload != current || payload != lifetime) { + return Err(integrity( + "generic payload is not its exact current physical lifetime", + )); + } + if current.is_some() { + let state = row + .try_get::>("", "lifetime_state") + .map_err(internal)?; + if current != lifetime + || current.is_some_and(|generation| generation <= 0) + || row + .try_get::>("", "graph_domain") + .map_err(internal)? + .as_deref() + != Some("generic-v1") + || !matches!(state.as_deref(), Some("RESERVED" | "LIVE")) + || row + .try_get::>("", "lifetime_codec") + .map_err(internal)? + != Some(codec) + || row + .try_get::>("", "lifetime_size") + .map_err(internal)? + != Some(size) + || (state.as_deref() == Some("LIVE") + && (graph.is_none() + || payload != current + || !row + .try_get::("", "payload_present") + .map_err(internal)?)) + { + return Err(unavailable( + "requested physical lifetime cannot be shared by legacy generic installation", + )); + } + } + Ok(()) +} + +fn member_digest(members: &BTreeMap<[u8; 32], (i32, Option)>) -> [u8; 32] { + let mut hash = Sha256::new(); + for (id, (size, generation)) in members { + hash.update(id); + hash.update(size.to_be_bytes()); + match generation { + None => hash.update([0]), + Some(generation) => { + hash.update([1]); + hash.update(generation.to_be_bytes()); + } + } + } + hash.finalize().into() +} + +fn scope_bytes(scope: &PrimaryStorageScope) -> Result, SnapshotError> { + let bytes = serde_json::to_vec(&( + &scope.storage_uuid, + &scope.database, + scope.database_oid, + &scope.schema, + scope.schema_oid, + &scope.server_address, + scope.server_port, + )) + .map_err(internal)?; + if bytes.is_empty() || bytes.len() > 16384 { + return Err(integrity("invalid primary install scope size")); + } + Ok(bytes) +} + +fn seal( + intent: &MetadataPrepareIntent, + members: &[u8; 32], + scope: &[u8], +) -> Result<[u8; 32], SnapshotError> { + let id = uuid::Uuid::parse_str(&intent.prepare_id).map_err(internal)?; + let mut hash = Sha256::new(); + hash.update(SEAL_DOMAIN); + hash.update(id.as_bytes()); + hash.update((intent.operation_id.len() as u32).to_be_bytes()); + hash.update(intent.operation_id.as_bytes()); + hash.update(intent.manifest_digest); + hash.update(members); + hash.update((scope.len() as u32).to_be_bytes()); + hash.update(scope); + Ok(hash.finalize().into()) +} diff --git a/src/jupiter/storage/native_metadata_install_capability_fault_tests.rs b/src/jupiter/storage/native_metadata_install_capability_fault_tests.rs new file mode 100644 index 00000000..9903085e --- /dev/null +++ b/src/jupiter/storage/native_metadata_install_capability_fault_tests.rs @@ -0,0 +1,217 @@ +use super::*; + +async fn count(db: &DatabaseConnection, sql: &str) -> i64 { + db.query_one_raw(statement(sql, [])) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap() +} + +async fn mint_fault(fault: Fault) { + let (direct, recovery, _schema, url) = fixture().await; + let proxy = PgCommitFaultProxy::start(&url, fault).await; + let mut options = sea_orm::ConnectOptions::new(proxy.url.clone()); + options.max_connections(1).min_connections(1); + let repository = + PostgresMetadataInstallRepository::new(Database::connect(options).await.unwrap()) + .await + .unwrap(); + let pages = prepared("/"); + let intent = repository + .begin_intent("cap-mint-fault", &pages) + .await + .unwrap(); + proxy.armed.store(true, Ordering::SeqCst); + let error = tokio::time::timeout( + Duration::from_secs(15), + repository.mint_legacy_install_capability(&intent), + ) + .await + .unwrap() + .unwrap_err(); + assert!(matches!( + error, + MetadataInstallError::CommitUncertain { + phase: MetadataCommitPhase::Intent, + .. + } + )); + proxy.wait_for_fault().await; + assert_eq!( + count(&direct, "SELECT count(*) FROM mst2_metadata_payload").await, + 0 + ); + assert_eq!( + count(&direct, "SELECT count(*) FROM mst2_metadata_install_seal").await, + if matches!(fault, Fault::AfterCommit) { + 1 + } else { + 0 + } + ); + assert_eq!( + proxy.commit_observed.load(Ordering::SeqCst), + matches!(fault, Fault::AfterCommit) + ); + let restarted = PostgresMetadataInstallRepository::new(recovery.clone()) + .await + .unwrap(); + let cap = restarted + .mint_legacy_install_capability(&intent) + .await + .unwrap(); + assert_eq!( + count(&direct, "SELECT count(*) FROM mst2_metadata_install_seal").await, + 1 + ); + restarted + .install_pages_validated(&cap, pages.dag().payloads()) + .await + .unwrap(); + restarted.finalize(&intent).await.unwrap(); +} + +async fn batch_fault(fault: Fault) { + let (direct, recovery, _schema, url) = fixture().await; + let proxy = PgCommitFaultProxy::start(&url, fault).await; + let mut options = sea_orm::ConnectOptions::new(proxy.url.clone()); + options.max_connections(1).min_connections(1); + let repository = + PostgresMetadataInstallRepository::new(Database::connect(options).await.unwrap()) + .await + .unwrap(); + let pages = prepared("/"); + let intent = repository + .begin_intent("cap-batch-fault", &pages) + .await + .unwrap(); + let cap = repository + .mint_legacy_install_capability(&intent) + .await + .unwrap(); + proxy.armed.store(true, Ordering::SeqCst); + let error = tokio::time::timeout( + Duration::from_secs(15), + repository.install_pages_validated(&cap, pages.dag().payloads()), + ) + .await + .unwrap() + .unwrap_err(); + assert!(matches!( + error, + MetadataInstallError::CommitUncertain { + phase: MetadataCommitPhase::Payload, + .. + } + )); + proxy.wait_for_fault().await; + assert_eq!( + count(&direct, "SELECT count(*) FROM mst2_metadata_install_seal").await, + 1 + ); + assert_eq!( + count(&direct, "SELECT count(*) FROM mst2_metadata_payload").await, + if matches!(fault, Fault::AfterCommit) { + pages.dag().payloads().len() as i64 + } else { + 0 + } + ); + let restarted = PostgresMetadataInstallRepository::new(recovery.clone()) + .await + .unwrap(); + let fresh = restarted + .mint_legacy_install_capability(&intent) + .await + .unwrap(); + restarted + .install_pages_validated(&fresh, pages.dag().payloads()) + .await + .unwrap(); + let receipt = restarted.finalize(&intent).await.unwrap(); + assert_eq!(receipt.metadata_root(), pages.dag().root()); + assert_eq!( + count( + &direct, + "SELECT count(*) FROM mst2_retention_root WHERE root_kind='prepare'" + ) + .await, + pages.dag().payloads().len() as i64 + ); + restarted + .install_pages_validated(&fresh, pages.dag().payloads()) + .await + .unwrap(); + assert_eq!(restarted.finalize(&intent).await.unwrap(), receipt); +} + +#[tokio::test] +async fn install_capability_real_registration_commit_response_loss_recovers_fresh_proof() { + mint_fault(Fault::AfterCommit).await; +} + +#[tokio::test] +async fn install_capability_real_registration_rollback_leaves_no_freeze_or_capability() { + mint_fault(Fault::BeforeCommit).await; +} + +#[tokio::test] +async fn install_capability_real_batch_commit_response_loss_preserves_exact_bytes() { + batch_fault(Fault::AfterCommit).await; +} + +#[tokio::test] +async fn install_capability_real_batch_connection_loss_rolls_back_without_false_success() { + batch_fault(Fault::BeforeCommit).await; +} + +#[tokio::test] +async fn install_capability_failed_registration_trigger_rolls_back_freeze_atomically() { + let (direct, recovery, _schema, _url) = fixture().await; + let repository = PostgresMetadataInstallRepository::new(direct.clone()) + .await + .unwrap(); + let pages = prepared("/"); + let intent = repository + .begin_intent("cap-register-rollback", &pages) + .await + .unwrap(); + direct.execute_unprepared("CREATE FUNCTION test_cap_registration_fault() RETURNS trigger LANGUAGE plpgsql AS $$ + BEGIN RAISE EXCEPTION 'registration fault after production guard'; END $$; + CREATE TRIGGER test_cap_registration_fault AFTER INSERT ON mst2_metadata_install_seal FOR EACH ROW + EXECUTE FUNCTION test_cap_registration_fault()").await.unwrap(); + assert!( + repository + .mint_legacy_install_capability(&intent) + .await + .is_err() + ); + assert_eq!( + count(&direct, "SELECT count(*) FROM mst2_metadata_install_seal").await, + 0 + ); + recovery + .execute_unprepared("UPDATE mst2_metadata_prepare_page SET expected_size=expected_size") + .await + .unwrap(); + direct + .execute_unprepared( + "DROP TRIGGER test_cap_registration_fault ON mst2_metadata_install_seal; + DROP FUNCTION test_cap_registration_fault()", + ) + .await + .unwrap(); + let restarted = PostgresMetadataInstallRepository::new(recovery) + .await + .unwrap(); + let cap = restarted + .mint_legacy_install_capability(&intent) + .await + .unwrap(); + restarted + .install_pages_validated(&cap, pages.dag().payloads()) + .await + .unwrap(); +} diff --git a/src/jupiter/storage/native_metadata_install_capability_tests.rs b/src/jupiter/storage/native_metadata_install_capability_tests.rs new file mode 100644 index 00000000..fb9df08f --- /dev/null +++ b/src/jupiter/storage/native_metadata_install_capability_tests.rs @@ -0,0 +1,782 @@ +use std::sync::Arc; + +use mst2_codec::metapage::{Entry, EntryKind}; +use sea_orm::{Database, PaginatorTrait}; +use sea_orm_migration::MigratorTrait; + +use super::*; +use crate::{ + ceres::snapshot::retention_dag::MetadataDagBuilder, + jupiter::{ + migration::Migrator, + tests::{TestSchemaGuard, test_db_config}, + }, +}; + +fn prepared(children: usize) -> PreparedNativeMetadataRetention { + let mut builder = MetadataDagBuilder::new(MetadataDagLimits::default()); + let mut entries = Vec::new(); + for index in 0..children { + let file = [Entry::file( + EntryKind::Regular, + b"file", + 3, + [index as u8; 32], + )]; + let child = Page::build(&file).unwrap(); + builder.add_directory(&child, &file).unwrap(); + entries.push(Entry::dir( + format!("dir-{index:03}").as_bytes(), + page_id(&child), + )); + } + let root = Page::build(&entries).unwrap(); + builder.add_directory(&root, &entries).unwrap(); + PreparedNativeMetadataRetention::test_installation( + Arc::new(builder.finish(page_id(&root)).unwrap()), + "/", + ) +} + +async fn fixture() -> (DatabaseConnection, DatabaseConnection, TestSchemaGuard) { + let temp = tempfile::tempdir().unwrap(); + let (config, schema) = test_db_config(temp.path()).await; + let first = Database::connect(config.db_url.clone()).await.unwrap(); + Migrator::up(&first, None).await.unwrap(); + let second = Database::connect(config.db_url).await.unwrap(); + (first, second, schema) +} + +async fn scalar(db: &C, sql: &str) -> i64 { + db.query_one_raw(statement(sql, [])) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap() +} + +async fn install( + repository: &PostgresMetadataInstallRepository, + capability: &ValidatedLegacyInstallCapability, + prepared: &PreparedNativeMetadataRetention, +) { + for batch in prepared.dag().payloads().chunks(64) { + repository + .install_pages_validated(capability, batch) + .await + .unwrap(); + } +} + +async fn assert_guard(db: &DatabaseConnection, sql: &str) { + let error = db.execute_unprepared(sql).await.unwrap_err(); + assert!( + error.to_string().contains("install capability") + || error.to_string().contains("registered metadata"), + "{error}" + ); +} + +#[tokio::test] +async fn install_capability_many_batches_concurrent_mint_and_fresh_repository_finalize() { + let (first, second, _schema) = fixture().await; + let a = PostgresMetadataInstallRepository::new(first.clone()) + .await + .unwrap(); + let b = PostgresMetadataInstallRepository::new(second.clone()) + .await + .unwrap(); + let pages = prepared(130); + assert_eq!(pages.dag().payloads().len(), 131); + let intent = a.begin_intent("many", &pages).await.unwrap(); + let (left, right) = tokio::join!( + a.mint_legacy_install_capability(&intent), + b.mint_legacy_install_capability(&intent) + ); + let left = left.unwrap(); + let right = right.unwrap(); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_install_seal").await, + 1 + ); + for (index, batch) in pages.dag().payloads().chunks(64).enumerate() { + if index == 0 { + a.install_pages_validated(&left, batch).await.unwrap(); + } else { + b.install_pages_validated(&right, batch).await.unwrap(); + } + } + drop(left); + drop(right); + drop(a); + drop(b); + let restarted = PostgresMetadataInstallRepository::new(second.clone()) + .await + .unwrap(); + let fresh = restarted + .mint_legacy_install_capability(&intent) + .await + .unwrap(); + install(&restarted, &fresh, &pages).await; + let receipt = restarted.finalize(&intent).await.unwrap(); + assert_eq!(receipt.metadata_root(), pages.dag().root()); + assert_eq!(receipt.payload_bytes(), pages.dag().payload_bytes()); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_payload WHERE generation IS NULL" + ) + .await, + 131 + ); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_retention_node WHERE state='LIVE'" + ) + .await, + 131 + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_retention_edge").await, + 130 + ); + let before = scalar( + &first, + "SELECT sum(incoming_refs)::bigint FROM mst2_retention_node", + ) + .await; + install(&restarted, &fresh, &pages).await; + assert_eq!(restarted.finalize(&intent).await.unwrap(), receipt); + assert_eq!( + scalar( + &first, + "SELECT sum(incoming_refs)::bigint FROM mst2_retention_node" + ) + .await, + before + ); +} + +#[tokio::test] +async fn install_capability_freezes_every_identity_member_anchor_and_truncate_path() { + let (first, _second, _schema) = fixture().await; + let repository = PostgresMetadataInstallRepository::new(first.clone()) + .await + .unwrap(); + let pages = prepared(2); + let intent = repository.begin_intent("freeze", &pages).await.unwrap(); + let other = repository + .begin_intent("unregistered", &pages) + .await + .unwrap(); + let cap = repository + .mint_legacy_install_capability(&intent) + .await + .unwrap(); + let before = mst2_metadata_prepare::Entity::find_by_id(intent.prepare_id().to_owned()) + .one(&first) + .await + .unwrap() + .unwrap(); + for mutation in [ + "operation_id='changed'", + "manifest_digest=decode(repeat('01',32),'hex')", + "canonical_plan=canonical_plan||decode('ff','hex')", + "source_domain='other'", + "tagged_root_tree_oid='sha1:bad'", + "scope='/changed'", + "schema_version=3", + "metadata_codec=2", + "materialization_policy=2", + "fs_semantics=2", + "access_projection=2", + "verification_revision=2", + "projection_revision=2", + "metadata_root=decode(repeat('01',32),'hex')", + "node_count=node_count+1", + "edge_count=edge_count+1", + "total_bytes=total_bytes+1", + "created_at=created_at+interval '1 second'", + ] { + assert_guard( + &first, + &format!( + "UPDATE mst2_metadata_prepare SET {mutation} WHERE prepare_id='{}'", + intent.prepare_id() + ), + ) + .await; + } + for sql in [ + format!( + "DELETE FROM mst2_metadata_prepare WHERE prepare_id='{}'", + intent.prepare_id() + ), + format!( + "UPDATE mst2_metadata_prepare_page SET expected_size=expected_size+1 WHERE prepare_id='{}'", + intent.prepare_id() + ), + format!( + "UPDATE mst2_metadata_prepare_page SET generation=1 WHERE prepare_id='{}'", + intent.prepare_id() + ), + format!( + "UPDATE mst2_metadata_prepare_page SET prepare_id='{}' WHERE prepare_id='{}'", + other.prepare_id(), + intent.prepare_id() + ), + format!( + "UPDATE mst2_metadata_prepare_page SET prepare_id='{}' WHERE prepare_id='{}'", + intent.prepare_id(), + other.prepare_id() + ), + format!( + "DELETE FROM mst2_metadata_prepare_page WHERE prepare_id='{}'", + intent.prepare_id() + ), + format!( + "INSERT INTO mst2_metadata_prepare_page(prepare_id,page_id,expected_size) VALUES('{}',decode(repeat('ff',32),'hex'),64)", + intent.prepare_id() + ), + "UPDATE mst2_metadata_install_seal SET members_digest=decode(repeat('ff',32),'hex')".into(), + "DELETE FROM mst2_metadata_install_seal".into(), + "TRUNCATE mst2_metadata_install_seal".into(), + "TRUNCATE mst2_metadata_prepare_page".into(), + "TRUNCATE mst2_metadata_prepare CASCADE".into(), + ] { + assert_guard(&first, &sql).await; + } + assert_eq!( + mst2_metadata_prepare::Entity::find_by_id(intent.prepare_id().to_owned()) + .one(&first) + .await + .unwrap() + .unwrap(), + before + ); + assert_eq!( + mst2_metadata_prepare_page::Entity::find() + .filter(mst2_metadata_prepare_page::Column::PrepareId.eq(intent.prepare_id())) + .count(&first) + .await + .unwrap(), + 3 + ); + install(&repository, &cap, &pages).await; + repository.finalize(&intent).await.unwrap(); +} + +#[tokio::test] +async fn install_capability_unregistered_corruption_and_seal_conflict_cannot_mint() { + let (first, _second, _schema) = fixture().await; + let repository = PostgresMetadataInstallRepository::new(first.clone()) + .await + .unwrap(); + let pages = prepared(2); + let intent = repository.begin_intent("corrupt", &pages).await.unwrap(); + first + .execute_unprepared("UPDATE mst2_metadata_prepare SET projection_revision=2") + .await + .unwrap(); + assert!( + repository + .mint_legacy_install_capability(&intent) + .await + .is_err() + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_install_seal").await, + 0 + ); + first.execute_unprepared("UPDATE mst2_metadata_prepare SET projection_revision=1,canonical_plan=canonical_plan||decode('ff','hex')").await.unwrap(); + assert!( + repository + .mint_legacy_install_capability(&intent) + .await + .is_err() + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_install_seal").await, + 0 + ); + let mapping = repository + .begin_intent("mapping-corrupt", &pages) + .await + .unwrap(); + first.execute_raw(statement( + "UPDATE mst2_metadata_prepare_page SET expected_size=expected_size+1 WHERE prepare_id=$1", + [mapping.prepare_id().into()], + )).await.unwrap(); + assert!( + repository + .mint_legacy_install_capability(&mapping) + .await + .is_err() + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_install_seal").await, + 0 + ); + first.execute_raw(statement( + "UPDATE mst2_metadata_prepare_page SET expected_size=expected_size-1 WHERE prepare_id=$1", + [mapping.prepare_id().into()], + )).await.unwrap(); + let intent = repository.begin_intent("conflict", &pages).await.unwrap(); + let sql=format!("INSERT INTO mst2_metadata_install_seal(prepare_id,operation_id,manifest_digest,members_digest,primary_scope,install_seal) + SELECT prepare_id,operation_id,manifest_digest,decode(repeat('00',32),'hex'),convert_to('[]','UTF8'),decode(repeat('00',32),'hex') + FROM mst2_metadata_prepare WHERE prepare_id='{}'",intent.prepare_id()); + assert_guard(&first, &sql).await; + let cap = repository + .mint_legacy_install_capability(&intent) + .await + .unwrap(); + install(&repository, &cap, &pages).await; +} + +#[tokio::test] +async fn install_capability_bounded_invalid_batches_are_atomic() { + let (first, _second, _schema) = fixture().await; + let repository = PostgresMetadataInstallRepository::new(first.clone()) + .await + .unwrap(); + let pages = prepared(70); + let intent = repository.begin_intent("invalid", &pages).await.unwrap(); + let cap = repository + .mint_legacy_install_capability(&intent) + .await + .unwrap(); + let payload = &pages.dag().payloads()[0]; + assert!(repository.install_pages_validated(&cap, &[]).await.is_err()); + assert!( + repository + .install_pages_validated(&cap, &pages.dag().payloads()[..65]) + .await + .is_err() + ); + assert!( + repository + .install_pages_validated(&cap, &[payload.clone(), payload.clone()]) + .await + .is_err() + ); + let mut wrong = payload.clone(); + wrong.size += 1; + assert!( + repository + .install_pages_validated(&cap, &[payload.clone(), wrong]) + .await + .is_err() + ); + let mut wrong = payload.clone(); + wrong.id = [255; 32]; + assert!( + repository + .install_pages_validated(&cap, &[wrong]) + .await + .is_err() + ); + let mut wrong = payload.clone(); + wrong.bytes = vec![0; PAGE_MAX_BYTES + 1]; + wrong.size = wrong.bytes.len() as u64; + wrong.id = page_id(&wrong.bytes); + assert!( + repository + .install_pages_validated(&cap, &[wrong]) + .await + .is_err() + ); + let mut wrong = payload.clone(); + wrong.bytes = vec![0; HEADER_LEN]; + wrong.size = wrong.bytes.len() as u64; + wrong.id = page_id(&wrong.bytes); + assert!( + repository + .install_pages_validated(&cap, &[wrong]) + .await + .is_err() + ); + let stranger = prepared(71); + let stranger = stranger + .dag() + .payloads() + .iter() + .find(|p| { + !pages + .dag() + .payloads() + .iter() + .any(|member| member.id == p.id) + }) + .unwrap(); + assert!( + repository + .install_pages_validated(&cap, std::slice::from_ref(stranger)) + .await + .is_err() + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_payload").await, + 0 + ); + let mut wrong = payload.bytes.clone(); + let last = wrong.len() - 1; + wrong[last] ^= 1; + first.execute_raw(statement("INSERT INTO mst2_metadata_payload(page_id,metadata_codec,byte_size,payload) VALUES($1,1,$2,$3)", + [payload.id.to_vec().into(),(payload.size as i32).into(),wrong.into()])).await.unwrap(); + assert!( + repository + .install_pages_validated(&cap, &pages.dag().payloads()[..64]) + .await + .is_err() + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_payload").await, + 1 + ); +} + +#[tokio::test] +async fn install_capability_raw_committed_missing_bytes_does_not_repair_and_retirement_rejects() { + let (first, second, _schema) = fixture().await; + let repository = PostgresMetadataInstallRepository::new(first.clone()) + .await + .unwrap(); + let pages = prepared(2); + let intent = repository.begin_intent("promoted", &pages).await.unwrap(); + let cap = repository + .mint_legacy_install_capability(&intent) + .await + .unwrap(); + second.execute_raw(statement("UPDATE mst2_metadata_prepare SET state='COMMITTED',committed_at=clock_timestamp() WHERE prepare_id=$1", + [intent.prepare_id().into()])).await.unwrap(); + assert!( + repository + .install_pages_validated(&cap, pages.dag().payloads()) + .await + .is_err() + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_payload").await, + 0 + ); + assert!( + repository + .mint_legacy_install_capability(&intent) + .await + .is_err() + ); + let intent = repository.begin_intent("retired", &pages).await.unwrap(); + let cap = repository + .mint_legacy_install_capability(&intent) + .await + .unwrap(); + install(&repository, &cap, &pages).await; + repository.finalize(&intent).await.unwrap(); + second.execute_raw(statement("UPDATE mst2_metadata_prepare SET coverage_retired_at=clock_timestamp() WHERE prepare_id=$1", + [intent.prepare_id().into()])).await.unwrap(); + assert!( + repository + .install_pages_validated(&cap, pages.dag().payloads()) + .await + .is_err() + ); + assert!( + repository + .mint_legacy_install_capability(&intent) + .await + .is_err() + ); +} + +#[tokio::test] +async fn install_capability_cross_primary_schema_repeatable_read_and_temp_spoof_refuse() { + let (first, second, _schema) = fixture().await; + let (alien, _unused, _alien_schema) = fixture().await; + let a = PostgresMetadataInstallRepository::new(first.clone()) + .await + .unwrap(); + let b = PostgresMetadataInstallRepository::new(alien.clone()) + .await + .unwrap(); + let pages = prepared(2); + let intent = a.begin_intent("scope", &pages).await.unwrap(); + let cap = a.mint_legacy_install_capability(&intent).await.unwrap(); + let alien_intent = b.begin_intent("scope", &pages).await.unwrap(); + assert_ne!(alien_intent.prepare_id(), intent.prepare_id()); + let copied_prepare = first.query_one_raw(statement( + "SELECT row_to_json(p)::text AS copied FROM mst2_metadata_prepare p WHERE prepare_id=$1", + [intent.prepare_id().into()], + )).await.unwrap().unwrap().try_get::("", "copied").unwrap(); + let copied_members = first.query_one_raw(statement( + "SELECT json_agg(m)::text AS copied FROM mst2_metadata_prepare_page m WHERE prepare_id=$1", + [intent.prepare_id().into()], + )).await.unwrap().unwrap().try_get::("", "copied").unwrap(); + alien + .execute_raw(statement( + "DELETE FROM mst2_metadata_prepare_page WHERE prepare_id=$1", + [alien_intent.prepare_id().into()], + )) + .await + .unwrap(); + alien + .execute_raw(statement( + "DELETE FROM mst2_metadata_prepare WHERE prepare_id=$1", + [alien_intent.prepare_id().into()], + )) + .await + .unwrap(); + alien.execute_raw(statement("INSERT INTO mst2_metadata_prepare SELECT * FROM json_populate_record(NULL::mst2_metadata_prepare,$1::json)", + [copied_prepare.into()])).await.unwrap(); + alien.execute_raw(statement("INSERT INTO mst2_metadata_prepare_page SELECT * FROM json_populate_recordset(NULL::mst2_metadata_prepare_page,$1::json)", + [copied_members.into()])).await.unwrap(); + let own_scope_cap = b.mint_legacy_install_capability(&intent).await.unwrap(); + assert!( + b.install_pages_validated(&cap, pages.dag().payloads()) + .await + .is_err() + ); + assert_eq!( + scalar(&alien, "SELECT count(*) FROM mst2_metadata_payload").await, + 0 + ); + install(&b, &own_scope_cap, &pages).await; + let txn = second + .begin_with_config(Some(IsolationLevel::RepeatableRead), None) + .await + .unwrap(); + assert!(a.capability_barrier(&txn).await.is_err()); + assert!( + txn.execute_unprepared("UPDATE mst2_metadata_prepare SET scope='/rr'") + .await + .is_err() + ); + txn.rollback().await.unwrap(); + let txn = second.begin().await.unwrap(); + let schema = a.storage_scope.schema.clone(); + txn.execute_unprepared("SET LOCAL search_path=pg_catalog") + .await + .unwrap(); + assert!(a.capability_barrier(&txn).await.is_err()); + let sql = format!( + "UPDATE \"{}\".mst2_metadata_prepare SET scope='/wrongpath'", + schema.replace('"', "\"\"") + ); + assert!(txn.execute_unprepared(&sql).await.is_err()); + txn.rollback().await.unwrap(); + let txn = second.begin().await.unwrap(); + txn.execute_unprepared( + "CREATE TEMP TABLE mst2_metadata_install_seal(prepare_id text); + CREATE TEMP TABLE mst2_metadata_storage_scope(singleton integer,storage_uuid text)", + ) + .await + .unwrap(); + let raw_sql = format!( + "UPDATE \"{}\".mst2_metadata_prepare SET scope='/raw-temp-spoof'", + schema.replace('"', "\"\"") + ); + assert!( + txn.execute_unprepared(&raw_sql) + .await + .unwrap_err() + .to_string() + .contains("registered metadata") + ); + txn.rollback().await.unwrap(); + let txn = second.begin().await.unwrap(); + txn.execute_unprepared( + "CREATE TEMP TABLE mst2_metadata_install_seal(prepare_id text); + CREATE TEMP TABLE mst2_metadata_storage_scope(singleton integer,storage_uuid text)", + ) + .await + .unwrap(); + a.capability_barrier(&txn).await.unwrap(); + assert_eq!( + scalar(&txn, "SELECT count(*) FROM mst2_metadata_install_seal").await, + 1 + ); + assert!( + txn.execute_unprepared("UPDATE mst2_metadata_prepare SET scope='/spoof'") + .await + .is_err() + ); + txn.rollback().await.unwrap(); + install(&a, &cap, &pages).await; +} + +#[tokio::test] +async fn install_capability_reuses_bound_generic_bytes_without_adopting_qualified_generation() { + let (first, _second, _schema) = fixture().await; + let generic = generations::PostgresMetadataGenerationRepository::new(first.clone()) + .await + .unwrap(); + let legacy = PostgresMetadataInstallRepository::new(first.clone()) + .await + .unwrap(); + let pages = prepared(2); + let bound = generic.begin_intent("bound-generic", &pages).await.unwrap(); + generic + .install_pages(&bound, pages.dag().payloads()) + .await + .unwrap(); + generic.finalize(&bound).await.unwrap(); + let intent = legacy.begin_intent("legacy-share", &pages).await.unwrap(); + let cap = legacy + .mint_legacy_install_capability(&intent) + .await + .unwrap(); + install(&legacy, &cap, &pages).await; + legacy.finalize(&intent).await.unwrap(); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_payload WHERE generation IS NOT NULL" + ) + .await, + 3 + ); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_prepare_page WHERE generation IS NULL" + ) + .await, + 3 + ); + let bound_intent = MetadataPrepareIntent { + prepare_id: bound.prepare_id().into(), + operation_id: bound.operation_id().into(), + manifest_digest: bound.manifest_digest(), + }; + assert!( + legacy + .mint_legacy_install_capability(&bound_intent) + .await + .is_err() + ); + let (qualified_db, _other, _q_schema) = fixture().await; + let qualified = + generations::qualified::PostgresQualifiedMetadataRepository::new(qualified_db.clone()) + .await + .unwrap(); + let q_legacy = PostgresMetadataInstallRepository::new(qualified_db.clone()) + .await + .unwrap(); + let bound = qualified.begin_intent("qualified", &pages).await.unwrap(); + qualified + .install_pages(&bound, pages.dag().payloads()) + .await + .unwrap(); + let intent = q_legacy + .begin_intent("generic-refusal", &pages) + .await + .unwrap(); + let cap = q_legacy + .mint_legacy_install_capability(&intent) + .await + .unwrap(); + assert!( + q_legacy + .install_pages_validated(&cap, pages.dag().payloads()) + .await + .is_err() + ); + assert_eq!( + scalar(&qualified_db, "SELECT count(*) FROM mst2_metadata_payload").await, + 3 + ); + assert_eq!( + scalar(&qualified_db, "SELECT count(*) FROM mst2_retention_node").await, + 0 + ); +} + +#[tokio::test] +async fn install_capability_batch_fault_rolls_back_and_raw_writer_waits_before_row_lock() { + let (first, second, _schema) = fixture().await; + let repository = PostgresMetadataInstallRepository::new(first.clone()) + .await + .unwrap(); + let pages = prepared(2); + let intent = repository + .begin_intent("batch-fault", &pages) + .await + .unwrap(); + let cap = repository + .mint_legacy_install_capability(&intent) + .await + .unwrap(); + let failed = hex::encode(pages.dag().payloads()[1].id); + first.execute_unprepared(&format!("CREATE FUNCTION test_install_batch_fault() RETURNS trigger LANGUAGE plpgsql AS $$ + BEGIN IF NEW.page_id=decode('{failed}','hex') THEN RAISE EXCEPTION 'injected batch fault'; END IF; RETURN NEW; END $$; + CREATE TRIGGER test_install_batch_fault AFTER INSERT ON mst2_metadata_payload FOR EACH ROW EXECUTE FUNCTION test_install_batch_fault()")).await.unwrap(); + assert!( + repository + .install_pages_validated(&cap, pages.dag().payloads()) + .await + .is_err() + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_payload").await, + 0 + ); + first.execute_unprepared("DROP TRIGGER test_install_batch_fault ON mst2_metadata_payload;DROP FUNCTION test_install_batch_fault()").await.unwrap(); + let held = first + .begin_with_config(Some(IsolationLevel::ReadCommitted), None) + .await + .unwrap(); + repository.capability_barrier(&held).await.unwrap(); + let worker = second + .begin_with_config(Some(IsolationLevel::ReadCommitted), None) + .await + .unwrap(); + let pid = scalar(&worker, "SELECT pg_backend_pid()::bigint").await; + let id = intent.prepare_id().to_owned(); + let blocked = tokio::spawn(async move { + let result = worker + .execute_raw(statement( + "UPDATE mst2_metadata_prepare SET scope='/blocked' WHERE prepare_id=$1", + [id.into()], + )) + .await; + worker.rollback().await.unwrap(); + result + }); + let deadline = tokio::time::Instant::now() + Duration::from_secs(4); + loop { + let waiting=held.query_one_raw(statement("SELECT EXISTS(SELECT 1 FROM pg_locks WHERE locktype='advisory' AND NOT granted + AND classid=$1::integer::oid AND objid=hashtext(current_schema())::oid + AND database=(SELECT oid FROM pg_database WHERE datname=current_database())) AS waiting",[RETENTION_LOCK_KEY.into()])).await.unwrap().unwrap().try_get::("","waiting").unwrap(); + if waiting { + break; + } + assert!( + tokio::time::Instant::now() < deadline, + "real writer must wait before frozen-row proof" + ); + tokio::task::yield_now().await; + } + let row_locked=held.query_one_raw(statement("SELECT EXISTS(SELECT 1 FROM pg_locks WHERE pid=$1 AND locktype='tuple' AND relation='mst2_metadata_prepare'::regclass) AS row_locked",[(pid as i32).into()])).await.unwrap().unwrap().try_get::("","row_locked").unwrap(); + assert!( + !row_locked, + "statement retention barrier precedes prepare row locking" + ); + held.query_one_raw(statement( + "SELECT prepare_id FROM mst2_metadata_prepare WHERE prepare_id=$1 FOR UPDATE", + [intent.prepare_id().into()], + )) + .await + .unwrap() + .unwrap(); + held.commit().await.unwrap(); + assert!( + blocked + .await + .unwrap() + .unwrap_err() + .to_string() + .contains("registered metadata") + ); + install(&repository, &cap, &pages).await; +} diff --git a/src/jupiter/storage/native_metadata_install_tests.rs b/src/jupiter/storage/native_metadata_install_tests.rs index f7662117..4145afd8 100644 --- a/src/jupiter/storage/native_metadata_install_tests.rs +++ b/src/jupiter/storage/native_metadata_install_tests.rs @@ -1279,3 +1279,6 @@ async fn native_metadata_query_loss_after_recovery_barrier_preserves_typed_unkno MetadataPrepareObservation::Committed(_) )); } + +#[path = "native_metadata_install_capability_fault_tests.rs"] +mod capability_fault_tests; diff --git a/src/jupiter/storage/native_snapshot_session.rs b/src/jupiter/storage/native_snapshot_session.rs index 415228c8..d84d075a 100644 --- a/src/jupiter/storage/native_snapshot_session.rs +++ b/src/jupiter/storage/native_snapshot_session.rs @@ -69,9 +69,13 @@ impl PostgresNativeSessionRepository { .begin_intent(&format!("http:{}", built.snapshot_id), prepared) .await .map_err(install_error)?; + let capability = installer + .mint_legacy_install_capability(&intent) + .await + .map_err(install_error)?; for pages in prepared.dag().payloads().chunks(64) { installer - .install_pages(&intent, pages) + .install_pages_validated(&capability, pages) .await .map_err(install_error)?; } From 1fc78c5ebfc47000d8a4aa90f76c31266ebacb57 Mon Sep 17 00:00:00 2001 From: Xiaoyang Han Date: Wed, 7 Oct 2026 23:01:35 +0800 Subject: [PATCH 14/50] test(mst2): repair native capability fixtures without relaxing guards (#64) Repair exactly diagnosed native capability regressions: mutate verification revision rather than a legal no-op; count the actual canonical radix DAG and batch partitions; inject one handoff corruption with the exact prepare guard restored transactionally; prove both legitimate generic/qualified refusal paths with complete state rollback and retain a real private physical-checker rejection. Production guards and original HTTP integrity/zero-read assertions remain unchanged. Native rerun is separate from source review. --- src/api/router/snapshot_session_tests.rs | 95 ++++++++- .../native_metadata_install_capability.rs | 50 +++++ ...ative_metadata_install_capability_tests.rs | 192 ++++++++++++++++-- 3 files changed, 316 insertions(+), 21 deletions(-) diff --git a/src/api/router/snapshot_session_tests.rs b/src/api/router/snapshot_session_tests.rs index 7cf495b4..17dfcb21 100644 --- a/src/api/router/snapshot_session_tests.rs +++ b/src/api/router/snapshot_session_tests.rs @@ -995,11 +995,7 @@ async fn mst2_durable_http_handoff_rejects_changed_plan_or_receipt_summary() { prepared.wait().await; let mono = fixture.state.storage.mono_storage(); let db = mono.get_connection(); - db.execute_unprepared(&format!( - "UPDATE mst2_metadata_prepare SET {mutation} WHERE scope='/'" - )) - .await - .unwrap(); + corrupt_registered_handoff_plan_for_test(db, mutation).await; release.wait().await; error(resolving.await.unwrap(), 502, "INTEGRITY_ERROR", false).await; assert_eq!( @@ -1037,6 +1033,95 @@ async fn mst2_durable_http_handoff_rejects_changed_plan_or_receipt_summary() { } } +async fn corrupt_registered_handoff_plan_for_test(db: &DatabaseConnection, mutation: &str) { + let guard_modes = handoff_trigger_modes_for_test(db).await; + assert!(guard_modes.iter().any(|(table, trigger, mode)| { + table == "mst2_metadata_prepare" + && trigger == "mst2_install_capability_prepare_guard" + && mode == "O" + })); + let prepares = db + .query_all_raw(Statement::from_string( + DbBackend::Postgres, + "SELECT prepare_id FROM mst2_metadata_prepare WHERE scope='/'", + )) + .await + .unwrap(); + assert_eq!(prepares.len(), 1); + let prepare_id: String = prepares[0].try_get("", "prepare_id").unwrap(); + let protected_update = || { + Statement::from_sql_and_values( + DbBackend::Postgres, + "UPDATE mst2_metadata_prepare SET total_bytes=total_bytes+1 WHERE prepare_id=$1", + [prepare_id.clone().into()], + ) + }; + assert!( + db.execute_raw(protected_update()) + .await + .unwrap_err() + .to_string() + .contains("registered metadata preparation identity is immutable") + ); + // The isolated fault injection restores this exact production guard in the + // same transaction. Acquire the statement barrier before ALTER's table lock. + let txn = db + .begin_with_config(Some(sea_orm::IsolationLevel::ReadCommitted), None) + .await + .unwrap(); + txn.execute_unprepared("SELECT pg_advisory_xact_lock(1296717362,hashtext(current_schema()))") + .await + .unwrap(); + txn.execute_unprepared( + "ALTER TABLE mst2_metadata_prepare DISABLE TRIGGER mst2_install_capability_prepare_guard", + ) + .await + .unwrap(); + let changed = txn + .execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + format!("UPDATE mst2_metadata_prepare SET {mutation} WHERE prepare_id=$1"), + [prepare_id.clone().into()], + )) + .await + .unwrap(); + assert_eq!(changed.rows_affected(), 1); + txn.execute_unprepared( + "ALTER TABLE mst2_metadata_prepare ENABLE TRIGGER mst2_install_capability_prepare_guard", + ) + .await + .unwrap(); + txn.commit().await.unwrap(); + assert_eq!(handoff_trigger_modes_for_test(db).await, guard_modes); + assert!( + db.execute_raw(protected_update()) + .await + .unwrap_err() + .to_string() + .contains("registered metadata preparation identity is immutable") + ); +} + +async fn handoff_trigger_modes_for_test(db: &DatabaseConnection) -> Vec<(String, String, String)> { + db.query_all_raw(Statement::from_string( + DbBackend::Postgres, + "SELECT c.relname,t.tgname,t.tgenabled::text AS mode FROM pg_trigger t + JOIN pg_class c ON c.oid=t.tgrelid JOIN pg_namespace n ON n.oid=c.relnamespace + WHERE n.nspname=current_schema() AND NOT t.tgisinternal ORDER BY c.relname,t.tgname", + )) + .await + .unwrap() + .into_iter() + .map(|row| { + ( + row.try_get("", "relname").unwrap(), + row.try_get("", "tgname").unwrap(), + row.try_get("", "mode").unwrap(), + ) + }) + .collect() +} + async fn overwrite_primary_scope_for_test(db: &DatabaseConnection, storage_uuid: String) { // Fault injection owns this isolated schema. Restore the production // immutable trigger before commit; failed injection rolls back its DDL. diff --git a/src/jupiter/storage/native_metadata_install_capability.rs b/src/jupiter/storage/native_metadata_install_capability.rs index 348405e1..98a08695 100644 --- a/src/jupiter/storage/native_metadata_install_capability.rs +++ b/src/jupiter/storage/native_metadata_install_capability.rs @@ -462,3 +462,53 @@ fn seal( hash.update(scope); Ok(hash.finalize().into()) } + +#[cfg(test)] +#[tokio::test] +async fn install_capability_physical_checker_rejects_real_qualified_owner() { + let (db, _other, _schema) = super::capability_tests::fixture().await; + let pages = super::capability_tests::prepared(2); + let qualified = + super::generations::qualified::PostgresQualifiedMetadataRepository::new(db.clone()) + .await + .unwrap(); + let bound = qualified + .begin_intent("physical-qualified", &pages) + .await + .unwrap(); + qualified + .install_pages(&bound, pages.dag().payloads()) + .await + .unwrap(); + let legacy = PostgresMetadataInstallRepository::new(db).await.unwrap(); + let txn = legacy.transaction().await.unwrap(); + legacy.capability_barrier(&txn).await.unwrap(); + for page in pages.dag().payloads() { + let row = txn.query_one_raw(statement( + "SELECT b.page_id IS NOT NULL AS payload_present,b.generation AS payload_generation, + c.generation AS current_generation,l.generation AS lifetime_generation,l.graph_domain, + l.state AS lifetime_state,l.metadata_codec AS lifetime_codec,l.expected_size AS lifetime_size, + n.state AS graph_state,n.kind AS graph_kind,n.bytes AS graph_bytes, + EXISTS(SELECT 1 FROM mst2_retention_gc_op g WHERE g.node_id=l.node_id + AND g.operation='REMOVE' AND g.state IN ('PENDING','APPLIED')) AS tombstone + FROM mst2_metadata_current c JOIN mst2_metadata_lifetime l USING(page_id,generation) + JOIN mst2_metadata_payload b USING(page_id,generation) + LEFT JOIN mst2_retention_node n ON n.node_id=l.node_id WHERE c.page_id=$1", + [page.id.to_vec().into()], + )).await.unwrap().unwrap(); + assert_eq!( + row.try_get::("", "graph_domain").unwrap(), + "qualified-v1" + ); + assert_eq!(row.try_get::("", "current_generation").unwrap(), 1); + assert_eq!(row.try_get::("", "payload_generation").unwrap(), 1); + assert!(row.try_get::("", "payload_present").unwrap()); + let error = check_physical_member(&row, 1, page.size as i32).unwrap_err(); + assert_eq!(error.code, SnapshotErrorCode::ObjectUnavailable); + assert_eq!( + error.message, + "requested physical lifetime cannot be shared by legacy generic installation" + ); + } + txn.rollback().await.unwrap(); +} diff --git a/src/jupiter/storage/native_metadata_install_capability_tests.rs b/src/jupiter/storage/native_metadata_install_capability_tests.rs index fb9df08f..cd2759ef 100644 --- a/src/jupiter/storage/native_metadata_install_capability_tests.rs +++ b/src/jupiter/storage/native_metadata_install_capability_tests.rs @@ -13,7 +13,7 @@ use crate::{ }, }; -fn prepared(children: usize) -> PreparedNativeMetadataRetention { +pub(super) fn prepared(children: usize) -> PreparedNativeMetadataRetention { let mut builder = MetadataDagBuilder::new(MetadataDagLimits::default()); let mut entries = Vec::new(); for index in 0..children { @@ -38,7 +38,7 @@ fn prepared(children: usize) -> PreparedNativeMetadataRetention { ) } -async fn fixture() -> (DatabaseConnection, DatabaseConnection, TestSchemaGuard) { +pub(super) async fn fixture() -> (DatabaseConnection, DatabaseConnection, TestSchemaGuard) { let temp = tempfile::tempdir().unwrap(); let (config, schema) = test_db_config(temp.path()).await; let first = Database::connect(config.db_url.clone()).await.unwrap(); @@ -70,7 +70,7 @@ async fn install( } async fn assert_guard(db: &DatabaseConnection, sql: &str) { - let error = db.execute_unprepared(sql).await.unwrap_err(); + let error = db.execute_unprepared(sql).await.expect_err(sql); assert!( error.to_string().contains("install capability") || error.to_string().contains("registered metadata"), @@ -88,7 +88,68 @@ async fn install_capability_many_batches_concurrent_mint_and_fresh_repository_fi .await .unwrap(); let pages = prepared(130); - assert_eq!(pages.dag().payloads().len(), 131); + assert_eq!(pages.dag().payloads().len(), 133); + assert_eq!(pages.dag().edges().len(), 132); + let root = pages + .dag() + .payloads() + .iter() + .find(|page| page.id == pages.dag().root()) + .unwrap(); + let ( + Page::Branch { + prefix, + terminal, + children, + }, + count, + ) = Page::decode(&root.bytes).unwrap() + else { + panic!("130 directory entries must use a canonical radix branch"); + }; + assert_eq!(prefix, b"dir-"); + assert!(terminal.is_none()); + assert_eq!(count, 130); + assert_eq!( + children + .iter() + .map(|child| (child.label, child.subtree_entries)) + .collect::>(), + [(b'0', 100), (b'1', 30)] + ); + for child in children { + let payload = pages + .dag() + .payloads() + .iter() + .find(|page| page.id == child.child_page_id) + .unwrap(); + let (Page::Leaf { entries }, count) = Page::decode(&payload.bytes).unwrap() else { + panic!("each root partition must be one canonical radix leaf"); + }; + assert_eq!(count, child.subtree_entries); + assert_eq!(entries.len() as u64, count); + for entry in entries { + assert_eq!(entry.kind, EntryKind::Directory); + assert_eq!(entry.name[4], child.label); + assert!( + pages + .dag() + .payloads() + .iter() + .any(|page| page.id == entry.child_root) + ); + } + } + assert_eq!( + pages + .dag() + .payloads() + .chunks(64) + .map(|batch| batch.len()) + .collect::>(), + [64, 64, 5] + ); let intent = a.begin_intent("many", &pages).await.unwrap(); let (left, right) = tokio::join!( a.mint_legacy_install_capability(&intent), @@ -128,7 +189,7 @@ async fn install_capability_many_batches_concurrent_mint_and_fresh_repository_fi "SELECT count(*) FROM mst2_metadata_payload WHERE generation IS NULL" ) .await, - 131 + 133 ); assert_eq!( scalar( @@ -136,11 +197,11 @@ async fn install_capability_many_batches_concurrent_mint_and_fresh_repository_fi "SELECT count(*) FROM mst2_retention_node WHERE state='LIVE'" ) .await, - 131 + 133 ); assert_eq!( scalar(&first, "SELECT count(*) FROM mst2_retention_edge").await, - 130 + 132 ); let before = scalar( &first, @@ -180,6 +241,13 @@ async fn install_capability_freezes_every_identity_member_anchor_and_truncate_pa .await .unwrap() .unwrap(); + assert_eq!( + first.execute_raw(statement( + "UPDATE mst2_metadata_prepare SET verification_revision=verification_revision WHERE prepare_id=$1", + [intent.prepare_id().into()], + )).await.unwrap().rows_affected(), + 1 + ); for mutation in [ "operation_id='changed'", "manifest_digest=decode(repeat('01',32),'hex')", @@ -192,7 +260,7 @@ async fn install_capability_freezes_every_identity_member_anchor_and_truncate_pa "materialization_policy=2", "fs_semantics=2", "access_projection=2", - "verification_revision=2", + "verification_revision=verification_revision+1", "projection_revision=2", "metadata_root=decode(repeat('01',32),'hex')", "node_count=node_count+1", @@ -668,30 +736,122 @@ async fn install_capability_reuses_bound_generic_bytes_without_adopting_qualifie .install_pages(&bound, pages.dag().payloads()) .await .unwrap(); - let intent = q_legacy + let q_state = domain_state_for_test(&qualified_db).await; + let error = q_legacy .begin_intent("generic-refusal", &pages) + .await + .unwrap_err(); + assert!( + error + .to_string() + .contains("qualified mapping cannot cross domain/current/state fence"), + "{error}" + ); + assert_eq!(domain_state_for_test(&qualified_db).await, q_state); + assert_eq!( + scalar(&qualified_db, "SELECT count(*) FROM mst2_metadata_payload").await, + 3 + ); + assert_eq!( + scalar(&qualified_db, "SELECT count(*) FROM mst2_retention_node").await, + 0 + ); + + let (legacy_db, _other, _legacy_schema) = fixture().await; + let legacy = PostgresMetadataInstallRepository::new(legacy_db.clone()) .await .unwrap(); - let cap = q_legacy + let qualified = + generations::qualified::PostgresQualifiedMetadataRepository::new(legacy_db.clone()) + .await + .unwrap(); + let intent = legacy.begin_intent("generic-first", &pages).await.unwrap(); + let cap = legacy .mint_legacy_install_capability(&intent) .await .unwrap(); + let generic_state = domain_state_for_test(&legacy_db).await; + let error = qualified + .begin_intent("qualified-refusal", &pages) + .await + .unwrap_err(); assert!( - q_legacy - .install_pages_validated(&cap, pages.dag().payloads()) - .await - .is_err() + error + .to_string() + .contains("metadata lifetime has durable or ambiguous coverage"), + "{error}" ); + assert_eq!(domain_state_for_test(&legacy_db).await, generic_state); + install(&legacy, &cap, &pages).await; assert_eq!( - scalar(&qualified_db, "SELECT count(*) FROM mst2_metadata_payload").await, + legacy.finalize(&intent).await.unwrap().metadata_root(), + pages.dag().root() + ); + assert_eq!( + scalar( + &legacy_db, + "SELECT count(*) FROM mst2_metadata_payload WHERE generation IS NULL" + ) + .await, 3 ); assert_eq!( - scalar(&qualified_db, "SELECT count(*) FROM mst2_retention_node").await, + scalar( + &legacy_db, + "SELECT count(*) FROM mst2_retention_node WHERE state='LIVE'" + ) + .await, + 3 + ); + assert_eq!( + scalar(&legacy_db, "SELECT count(*) FROM mst2_metadata_lifetime").await, + 0 + ); + assert_eq!( + scalar(&legacy_db, "SELECT count(*) FROM mst2_metadata_current").await, + 0 + ); + assert_eq!( + scalar(&legacy_db, "SELECT count(*) FROM mst2_metadata_graph_node").await, 0 ); } +async fn domain_state_for_test(db: &DatabaseConnection) -> Vec<(String, Vec)> { + let mut state = Vec::new(); + for table in [ + "mst2_metadata_prepare", + "mst2_metadata_prepare_page", + "mst2_metadata_install_seal", + "mst2_metadata_payload", + "mst2_metadata_current", + "mst2_metadata_lifetime", + "mst2_metadata_graph_node", + "mst2_metadata_graph_edge", + "mst2_metadata_graph_root", + "mst2_metadata_gc_op", + "mst2_retention_node", + "mst2_retention_edge", + "mst2_retention_root", + "mst2_retention_gc_op", + "mst2_snapshot_context", + "mst2_snapshot_lease", + ] { + let rows = db + .query_all_raw(statement( + &format!("SELECT to_jsonb(t)::text FROM {table} t ORDER BY 1"), + [], + )) + .await + .unwrap() + .into_iter() + .map(|row| row.try_get_by_index(0).unwrap()) + .collect(); + state.push((table.into(), rows)); + } + state +} + #[tokio::test] async fn install_capability_batch_fault_rolls_back_and_raw_writer_waits_before_row_lock() { let (first, second, _schema) = fixture().await; From 8bc5100cc40930d54744807e5a618b3f33e7451e Mon Sep 17 00:00:00 2001 From: Xiaoyang Han Date: Wed, 7 Oct 2026 23:19:11 +0800 Subject: [PATCH 15/50] perf(mst2): share concurrent cold chunk projections (#65) Deduplicate concurrent same-digest cold projection loads, retain verified results for active participants after cache eviction, and preserve full SHA/map/slice validation. Bound registry keys and clean up exact last-owner gates without dropping large results under the global lock. Add ten failure/cancellation/eviction/capacity regressions. Native and performance evidence remain separate. --- .../snapshot/chunk_singleflight_tests.rs | 487 ++++++++++++++++++ src/ceres/snapshot/chunks.rs | 121 ++++- 2 files changed, 603 insertions(+), 5 deletions(-) create mode 100644 src/ceres/snapshot/chunk_singleflight_tests.rs diff --git a/src/ceres/snapshot/chunk_singleflight_tests.rs b/src/ceres/snapshot/chunk_singleflight_tests.rs new file mode 100644 index 00000000..2534acf9 --- /dev/null +++ b/src/ceres/snapshot/chunk_singleflight_tests.rs @@ -0,0 +1,487 @@ +use std::{ + sync::{ + Barrier, + atomic::{AtomicUsize, Ordering}, + }, + time::Duration, +}; + +use tokio::{sync::Notify, time::timeout}; + +use super::*; + +fn unique_data(size: usize) -> ([u8; 32], Arc>) { + let seed = uuid::Uuid::new_v4(); + let raw: Vec = (0..size).map(|i| seed.as_bytes()[i % 16]).collect(); + (Sha256::digest(&raw).into(), Arc::new(raw)) +} + +async fn started(signal: &Notify) { + timeout(Duration::from_secs(5), signal.notified()) + .await + .unwrap(); +} + +async fn participants(content_id: [u8; 32], expected: usize) { + timeout(Duration::from_secs(5), async { + loop { + let count = FLIGHTS + .get() + .unwrap() + .lock() + .unwrap() + .entries + .get(&content_id) + .map_or(0, Weak::strong_count); + if count == expected { + break; + } + tokio::task::yield_now().await; + } + }) + .await + .unwrap(); +} + +fn assert_flight_released(content_id: [u8; 32]) { + assert!( + !FLIGHTS + .get() + .unwrap() + .lock() + .unwrap() + .entries + .contains_key(&content_id) + ); +} + +fn assert_uncached(content_id: [u8; 32]) { + assert!( + STAGED + .get() + .unwrap() + .lock() + .unwrap() + .get(content_id) + .is_none() + ); +} + +fn evict(content_id: [u8; 32]) { + let mut cache = STAGED.get().unwrap().lock().unwrap(); + let projection = cache.entries.remove(&content_id).unwrap(); + cache.total_bytes -= projection.raw.len(); + let position = cache.order.iter().position(|id| *id == content_id).unwrap(); + cache.order.remove(position); +} + +#[tokio::test] +async fn cold_same_digest_loads_once_and_warm_cache_skips_loader() { + let (content_id, raw) = unique_data(CHUNK_SIZE as usize + 7); + let calls = Arc::new(AtomicUsize::new(0)); + let entered = Arc::new(Notify::new()); + let release = Arc::new(Notify::new()); + let leader = tokio::spawn({ + let raw = raw.clone(); + let calls = calls.clone(); + let entered = entered.clone(); + let release = release.clone(); + async move { + get_or_project(content_id, || async move { + calls.fetch_add(1, Ordering::SeqCst); + entered.notify_one(); + release.notified().await; + Ok(raw.as_ref().clone()) + }) + .await + } + }); + started(&entered).await; + let mut waiters = Vec::new(); + for _ in 0..7 { + let raw = raw.clone(); + let calls = calls.clone(); + waiters.push(tokio::spawn(async move { + get_or_project(content_id, || async move { + calls.fetch_add(1, Ordering::SeqCst); + Ok(raw.as_ref().clone()) + }) + .await + })); + } + participants(content_id, 8).await; + release.notify_one(); + let projection = leader.await.unwrap().unwrap(); + for waiter in waiters { + let shared = waiter.await.unwrap().unwrap(); + assert!(Arc::ptr_eq(&projection, &shared)); + } + assert_eq!(calls.load(Ordering::SeqCst), 1); + assert_eq!(projection.map.file_content_id, content_id); + assert_eq!(projection.map.chunk_count, 2); + assert_eq!( + projection.chunk_bytes(0).unwrap(), + &raw[..CHUNK_SIZE as usize] + ); + assert_eq!( + projection.chunk_bytes(1).unwrap(), + &raw[CHUNK_SIZE as usize..] + ); + let (leaf, proof) = projection.leaf_and_proof(0).unwrap(); + mst2_codec::chunkmap::verify_leaf( + projection.page_count(), + 0, + leaf.leaf_hash().unwrap(), + &proof, + projection.map.pages_root, + ) + .unwrap(); + let warm = get_or_project(content_id, || async { + panic!("warm projection must not invoke its loader"); + }) + .await + .unwrap(); + assert!(Arc::ptr_eq(&projection, &warm)); + assert_flight_released(content_id); +} + +#[tokio::test] +async fn different_digests_enter_their_loaders_independently() { + let mut requests = Vec::new(); + let mut signals = Vec::new(); + for _ in 0..2 { + let (content_id, raw) = unique_data(97); + let entered = Arc::new(Notify::new()); + let release = Arc::new(Notify::new()); + signals.push((content_id, entered.clone(), release.clone())); + requests.push(tokio::spawn(async move { + get_or_project(content_id, || async move { + entered.notify_one(); + release.notified().await; + Ok(raw.as_ref().clone()) + }) + .await + })); + } + for (_, entered, _) in &signals { + started(entered).await; + } + for (_, _, release) in &signals { + release.notify_one(); + } + for request in requests { + request.await.unwrap().unwrap(); + } + for (content_id, _, _) in signals { + assert_flight_released(content_id); + } +} + +async fn failed_leader_then_waiter(wrong_digest: bool) { + let (content_id, raw) = unique_data(103); + let calls = Arc::new(AtomicUsize::new(0)); + let first_entered = Arc::new(Notify::new()); + let first_release = Arc::new(Notify::new()); + let second_entered = Arc::new(Notify::new()); + let second_release = Arc::new(Notify::new()); + let leader = tokio::spawn({ + let calls = calls.clone(); + let entered = first_entered.clone(); + let release = first_release.clone(); + async move { + get_or_project(content_id, || async move { + calls.fetch_add(1, Ordering::SeqCst); + entered.notify_one(); + release.notified().await; + if wrong_digest { + Ok(vec![0; 103]) + } else { + Err(SnapshotError::new( + SnapshotErrorCode::ObjectUnavailable, + "failed test loader", + )) + } + }) + .await + } + }); + started(&first_entered).await; + let waiter = tokio::spawn({ + let raw = raw.clone(); + let calls = calls.clone(); + let entered = second_entered.clone(); + let release = second_release.clone(); + async move { + get_or_project(content_id, || async move { + calls.fetch_add(1, Ordering::SeqCst); + entered.notify_one(); + release.notified().await; + Ok(raw.as_ref().clone()) + }) + .await + } + }); + participants(content_id, 2).await; + first_release.notify_one(); + let failure = leader.await.unwrap().err().unwrap(); + assert_eq!( + failure.code, + if wrong_digest { + SnapshotErrorCode::DigestMismatch + } else { + SnapshotErrorCode::ObjectUnavailable + } + ); + started(&second_entered).await; + assert_eq!(calls.load(Ordering::SeqCst), 2); + assert_uncached(content_id); + second_release.notify_one(); + let projection = waiter.await.unwrap().unwrap(); + assert_eq!(projection.chunk_bytes(0).unwrap(), raw.as_slice()); + assert_flight_released(content_id); +} + +#[tokio::test] +async fn load_failure_is_unpublished_and_waiter_uses_its_loader() { + failed_leader_then_waiter(false).await; +} + +#[tokio::test] +async fn digest_failure_is_unpublished_and_waiter_uses_its_loader() { + failed_leader_then_waiter(true).await; +} + +#[tokio::test] +async fn cancelled_leader_releases_gate_for_waiting_loader() { + let (content_id, raw) = unique_data(113); + let entered = Arc::new(Notify::new()); + let release = Arc::new(Notify::new()); + let takeover = Arc::new(Notify::new()); + let takeover_release = Arc::new(Notify::new()); + let calls = Arc::new(AtomicUsize::new(0)); + let leader = tokio::spawn({ + let raw = raw.clone(); + let calls = calls.clone(); + let entered = entered.clone(); + let release = release.clone(); + async move { + get_or_project(content_id, || async move { + calls.fetch_add(1, Ordering::SeqCst); + entered.notify_one(); + release.notified().await; + Ok(raw.as_ref().clone()) + }) + .await + } + }); + started(&entered).await; + let waiter = tokio::spawn({ + let raw = raw.clone(); + let calls = calls.clone(); + let takeover = takeover.clone(); + let takeover_release = takeover_release.clone(); + async move { + get_or_project(content_id, || async move { + calls.fetch_add(1, Ordering::SeqCst); + takeover.notify_one(); + takeover_release.notified().await; + Ok(raw.as_ref().clone()) + }) + .await + } + }); + participants(content_id, 2).await; + leader.abort(); + assert!(leader.await.err().unwrap().is_cancelled()); + started(&takeover).await; + assert_eq!(calls.load(Ordering::SeqCst), 2); + assert_uncached(content_id); + takeover_release.notify_one(); + let projection = waiter.await.unwrap().unwrap(); + assert_eq!(projection.chunk_bytes(0).unwrap(), raw.as_slice()); + assert_flight_released(content_id); +} + +#[tokio::test] +async fn cancelled_waiter_does_not_cancel_or_leak_the_leader() { + let (content_id, raw) = unique_data(127); + let entered = Arc::new(Notify::new()); + let release = Arc::new(Notify::new()); + let leader = tokio::spawn({ + let entered = entered.clone(); + let release = release.clone(); + async move { + get_or_project(content_id, || async move { + entered.notify_one(); + release.notified().await; + Ok(raw.as_ref().clone()) + }) + .await + } + }); + started(&entered).await; + let waiter = tokio::spawn(async move { + get_or_project(content_id, || async { + panic!("cancelled waiter must not load while leader is active"); + }) + .await + }); + participants(content_id, 2).await; + waiter.abort(); + assert!(waiter.await.err().unwrap().is_cancelled()); + participants(content_id, 1).await; + release.notify_one(); + let projection = leader.await.unwrap().unwrap(); + assert_eq!(projection.map.file_content_id, content_id); + assert_flight_released(content_id); +} + +#[tokio::test] +async fn evicted_result_survives_leader_drop_until_joined_waiter_finishes() { + let (content_id, raw) = unique_data(131); + let calls = Arc::new(AtomicUsize::new(0)); + let entered = Arc::new(Notify::new()); + let release = Arc::new(Notify::new()); + let leader = tokio::spawn({ + let raw = raw.clone(); + let calls = calls.clone(); + let entered = entered.clone(); + let release = release.clone(); + async move { + get_or_project(content_id, || async move { + calls.fetch_add(1, Ordering::SeqCst); + entered.notify_one(); + release.notified().await; + Ok(raw.as_ref().clone()) + }) + .await + } + }); + started(&entered).await; + let registry = FLIGHTS.get().unwrap(); + let retained = FlightClaim::acquire(registry, content_id).unwrap(); + let result_weak; + { + // Queue a test-only lock before the real waiter, keeping its turn + // blocked until eviction and the leader's return owner are gone. + let queued = retained.flight.as_ref().unwrap().result.lock(); + tokio::pin!(queued); + assert!(futures::poll!(&mut queued).is_pending()); + let waiter = tokio::spawn({ + let raw = raw.clone(); + let calls = calls.clone(); + async move { + get_or_project(content_id, || async move { + calls.fetch_add(1, Ordering::SeqCst); + Ok(raw.as_ref().clone()) + }) + .await + } + }); + participants(content_id, 3).await; + release.notify_one(); + let projection = leader.await.unwrap().unwrap(); + let guard = queued.await; + result_weak = Arc::downgrade(&projection); + evict(content_id); + drop(projection); + assert_uncached(content_id); + assert!(result_weak.upgrade().is_some()); + drop(guard); + let shared = waiter.await.unwrap().unwrap(); + assert!(result_weak.ptr_eq(&Arc::downgrade(&shared))); + assert_eq!(shared.chunk_bytes(0).unwrap(), raw.as_slice()); + assert_eq!(calls.load(Ordering::SeqCst), 1); + drop(shared); + } + assert!(result_weak.upgrade().is_some()); + drop(retained); + assert!(result_weak.upgrade().is_none()); + assert_flight_released(content_id); + assert_uncached(content_id); + let reloaded = get_or_project(content_id, || async { + calls.fetch_add(1, Ordering::SeqCst); + Ok(raw.as_ref().clone()) + }) + .await + .unwrap(); + assert_eq!(calls.load(Ordering::SeqCst), 2); + assert_eq!(reloaded.chunk_bytes(0).unwrap(), raw.as_slice()); + assert_flight_released(content_id); +} + +#[test] +fn distinct_flights_are_bounded_but_existing_digest_can_join_at_capacity() { + let registry = Mutex::new(FlightRegistry::default()); + let mut claims = Vec::new(); + for index in 0..PROJECTION_FLIGHT_CAP { + claims.push(FlightClaim::acquire(®istry, [index as u8; 32]).unwrap()); + } + let joined = FlightClaim::acquire(®istry, [42; 32]).unwrap(); + assert!(Arc::ptr_eq( + claims[42].flight.as_ref().unwrap(), + joined.flight.as_ref().unwrap() + )); + let error = FlightClaim::acquire(®istry, [255; 32]).err().unwrap(); + assert_eq!(error.code, SnapshotErrorCode::LimitExceeded); + drop(claims); + assert_eq!(registry.lock().unwrap().entries.len(), 1); + drop(joined); + assert!(registry.lock().unwrap().entries.is_empty()); + let fresh = FlightClaim::acquire(®istry, [255; 32]).unwrap(); + drop(fresh); + assert!(registry.lock().unwrap().entries.is_empty()); +} + +#[test] +fn last_claim_removes_only_its_exact_registered_gate() { + let registry = Mutex::new(FlightRegistry::default()); + let content_id = [91; 32]; + let old = FlightClaim::acquire(®istry, content_id).unwrap(); + let replacement = Arc::new(ProjectionFlight { + result: tokio::sync::Mutex::new(None), + }); + registry + .lock() + .unwrap() + .entries + .insert(content_id, Arc::downgrade(&replacement)); + drop(old); + assert!(registry.lock().unwrap().entries[&content_id].ptr_eq(&Arc::downgrade(&replacement))); + let joined = FlightClaim::acquire(®istry, content_id).unwrap(); + assert!(Arc::ptr_eq(joined.flight.as_ref().unwrap(), &replacement)); + drop(replacement); + drop(joined); + assert!(registry.lock().unwrap().entries.is_empty()); + registry + .lock() + .unwrap() + .entries + .insert(content_id, Weak::new()); + let reclaimed = FlightClaim::acquire(®istry, content_id).unwrap(); + drop(reclaimed); + assert!(registry.lock().unwrap().entries.is_empty()); +} + +#[test] +fn concurrent_final_claim_drops_release_registry_capacity() { + let registry = Mutex::new(FlightRegistry::default()); + for index in 0..64 { + let content_id = [index; 32]; + let first = FlightClaim::acquire(®istry, content_id).unwrap(); + let second = FlightClaim::acquire(®istry, content_id).unwrap(); + let barrier = Barrier::new(2); + std::thread::scope(|scope| { + let barrier = &barrier; + scope.spawn(move || { + barrier.wait(); + drop(first); + }); + scope.spawn(move || { + barrier.wait(); + drop(second); + }); + }); + assert!(registry.lock().unwrap().entries.is_empty()); + } +} diff --git a/src/ceres/snapshot/chunks.rs b/src/ceres/snapshot/chunks.rs index 18d9027b..e4182c01 100644 --- a/src/ceres/snapshot/chunks.rs +++ b/src/ceres/snapshot/chunks.rs @@ -13,7 +13,7 @@ use std::{ collections::{HashMap, VecDeque}, - sync::{Mutex, OnceLock}, + sync::{Arc, Mutex, OnceLock, Weak}, }; use mst2_codec::chunkmap::{CHUNK_SIZE, CHUNKS_PER_PAGE, ChunkLeaf, ChunkMap}; @@ -202,7 +202,7 @@ impl ProjectionCache { return; } // Reclaim oldest entries until the new one fits. A file larger than - // the cap alone is not cached (rebuild per request) but still + // the cap alone is not cached between flights but still // served correctly. while self.total_bytes + proj.raw.len() > STAGED_CAP_BYTES && let Some(victim) = self.order.pop_front() @@ -219,8 +219,88 @@ impl ProjectionCache { } } +static FLIGHTS: OnceLock> = OnceLock::new(); +const PROJECTION_FLIGHT_CAP: usize = 128; + +struct ProjectionFlight { + result: tokio::sync::Mutex>>, +} + +#[derive(Default)] +struct FlightRegistry { + entries: HashMap<[u8; 32], Weak>, +} + +struct FlightClaim<'a> { + content_id: [u8; 32], + flight: Option>, + registry: &'a Mutex, +} + +impl<'a> FlightClaim<'a> { + fn acquire( + registry: &'a Mutex, + content_id: [u8; 32], + ) -> Result { + let mut state = registry + .lock() + .map_err(|_| internal("chunk projection flight registry lock poisoned"))?; + if let Some(flight) = state.entries.get(&content_id).and_then(Weak::upgrade) { + return Ok(Self { + content_id, + flight: Some(flight), + registry, + }); + } + state.entries.remove(&content_id); + if state.entries.len() >= PROJECTION_FLIGHT_CAP { + return Err(SnapshotError::new( + SnapshotErrorCode::LimitExceeded, + "too many distinct chunk projections in flight", + )); + } + let flight = Arc::new(ProjectionFlight { + result: tokio::sync::Mutex::new(None), + }); + state.entries.insert(content_id, Arc::downgrade(&flight)); + Ok(Self { + content_id, + flight: Some(flight), + registry, + }) + } +} + +impl Drop for FlightClaim<'_> { + fn drop(&mut self) { + let Some(flight) = self.flight.take() else { + return; + }; + let Ok(mut state) = self.registry.lock() else { + return; + }; + if Arc::strong_count(&flight) == 1 { + if state + .entries + .get(&self.content_id) + .is_some_and(|registered| registered.ptr_eq(&Arc::downgrade(&flight))) + { + state.entries.remove(&self.content_id); + } + drop(state); + drop(flight); + } else { + // Serialize owner release with admission and other final drops. + // Another participant keeps the potentially large result alive. + drop(flight); + drop(state); + } + } +} + /// Return the cached projection, or build one via `load` (which must resolve -/// the file in the fixed view and return its verified bytes). +/// the file in the fixed view and return its verified bytes). Concurrent +/// misses for the same digest share one successful load and projection. pub async fn get_or_project( content_id: [u8; 32], load: F, @@ -230,15 +310,46 @@ where Fut: std::future::Future, SnapshotError>>, { let cache = STAGED.get_or_init(|| Mutex::new(ProjectionCache::new())); - if let Some(p) = cache.lock().unwrap().get(content_id) { + if let Some(p) = cache + .lock() + .map_err(|_| internal("chunk projection cache lock poisoned"))? + .get(content_id) + { + return Ok(p); + } + + let registry = FLIGHTS.get_or_init(|| Mutex::new(FlightRegistry::default())); + let claim = FlightClaim::acquire(registry, content_id)?; + let flight = claim + .flight + .as_ref() + .ok_or_else(|| internal("chunk projection flight claim released"))?; + let mut result = flight.result.lock().await; + if let Some(p) = cache + .lock() + .map_err(|_| internal("chunk projection cache lock poisoned"))? + .get(content_id) + { + *result = Some(p.clone()); return Ok(p); } + if let Some(p) = result.as_ref() { + return Ok(p.clone()); + } let raw = load().await?; let proj = std::sync::Arc::new(ChunkProjection::build(content_id, raw)?); - cache.lock().unwrap().put(proj.clone()); + cache + .lock() + .map_err(|_| internal("chunk projection cache lock poisoned"))? + .put(proj.clone()); + *result = Some(proj.clone()); Ok(proj) } +#[cfg(test)] +#[path = "chunk_singleflight_tests.rs"] +mod singleflight_tests; + #[cfg(test)] mod tests { use super::*; From 2a8a001866fa9cd2d8fbc8a078e54612870d2abb Mon Sep 17 00:00:00 2001 From: Xiaoyang Han Date: Wed, 7 Oct 2026 23:41:37 +0800 Subject: [PATCH 16/50] perf(mst2): admit OBJECT batches before bounded body reads (#66) Admit all fixed paths and current verified facts before body I/O, reject item/batch caps early, and collect actual raw streams with exact EOF/length/full SHA. Share only exact OID/digest/size bodies; verify every different source. Preserve frames/END/current source and lease guards, remove replaced full-body OBJECT helper, and add thirteen HTTP regressions. Native and performance evidence remain separate. --- src/api/router/snapshot_content.rs | 171 +++-- src/api/router/snapshot_content_tests.rs | 29 + .../router/snapshot_objects_bounded_tests.rs | 645 ++++++++++++++++++ src/ceres/api_service/mod.rs | 13 + src/jupiter/service/git_service.rs | 12 + 5 files changed, 806 insertions(+), 64 deletions(-) create mode 100644 src/api/router/snapshot_objects_bounded_tests.rs diff --git a/src/api/router/snapshot_content.rs b/src/api/router/snapshot_content.rs index 6e419a61..a6d9efc4 100644 --- a/src/api/router/snapshot_content.rs +++ b/src/api/router/snapshot_content.rs @@ -12,6 +12,7 @@ use axum::{ use futures::stream::StreamExt; use serde::Deserialize; use serde_json::json; +use sha2::{Digest, Sha256}; use super::{ abs_view_path, guarded_treeframe_response, internal, mst2_error_response, request::Mst2Bytes, @@ -19,21 +20,11 @@ use super::{ use crate::ceres::snapshot::{ chunks::{ChunkProjection, get_or_project}, error::{SnapshotError, SnapshotErrorCode}, - pages::{ - MetadataWalkOutcome, WalkOutcome, base64_of, fetch_raw_blob, hex_of, resolve_abs, - resolve_abs_metadata, - }, + pages::{MetadataWalkOutcome, base64_of, fetch_raw_blob, hex_of, resolve_abs_metadata}, resolver::FsKind, view::validate_scope_relative_path, }; -/// One file resolved at a fixed path with verified content. -struct ResolvedFile { - digest: [u8; 32], - size: u64, - raw: Vec, -} - pub(super) struct ResolvedFileMetadata { fs_kind: FsKind, oid: String, @@ -173,52 +164,72 @@ pub(super) fn fs_kind_str(k: FsKind) -> &'static str { } } -/// Resolve a scope-relative path against the fixed tree and verify the -/// optional `expected_digest`. Absence/directory/intermediate outcomes stay -/// typed errors, never an empty body. #[allow(clippy::result_large_err)] -async fn resolve_file( +async fn read_object( handler: &T, - root_tree: &git_internal::internal::object::tree::Tree, - scope: &str, + file: &ResolvedFileMetadata, path: &str, - expected_digest: Option<&str>, -) -> Result { - let abs_path = abs_view_path(scope, path); - match resolve_abs(handler, root_tree, &abs_path) +) -> Result, Response> { + if file.size > OBJECT_ITEM_MAX { + return Err(mst2_error_response(SnapshotError::new( + SnapshotErrorCode::ScopeInvalid, + "object exceeds the 256KiB item cap", + ))); + } + let expected_size = file.size as usize; + let mut input = handler + .get_raw_blob_stream_by_hash(&file.oid) .await - .map_err(mst2_error_response)? - { - WalkOutcome::FoundFile { - raw, size, digest, .. - } => { - if let Some(expected) = expected_digest - && expected != format!("sha256:{}", hex_of(&digest)) - { - return Err(mst2_error_response(SnapshotError::new( - SnapshotErrorCode::DigestMismatch, - format!("{path}: content does not match expected_digest"), - ))); - } - Ok(ResolvedFile { digest, size, raw }) + .map_err(|error| { + let code = match error { + crate::common::errors::MegaError::ObjStorageNotFound(_) => { + SnapshotErrorCode::ObjectUnavailable + } + crate::common::errors::MegaError::ObjStorageInconsistent(_) => { + SnapshotErrorCode::IntegrityError + } + _ => SnapshotErrorCode::Internal, + }; + tracing::warn!(error = %error, "fixed-view blob fetch failed"); + mst2_error_response(SnapshotError::new( + code, + "fixed-view content could not be read", + )) + })?; + let mut raw = Vec::with_capacity(expected_size); + let mut hash = Sha256::new(); + while let Some(part) = input.next().await { + let bytes = part.map_err(|error| { + tracing::warn!(error = %error, "fixed-view blob stream failed"); + mst2_error_response(SnapshotError::new( + SnapshotErrorCode::Internal, + "fixed-view content could not be read", + )) + })?; + // Reject oversized producer chunks before copying or hashing them. + if bytes.len() > expected_size - raw.len() { + return Err(mst2_error_response(SnapshotError::new( + SnapshotErrorCode::IntegrityError, + "fixed blob length disagrees with its verified size fact", + ))); } - WalkOutcome::FoundDir => Err(mst2_error_response(SnapshotError::new( - SnapshotErrorCode::NotDirectory, - format!("{path} is a directory"), - ))), - WalkOutcome::Absent => Err(mst2_error_response(SnapshotError::new( - SnapshotErrorCode::PathNotFound, - format!("{path} absent in the fixed view"), - ))), - WalkOutcome::NotDirectory { symlink } => Err(mst2_error_response(SnapshotError::new( - if symlink { - SnapshotErrorCode::SymlinkTraversal - } else { - SnapshotErrorCode::NotDirectory - }, - format!("{path}: intermediate component is not a directory"), - ))), + hash.update(&bytes); + raw.extend_from_slice(&bytes); } + if raw.len() != expected_size { + return Err(mst2_error_response(SnapshotError::new( + SnapshotErrorCode::IntegrityError, + "fixed blob length disagrees with its verified size fact", + ))); + } + let digest: [u8; 32] = hash.finalize().into(); + if digest != file.digest { + return Err(mst2_error_response(SnapshotError::new( + SnapshotErrorCode::DigestMismatch, + format!("{path}: content does not match expected_digest"), + ))); + } + Ok(raw) } #[derive(Deserialize, Debug)] @@ -312,9 +323,7 @@ pub(super) async fn objects( .map_err(mst2_error_response)? .unwrap_or(crate::ceres::snapshot::frame_stream::Encoding::Identity); - // Verify every member at its fixed path before any 200 is produced. - // Members resolve concurrently: a batch is up to 128 files, and a - // sequential S3 read per member dominated cold-mount time. + // Admit the whole fixed-path batch before opening any object body. let handler = state .api_handler(std::path::Path::new("/")) .await @@ -329,7 +338,7 @@ pub(super) async fn objects( let resolved: Vec> = futures::stream::iter(req.items.clone()) .map(move |item| async move { validate_scope_relative_path(&item.path).map_err(mst2_error_response)?; - let f = resolve_file( + let f = resolve_file_metadata( handler_ref, root_ref, scope_ref, @@ -351,21 +360,55 @@ pub(super) async fn objects( .buffered(16) .collect() .await; - let mut unique: Vec<([u8; 32], Vec)> = Vec::new(); - let mut seen: Vec<[u8; 32]> = Vec::new(); - let mut logical_bytes = 0u64; + let mut sources: Vec<(ObjectItem, ResolvedFileMetadata)> = Vec::new(); + let mut content_sizes: Vec<([u8; 32], u64)> = Vec::new(); + let mut planned_bytes = 0u64; for pair in resolved { - let (_, f) = pair?; - if !seen.contains(&f.digest) { - if logical_bytes as usize + f.raw.len() > OBJECT_TOTAL_MAX { + let (item, f) = pair?; + if let Some((_, size)) = content_sizes.iter().find(|(digest, _)| *digest == f.digest) { + if *size != f.size { + return Err(mst2_error_response(SnapshotError::new( + SnapshotErrorCode::IntegrityError, + "fixed content digest has conflicting verified sizes", + ))); + } + } else { + if planned_bytes + f.size > OBJECT_TOTAL_MAX as u64 { return Err(mst2_error_response(SnapshotError::new( SnapshotErrorCode::ScopeInvalid, "unique object content exceeds the 8MiB batch cap", ))); } - logical_bytes += f.raw.len() as u64; - seen.push(f.digest); - unique.push((f.digest, f.raw)); + planned_bytes += f.size; + content_sizes.push((f.digest, f.size)); + } + if let Some((_, source)) = sources.iter().find(|(_, source)| source.oid == f.oid) { + if source.digest != f.digest || source.size != f.size { + return Err(mst2_error_response(SnapshotError::new( + SnapshotErrorCode::IntegrityError, + "fixed object has conflicting verified facts", + ))); + } + } else { + sources.push((item, f)); + } + } + let loaded = futures::stream::iter(sources) + .map(|(item, file)| async move { + let raw = read_object(handler_ref, &file, &item.path).await?; + Ok::<_, Response>((file.digest, raw)) + }) + .buffered(16); + tokio::pin!(loaded); + let mut unique: Vec<([u8; 32], Vec)> = Vec::new(); + let mut seen: Vec<[u8; 32]> = Vec::new(); + let mut logical_bytes = 0u64; + while let Some(pair) = loaded.next().await { + let (digest, raw) = pair?; + if !seen.contains(&digest) { + logical_bytes += raw.len() as u64; + seen.push(digest); + unique.push((digest, raw)); } } diff --git a/src/api/router/snapshot_content_tests.rs b/src/api/router/snapshot_content_tests.rs index 5399ad42..1e0dcd55 100644 --- a/src/api/router/snapshot_content_tests.rs +++ b/src/api/router/snapshot_content_tests.rs @@ -79,6 +79,9 @@ use crate::{ const TOKEN: &str = "mst2-fixed-content-test"; +#[path = "snapshot_objects_bounded_tests.rs"] +mod bounded_objects; + #[path = "snapshot_session_tests.rs"] mod durable_sessions; @@ -99,6 +102,7 @@ struct ReadCounts { whole: AtomicUsize, range: AtomicUsize, bytes: AtomicUsize, + object_fault: std::sync::Mutex>, } impl ReadCounts { @@ -134,6 +138,11 @@ impl MegaObjectStorage for CountingStorage { async fn get_stream(&self, key: &ObjectKey) -> OrbitResult<(ObjectByteStream, ObjectMeta)> { self.counts.whole.fetch_add(1, Ordering::SeqCst); let (stream, meta) = self.inner.inner.get_stream(key).await?; + let fault = self.counts.object_fault.lock().unwrap().clone(); + let stream = match fault { + Some(fault) if fault.oid == key.key => fault.stream(), + _ => stream, + }; let counts = self.counts.clone(); let stream = stream.map(move |part| { if let Ok(bytes) = &part { @@ -328,6 +337,14 @@ impl Fixture { } async fn new_with_pg_config_and_directories(rebuildable: bool, directory_count: usize) -> Self { + Self::new_with_pg_config_directories_and_objects(rebuildable, directory_count, &[]).await + } + + async fn new_with_pg_config_directories_and_objects( + rebuildable: bool, + directory_count: usize, + objects: &[(String, Vec)], + ) -> Self { let temp = tempfile::tempdir().unwrap(); let mut config = isolated_config(temp.path().join("config")); config.monorepo.push_policy = PushPolicy::Trunk; @@ -414,6 +431,18 @@ impl Fixture { )); extra_trees.push(child); } + for (name, raw) in objects { + let oid = storage + .git_service + .save_object_from_raw(Bytes::copy_from_slice(raw)) + .await + .unwrap(); + project_items.push(item( + TreeItemMode::Blob, + ObjectHash::from_hex_for_kind(HashKind::Sha1, &oid).unwrap(), + name, + )); + } let project = tree(project_items); let old_tip = Commit::from_tree_id_with_kind( HashKind::Sha1, diff --git a/src/api/router/snapshot_objects_bounded_tests.rs b/src/api/router/snapshot_objects_bounded_tests.rs new file mode 100644 index 00000000..6475c53a --- /dev/null +++ b/src/api/router/snapshot_objects_bounded_tests.rs @@ -0,0 +1,645 @@ +use std::io; + +use tokio::{sync::Notify, time::timeout}; + +use super::*; + +#[derive(Clone)] +pub(super) struct StreamFault { + pub(super) oid: String, + kind: FaultKind, +} + +#[derive(Clone)] +enum FaultKind { + Parts(Vec), + LateError(Bytes), + Oversized(Bytes, Arc), + Held { + raw: Bytes, + entered: Arc, + release: Arc, + drops: Arc, + }, +} + +struct DropCount(Arc); + +impl Drop for DropCount { + fn drop(&mut self) { + self.0.fetch_add(1, Ordering::SeqCst); + } +} + +impl StreamFault { + pub(super) fn stream(self) -> ObjectByteStream { + match self.kind { + FaultKind::Parts(parts) => Box::pin(futures::stream::iter(parts.into_iter().map(Ok))), + FaultKind::LateError(raw) => Box::pin(futures::stream::iter([ + Ok(raw), + Err(io::Error::other("test late object stream failure")), + ])), + FaultKind::Oversized(raw, tail_polls) => Box::pin(futures::stream::unfold( + (Some(raw), tail_polls), + |(raw, tail_polls)| async move { + match raw { + Some(raw) => Some((Ok(raw), (None, tail_polls))), + None => { + tail_polls.fetch_add(1, Ordering::SeqCst); + Some(( + Err(io::Error::other("oversized tail must not be polled")), + (None, tail_polls), + )) + } + } + }, + )), + FaultKind::Held { + raw, + entered, + release, + drops, + } => Box::pin(futures::stream::unfold( + (raw, entered, release, DropCount(drops), 0u8), + |(raw, entered, release, drop_count, turn)| async move { + match turn { + 0 => Some((Ok(raw.slice(..1)), (raw, entered, release, drop_count, 1))), + 1 => { + entered.notify_one(); + release.notified().await; + Some((Ok(raw.slice(1..)), (raw, entered, release, drop_count, 2))) + } + _ => None, + } + }, + )), + } + } +} + +fn files(count: usize, size: usize) -> Vec<(String, Vec)> { + let seed = uuid::Uuid::new_v4(); + (0..count) + .map(|index| { + let mut raw = vec![index as u8; size]; + let prefix = format!("blob 3\0abc\0{seed}-{index}"); + raw[..prefix.len()].copy_from_slice(prefix.as_bytes()); + (format!("object-{index:03}"), raw) + }) + .collect() +} + +async fn fixture(objects: &[(String, Vec)]) -> Fixture { + Fixture::new_with_pg_config_directories_and_objects(false, 0, objects).await +} + +fn body(objects: &[(String, Vec)]) -> Body { + Body::from(request_bytes(objects)) +} + +fn request_bytes(objects: &[(String, Vec)]) -> Vec { + serde_json::to_vec(&json!({ + "items":objects.iter().map(|(name, raw)| json!({ + "path":format!("/{name}"), + "expected_digest":format!("sha256:{}", hex_of(&digest(raw))), + })).collect::>(), + "encoding":"identity", + })) + .unwrap() +} + +async fn object_oid(fixture: &Fixture, path: &str) -> String { + let handler = MonoApiService::from(&fixture.state); + let context = fixture + .state + .storage + .mono_storage() + .get_main_ref("/project") + .await + .unwrap() + .unwrap(); + let tree = handler + .get_tree_by_hash(&context.ref_tree_hash) + .await + .unwrap(); + match resolve_abs_metadata(&handler, &tree, path).await.unwrap() { + MetadataWalkOutcome::FoundFile { oid, .. } => oid, + other => panic!("object test path did not resolve: {other:?}"), + } +} + +async fn install_fault(fixture: &Fixture, path: &str, kind: FaultKind) { + let oid = object_oid(fixture, path).await; + *fixture.counts.object_fault.lock().unwrap() = Some(StreamFault { oid, kind }); +} + +async fn assert_objects(response: Response, request: &[u8], objects: &[(String, Vec)]) { + assert_eq!(response.status(), 200); + assert_eq!( + response.headers()["x-mega-request-digest"], + format!("sha256:{}", hex_of(&digest(request))) + ); + let encoded = to_bytes(response.into_body(), 10 * 1024 * 1024) + .await + .unwrap(); + let frames = parse_stream(&encoded).unwrap(); + let mut actual = Vec::new(); + let mut end = None; + for frame in frames { + match frame { + Frame::Object(payload) => { + assert!(end.is_none(), "END must be terminal"); + actual.extend(payload.objects); + } + Frame::End(record) => { + assert!(end.is_none()); + end = Some(record); + } + other => panic!("unexpected OBJECT frame: {other:?}"), + } + } + let mut expected = Vec::new(); + for (_, raw) in objects { + let id = digest(raw); + if !expected.iter().any(|(digest, _)| *digest == id) { + expected.push((id, raw.clone())); + } + } + assert_eq!(actual, expected); + let end = end.unwrap(); + assert_eq!(end.request_item_count, objects.len() as u32); + assert_eq!(end.unique_unit_count, expected.len() as u32); + assert_eq!( + end.logical_bytes, + expected + .iter() + .map(|(_, bytes)| bytes.len() as u64) + .sum::() + ); + assert_eq!(end.request_body_sha256, digest(request)); +} + +#[tokio::test] +async fn oversized_item_and_later_invalid_path_reject_entire_batch_before_body_io() { + let fixture = Fixture::new().await; + let link_digest = format!("sha256:{}", hex_of(&digest(b"file"))); + for last in [ + json!({"path":"/file","expected_digest":fixture.digest_string()}), + json!({"path":"/missing","expected_digest":link_digest}), + ] { + let oversized = last["path"] == "/file"; + let request = json!({"items":[ + {"path":"/link","expected_digest":link_digest},last, + ]}); + error( + fixture + .send("POST", "objects", Body::from(request.to_string())) + .await, + if oversized { 400 } else { 404 }, + if oversized { + "SCOPE_INVALID" + } else { + "PATH_NOT_FOUND" + }, + false, + ) + .await; + fixture.counts.assert(0, 0); + } +} + +#[tokio::test] +async fn unique_batch_over_eight_mib_rejects_before_any_body_io() { + let objects = files(33, 256 * 1024); + let fixture = fixture(&objects).await; + error( + fixture.send("POST", "objects", body(&objects)).await, + 400, + "SCOPE_INVALID", + false, + ) + .await; + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn cap_boundaries_and_empty_object_keep_exact_raw_bytes_and_end_counts() { + let mut objects = files(32, 256 * 1024); + objects.push(("empty".into(), Vec::new())); + let fixture = fixture(&objects[..32]).await; + let request = request_bytes(&objects); + assert_objects( + fixture + .send("POST", "objects", Body::from(request.clone())) + .await, + &request, + &objects, + ) + .await; + fixture.counts.assert(33, 8 * 1024 * 1024); +} + +#[tokio::test] +async fn every_alias_is_admitted_and_exact_oid_body_is_loaded_once() { + let raw = files(1, 8192).remove(0).1; + let objects: Vec<_> = (0..128) + .map(|i| (format!("alias-{i:03}"), raw.clone())) + .collect(); + let fixture = fixture(&objects).await; + let request = request_bytes(&objects); + assert_objects( + fixture + .send("POST", "objects", Body::from(request.clone())) + .await, + &request, + &objects, + ) + .await; + fixture.counts.assert(1, raw.len()); + fixture.counts.reset(); + let mut request: Value = serde_json::from_slice(&request).unwrap(); + request["items"][127]["expected_digest"] = json!(format!("sha256:{}", "00".repeat(32))); + error( + fixture + .send("POST", "objects", Body::from(request.to_string())) + .await, + 409, + "DIGEST_MISMATCH", + false, + ) + .await; + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn conflicting_sizes_reject_before_io_and_distinct_oids_still_verify_each_body() { + let objects = files(2, 8192); + let fixture = fixture(&objects).await; + let oid = object_oid(&fixture, "/object-001").await; + let db = fixture + .state + .storage + .mono_storage() + .get_connection() + .clone(); + let fact = mst2_verified_object::Entity::find() + .filter(mst2_verified_object::Column::GitOid.eq(oid)) + .one(&db) + .await + .unwrap() + .unwrap(); + let mut fact = fact.into_active_model(); + fact.raw_sha256 = Set(digest(&objects[0].1).to_vec()); + fact.size = Set(8193); + let fact = fact.update(&db).await.unwrap(); + let request = json!({"items":[ + {"path":"/object-000","expected_digest":format!("sha256:{}", hex_of(&digest(&objects[0].1)))}, + {"path":"/object-001","expected_digest":format!("sha256:{}", hex_of(&digest(&objects[0].1)))}, + ]}); + error( + fixture + .send("POST", "objects", Body::from(request.to_string())) + .await, + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + fixture.counts.assert(0, 0); + let mut fact = fact.into_active_model(); + fact.size = Set(8192); + fact.update(&db).await.unwrap(); + error( + fixture + .send("POST", "objects", Body::from(request.to_string())) + .await, + 409, + "DIGEST_MISMATCH", + false, + ) + .await; + fixture.counts.assert(2, 16384); +} + +#[tokio::test] +async fn missing_or_invalid_current_facts_fail_before_body_io() { + let fixture = Fixture::new().await; + let request = json!({"items":[{"path":"/file","expected_digest":fixture.digest_string()}]}); + let fact = fixture.fact().await; + fixture.delete_fact().await; + error( + fixture + .send("POST", "objects", Body::from(request.to_string())) + .await, + 503, + "METADATA_NOT_READY", + true, + ) + .await; + fixture.counts.assert(0, 0); + let mut invalid = fact; + invalid.raw_sha256 = vec![0; 31]; + fixture.replace_fact(invalid).await; + error( + fixture + .send("POST", "objects", Body::from(request.to_string())) + .await, + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn oversized_stream_chunk_is_rejected_without_copying_or_polling_tail() { + let objects = files(1, 8192); + let fixture = fixture(&objects).await; + let tail_polls = Arc::new(AtomicUsize::new(0)); + install_fault( + &fixture, + "/object-000", + FaultKind::Oversized(Bytes::from(vec![7; 8193]), tail_polls.clone()), + ) + .await; + error( + fixture.send("POST", "objects", body(&objects)).await, + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + fixture.counts.assert(1, 8193); + assert_eq!(tail_polls.load(Ordering::SeqCst), 0); +} + +#[tokio::test] +async fn truncated_wrong_sha_and_late_stream_error_never_produce_200() { + let objects = files(1, 8192); + let fixture = fixture(&objects).await; + for (fault, status, code, bytes) in [ + ( + FaultKind::Parts(vec![Bytes::copy_from_slice(&objects[0].1[..8191])]), + 502, + "INTEGRITY_ERROR", + 8191, + ), + ( + FaultKind::Parts(vec![Bytes::from(vec![0; 8192])]), + 409, + "DIGEST_MISMATCH", + 8192, + ), + ( + FaultKind::LateError(Bytes::copy_from_slice(&objects[0].1)), + 500, + "INTERNAL", + 8192, + ), + ( + FaultKind::Parts(vec![ + Bytes::copy_from_slice(&objects[0].1), + Bytes::from_static(b"x"), + ]), + 502, + "INTEGRITY_ERROR", + 8193, + ), + ] { + fixture.counts.reset(); + install_fault(&fixture, "/object-000", fault).await; + error( + fixture.send("POST", "objects", body(&objects)).await, + status, + code, + code == "INTERNAL", + ) + .await; + fixture.counts.assert(1, bytes); + } +} + +#[tokio::test] +async fn a_later_object_stream_failure_keeps_earlier_verified_data_unpublished() { + let objects = files(2, 8192); + let fixture = fixture(&objects).await; + install_fault( + &fixture, + "/object-001", + FaultKind::LateError(Bytes::copy_from_slice(&objects[1].1)), + ) + .await; + error( + fixture.send("POST", "objects", body(&objects)).await, + 500, + "INTERNAL", + true, + ) + .await; + fixture.counts.assert(2, 16384); +} + +#[tokio::test] +async fn missing_actual_object_body_keeps_source_error_classification_and_can_retry() { + let objects = files(1, 8192); + let fixture = fixture(&objects).await; + let oid = object_oid(&fixture, "/object-000").await; + fixture + .state + .storage + .git_service + .obj_storage + .inner + .delete(&ObjectKey { + namespace: ObjectNamespace::Git, + key: oid.clone(), + }) + .await + .unwrap(); + error( + fixture.send("POST", "objects", body(&objects)).await, + 503, + "OBJECT_UNAVAILABLE", + false, + ) + .await; + fixture.counts.assert(1, 0); + fixture + .state + .storage + .git_service + .save_object_from_model(objects[0].1.clone(), &oid) + .await + .unwrap(); + fixture.counts.reset(); + let request = request_bytes(&objects); + assert_objects( + fixture + .send("POST", "objects", Body::from(request.clone())) + .await, + &request, + &objects, + ) + .await; + fixture.counts.assert(1, 8192); +} + +#[tokio::test] +async fn old_snapshot_objects_keep_fixed_oid_after_real_publication_advances() { + let objects = files(1, 8192); + let fixture = fixture(&objects).await; + let mono = fixture.state.storage.mono_storage(); + let old = mono.get_main_ref("/project").await.unwrap().unwrap(); + let old_commit = ObjectHash::from_hex_for_kind(HashKind::Sha1, &old.ref_commit_hash).unwrap(); + let oid = object_oid(&fixture, "/object-000").await; + let next_tree = tree(vec![item( + TreeItemMode::Blob, + ObjectHash::from_hex_for_kind(HashKind::Sha1, &oid).unwrap(), + "new-only", + )]); + let next_commit = Commit::from_tree_id_with_kind( + HashKind::Sha1, + next_tree.id, + vec![old_commit], + "OBJECT old snapshot must retain its fixed path", + ) + .unwrap(); + mono.save_mega_trees(vec![next_tree], next_commit.id, None) + .await + .unwrap(); + mono.save_mega_commits(vec![next_commit.clone()], None) + .await + .unwrap(); + publish_native_push(&fixture.state.storage, "/project", old_commit, &next_commit).await; + let head = mono + .read_native_publication_head( + fixture + .state + .storage + .config() + .mst2 + .instance_uuid + .as_deref() + .unwrap(), + ) + .await + .unwrap(); + assert_eq!(head.token.sequence, 2); + assert!(head.token.certificate.is_some()); + assert_ne!( + mono.get_main_ref("/project") + .await + .unwrap() + .unwrap() + .ref_tree_hash, + old.ref_tree_hash + ); + fixture.counts.reset(); + let request = request_bytes(&objects); + assert_objects( + fixture + .send("POST", "objects", Body::from(request.clone())) + .await, + &request, + &objects, + ) + .await; + fixture.counts.assert(1, 8192); +} + +#[tokio::test] +async fn cancelling_actual_object_request_drops_held_stream_and_retry_succeeds() { + let objects = files(1, 8192); + let fixture = fixture(&objects).await; + let entered = Arc::new(Notify::new()); + let release = Arc::new(Notify::new()); + let drops = Arc::new(AtomicUsize::new(0)); + install_fault( + &fixture, + "/object-000", + FaultKind::Held { + raw: Bytes::copy_from_slice(&objects[0].1), + entered: entered.clone(), + release, + drops: drops.clone(), + }, + ) + .await; + let task = tokio::spawn(fixture.app.clone().oneshot(fixture.request( + "POST", + "objects", + body(&objects), + ))); + timeout(Duration::from_secs(10), entered.notified()) + .await + .unwrap(); + assert!(!task.is_finished()); + fixture.counts.assert(1, 1); + task.abort(); + assert!(task.await.err().unwrap().is_cancelled()); + assert_eq!(drops.load(Ordering::SeqCst), 1); + *fixture.counts.object_fault.lock().unwrap() = None; + fixture.counts.reset(); + let request = request_bytes(&objects); + assert_objects( + fixture + .send("POST", "objects", Body::from(request.clone())) + .await, + &request, + &objects, + ) + .await; + fixture.counts.assert(1, 8192); +} + +#[tokio::test] +async fn lease_revoked_during_object_load_cannot_deliver_verified_frames() { + let objects = files(1, 8192); + let fixture = fixture(&objects).await; + let entered = Arc::new(Notify::new()); + let release = Arc::new(Notify::new()); + let drops = Arc::new(AtomicUsize::new(0)); + install_fault( + &fixture, + "/object-000", + FaultKind::Held { + raw: Bytes::copy_from_slice(&objects[0].1), + entered: entered.clone(), + release: release.clone(), + drops: drops.clone(), + }, + ) + .await; + let task = tokio::spawn(fixture.app.clone().oneshot(fixture.request( + "POST", + "objects", + body(&objects), + ))); + timeout(Duration::from_secs(10), entered.notified()) + .await + .unwrap(); + let revoked = fixture + .app + .clone() + .oneshot( + Request::builder() + .method("DELETE") + .uri(format!("/api/v2/snapshots/leases/{}", fixture.lease)) + .header("authorization", format!("Bearer {TOKEN}")) + .body(Body::empty()) + .unwrap(), + ) + .await + .unwrap(); + assert_eq!(revoked.status(), 200); + release.notify_one(); + let response = timeout(Duration::from_secs(10), task) + .await + .unwrap() + .unwrap() + .unwrap(); + error(response, 410, "LEASE_EXPIRED", false).await; + fixture.counts.assert(1, 8192); + assert_eq!(drops.load(Ordering::SeqCst), 1); +} diff --git a/src/ceres/api_service/mod.rs b/src/ceres/api_service/mod.rs index e5e0f4be..32bbf780 100644 --- a/src/ceres/api_service/mod.rs +++ b/src/ceres/api_service/mod.rs @@ -178,6 +178,19 @@ pub trait ApiHandler: Send + Sync { } } + async fn get_raw_blob_stream_by_hash( + &self, + hash: &str, + ) -> Result { + let storage = self.get_context(); + match storage.git_service.get_object_stream(hash).await { + Ok(stream) => Ok(stream), + Err(error) => Err(storage + .classify_blob_objstorage_not_found(hash, error) + .await), + } + } + /// Preview unified diff for a single file change async fn preview_file_diff( &self, diff --git a/src/jupiter/service/git_service.rs b/src/jupiter/service/git_service.rs index 17cf70f4..bf42b5ee 100644 --- a/src/jupiter/service/git_service.rs +++ b/src/jupiter/service/git_service.rs @@ -107,6 +107,18 @@ impl GitService { Ok(data) } + pub async fn get_object_stream(&self, hash: &str) -> Result { + if !is_full_hex_object_id(hash) { + return Err(MegaError::Other("Invalid object ID format".to_string())); + } + let key = ObjectKey { + namespace: ObjectNamespace::Git, + key: hash.to_string(), + }; + let (stream, _) = self.obj_storage.inner.get_stream(&key).await?; + Ok(stream) + } + pub fn get_objects_stream(&self, hashes: Vec) -> MultiObjectByteStream<'_> { // Filter out obviously invalid object ids early to avoid spurious backend requests. // Callers that need strict validation should validate up-front and return 4xx. From 659a8ea0079130c07b12af576db6afcb68222f54 Mon Sep 17 00:00:00 2001 From: Xiaoyang Han Date: Wed, 7 Oct 2026 23:54:26 +0800 Subject: [PATCH 17/50] feat(mst2): bind generic sessions to immutable storage routes (#67) Derive permanent semantic SID routes, separate generic physical bindings and immutable original lease routes on actual v3 open/read/renew/release paths. Serialize context/lease insert before row locks via captured core mono, route and namespace retention barriers; backfill complete inventory with fixed-scope proof. Authenticate generic prepare domain at fixed-core registration, preserve original domain immutability and old upgrade reads, and add ten actual PostgreSQL/HTTP regressions. Qualified serving and collector remain closed; native and performance evidence remain separate. --- src/api/router/snapshot_content_tests.rs | 6 + .../router/snapshot_storage_route_fixture.rs | 165 +++ .../router/snapshot_storage_route_tests.rs | 1183 +++++++++++++++++ ...20261007_000600_add_mst2_storage_routes.rs | 42 + .../m20261007_000600_storage_routes.sql | 329 +++++ src/jupiter/migration/mod.rs | 16 +- .../storage/native_metadata_install.rs | 4 + src/jupiter/storage/native_snapshot_routes.rs | 122 ++ .../storage/native_snapshot_session.rs | 44 +- 9 files changed, 1900 insertions(+), 11 deletions(-) create mode 100644 src/api/router/snapshot_storage_route_fixture.rs create mode 100644 src/api/router/snapshot_storage_route_tests.rs create mode 100644 src/jupiter/migration/m20261007_000600_add_mst2_storage_routes.rs create mode 100644 src/jupiter/migration/m20261007_000600_storage_routes.sql create mode 100644 src/jupiter/storage/native_snapshot_routes.rs diff --git a/src/api/router/snapshot_content_tests.rs b/src/api/router/snapshot_content_tests.rs index 1e0dcd55..add40d69 100644 --- a/src/api/router/snapshot_content_tests.rs +++ b/src/api/router/snapshot_content_tests.rs @@ -1346,3 +1346,9 @@ async fn mst2_fixed_metadata_walker_rejects_real_tree_gitlinks_without_body_read )); fixture.counts.assert(0, 0); } + +#[path = "snapshot_storage_route_tests.rs"] +mod storage_routes; + +#[path = "snapshot_storage_route_fixture.rs"] +mod storage_route_fixture; diff --git a/src/api/router/snapshot_storage_route_fixture.rs b/src/api/router/snapshot_storage_route_fixture.rs new file mode 100644 index 00000000..751502b1 --- /dev/null +++ b/src/api/router/snapshot_storage_route_fixture.rs @@ -0,0 +1,165 @@ +use sea_orm::{ConnectionTrait, DatabaseConnection, DbBackend, Statement, TransactionTrait}; + +async fn trigger_modes(db: &C) -> String { + db.query_one_raw(Statement::from_string( + DbBackend::Postgres, + "SELECT coalesce(string_agg(c.relname||':'||t.tgname||':'||t.tgenabled::text,',' + ORDER BY c.relname,t.tgname),'') AS modes FROM pg_trigger t + JOIN pg_class c ON c.oid=t.tgrelid JOIN pg_namespace n ON n.oid=c.relnamespace + WHERE n.nspname=current_schema() AND NOT t.tgisinternal", + )) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap() +} + +pub(super) async fn restore_pre_route_schema(db: &DatabaseConnection) { + let txn = db.begin().await.unwrap(); + txn.execute_unprepared("SELECT mst2_route_enter(current_schema())") + .await + .unwrap(); + let untouched = txn.query_one_raw(Statement::from_string(DbBackend::Postgres, + "SELECT coalesce(string_agg(c.relname||':'||t.tgname||':'||t.tgenabled::text,',' ORDER BY c.relname,t.tgname),'') + FROM pg_trigger t JOIN pg_class c ON c.oid=t.tgrelid + JOIN pg_namespace n ON n.oid=c.relnamespace JOIN pg_proc p ON p.oid=t.tgfoid + WHERE n.nspname=current_schema() AND NOT t.tgisinternal AND left(p.proname,11)<>'mst2_route_'", + )).await.unwrap().unwrap().try_get_by_index::(0).unwrap(); + // Remove only this additive layer in an isolated deployment-upgrade fixture. + // The original contexts, leases, plans, payloads and protection stay intact. + txn.execute_unprepared( + "DO $$ DECLARE r record; functions text; BEGIN + FOR r IN SELECT c.relname,t.tgname FROM pg_trigger t + JOIN pg_class c ON c.oid=t.tgrelid JOIN pg_namespace n ON n.oid=c.relnamespace + JOIN pg_proc p ON p.oid=t.tgfoid + WHERE n.nspname=current_schema() AND NOT t.tgisinternal AND left(p.proname,11)='mst2_route_' + LOOP EXECUTE format('DROP TRIGGER %I ON %I.%I',r.tgname,current_schema(),r.relname); END LOOP; + DROP TABLE mst2_lease_storage_route,mst2_generic_session_storage_binding,mst2_snapshot_storage_route,mst2_metadata_namespace; + FOR r IN SELECT conname FROM pg_constraint + WHERE conrelid='mst2_snapshot_context'::regclass AND contype='u' + AND pg_get_constraintdef(oid)='UNIQUE (snapshot_id, prepare_id, metadata_root)' + LOOP EXECUTE format('ALTER TABLE mst2_snapshot_context DROP CONSTRAINT %I',r.conname); END LOOP; + SELECT string_agg(format('%I.%I(%s)',n.nspname,p.proname,pg_get_function_identity_arguments(p.oid)),',') + INTO functions FROM pg_proc p JOIN pg_namespace n ON n.oid=p.pronamespace + WHERE n.nspname=current_schema() AND left(p.proname,11)='mst2_route_'; + IF functions IS NULL THEN RAISE EXCEPTION 'route upgrade fixture has no route functions'; END IF; + EXECUTE 'DROP FUNCTION '||functions; + END $$; + DELETE FROM seaql_migrations WHERE version='m20261007_000600_add_mst2_storage_routes'", + ) + .await + .unwrap(); + assert_eq!(trigger_modes(&txn).await, untouched); + txn.commit().await.unwrap(); +} + +pub(super) async fn reject_half_lease_commit_for_test( + db: &DatabaseConnection, + lease_id: &str, + insert_sql: &str, +) { + let before = trigger_modes(db).await; + let txn = db.begin().await.unwrap(); + txn.execute_unprepared("SELECT mst2_route_enter(current_schema())") + .await + .unwrap(); + // Simulate an omitted derived write; all statement, identity and deferred + // completeness guards remain active, and the deriving trigger is restored + // before the actual commit attempt. + txn.execute_unprepared( + "ALTER TABLE mst2_snapshot_lease DISABLE TRIGGER mst2_route_lease_insert", + ) + .await + .unwrap(); + assert_eq!( + txn.execute_unprepared(insert_sql) + .await + .unwrap() + .rows_affected(), + 1 + ); + let half = txn + .query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT (SELECT count(*) FROM mst2_snapshot_lease WHERE lease_id=$1)::bigint AS actual, + (SELECT count(*) FROM mst2_lease_storage_route WHERE lease_id=$1)::bigint AS routes", + [lease_id.into()], + )) + .await + .unwrap() + .unwrap(); + assert_eq!(half.try_get::("", "actual").unwrap(), 1); + assert_eq!(half.try_get::("", "routes").unwrap(), 0); + txn.execute_unprepared( + "ALTER TABLE mst2_snapshot_lease ENABLE TRIGGER mst2_route_lease_insert", + ) + .await + .unwrap(); + assert_eq!(trigger_modes(&txn).await, before); + let rejected = txn.commit().await.unwrap_err(); + assert!( + rejected + .to_string() + .contains("storage route lease committed without exact routing"), + "{rejected}" + ); + assert_eq!(trigger_modes(db).await, before); +} + +pub(super) async fn overwrite_lease_route_incarnation_for_test( + db: &DatabaseConnection, + lease_id: &str, +) { + let sql = "UPDATE mst2_lease_storage_route SET session_incarnation=$2::uuid WHERE lease_id=$1"; + let incarnation = uuid::Uuid::new_v4().to_string(); + let mutation = || { + Statement::from_sql_and_values( + DbBackend::Postgres, + sql, + [lease_id.into(), incarnation.clone().into()], + ) + }; + assert!( + db.execute_raw(mutation()) + .await + .unwrap_err() + .to_string() + .contains("immutable") + ); + let before = trigger_modes(db).await; + let txn = db.begin().await.unwrap(); + txn.execute_unprepared("SELECT mst2_route_enter(current_schema())") + .await + .unwrap(); + txn.execute_unprepared( + "ALTER TABLE mst2_lease_storage_route DISABLE TRIGGER mst2_route_immutable", + ) + .await + .unwrap(); + assert_eq!( + txn.execute_raw(mutation()).await.unwrap().rows_affected(), + 1 + ); + txn.execute_unprepared( + "ALTER TABLE mst2_lease_storage_route ENABLE TRIGGER mst2_route_immutable", + ) + .await + .unwrap(); + assert_eq!(trigger_modes(&txn).await, before); + txn.commit().await.unwrap(); + assert_eq!(trigger_modes(db).await, before); + let forbidden = uuid::Uuid::new_v4().to_string(); + assert_ne!(forbidden, incarnation); + assert!( + db.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + sql, + [lease_id.into(), forbidden.into()], + )) + .await + .unwrap_err() + .to_string() + .contains("immutable") + ); +} diff --git a/src/api/router/snapshot_storage_route_tests.rs b/src/api/router/snapshot_storage_route_tests.rs new file mode 100644 index 00000000..bd8df6f9 --- /dev/null +++ b/src/api/router/snapshot_storage_route_tests.rs @@ -0,0 +1,1183 @@ +use mst2_codec::{ + descriptor::ServingDescriptor, + metapage::{Entry, EntryKind, Page, page_id}, +}; +use sea_orm::{ + ConnectionTrait, DatabaseConnection, DatabaseTransaction, DbBackend, IsolationLevel, Statement, + TransactionTrait, +}; +use sea_orm_migration::MigratorTrait; + +use super::*; +use crate::{ + ceres::snapshot::{ + pages::PreparedNativeMetadataRetention, + retention::RetentionRoot, + retention_dag::{MetadataDagBuilder, MetadataDagLimits}, + }, + jupiter::{ + migration::Migrator, + storage::{ + mst2_retention::{PostgresRetentionRepository, RETENTION_LOCK_KEY}, + native_metadata_install::generations::qualified::PostgresQualifiedMetadataRepository, + push_queue_storage::MONO_WRITE_LOCK_KEY1, + }, + }, +}; + +const ROUTE_LOCK_KEY: i32 = 1_296_718_001; +const ROUTE_TABLES: [&str; 4] = [ + "mst2_metadata_namespace", + "mst2_snapshot_storage_route", + "mst2_generic_session_storage_binding", + "mst2_lease_storage_route", +]; + +fn statement(sql: &str, values: impl IntoIterator) -> Statement { + Statement::from_sql_and_values(DbBackend::Postgres, sql, values) +} + +async fn scalar(db: &C, sql: &str) -> i64 { + db.query_one_raw(Statement::from_string(DbBackend::Postgres, sql)) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap() +} + +async fn json_sql(db: &C, sql: &str) -> Value { + let value: String = db + .query_one_raw(Statement::from_string(DbBackend::Postgres, sql)) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap(); + serde_json::from_str(&value).unwrap() +} + +async fn routes(db: &C) -> Value { + json_sql( + db, + "SELECT jsonb_build_object( + 'namespace',(SELECT jsonb_agg(to_jsonb(n) ORDER BY namespace_uuid) FROM mst2_metadata_namespace n), + 'snapshot',(SELECT jsonb_agg(to_jsonb(r) ORDER BY snapshot_id) FROM mst2_snapshot_storage_route r), + 'binding',(SELECT jsonb_agg(to_jsonb(b) ORDER BY snapshot_id) FROM mst2_generic_session_storage_binding b), + 'lease',(SELECT jsonb_agg(to_jsonb(l) ORDER BY lease_id) FROM mst2_lease_storage_route l))::text", + ) + .await +} + +async fn sources(db: &C) -> Value { + json_sql( + db, + "SELECT jsonb_build_object( + 'context',(SELECT jsonb_agg(to_jsonb(s) ORDER BY snapshot_id) FROM mst2_snapshot_context s), + 'lease',(SELECT jsonb_agg(to_jsonb(l) ORDER BY lease_id) FROM mst2_snapshot_lease l), + 'prepare',(SELECT jsonb_agg(to_jsonb(p) ORDER BY prepare_id) FROM mst2_metadata_prepare p), + 'member',(SELECT jsonb_agg(to_jsonb(m) ORDER BY prepare_id,page_id) FROM mst2_metadata_prepare_page m), + 'payload',(SELECT jsonb_agg(to_jsonb(p) ORDER BY page_id) FROM mst2_metadata_payload p), + 'root',(SELECT jsonb_agg(to_jsonb(r) ORDER BY root_key,node_id) FROM mst2_retention_root r))::text", + ) + .await +} + +async fn roots(db: &C) -> Value { + json_sql( + db, + "SELECT coalesce(jsonb_agg(to_jsonb(r) ORDER BY root_key,node_id),'[]'::jsonb)::text FROM mst2_retention_root r", + ) + .await +} + +async fn domain_boundary_digests(db: &DatabaseConnection) -> Value { + let mut inventory = serde_json::Map::new(); + for (table, keys) in [ + ("mst2_snapshot_context", "r.snapshot_id"), + ("mst2_snapshot_lease", "r.lease_id"), + ("mst2_metadata_prepare", "r.prepare_id"), + ("mst2_metadata_prepare_page", "r.prepare_id,r.page_id"), + ("mst2_metadata_payload", "r.page_id"), + ("mst2_metadata_lifetime", "r.page_id,r.generation"), + ("mst2_metadata_current", "r.page_id"), + ("mst2_metadata_graph_node", "r.page_id,r.generation"), + ( + "mst2_metadata_graph_edge", + "r.parent_page,r.parent_generation,r.child_page,r.child_generation", + ), + ( + "mst2_metadata_graph_root", + "r.prepare_id,r.page_id,r.generation", + ), + ("mst2_metadata_gc_op", "r.operation_id"), + ("mst2_retention_node", "r.node_id"), + ("mst2_retention_edge", "r.parent_id,r.child_id"), + ("mst2_retention_root", "r.node_id,r.root_key"), + ("mst2_retention_gc_op", "r.operation_id"), + ("mst2_metadata_install_seal", "r.prepare_id"), + ("mst2_metadata_storage_scope", "r.singleton"), + ("mst2_metadata_namespace", "r.namespace_uuid"), + ("mst2_snapshot_storage_route", "r.snapshot_id"), + ( + "mst2_generic_session_storage_binding", + "r.session_incarnation", + ), + ("mst2_lease_storage_route", "r.lease_id"), + ] { + // Keep plans, binding bytes and payloads in PostgreSQL; only their + // keyed row digests leave the database for the rollback oracle. + inventory.insert( + table.to_owned(), + json_sql( + db, + &format!( + "SELECT coalesce(jsonb_agg(jsonb_build_object('key',jsonb_build_array({keys}), + 'sha256',encode(sha256(convert_to(to_jsonb(r)::text,'UTF8')),'hex')) + ORDER BY {keys}),'[]'::jsonb)::text FROM {table} r", + ), + ) + .await, + ); + } + Value::Object(inventory) +} + +fn resolve_request() -> Request { + Request::builder() + .method("POST") + .uri("/api/v2/snapshots/resolve") + .header("authorization", format!("Bearer {TOKEN}")) + .header("content-type", "application/json") + .body(Body::from( + json!({"target":{"kind":"latest"},"scope":"/project"}).to_string(), + )) + .unwrap() +} + +async fn lease_control(fixture: &Fixture, lease: &str, renew: bool) -> Response { + fixture + .app + .clone() + .oneshot( + Request::builder() + .method(if renew { "POST" } else { "DELETE" }) + .uri(format!( + "/api/v2/snapshots/leases/{lease}{}", + if renew { "/renew" } else { "" } + )) + .header("authorization", format!("Bearer {TOKEN}")) + .body(Body::empty()) + .unwrap(), + ) + .await + .unwrap() +} + +async fn rebuilt(fixture: &Fixture) -> Router { + let state = rebuilt_state(fixture).await; + Router::new().nest("/api/v2", routers(state.clone()).with_state(state)) +} + +async fn rebuilt_state(fixture: &Fixture) -> MonoApiServiceState { + let config = fixture.state.storage.config(); + let mut database = config.database.clone(); + database.max_connection = 1; + database.min_connection = 1; + let connection = crate::jupiter::storage::init::postgres_connection(&database) + .await + .unwrap(); + let storage = crate::jupiter::storage::Storage::new_with_connection( + config, + Arc::new(connection), + fixture.state.storage.git_service.obj_storage.clone(), + ) + .await + .unwrap(); + MonoApiServiceState { + storage, + git_object_cache: Arc::new(GitObjectCache { + connection: fixture.state.git_object_cache.connection.clone(), + prefix: uuid::Uuid::new_v4().to_string(), + }), + ..fixture.state.clone() + } +} + +async fn assert_exact_bindings(db: &C, leases: i64) { + assert_eq!( + scalar(db, "SELECT count(*) FROM mst2_metadata_namespace").await, + 1 + ); + assert_eq!( + scalar(db, "SELECT count(*) FROM mst2_snapshot_storage_route").await, + 1 + ); + assert_eq!( + scalar( + db, + "SELECT count(*) FROM mst2_generic_session_storage_binding" + ) + .await, + 1 + ); + assert_eq!( + scalar(db, "SELECT count(*) FROM mst2_lease_storage_route").await, + leases + ); + assert_eq!( + scalar( + db, + "SELECT count(*) FROM mst2_snapshot_context s + JOIN mst2_metadata_prepare p ON p.prepare_id=s.prepare_id + JOIN mst2_snapshot_storage_route r ON r.snapshot_id=s.snapshot_id + JOIN mst2_generic_session_storage_binding b ON b.snapshot_id=s.snapshot_id + JOIN mst2_metadata_namespace n ON n.namespace_uuid=r.namespace_uuid + WHERE r.canonical_descriptor=s.canonical_descriptor + AND r.instance_id=s.instance_id AND r.commit_oid=s.commit_oid + AND r.root_tree_oid=s.root_tree_oid AND r.metadata_root=s.metadata_root + AND b.namespace_uuid=r.namespace_uuid AND b.prepare_id=s.prepare_id + AND b.metadata_root=s.metadata_root AND p.metadata_root=s.metadata_root + AND r.source_profile=jsonb_build_object('source_domain',p.source_domain, + 'tagged_root_tree_oid',p.tagged_root_tree_oid,'scope',p.scope, + 'schema_version',p.schema_version,'metadata_codec',p.metadata_codec, + 'materialization_policy',p.materialization_policy,'fs_semantics',p.fs_semantics, + 'access_projection',p.access_projection,'verification_revision',p.verification_revision, + 'projection_revision',p.projection_revision)", + ) + .await, + 1, + ); + assert_eq!( + scalar( + db, + "SELECT count(*) FROM mst2_snapshot_lease l + JOIN mst2_lease_storage_route r ON r.lease_id=l.lease_id + JOIN mst2_generic_session_storage_binding b ON b.snapshot_id=l.snapshot_id + WHERE r.snapshot_id=l.snapshot_id AND r.namespace_uuid=b.namespace_uuid + AND r.session_incarnation=b.session_incarnation AND r.prepare_id=b.prepare_id + AND r.metadata_root=b.metadata_root AND r.authorization_epoch=l.authorization_epoch + AND r.publication_sequence=l.publication_sequence AND r.writer_epoch=l.writer_epoch + AND r.certificate_receipt_id=l.certificate_receipt_id", + ) + .await, + leases, + ); + let stored = routes(db).await; + uuid::Uuid::parse_str( + stored["binding"][0]["session_incarnation"] + .as_str() + .unwrap(), + ) + .unwrap(); + let namespace = &stored["namespace"][0]; + assert_eq!(namespace["graph_domain"], "generic-v1"); + assert_eq!(namespace["family_identity"], "v3-generic-session-1"); + assert_eq!(namespace["admission_state"], "G_ADMITTED_Q_CLOSED"); + assert_eq!(namespace["collector_state"], "CLOSED"); + assert_eq!(scalar(db, + "SELECT count(*) FROM mst2_metadata_namespace n JOIN mst2_metadata_storage_scope s ON s.singleton=1 + JOIN pg_namespace c ON c.nspname=current_schema() JOIN pg_database d ON d.datname=current_database() + WHERE n.core_schema=c.nspname AND n.core_schema_oid=c.oid + AND n.metadata_schema=c.nspname AND n.metadata_schema_oid=c.oid + AND n.database_name=d.datname AND n.database_oid=d.oid AND n.storage_uuid=s.storage_uuid + AND n.mono_lock_key2=hashtext(current_schema()) + AND n.server_address IS NOT DISTINCT FROM inet_server_addr()::text + AND n.server_port IS NOT DISTINCT FROM inet_server_port()", + ).await, 1); + assert_eq!( + scalar( + db, + "SELECT count(*) FROM information_schema.columns + WHERE table_schema=current_schema() AND table_name='mst2_snapshot_storage_route' + AND column_name IN ('prepare_id','generation','root_generation','session_incarnation')", + ) + .await, + 0, + "the permanent SID route must not freeze a physical incarnation", + ); + assert_eq!( + scalar( + db, + "SELECT count(*) FROM pg_constraint WHERE conrelid='mst2_lease_storage_route'::regclass + AND contype='f' AND confrelid='mst2_generic_session_storage_binding'::regclass", + ) + .await, + 0, + "future namespace-specific incarnations must not have a universal G ledger FK" + ); +} + +async fn rejected(db: &DatabaseConnection, sql: &str) { + let txn = db.begin().await.unwrap(); + match txn.execute_unprepared(sql).await { + Ok(_) => { + assert!(txn.commit().await.is_err(), "mutation committed: {sql}"); + } + Err(error) => { + assert!( + error.to_string().contains("storage route"), + "wrong rejection for {sql}: {error}" + ); + txn.rollback().await.unwrap(); + } + } +} + +#[tokio::test] +async fn mst2_generic_storage_routes_actual_resolve_warm_and_fresh_service_keep_exact_tuple() { + let fixture = Fixture::new_with_pg_config(true).await; + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + assert_exact_bindings(db, 1).await; + let initial = routes(db).await; + assert_eq!(initial["snapshot"][0]["snapshot_id"], fixture.snapshot); + assert_eq!(initial["lease"][0]["lease_id"], fixture.lease); + let original = success_json(fixture.send("GET", "descriptor", Body::empty()).await).await; + let warm = success_json( + fixture + .app + .clone() + .oneshot(resolve_request()) + .await + .unwrap(), + ) + .await; + assert_eq!(warm["descriptor"]["snapshot_id"], fixture.snapshot); + assert_ne!(warm["lease_id"], fixture.lease); + assert_exact_bindings(db, 2).await; + let after = routes(db).await; + for key in ["namespace", "snapshot", "binding"] { + assert_eq!(after[key], initial[key], "warm resolve changed {key}"); + } + let cold = rebuilt(&fixture).await; + assert_eq!( + success_json( + cold.clone() + .oneshot(fixture.request("GET", "descriptor", Body::empty())) + .await + .unwrap() + ) + .await, + original, + ); + assert_eq!( + cold.oneshot(fixture.request("HEAD", "blob?path=/file", Body::empty())) + .await + .unwrap() + .status(), + 200, + ); + assert_eq!(routes(db).await, after); + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn mst2_generic_storage_routes_every_member_delete_and_truncate_are_immutable() { + let fixture = Fixture::new_with_pg_config(true).await; + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + let original = routes(db).await; + let source = sources(db).await; + for table in ROUTE_TABLES { + let columns = db + .query_all_raw(statement( + "SELECT column_name,udt_name FROM information_schema.columns + WHERE table_schema=current_schema() AND table_name=$1 ORDER BY ordinal_position", + [table.into()], + )) + .await + .unwrap(); + assert!(!columns.is_empty(), "missing route relation: {table}"); + for column in columns { + let name: String = column.try_get("", "column_name").unwrap(); + let kind: String = column.try_get("", "udt_name").unwrap(); + let quoted = format!("\"{}\"", name.replace('"', "\"\"")); + let mutation = match kind.as_str() { + "uuid" => format!("'{}'::uuid", uuid::Uuid::new_v4()), + "text" | "varchar" | "bpchar" => format!("coalesce({quoted},'')||'-mutated'"), + "int2" | "int4" | "int8" => format!("coalesce({quoted},0)+1"), + "oid" => format!("({quoted}::bigint+1)::oid"), + "bytea" => format!("set_byte({quoted},0,(get_byte({quoted},0)+1)%256)"), + "jsonb" => format!("{quoted}||jsonb_build_object('_mutation',true)"), + "timestamptz" => format!("{quoted}+interval '1 second'"), + "bool" => format!("NOT {quoted}"), + other => panic!("uncovered immutable route member {table}.{name}: {other}"), + }; + rejected(db, &format!("UPDATE {table} SET {quoted}={mutation}")).await; + assert_eq!(routes(db).await, original, "{table}.{name}"); + } + rejected(db, &format!("DELETE FROM {table}")).await; + rejected(db, &format!("TRUNCATE {table} CASCADE")).await; + assert_eq!(routes(db).await, original, "{table}"); + } + for table in ["mst2_snapshot_context", "mst2_snapshot_lease"] { + rejected(db, &format!("DELETE FROM {table}")).await; + rejected(db, &format!("TRUNCATE {table} CASCADE")).await; + } + assert_eq!(sources(db).await, source); + fixture.counts.assert(0, 0); +} + +fn lease_insert(lease: &str) -> String { + format!( + "INSERT INTO mst2_snapshot_lease(lease_id,snapshot_id,authorization_epoch, + publication_sequence,writer_epoch,certificate_receipt_id,expires_at_unix,state) + SELECT '{lease}',snapshot_id,authorization_epoch,publication_sequence,writer_epoch, + certificate_receipt_id,floor(extract(epoch FROM clock_timestamp()))::bigint+3600,'ACTIVE' + FROM mst2_snapshot_context", + ) +} + +#[tokio::test] +async fn mst2_generic_storage_routes_raw_lease_derives_atomically_and_half_rows_roll_back() { + let fixture = Fixture::new_with_pg_config(true).await; + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + let original = routes(db).await; + let source = sources(db).await; + let lease = uuid::Uuid::new_v4().to_string(); + let txn = db.begin().await.unwrap(); + assert_eq!( + txn.execute_unprepared(&lease_insert(&lease)) + .await + .unwrap() + .rows_affected(), + 1 + ); + assert_exact_bindings(&txn, 2).await; + txn.rollback().await.unwrap(); + assert_eq!(routes(db).await, original); + assert_eq!(sources(db).await, source); + super::storage_route_fixture::reject_half_lease_commit_for_test( + db, + &lease, + &lease_insert(&lease), + ) + .await; + assert_eq!(routes(db).await, original); + assert_eq!(sources(db).await, source); + for table in [ + "mst2_snapshot_storage_route", + "mst2_generic_session_storage_binding", + ] { + assert_eq!(routes(db).await, original); + rejected( + db, + &format!( + "INSERT INTO {table} SELECT (jsonb_populate_record(NULL::{table}, + to_jsonb(r)||jsonb_build_object('snapshot_id','sha256:{}'))).* FROM {table} r", + hex::encode([7; 32]), + ), + ) + .await; + } + rejected(db, &format!( + "INSERT INTO mst2_lease_storage_route + SELECT (jsonb_populate_record(NULL::mst2_lease_storage_route, + to_jsonb(r)||jsonb_build_object('lease_id','{lease}'))).* FROM mst2_lease_storage_route r", + )).await; + rejected(db, + "INSERT INTO mst2_metadata_namespace SELECT (jsonb_populate_record(NULL::mst2_metadata_namespace, + to_jsonb(n)||jsonb_build_object('namespace_uuid','00000000-0000-4000-8000-000000000001', + 'graph_domain','qualified-v1'))).* FROM mst2_metadata_namespace n", + ).await; + assert_eq!(routes(db).await, original); + assert_eq!(sources(db).await, source); +} + +#[tokio::test] +async fn mst2_generic_storage_routes_renew_terminal_and_unknown_release_preserve_route_and_roots() { + let fixture = Fixture::new_with_pg_config(true).await; + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + let original = routes(db).await; + let unknown = uuid::Uuid::new_v4().to_string(); + let root: Vec = db + .query_one_raw(Statement::from_string( + DbBackend::Postgres, + "SELECT metadata_root FROM mst2_snapshot_context", + )) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap(); + let txn = db.begin().await.unwrap(); + PostgresRetentionRepository::acquire_existing_roots_in_txn( + &txn, + &format!("page:sha256:{}", hex::encode(root)), + &[RetentionRoot::Lease(unknown.clone())], + ) + .await + .unwrap(); + txn.commit().await.unwrap(); + let protected = roots(db).await; + assert_eq!( + success_json(lease_control(&fixture, &unknown, false).await).await["released"], + false + ); + assert_eq!( + roots(db).await, + protected, + "unknown release must not remove an unrelated protection root" + ); + assert_eq!(routes(db).await, original); + let renewed = success_json(lease_control(&fixture, &fixture.lease, true).await).await; + assert_eq!(renewed["lease_id"], fixture.lease); + assert_eq!(renewed["snapshot_id"], fixture.snapshot); + assert_eq!(routes(db).await, original); + assert_eq!( + success_json(lease_control(&fixture, &fixture.lease, false).await).await["released"], + true + ); + let terminal_roots = roots(db).await; + assert_eq!(routes(db).await, original); + assert_eq!( + success_json(lease_control(&fixture, &fixture.lease, false).await).await["released"], + false + ); + assert_eq!(roots(db).await, terminal_roots); + error( + lease_control(&fixture, &fixture.lease, true).await, + 410, + "LEASE_EXPIRED", + false, + ) + .await; + error( + fixture.send("GET", "descriptor", Body::empty()).await, + 410, + "LEASE_EXPIRED", + false, + ) + .await; + assert_eq!( + fixture + .send("HEAD", "blob?path=/file", Body::empty()) + .await + .status(), + 410 + ); + assert_eq!(routes(db).await, original); + assert_eq!(roots(db).await, terminal_roots); + assert_exact_bindings(db, 1).await; + fixture.counts.assert(0, 0); +} + +async fn wait_for_advisory_waiter(held: &DatabaseTransaction, key: i32) -> i64 { + tokio::time::timeout(Duration::from_secs(30), async { + loop { + held.execute_unprepared("SELECT pg_stat_clear_snapshot()").await.unwrap(); + let row = held.query_one_raw(statement( + "SELECT l.pid::bigint AS pid FROM pg_locks l JOIN pg_stat_activity a ON a.pid=l.pid + WHERE l.locktype='advisory' AND l.classid=$1::bigint::oid AND l.objid=hashtext(current_schema())::oid + AND l.objsubid=2 AND NOT l.granted AND a.datname=current_database() + AND a.application_name=current_schema() LIMIT 1", + [i64::from(key).into()], + )).await.unwrap(); + if let Some(row) = row { break row.try_get("", "pid").unwrap(); } + tokio::task::yield_now().await; + } + }).await.expect("raw lease statement did not wait on the expected advisory barrier") +} + +async fn lock_count(held: &DatabaseTransaction, pid: i64, key: i32) -> i64 { + held.query_one_raw(statement( + "SELECT count(*)::bigint FROM pg_locks WHERE pid::bigint=$1 AND locktype='advisory' + AND classid=$2::bigint::oid AND objid=hashtext(current_schema())::oid AND objsubid=2 AND granted", + [pid.into(), i64::from(key).into()], + )).await.unwrap().unwrap().try_get_by_index(0).unwrap() +} + +#[tokio::test] +async fn mst2_generic_storage_routes_raw_statement_waits_mono_then_route_then_retention() { + let fixture = Fixture::new_with_pg_config(true).await; + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + for held_key in [MONO_WRITE_LOCK_KEY1, ROUTE_LOCK_KEY, RETENTION_LOCK_KEY] { + let held = db.begin().await.unwrap(); + held.execute_raw(statement( + "SELECT pg_advisory_xact_lock($1,hashtext(current_schema()))", + [held_key.into()], + )) + .await + .unwrap(); + let writing = { + let db = db.clone(); + let lease = fixture.lease.clone(); + tokio::spawn(async move { + let txn = db.begin().await.unwrap(); + let result = txn.execute_unprepared(&format!( + "{} ON CONFLICT(lease_id) DO UPDATE SET expires_at_unix=EXCLUDED.expires_at_unix", + lease_insert(&lease), + )).await; + txn.rollback().await.unwrap(); + result + }) + }; + let pid = wait_for_advisory_waiter(&held, held_key).await; + assert_eq!( + lock_count(&held, pid, MONO_WRITE_LOCK_KEY1).await, + if held_key == MONO_WRITE_LOCK_KEY1 { + 0 + } else { + 1 + } + ); + assert_eq!( + lock_count(&held, pid, ROUTE_LOCK_KEY).await, + if held_key == RETENTION_LOCK_KEY { 1 } else { 0 } + ); + assert_eq!(lock_count(&held, pid, RETENTION_LOCK_KEY).await, 0); + assert!( + held.query_one_raw(statement( + "SELECT lease_id FROM mst2_snapshot_lease WHERE lease_id=$1 FOR UPDATE NOWAIT", + [fixture.lease.clone().into()], + )) + .await + .unwrap() + .is_some(), + "the blocked statement locked its existing lease before the barrier" + ); + assert_eq!( + held.query_one_raw(statement( + "SELECT count(*)::bigint FROM pg_locks WHERE pid::bigint=$1 AND locktype='tuple' + AND relation IN ('mst2_snapshot_context'::regclass,'mst2_snapshot_lease'::regclass)", + [pid.into()], + )) + .await + .unwrap() + .unwrap() + .try_get_by_index::(0) + .unwrap(), + 0 + ); + held.rollback().await.unwrap(); + assert_eq!(writing.await.unwrap().unwrap().rows_affected(), 1); + assert_exact_bindings(db, 1).await; + } + for (key, message) in [ + ( + RETENTION_LOCK_KEY, + "cannot acquire core locks after retention", + ), + (ROUTE_LOCK_KEY, "lock was acquired before core mono"), + ] { + for lock in ["pg_advisory_xact_lock", "pg_advisory_xact_lock_shared"] { + let held = db.begin().await.unwrap(); + held.execute_raw(statement( + &format!("SELECT {lock}($1,hashtext(current_schema()))"), + [key.into()], + )) + .await + .unwrap(); + let result = tokio::time::timeout( + Duration::from_secs(5), + held.execute_unprepared(&lease_insert(&uuid::Uuid::new_v4().to_string())), + ) + .await + .expect("reverse-order statement waited instead of rejecting"); + let error = result.unwrap_err(); + assert!( + error.to_string().contains(message), + "wrong reverse-order {lock} rejection: {error}" + ); + held.rollback().await.unwrap(); + } + } + assert_exact_bindings(db, 1).await; +} + +#[tokio::test] +async fn mst2_generic_storage_routes_schema_isolation_wrong_caller_rr_and_temp_shadow_fail_closed() +{ + let fixture = Fixture::new_with_pg_config(true).await; + let alien = Fixture::new_with_pg_config(true).await; + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + let original = routes(db).await; + let other = routes(alien.state.storage.mono_storage().get_connection()).await; + assert_ne!( + original["namespace"][0]["namespace_uuid"], + other["namespace"][0]["namespace_uuid"] + ); + let transplanted = original["snapshot"][0].clone(); + let alien_db = alien.state.storage.mono_storage(); + let txn = alien_db.get_connection().begin().await.unwrap(); + let transplant = txn.execute_raw(statement( + "INSERT INTO mst2_snapshot_storage_route SELECT (jsonb_populate_record(NULL::mst2_snapshot_storage_route, + $1::jsonb||jsonb_build_object('namespace_uuid',(SELECT namespace_uuid FROM mst2_metadata_namespace)))).*", + [transplanted.to_string().into()], + )).await.unwrap_err(); + assert!(transplant.to_string().contains("actual generic context")); + txn.rollback().await.unwrap(); + let schema = fixture._schema.as_ref().unwrap().schema(); + let qualified = format!("\"{}\".mst2_route_enter", schema.replace('"', "\"\"")); + let txn = db + .begin_with_config(Some(IsolationLevel::RepeatableRead), None) + .await + .unwrap(); + assert!( + txn.execute_unprepared(&lease_insert(&uuid::Uuid::new_v4().to_string())) + .await + .is_err() + ); + txn.rollback().await.unwrap(); + let txn = db.begin().await.unwrap(); + txn.execute_unprepared("SET LOCAL search_path=pg_catalog") + .await + .unwrap(); + assert!( + txn.execute_raw(statement( + &format!("SELECT {qualified}($1)"), + ["pg_catalog".into()] + )) + .await + .is_err() + ); + txn.rollback().await.unwrap(); + let txn = db.begin().await.unwrap(); + assert!( + txn.execute_raw(statement( + &format!("SELECT {qualified}($1)"), + [alien._schema.as_ref().unwrap().schema().into()] + )) + .await + .is_err() + ); + txn.rollback().await.unwrap(); + let txn = db.begin().await.unwrap(); + txn.execute_unprepared( + "CREATE TEMP TABLE mst2_metadata_namespace(namespace_uuid uuid); + CREATE TEMP TABLE mst2_snapshot_storage_route(snapshot_id text); + CREATE TEMP TABLE mst2_generic_session_storage_binding(snapshot_id text); + CREATE TEMP TABLE mst2_lease_storage_route(lease_id text)", + ) + .await + .unwrap(); + txn.execute_raw(statement( + &format!("SELECT {qualified}($1)"), + [schema.into()], + )) + .await + .unwrap(); + assert_eq!( + scalar( + &txn, + &format!( + "SELECT count(*) FROM \"{}\".mst2_metadata_namespace", + schema.replace('"', "\"\"") + ) + ) + .await, + 1 + ); + assert_eq!( + scalar(&txn, "SELECT count(*) FROM pg_temp.mst2_metadata_namespace").await, + 0 + ); + txn.rollback().await.unwrap(); + error( + rebuilt(&alien) + .await + .oneshot(fixture.request("GET", "descriptor", Body::empty())) + .await + .unwrap(), + 410, + "LEASE_EXPIRED", + false, + ) + .await; + assert_eq!(routes(db).await, original); + assert_eq!( + routes(alien.state.storage.mono_storage().get_connection()).await, + other + ); + fixture.counts.assert(0, 0); + alien.counts.assert(0, 0); +} + +#[tokio::test] +async fn mst2_generic_storage_routes_additive_backfill_keeps_active_terminal_sources_bytes_and_roots() + { + let fixture = Fixture::new_with_pg_config(true).await; + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + let warm = success_json( + fixture + .app + .clone() + .oneshot(resolve_request()) + .await + .unwrap(), + ) + .await; + let second = warm["lease_id"].as_str().unwrap(); + assert_eq!( + success_json(lease_control(&fixture, second, false).await).await["released"], + true + ); + assert_exact_bindings(db, 2).await; + let source = sources(db).await; + let original = routes(db).await; + let descriptor = success_json(fixture.send("GET", "descriptor", Body::empty()).await).await; + super::storage_route_fixture::restore_pre_route_schema(db).await; + assert_eq!(sources(db).await, source); + Migrator::up(db, None).await.unwrap(); + assert_eq!(sources(db).await, source); + assert_exact_bindings(db, 2).await; + let restored = routes(db).await; + for key in [ + "canonical_descriptor", + "instance_id", + "commit_oid", + "root_tree_oid", + "metadata_root", + "source_profile", + ] { + assert_eq!( + restored["snapshot"][0][key], original["snapshot"][0][key], + "backfill changed {key}" + ); + } + assert_eq!( + success_json( + rebuilt(&fixture) + .await + .oneshot(fixture.request("GET", "descriptor", Body::empty()),) + .await + .unwrap() + ) + .await, + descriptor + ); + assert_eq!( + success_json(lease_control(&fixture, second, false).await).await["released"], + false + ); + assert_eq!(sources(db).await, source); + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn mst2_generic_storage_routes_actual_http_ignores_temp_source_and_ledger_shadows() { + let fixture = Fixture::new_with_pg_config(true).await; + let state = rebuilt_state(&fixture).await; + let app = Router::new().nest("/api/v2", routers(state.clone()).with_state(state.clone())); + let mono = state.storage.mono_storage(); + let db = mono.get_connection(); + let original = routes(db).await; + assert_eq!( + app.clone() + .oneshot(fixture.request("HEAD", "blob?path=/file", Body::empty())) + .await + .unwrap() + .status(), + 200 + ); + let schema = fixture + ._schema + .as_ref() + .unwrap() + .schema() + .replace('"', "\"\""); + db.execute_unprepared(&format!( + "CREATE TEMP TABLE mst2_snapshot_context(LIKE \"{schema}\".mst2_snapshot_context); + CREATE TEMP TABLE mst2_snapshot_lease(LIKE \"{schema}\".mst2_snapshot_lease); + CREATE TEMP TABLE mst2_metadata_namespace(namespace_uuid uuid); + CREATE TEMP TABLE mst2_snapshot_storage_route(snapshot_id text); + CREATE TEMP TABLE mst2_generic_session_storage_binding(snapshot_id text); + CREATE TEMP TABLE mst2_lease_storage_route(lease_id text)", + )) + .await + .unwrap(); + assert_eq!( + app.clone() + .oneshot(fixture.request("HEAD", "blob?path=/file", Body::empty())) + .await + .unwrap() + .status(), + 200 + ); + let renewed = success_json( + app.oneshot( + Request::builder() + .method("POST") + .uri(format!("/api/v2/snapshots/leases/{}/renew", fixture.lease)) + .header("authorization", format!("Bearer {TOKEN}")) + .body(Body::empty()) + .unwrap(), + ) + .await + .unwrap(), + ) + .await; + assert_eq!(renewed["lease_id"], fixture.lease); + assert_eq!(renewed["snapshot_id"], fixture.snapshot); + for table in [ + "mst2_snapshot_context", + "mst2_snapshot_lease", + "mst2_metadata_namespace", + "mst2_snapshot_storage_route", + "mst2_generic_session_storage_binding", + "mst2_lease_storage_route", + ] { + assert_eq!( + scalar(db, &format!("SELECT count(*) FROM pg_temp.{table}")).await, + 0 + ); + } + let real = fixture.state.storage.mono_storage(); + assert_eq!(routes(real.get_connection()).await, original); + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn mst2_generic_storage_routes_temp_prepare_shadow_rejects_qualified_context_and_registered_generic_rebind() + { + let fixture = Fixture::new_with_pg_config(true).await; + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + let original_routes = routes(db).await; + let unique = uuid::Uuid::new_v4().to_string(); + let entries = [Entry::file( + EntryKind::Regular, + unique.as_bytes(), + 3, + digest(unique.as_bytes()), + )]; + let child = Page::build(&entries).unwrap(); + let roots = [Entry::dir(b"qualified-route-boundary", page_id(&child))]; + let root = Page::build(&roots).unwrap(); + let mut builder = MetadataDagBuilder::new(MetadataDagLimits::default()); + builder.add_directory(&child, &entries).unwrap(); + builder.add_directory(&root, &roots).unwrap(); + let prepared = PreparedNativeMetadataRetention::test_installation( + Arc::new(builder.finish(page_id(&root)).unwrap()), + "/", + ); + let tree = "a".repeat(40); + assert_eq!(prepared.fixed_root_tree_oid(), format!("sha1:{tree}")); + let qualified = PostgresQualifiedMetadataRepository::new(db.clone()) + .await + .unwrap(); + let intent = qualified + .begin_intent(&format!("qualified-route-boundary:{unique}"), &prepared) + .await + .unwrap(); + qualified + .install_pages(&intent, prepared.dag().payloads()) + .await + .unwrap(); + let receipt = qualified.finalize(&intent).await.unwrap(); + assert_eq!(receipt.metadata_root(), prepared.dag().root()); + let committed = db.query_one_raw(statement( + "SELECT state,graph_domain,metadata_root,tagged_root_tree_oid FROM mst2_metadata_prepare WHERE prepare_id=$1", + [intent.prepare_id().into()], + )).await.unwrap().unwrap(); + assert_eq!( + committed.try_get::("", "state").unwrap(), + "COMMITTED" + ); + assert_eq!( + committed.try_get::("", "graph_domain").unwrap(), + "qualified-v1" + ); + assert_eq!( + committed.try_get::>("", "metadata_root").unwrap(), + receipt.metadata_root().to_vec() + ); + assert_eq!( + committed + .try_get::("", "tagged_root_tree_oid") + .unwrap(), + format!("sha1:{tree}") + ); + assert_eq!(scalar(db, + "SELECT count(*) FROM mst2_metadata_lifetime WHERE graph_domain='qualified-v1' AND state='LIVE'", + ).await, prepared.dag().payloads().len() as i64); + assert_eq!(routes(db).await, original_routes); + let before = domain_boundary_digests(db).await; + let descriptor = ServingDescriptor { + instance_uuid: *uuid::Uuid::parse_str( + fixture + .state + .storage + .config() + .mst2 + .instance_uuid + .as_deref() + .unwrap(), + ) + .unwrap() + .as_bytes(), + namespace_view_id: digest(unique.as_bytes()), + scope: prepared.scope().to_owned(), + metadata_root: receipt.metadata_root(), + }; + let sid = format!("sha256:{}", hex::encode(descriptor.snapshot_id().unwrap())); + assert_ne!(sid, fixture.snapshot); + let schema = format!( + "\"{}\"", + fixture + ._schema + .as_ref() + .unwrap() + .schema() + .replace('"', "\"\"") + ); + let shadow = + format!("CREATE TEMP TABLE mst2_metadata_prepare(LIKE {schema}.mst2_metadata_prepare)"); + let txn = db.begin().await.unwrap(); + txn.execute_unprepared(&shadow).await.unwrap(); + txn.execute_unprepared(&format!("SET LOCAL search_path={schema},pg_catalog")) + .await + .unwrap(); + assert_eq!( + scalar(&txn, "SELECT count(*) FROM mst2_metadata_prepare").await, + 0 + ); + assert_eq!(scalar(&txn, "SELECT (to_regclass('mst2_metadata_prepare')='pg_temp.mst2_metadata_prepare'::regclass)::bigint").await, 1); + let rejected = txn.execute_raw(statement(&format!( + "INSERT INTO {schema}.mst2_snapshot_context(snapshot_id,canonical_descriptor,instance_id,commit_oid, + root_tree_oid,metadata_root,prepare_id,publication_sequence,writer_epoch,certificate_receipt_id,authorization_epoch,state) + SELECT $1,$2,instance_id,commit_oid,$3,$4,$5,publication_sequence,writer_epoch,certificate_receipt_id,authorization_epoch,'READY' + FROM {schema}.mst2_snapshot_context WHERE snapshot_id=$6", + ), [sid.into(),descriptor.encode().unwrap().into(),tree.into(),receipt.metadata_root().to_vec().into(), + intent.prepare_id().into(),fixture.snapshot.clone().into()], + )).await.unwrap_err(); + assert!( + rejected + .to_string() + .contains("storage route is not derived from its actual generic context"), + "{rejected}" + ); + txn.rollback().await.unwrap(); + assert_eq!( + domain_boundary_digests(db).await, + before, + "rejected Q adoption changed source, bytes, graph roots or route inventory" + ); + let generic = db + .query_one_raw(statement( + &format!( + "SELECT p.prepare_id,p.graph_domain FROM {schema}.mst2_metadata_prepare p + JOIN {schema}.mst2_metadata_install_seal i ON i.prepare_id=p.prepare_id + JOIN {schema}.mst2_snapshot_context s ON s.prepare_id=p.prepare_id WHERE s.snapshot_id=$1", + ), + [fixture.snapshot.clone().into()], + )) + .await + .unwrap() + .unwrap(); + let prepare_id: String = generic.try_get("", "prepare_id").unwrap(); + let domain: Option = generic.try_get("", "graph_domain").unwrap(); + assert!( + domain + .as_deref() + .is_none_or(|domain| domain == "generic-v1") + ); + let txn = db.begin().await.unwrap(); + txn.execute_unprepared(&shadow).await.unwrap(); + assert_eq!( + scalar(&txn, "SELECT count(*) FROM mst2_metadata_prepare").await, + 0 + ); + let rejected = txn.execute_raw(statement(&format!( + "UPDATE {schema}.mst2_metadata_prepare SET graph_domain='qualified-v1' WHERE prepare_id=$1", + ), [prepare_id.clone().into()])).await.unwrap_err(); + let error = rejected.to_string(); + assert!( + error.contains("registered metadata preparation identity is immutable") + || error.contains("metadata generation seal cannot be rebound"), + "{error}" + ); + txn.rollback().await.unwrap(); + let after_domain: Option = db.query_one_raw(statement(&format!( + "SELECT graph_domain FROM {schema}.mst2_metadata_prepare WHERE prepare_id=$1", + ), [prepare_id.into()])).await.unwrap().unwrap().try_get_by_index(0).unwrap(); + assert_eq!(after_domain, domain); + assert_eq!( + domain_boundary_digests(db).await, + before, + "registered G rebind changed source, bytes, graph roots or route inventory" + ); + assert_eq!(routes(db).await, original_routes); + assert_exact_bindings(db, 1).await; + assert_eq!( + fixture + .send("HEAD", "blob?path=/file", Body::empty()) + .await + .status(), + 200 + ); + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn mst2_generic_storage_routes_corrupt_original_incarnation_rejects_reads_renew_release_without_repair() + { + let fixture = Fixture::new_with_pg_config(true).await; + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + let source = sources(db).await; + let original = routes(db).await; + super::storage_route_fixture::overwrite_lease_route_incarnation_for_test(db, &fixture.lease) + .await; + let corrupted = routes(db).await; + assert_ne!(corrupted["lease"], original["lease"]); + for key in ["namespace", "snapshot", "binding"] { + assert_eq!(corrupted[key], original[key]); + } + error( + fixture.send("GET", "descriptor", Body::empty()).await, + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + assert_eq!( + fixture + .send("HEAD", "blob?path=/file", Body::empty()) + .await + .status(), + 502 + ); + error( + lease_control(&fixture, &fixture.lease, true).await, + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + error( + lease_control(&fixture, &fixture.lease, false).await, + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + error( + rebuilt(&fixture) + .await + .oneshot(fixture.request("GET", "descriptor", Body::empty())) + .await + .unwrap(), + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + assert_eq!( + routes(db).await, + corrupted, + "route corruption must not be silently repaired" + ); + assert_eq!( + sources(db).await, + source, + "failed access must not change leases, plans, bytes or roots" + ); + fixture.counts.assert(0, 0); +} diff --git a/src/jupiter/migration/m20261007_000600_add_mst2_storage_routes.rs b/src/jupiter/migration/m20261007_000600_add_mst2_storage_routes.rs new file mode 100644 index 00000000..541ce531 --- /dev/null +++ b/src/jupiter/migration/m20261007_000600_add_mst2_storage_routes.rs @@ -0,0 +1,42 @@ +//! Permanent semantic routes, with a separate existing generic session ledger. + +use sea_orm::{ConnectionTrait, DbBackend, Statement}; +use sea_orm_migration::prelude::*; + +#[derive(DeriveMigrationName)] +pub struct Migration; + +#[async_trait::async_trait] +impl MigrationTrait for Migration { + async fn up(&self, manager: &SchemaManager) -> Result<(), DbErr> { + let connection = manager.get_connection(); + let row = connection + .query_one_raw(Statement::from_string( + DbBackend::Postgres, + "SELECT current_schema() AS schema,n.oid::bigint AS oid + FROM pg_catalog.pg_namespace n WHERE n.nspname=current_schema()", + )) + .await? + .ok_or_else(|| DbErr::Custom("storage route core schema is missing".into()))?; + let schema: String = row.try_get("", "schema")?; + let oid: i64 = row.try_get("", "oid")?; + let quoted = format!("\"{}\"", schema.replace('"', "\"\"")); + let literal = format!("'{}'", schema.replace('\'', "''")); + #[cfg(test)] + let mono_key2 = format!("pg_catalog.hashtext({literal})"); + #[cfg(not(test))] + let mono_key2 = super::super::storage::push_queue_storage::MONO_WRITE_LOCK_KEY2.to_string(); + let sql = include_str!("m20261007_000600_storage_routes.sql") + .replace("$CORE_SCHEMA$", "ed) + .replace("$CORE_LITERAL$", &literal) + .replace("$CORE_OID$", &oid.to_string()) + .replace("$MONO_KEY2$", &mono_key2) + .replace("$NAMESPACE_UUID$", &uuid::Uuid::new_v4().to_string()); + connection.execute_unprepared(&sql).await?; + Ok(()) + } + + async fn down(&self, _manager: &SchemaManager) -> Result<(), DbErr> { + Ok(()) + } +} diff --git a/src/jupiter/migration/m20261007_000600_storage_routes.sql b/src/jupiter/migration/m20261007_000600_storage_routes.sql new file mode 100644 index 00000000..13e8b18c --- /dev/null +++ b/src/jupiter/migration/m20261007_000600_storage_routes.sql @@ -0,0 +1,329 @@ +DO $$ BEGIN + IF pg_catalog.pg_is_in_recovery() OR pg_catalog.current_setting('transaction_isolation')<>'read committed' + OR pg_catalog.current_schema()<>$CORE_LITERAL$ + OR NOT EXISTS(SELECT 1 FROM pg_catalog.pg_namespace WHERE oid=$CORE_OID$ AND nspname=$CORE_LITERAL$) THEN + RAISE EXCEPTION 'storage route migration requires its captured primary core schema'; + END IF; + IF EXISTS(SELECT 1 FROM pg_catalog.pg_locks WHERE locktype='advisory' AND pid=pg_catalog.pg_backend_pid() + AND classid=1296718001::oid AND objid=pg_catalog.hashtext($CORE_LITERAL$)::oid AND objsubid=2 AND granted) + AND NOT EXISTS(SELECT 1 FROM pg_catalog.pg_locks WHERE locktype='advisory' AND pid=pg_catalog.pg_backend_pid() + AND classid=1297043024::oid AND objid=($MONO_KEY2$)::oid AND objsubid=2 AND granted AND mode='ExclusiveLock') THEN + RAISE EXCEPTION 'storage route migration cannot acquire mono after route'; + END IF; + IF EXISTS(SELECT 1 FROM pg_catalog.pg_locks WHERE locktype='advisory' AND pid=pg_catalog.pg_backend_pid() + AND database=(SELECT oid FROM pg_catalog.pg_database WHERE datname=pg_catalog.current_database()) + AND classid=1296717362::oid AND objid=pg_catalog.hashtext($CORE_LITERAL$)::oid AND objsubid=2 AND granted) + AND NOT (EXISTS(SELECT 1 FROM pg_catalog.pg_locks WHERE locktype='advisory' AND pid=pg_catalog.pg_backend_pid() + AND classid=1297043024::oid AND objid=($MONO_KEY2$)::oid AND objsubid=2 AND granted AND mode='ExclusiveLock') + AND EXISTS(SELECT 1 FROM pg_catalog.pg_locks WHERE locktype='advisory' AND pid=pg_catalog.pg_backend_pid() + AND classid=1296718001::oid AND objid=pg_catalog.hashtext($CORE_LITERAL$)::oid + AND objsubid=2 AND granted AND mode='ExclusiveLock')) THEN + RAISE EXCEPTION 'storage route migration cannot acquire core locks after retention'; + END IF; +END $$; +SELECT pg_catalog.set_config('lock_timeout','5000ms',true); +SELECT pg_catalog.pg_advisory_xact_lock(1297043024,$MONO_KEY2$); +SELECT pg_catalog.pg_advisory_xact_lock(1296718001,pg_catalog.hashtext($CORE_LITERAL$)); +SELECT pg_catalog.pg_advisory_xact_lock(1296717362,pg_catalog.hashtext($CORE_LITERAL$)); +SET LOCAL search_path=$CORE_SCHEMA$,pg_catalog,pg_temp; +LOCK TABLE mst2_snapshot_context,mst2_snapshot_lease,mst2_metadata_prepare, + mst2_metadata_storage_scope IN ACCESS EXCLUSIVE MODE; + +CREATE TABLE mst2_metadata_namespace ( + singleton integer PRIMARY KEY CHECK (singleton=1), + namespace_uuid uuid NOT NULL UNIQUE, + core_schema text NOT NULL, + core_schema_oid oid NOT NULL, + database_name text NOT NULL, + database_oid oid NOT NULL, + storage_uuid text NOT NULL, + server_address text, + server_port integer, + mono_lock_key2 integer NOT NULL, + metadata_schema text NOT NULL, + metadata_schema_oid oid NOT NULL, + family_identity text NOT NULL CHECK (family_identity='v3-generic-session-1'), + graph_domain text NOT NULL CHECK (graph_domain='generic-v1'), + admission_state text NOT NULL CHECK (admission_state='G_ADMITTED_Q_CLOSED'), + collector_state text NOT NULL CHECK (collector_state='CLOSED'), + CHECK (core_schema=metadata_schema AND core_schema_oid=metadata_schema_oid) +); +INSERT INTO mst2_metadata_namespace +SELECT 1,'$NAMESPACE_UUID$'::uuid,$CORE_LITERAL$,$CORE_OID$,pg_catalog.current_database(),d.oid,s.storage_uuid, + pg_catalog.inet_server_addr()::text,pg_catalog.inet_server_port(),$MONO_KEY2$,$CORE_LITERAL$,$CORE_OID$, + 'v3-generic-session-1','generic-v1','G_ADMITTED_Q_CLOSED','CLOSED' +FROM mst2_metadata_storage_scope s JOIN pg_catalog.pg_database d ON d.datname=pg_catalog.current_database() +WHERE s.singleton=1; + +CREATE TABLE mst2_snapshot_storage_route ( + snapshot_id text PRIMARY KEY, + namespace_uuid uuid NOT NULL REFERENCES mst2_metadata_namespace(namespace_uuid), + canonical_descriptor bytea NOT NULL, + instance_id text NOT NULL, + commit_oid text NOT NULL, + root_tree_oid text NOT NULL, + metadata_root bytea NOT NULL CHECK (octet_length(metadata_root)=32), + source_profile jsonb NOT NULL, + UNIQUE(snapshot_id,namespace_uuid), + CHECK (snapshot_id='sha256:'||pg_catalog.encode(pg_catalog.sha256( + pg_catalog.convert_to('mega.mst2.descriptor','UTF8')||pg_catalog.decode('00','hex')||canonical_descriptor),'hex')) +); +ALTER TABLE mst2_snapshot_context ADD UNIQUE(snapshot_id,prepare_id,metadata_root); +CREATE TABLE mst2_generic_session_storage_binding ( + session_incarnation uuid PRIMARY KEY, + snapshot_id text NOT NULL UNIQUE, + namespace_uuid uuid NOT NULL, + prepare_id text NOT NULL, + metadata_root bytea NOT NULL, + FOREIGN KEY(snapshot_id,namespace_uuid) REFERENCES mst2_snapshot_storage_route(snapshot_id,namespace_uuid), + FOREIGN KEY(snapshot_id,prepare_id,metadata_root) REFERENCES mst2_snapshot_context(snapshot_id,prepare_id,metadata_root), + UNIQUE(snapshot_id,namespace_uuid,session_incarnation,prepare_id,metadata_root) +); +CREATE TABLE mst2_lease_storage_route ( + lease_id text PRIMARY KEY, + snapshot_id text NOT NULL, + namespace_uuid uuid NOT NULL, + session_incarnation uuid NOT NULL, + prepare_id text NOT NULL, + metadata_root bytea NOT NULL CHECK (octet_length(metadata_root)=32), + authorization_epoch bigint NOT NULL, + publication_sequence bigint NOT NULL, + writer_epoch bigint NOT NULL, + certificate_receipt_id bigint NOT NULL, + FOREIGN KEY(snapshot_id,namespace_uuid) REFERENCES mst2_snapshot_storage_route(snapshot_id,namespace_uuid) +); +CREATE INDEX mst2_lease_storage_route_incarnation ON mst2_lease_storage_route(namespace_uuid,session_incarnation,lease_id); + +CREATE FUNCTION mst2_route_scope_valid(caller_schema text) RETURNS boolean LANGUAGE sql VOLATILE +SET search_path=$CORE_SCHEMA$,pg_catalog,pg_temp AS $$ + SELECT NOT pg_catalog.pg_is_in_recovery() AND pg_catalog.current_setting('transaction_isolation')='read committed' + AND EXISTS(SELECT 1 FROM mst2_metadata_namespace n JOIN mst2_metadata_storage_scope s ON s.singleton=1 + JOIN pg_catalog.pg_database d ON d.datname=pg_catalog.current_database() + JOIN pg_catalog.pg_namespace c ON c.oid=n.core_schema_oid AND c.nspname=n.core_schema + WHERE n.singleton=1 AND caller_schema=n.core_schema AND n.core_schema=$CORE_LITERAL$ AND n.core_schema_oid=$CORE_OID$ + AND n.database_name=d.datname AND n.database_oid=d.oid AND n.storage_uuid=s.storage_uuid + AND n.server_address IS NOT DISTINCT FROM pg_catalog.inet_server_addr()::text + AND n.server_port IS NOT DISTINCT FROM pg_catalog.inet_server_port() + AND n.metadata_schema=n.core_schema AND n.metadata_schema_oid=n.core_schema_oid + AND n.graph_domain='generic-v1' AND n.family_identity='v3-generic-session-1' + AND n.admission_state='G_ADMITTED_Q_CLOSED' AND n.collector_state='CLOSED') +$$; + +CREATE FUNCTION mst2_route_lock_held(k1 integer,k2 integer,exclusive_only boolean DEFAULT true) RETURNS boolean LANGUAGE sql VOLATILE +SET search_path=$CORE_SCHEMA$,pg_catalog,pg_temp AS $$ + SELECT EXISTS(SELECT 1 FROM pg_catalog.pg_locks WHERE locktype='advisory' AND pid=pg_catalog.pg_backend_pid() + AND database=(SELECT oid FROM pg_catalog.pg_database WHERE datname=pg_catalog.current_database()) + AND classid=k1::oid AND objid=k2::oid AND objsubid=2 AND granted AND (NOT exclusive_only OR mode='ExclusiveLock')) +$$; + +CREATE FUNCTION mst2_route_enter(caller_schema text) RETURNS uuid LANGUAGE plpgsql VOLATILE +SET search_path=$CORE_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE n mst2_metadata_namespace%ROWTYPE; mono_held boolean; route_held boolean; +BEGIN + IF NOT mst2_route_scope_valid(caller_schema) THEN RAISE EXCEPTION 'storage route captured primary scope is unavailable'; END IF; + SELECT * INTO STRICT n FROM mst2_metadata_namespace WHERE singleton=1; + mono_held:=mst2_route_lock_held(1297043024,n.mono_lock_key2); + route_held:=mst2_route_lock_held(1296718001,pg_catalog.hashtext(n.core_schema)); + IF mst2_route_lock_held(1296718001,pg_catalog.hashtext(n.core_schema),false) AND NOT mono_held THEN + RAISE EXCEPTION 'storage route lock was acquired before core mono'; + END IF; + IF EXISTS(SELECT 1 FROM mst2_metadata_namespace x + WHERE mst2_route_lock_held(1296717362,pg_catalog.hashtext(x.metadata_schema),false)) + AND NOT (mono_held AND route_held) THEN + RAISE EXCEPTION 'storage route cannot acquire core locks after retention'; + END IF; + PERFORM pg_catalog.set_config('lock_timeout','5000ms',true); + PERFORM pg_catalog.pg_advisory_xact_lock(1297043024,n.mono_lock_key2); + PERFORM pg_catalog.pg_advisory_xact_lock(1296718001,pg_catalog.hashtext(n.core_schema)); + FOR n IN SELECT * FROM mst2_metadata_namespace ORDER BY namespace_uuid LOOP + PERFORM pg_catalog.pg_advisory_xact_lock(1296717362,pg_catalog.hashtext(n.metadata_schema)); + END LOOP; + RETURN n.namespace_uuid; +END $$; + +-- This wrapper observes the caller before entering a fixed trusted path. It +-- uses only pg_catalog and the explicitly captured core function/relation. +CREATE FUNCTION mst2_route_statement_barrier() RETURNS trigger LANGUAGE plpgsql VOLATILE AS $$ +BEGIN + IF TG_TABLE_SCHEMA<>$CORE_LITERAL$ OR NOT EXISTS(SELECT 1 FROM pg_catalog.pg_class + WHERE oid=TG_RELID AND relnamespace=$CORE_OID$) THEN + RAISE EXCEPTION 'storage route mutation is outside its captured core schema'; + END IF; + PERFORM $CORE_SCHEMA$.mst2_route_enter(pg_catalog.current_schema()); + RETURN NULL; +END $$; + +CREATE FUNCTION mst2_route_profile(source_domain text,tagged_root_tree_oid text,scope text,schema_version smallint, + metadata_codec smallint,materialization_policy smallint,fs_semantics smallint,access_projection smallint, + verification_revision integer,projection_revision smallint) RETURNS jsonb LANGUAGE sql IMMUTABLE STRICT +SET search_path=$CORE_SCHEMA$,pg_catalog,pg_temp AS $$ + SELECT pg_catalog.jsonb_build_object('source_domain',source_domain,'tagged_root_tree_oid',tagged_root_tree_oid, + 'scope',scope,'schema_version',schema_version,'metadata_codec',metadata_codec, + 'materialization_policy',materialization_policy,'fs_semantics',fs_semantics, + 'access_projection',access_projection,'verification_revision',verification_revision, + 'projection_revision',projection_revision) +$$; + +CREATE FUNCTION mst2_route_snapshot_proof(sid text) RETURNS boolean LANGUAGE sql VOLATILE +SET search_path=$CORE_SCHEMA$,pg_catalog,pg_temp AS $$ + SELECT EXISTS(SELECT 1 FROM mst2_snapshot_storage_route r JOIN mst2_snapshot_context s USING(snapshot_id) + JOIN mst2_generic_session_storage_binding b USING(snapshot_id,namespace_uuid) + JOIN mst2_metadata_namespace n USING(namespace_uuid) + JOIN mst2_metadata_prepare p ON p.prepare_id=s.prepare_id + WHERE r.snapshot_id=sid AND r.canonical_descriptor=s.canonical_descriptor + AND r.instance_id=s.instance_id AND r.commit_oid=s.commit_oid AND r.root_tree_oid=s.root_tree_oid + AND r.metadata_root=s.metadata_root AND r.source_profile=mst2_route_profile(p.source_domain,p.tagged_root_tree_oid, + p.scope,p.schema_version,p.metadata_codec,p.materialization_policy,p.fs_semantics,p.access_projection, + p.verification_revision,p.projection_revision) + AND b.prepare_id=s.prepare_id AND b.metadata_root=s.metadata_root + AND p.state='COMMITTED' AND p.metadata_root=s.metadata_root AND p.source_domain='native-git' + AND p.tagged_root_tree_oid IN ('sha1:'||s.root_tree_oid,'sha256:'||s.root_tree_oid) + AND n.graph_domain='generic-v1') +$$; + +CREATE FUNCTION mst2_route_lease_proof(lid text) RETURNS boolean LANGUAGE sql VOLATILE +SET search_path=$CORE_SCHEMA$,pg_catalog,pg_temp AS $$ + SELECT EXISTS(SELECT 1 FROM mst2_lease_storage_route r JOIN mst2_snapshot_lease l USING(lease_id) + JOIN mst2_generic_session_storage_binding b ON b.snapshot_id=r.snapshot_id AND b.namespace_uuid=r.namespace_uuid + AND b.session_incarnation=r.session_incarnation AND b.prepare_id=r.prepare_id AND b.metadata_root=r.metadata_root + WHERE r.lease_id=lid AND r.snapshot_id=l.snapshot_id AND r.authorization_epoch=l.authorization_epoch + AND r.publication_sequence=l.publication_sequence AND r.writer_epoch=l.writer_epoch + AND r.certificate_receipt_id=l.certificate_receipt_id AND mst2_route_snapshot_proof(r.snapshot_id)) +$$; + +CREATE FUNCTION mst2_route_select_snapshot(sid text,caller_schema text) +RETURNS TABLE(context_present boolean,route_present boolean,valid boolean) LANGUAGE plpgsql VOLATILE +SET search_path=$CORE_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + IF NOT mst2_route_scope_valid(caller_schema) THEN RAISE EXCEPTION 'storage route captured primary scope is unavailable'; END IF; + RETURN QUERY SELECT EXISTS(SELECT 1 FROM mst2_snapshot_context WHERE snapshot_id=sid), + EXISTS(SELECT 1 FROM mst2_snapshot_storage_route WHERE snapshot_id=sid),mst2_route_snapshot_proof(sid); +END $$; + +CREATE FUNCTION mst2_route_select_lease(lid text,caller_schema text) +RETURNS TABLE(actual_sid text,route_present boolean,valid boolean) LANGUAGE plpgsql VOLATILE +SET search_path=$CORE_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + IF NOT mst2_route_scope_valid(caller_schema) THEN RAISE EXCEPTION 'storage route captured primary scope is unavailable'; END IF; + RETURN QUERY SELECT (SELECT snapshot_id FROM mst2_snapshot_lease WHERE lease_id=lid), + EXISTS(SELECT 1 FROM mst2_lease_storage_route WHERE lease_id=lid),mst2_route_lease_proof(lid); +END $$; + +CREATE FUNCTION mst2_route_immutable() RETURNS trigger LANGUAGE plpgsql +SET search_path=$CORE_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN RAISE EXCEPTION 'storage route identity and historical ledger are immutable'; END $$; + +CREATE FUNCTION mst2_route_insert_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$CORE_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + IF TG_TABLE_NAME='mst2_snapshot_storage_route' THEN + IF NOT EXISTS(SELECT 1 FROM mst2_snapshot_context s JOIN mst2_metadata_prepare p ON p.prepare_id=s.prepare_id + JOIN mst2_metadata_namespace n ON n.namespace_uuid=NEW.namespace_uuid + WHERE s.snapshot_id=NEW.snapshot_id AND NEW.canonical_descriptor=s.canonical_descriptor + AND NEW.instance_id=s.instance_id AND NEW.commit_oid=s.commit_oid AND NEW.root_tree_oid=s.root_tree_oid + AND NEW.metadata_root=s.metadata_root AND NEW.source_profile=mst2_route_profile(p.source_domain,p.tagged_root_tree_oid, + p.scope,p.schema_version,p.metadata_codec,p.materialization_policy,p.fs_semantics,p.access_projection, + p.verification_revision,p.projection_revision) + AND p.state='COMMITTED' AND p.metadata_root=s.metadata_root AND p.source_domain='native-git' + AND (p.graph_domain IS NULL OR p.graph_domain='generic-v1') + AND p.tagged_root_tree_oid IN ('sha1:'||s.root_tree_oid,'sha256:'||s.root_tree_oid)) THEN + RAISE EXCEPTION 'storage route is not derived from its actual generic context'; + END IF; + ELSIF TG_TABLE_NAME='mst2_generic_session_storage_binding' THEN + IF NOT EXISTS(SELECT 1 FROM mst2_snapshot_context s JOIN mst2_snapshot_storage_route r USING(snapshot_id) + WHERE s.snapshot_id=NEW.snapshot_id AND r.namespace_uuid=NEW.namespace_uuid + AND s.prepare_id=NEW.prepare_id AND s.metadata_root=NEW.metadata_root) THEN + RAISE EXCEPTION 'storage route generic incarnation is not its actual context'; + END IF; + ELSIF TG_TABLE_NAME='mst2_lease_storage_route' THEN + IF NOT EXISTS(SELECT 1 FROM mst2_snapshot_lease l JOIN mst2_generic_session_storage_binding b USING(snapshot_id) + WHERE l.lease_id=NEW.lease_id AND l.snapshot_id=NEW.snapshot_id AND b.namespace_uuid=NEW.namespace_uuid + AND b.session_incarnation=NEW.session_incarnation AND b.prepare_id=NEW.prepare_id AND b.metadata_root=NEW.metadata_root + AND l.authorization_epoch=NEW.authorization_epoch AND l.publication_sequence=NEW.publication_sequence + AND l.writer_epoch=NEW.writer_epoch AND l.certificate_receipt_id=NEW.certificate_receipt_id + AND mst2_route_snapshot_proof(l.snapshot_id)) THEN + RAISE EXCEPTION 'storage route lease is not its exact generic incarnation and source'; + END IF; + ELSE RAISE EXCEPTION 'storage route insertion target is not registered'; END IF; + RETURN NEW; +END $$; + +CREATE FUNCTION mst2_route_context_insert() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$CORE_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + INSERT INTO mst2_snapshot_storage_route + SELECT s.snapshot_id,n.namespace_uuid,s.canonical_descriptor,s.instance_id,s.commit_oid,s.root_tree_oid, + s.metadata_root,mst2_route_profile(p.source_domain,p.tagged_root_tree_oid,p.scope,p.schema_version,p.metadata_codec, + p.materialization_policy,p.fs_semantics,p.access_projection,p.verification_revision,p.projection_revision) + FROM mst2_snapshot_context s JOIN mst2_metadata_prepare p ON p.prepare_id=s.prepare_id + CROSS JOIN mst2_metadata_namespace n WHERE s.snapshot_id=NEW.snapshot_id AND n.singleton=1 + ON CONFLICT(snapshot_id) DO NOTHING; + INSERT INTO mst2_generic_session_storage_binding + SELECT pg_catalog.gen_random_uuid(),s.snapshot_id,r.namespace_uuid,s.prepare_id,s.metadata_root + FROM mst2_snapshot_context s JOIN mst2_snapshot_storage_route r USING(snapshot_id) WHERE s.snapshot_id=NEW.snapshot_id + ON CONFLICT(snapshot_id) DO NOTHING; + RETURN NULL; +END $$; +CREATE FUNCTION mst2_route_lease_insert() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$CORE_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + INSERT INTO mst2_lease_storage_route + SELECT l.lease_id,l.snapshot_id,b.namespace_uuid,b.session_incarnation,b.prepare_id,b.metadata_root, + l.authorization_epoch,l.publication_sequence,l.writer_epoch,l.certificate_receipt_id + FROM mst2_snapshot_lease l JOIN mst2_generic_session_storage_binding b USING(snapshot_id) WHERE l.lease_id=NEW.lease_id + ON CONFLICT(lease_id) DO NOTHING; + RETURN NULL; +END $$; +CREATE FUNCTION mst2_route_complete() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$CORE_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + IF TG_TABLE_NAME='mst2_snapshot_context' THEN + IF NOT mst2_route_snapshot_proof(NEW.snapshot_id) THEN RAISE EXCEPTION 'storage route context committed without exact routing'; END IF; + ELSIF NOT mst2_route_lease_proof(NEW.lease_id) THEN RAISE EXCEPTION 'storage route lease committed without exact routing'; END IF; + RETURN NULL; +END $$; + +DO $$ DECLARE t text; BEGIN + FOREACH t IN ARRAY ARRAY['mst2_metadata_namespace','mst2_snapshot_storage_route', + 'mst2_generic_session_storage_binding','mst2_lease_storage_route'] LOOP + EXECUTE pg_catalog.format('CREATE TRIGGER mst2_00_route_statement_barrier BEFORE INSERT OR UPDATE OR DELETE ON %I FOR EACH STATEMENT EXECUTE FUNCTION mst2_route_statement_barrier()',t); + EXECUTE pg_catalog.format('CREATE TRIGGER mst2_route_immutable BEFORE UPDATE OR DELETE ON %I FOR EACH ROW EXECUTE FUNCTION mst2_route_immutable()',t); + EXECUTE pg_catalog.format('CREATE TRIGGER mst2_route_truncate_guard BEFORE TRUNCATE ON %I FOR EACH STATEMENT EXECUTE FUNCTION mst2_route_immutable()',t); + IF t<>'mst2_metadata_namespace' THEN + EXECUTE pg_catalog.format('CREATE TRIGGER mst2_route_insert_guard BEFORE INSERT ON %I FOR EACH ROW EXECUTE FUNCTION mst2_route_insert_guard()',t); + END IF; + END LOOP; + FOREACH t IN ARRAY ARRAY['mst2_snapshot_context','mst2_snapshot_lease'] LOOP + EXECUTE pg_catalog.format('CREATE TRIGGER mst2_00_route_statement_barrier BEFORE INSERT ON %I FOR EACH STATEMENT EXECUTE FUNCTION mst2_route_statement_barrier()',t); + EXECUTE pg_catalog.format('CREATE TRIGGER mst2_route_source_delete_guard BEFORE DELETE ON %I FOR EACH ROW EXECUTE FUNCTION mst2_route_immutable()',t); + EXECUTE pg_catalog.format('CREATE TRIGGER mst2_route_source_truncate_guard BEFORE TRUNCATE ON %I FOR EACH STATEMENT EXECUTE FUNCTION mst2_route_immutable()',t); + EXECUTE pg_catalog.format('CREATE CONSTRAINT TRIGGER mst2_route_complete AFTER INSERT ON %I DEFERRABLE INITIALLY DEFERRED FOR EACH ROW EXECUTE FUNCTION mst2_route_complete()',t); + END LOOP; +END $$; +CREATE TRIGGER mst2_route_namespace_registration_closed BEFORE INSERT ON mst2_metadata_namespace + FOR EACH ROW EXECUTE FUNCTION mst2_route_immutable(); +CREATE TRIGGER mst2_route_context_insert AFTER INSERT ON mst2_snapshot_context + FOR EACH ROW EXECUTE FUNCTION mst2_route_context_insert(); +CREATE TRIGGER mst2_route_lease_insert AFTER INSERT ON mst2_snapshot_lease + FOR EACH ROW EXECUTE FUNCTION mst2_route_lease_insert(); + +INSERT INTO mst2_snapshot_storage_route +SELECT s.snapshot_id,n.namespace_uuid,s.canonical_descriptor,s.instance_id,s.commit_oid,s.root_tree_oid, + s.metadata_root,mst2_route_profile(p.source_domain,p.tagged_root_tree_oid,p.scope,p.schema_version,p.metadata_codec, + p.materialization_policy,p.fs_semantics,p.access_projection,p.verification_revision,p.projection_revision) +FROM mst2_snapshot_context s JOIN mst2_metadata_prepare p ON p.prepare_id=s.prepare_id +CROSS JOIN mst2_metadata_namespace n; +INSERT INTO mst2_generic_session_storage_binding +SELECT pg_catalog.gen_random_uuid(),s.snapshot_id,r.namespace_uuid,s.prepare_id,s.metadata_root +FROM mst2_snapshot_context s JOIN mst2_snapshot_storage_route r USING(snapshot_id); +INSERT INTO mst2_lease_storage_route +SELECT l.lease_id,l.snapshot_id,b.namespace_uuid,b.session_incarnation,b.prepare_id,b.metadata_root, + l.authorization_epoch,l.publication_sequence,l.writer_epoch,l.certificate_receipt_id +FROM mst2_snapshot_lease l JOIN mst2_generic_session_storage_binding b USING(snapshot_id); +DO $$ BEGIN + IF NOT mst2_route_scope_valid($CORE_LITERAL$) + OR EXISTS(SELECT 1 FROM mst2_snapshot_context s WHERE NOT mst2_route_snapshot_proof(s.snapshot_id)) + OR EXISTS(SELECT 1 FROM mst2_snapshot_lease l WHERE NOT mst2_route_lease_proof(l.lease_id)) THEN + RAISE EXCEPTION 'storage route backfill did not cover the complete durable inventory'; + END IF; +END $$; diff --git a/src/jupiter/migration/mod.rs b/src/jupiter/migration/mod.rs index fa611fce..2d667a39 100644 --- a/src/jupiter/migration/mod.rs +++ b/src/jupiter/migration/mod.rs @@ -150,6 +150,7 @@ mod m20261007_000200_add_mst2_metadata_generations; mod m20261007_000300_add_mst2_metadata_lifetime_history; mod m20261007_000400_add_mst2_qualified_metadata_gc; mod m20261007_000500_add_mst2_install_capability; +mod m20261007_000600_add_mst2_storage_routes; mod runner; pub use m20260905_000100_add_push_queue::ensure_queue_control_seed; pub use runner::apply_migrations; @@ -288,6 +289,7 @@ impl MigratorTrait for Migrator { Box::new(m20261007_000300_add_mst2_metadata_lifetime_history::Migration), Box::new(m20261007_000400_add_mst2_qualified_metadata_gc::Migration), Box::new(m20261007_000500_add_mst2_install_capability::Migration), + Box::new(m20261007_000600_add_mst2_storage_routes::Migration), ] } } @@ -1198,7 +1200,7 @@ mod tests { async fn import_repo_alias_rows_canonicalized() { let names = migration_names(); assert_eq!( - &names[names.len() - 13..names.len() - 4], + &names[names.len() - 14..names.len() - 5], &[ "m20260923_000200_canonicalize_import_repo_paths".to_string(), "m20260925_000100_media_paging".to_string(), @@ -1213,21 +1215,25 @@ mod tests { "native retention, metadata installation and view tables follow media paging" ); assert_eq!( - &names[names.len() - 4], + &names[names.len() - 5], "m20261007_000200_add_mst2_metadata_generations" ); assert_eq!( - &names[names.len() - 3], + &names[names.len() - 4], "m20261007_000300_add_mst2_metadata_lifetime_history" ); assert_eq!( - &names[names.len() - 2], + &names[names.len() - 3], "m20261007_000400_add_mst2_qualified_metadata_gc" ); assert_eq!( - names.last().unwrap(), + &names[names.len() - 2], "m20261007_000500_add_mst2_install_capability" ); + assert_eq!( + names.last().unwrap(), + "m20261007_000600_add_mst2_storage_routes" + ); let db = alias_db().await; insert_repo(&db, 1, "/third-party//a").await; diff --git a/src/jupiter/storage/native_metadata_install.rs b/src/jupiter/storage/native_metadata_install.rs index 735dcc54..6a1c8500 100644 --- a/src/jupiter/storage/native_metadata_install.rs +++ b/src/jupiter/storage/native_metadata_install.rs @@ -163,6 +163,10 @@ impl PostgresMetadataInstallRepository { }) } + pub(crate) fn captured_schema(&self) -> &str { + &self.storage_scope.schema + } + /// These columns are returned by the same query that authorizes a lease, /// so a warm cache cannot authorize a stale replica or another schema. pub(crate) fn verify_primary_scope_row(&self, row: &QueryResult) -> Result<(), SnapshotError> { diff --git a/src/jupiter/storage/native_snapshot_routes.rs b/src/jupiter/storage/native_snapshot_routes.rs new file mode 100644 index 00000000..ac818034 --- /dev/null +++ b/src/jupiter/storage/native_snapshot_routes.rs @@ -0,0 +1,122 @@ +//! Permanent namespace selection precedes the generic physical session proof. + +use super::{ + ConnectionTrait, DatabaseTransaction, PostgresMetadataInstallRepository, SESSION_SQL, + SnapshotError, integrity, internal, statement, +}; + +fn function(installer: &PostgresMetadataInstallRepository, name: &str) -> String { + format!( + "\"{}\".{name}", + installer.captured_schema().replace('"', "\"\"") + ) +} + +pub(super) async fn enter( + txn: &DatabaseTransaction, + installer: &PostgresMetadataInstallRepository, +) -> Result<(), SnapshotError> { + installer.verify_primary_connection(txn).await?; + txn.execute_raw(statement( + &format!( + "SELECT {}(current_schema())", + function(installer, "mst2_route_enter") + ), + [], + )) + .await + .map_err(internal)?; + generic_path(txn, installer).await?; + Ok(()) +} + +pub(super) async fn generic_path( + txn: &DatabaseTransaction, + installer: &PostgresMetadataInstallRepository, +) -> Result<(), SnapshotError> { + let schema = installer.captured_schema().replace('"', "\"\""); + txn.execute_unprepared(&format!( + "SET LOCAL search_path=\"{schema}\",pg_catalog,pg_temp" + )) + .await + .map_err(internal)?; + Ok(()) +} + +pub(super) fn session_sql(installer: &PostgresMetadataInstallRepository) -> String { + let schema = installer.captured_schema().replace('"', "\"\""); + let mut sql = SESSION_SQL.to_owned(); + for table in [ + "mst2_metadata_storage_scope", + "mst2_snapshot_context", + "mst2_snapshot_lease", + "mst2_retention_node", + "mst2_retention_root", + "mst2_metadata_prepare", + "mst2_native_publication", + "mst2_publication", + "mst2_publication_outbox", + ] { + for join in ["FROM", "JOIN"] { + sql = sql.replace( + &format!("{join} {table} "), + &format!("{join} \"{schema}\".{table} "), + ); + } + } + sql +} + +pub(super) async fn snapshot( + connection: &C, + installer: &PostgresMetadataInstallRepository, + sid: &str, +) -> Result<(), SnapshotError> { + let row = connection + .query_one_raw(statement( + &format!( + "SELECT * FROM {}($1,current_schema())", + function(installer, "mst2_route_select_snapshot") + ), + [sid.into()], + )) + .await + .map_err(internal)? + .ok_or_else(|| integrity("snapshot storage route selection is missing"))?; + let context: bool = row.try_get("", "context_present").map_err(internal)?; + let route: bool = row.try_get("", "route_present").map_err(internal)?; + let valid: bool = row.try_get("", "valid").map_err(internal)?; + if context != route || context && !valid { + return Err(integrity( + "snapshot storage route conflicts with its immutable generic session", + )); + } + Ok(()) +} + +pub(super) async fn lease( + connection: &C, + installer: &PostgresMetadataInstallRepository, + lease_id: &str, +) -> Result, SnapshotError> { + let row = connection + .query_one_raw(statement( + &format!( + "SELECT * FROM {}($1,current_schema())", + function(installer, "mst2_route_select_lease") + ), + [lease_id.into()], + )) + .await + .map_err(internal)? + .ok_or_else(|| integrity("lease storage route selection is missing"))?; + let sid: Option = row.try_get("", "actual_sid").map_err(internal)?; + let route: bool = row.try_get("", "route_present").map_err(internal)?; + let valid: bool = row.try_get("", "valid").map_err(internal)?; + if sid.is_some() != route || sid.is_some() && !valid { + return Err(integrity( + "lease storage route conflicts with its exact generic incarnation", + )); + } + Ok(sid) +} diff --git a/src/jupiter/storage/native_snapshot_session.rs b/src/jupiter/storage/native_snapshot_session.rs index d84d075a..bb374236 100644 --- a/src/jupiter/storage/native_snapshot_session.rs +++ b/src/jupiter/storage/native_snapshot_session.rs @@ -17,7 +17,6 @@ use super::{ MetadataInstallError, PostgresMetadataInstallRepository, PreparedMetadataReceipt, }, native_publication_storage::{NativePublicationHead, decode_native_observation}, - push_queue_storage::PushQueueStorage, }; use crate::{ callisto::mst2_snapshot_context, @@ -34,6 +33,7 @@ use crate::{ pub(crate) struct PostgresNativeSessionRepository { connection: DatabaseConnection, installer: OnceCell, + qualified_session_sql: OnceCell, restored: Mutex>>>, } @@ -42,6 +42,7 @@ impl PostgresNativeSessionRepository { Self { connection, installer: OnceCell::new(), + qualified_session_sql: OnceCell::new(), restored: Mutex::new(HashMap::new()), } } @@ -52,6 +53,12 @@ impl PostgresNativeSessionRepository { .await } + async fn session_sql(&self, installer: &PostgresMetadataInstallRepository) -> &str { + self.qualified_session_sql + .get_or_init(|| async { routes::session_sql(installer) }) + .await + } + pub(crate) async fn install( &self, built: &BuiltDescriptor, @@ -95,14 +102,14 @@ impl PostgresNativeSessionRepository { let installer = self.installer().await?; let txn = self.transaction().await?; let result = async { - PushQueueStorage::acquire_mono_write_lock(&txn).await.map_err(internal)?; - installer.verify_primary_connection(&txn).await?; + routes::enter(&txn, installer).await?; let current = MonoStorage::read_native_publication_head_from(&txn, &expected.instance_id) .await.map_err(|_| not_ready("native publication is not ready"))?; if current.root != expected.root || current.token != expected.token { return Err(not_ready("native publication advanced during preparation; retry resolve")); } retention_lock(&txn).await?; + routes::snapshot(&txn, installer, &built.snapshot_id).await?; let existing = mst2_snapshot_context::Entity::find_by_id(built.snapshot_id.clone()) .one(&txn).await.map_err(internal)?; let prepare_id = if let Some(existing) = existing { @@ -175,10 +182,17 @@ impl PostgresNativeSessionRepository { instance_id: &str, ) -> Result { let installer = self.installer().await?; + if routes::lease(&self.connection, installer, lease_id) + .await? + .as_deref() + != Some(snapshot_id) + { + return Err(expired()); + } let row = self .connection .query_one_raw(statement( - SESSION_SQL, + self.session_sql(installer).await, [snapshot_id.into(), lease_id.into()], )) .await @@ -191,6 +205,9 @@ impl PostgresNativeSessionRepository { if error.code == SnapshotErrorCode::LeaseExpired { let txn = self.transaction().await?; let result = async { + installer.verify_primary_connection(&txn).await?; + routes::lease(&txn, installer, lease_id).await?; + routes::generic_path(&txn, installer).await?; retention_lock(&txn).await?; expire_specific_locked(&txn, lease_id).await } @@ -218,15 +235,23 @@ impl PostgresNativeSessionRepository { seconds: u64, instance: &str, ) -> Result { + let installer = self.installer().await?; let txn = self.transaction().await?; let result = async { + installer.verify_primary_connection(&txn).await?; + let selected_sid = routes::lease(&txn, installer, lease_id).await? + .ok_or_else(|| SnapshotError::new(SnapshotErrorCode::LeaseUnknown,"unknown lease_id"))?; + routes::generic_path(&txn, installer).await?; retention_lock(&txn).await?; let lease = txn.query_one_raw(statement( "SELECT snapshot_id FROM mst2_snapshot_lease WHERE lease_id=$1 FOR UPDATE", [lease_id.into()], )).await.map_err(internal)?.ok_or_else(|| SnapshotError::new(SnapshotErrorCode::LeaseUnknown,"unknown lease_id"))?; let sid: String = lease.try_get("","snapshot_id").map_err(internal)?; - let row = txn.query_one_raw(statement(SESSION_SQL,[sid.clone().into(),lease_id.into()])) + if sid != selected_sid { + return Err(integrity("lease changed after immutable route selection")); + } + let row = txn.query_one_raw(statement(self.session_sql(installer).await,[sid.clone().into(),lease_id.into()])) .await.map_err(internal)?.ok_or_else(expired)?; self.installer().await?.verify_primary_scope_row(&row)?; if let Err(error)=decode_context(&row,&sid,lease_id,instance) { @@ -257,8 +282,12 @@ impl PostgresNativeSessionRepository { let installer = self.installer().await?; let txn = self.transaction().await?; let result = async { - retention_lock(&txn).await?; installer.verify_primary_connection(&txn).await?; + if routes::lease(&txn, installer, lease_id).await?.is_none() { + return Ok(false); + } + routes::generic_path(&txn, installer).await?; + retention_lock(&txn).await?; let changed=txn.query_one_raw(statement( "UPDATE mst2_snapshot_lease SET state='RELEASED' WHERE lease_id=$1 AND state='ACTIVE' RETURNING snapshot_id", [lease_id.into()], @@ -293,6 +322,9 @@ impl PostgresNativeSessionRepository { } } +#[path = "native_snapshot_routes.rs"] +mod routes; + const SESSION_SQL: &str = "SELECT s.snapshot_id,s.canonical_descriptor,s.commit_oid,s.root_tree_oid, (SELECT storage_uuid FROM mst2_metadata_storage_scope WHERE singleton=1) AS authority_storage_uuid, current_database() AS authority_database, From fa1bdffa5ccdf0806eb143d1783fc724343eec4b Mon Sep 17 00:00:00 2001 From: Xiaoyang Han Date: Thu, 8 Oct 2026 00:20:15 +0800 Subject: [PATCH 18/50] test(mst2): cover FK and capability truncate barriers separately (#68) The exact f3e native run passed 2346 tests and failed one expectation: PostgreSQL FK preflight rejects plain prepare-page TRUNCATE before the install-capability trigger. Preserve plain FK rejection, add CASCADE to reach the unchanged capability guard, and prove all prepare mappings and seals survive. Existing identity assertions and actual installation/finalize remain. No production code changes. --- ...ative_metadata_install_capability_tests.rs | 23 ++++++++++++++++++- 1 file changed, 22 insertions(+), 1 deletion(-) diff --git a/src/jupiter/storage/native_metadata_install_capability_tests.rs b/src/jupiter/storage/native_metadata_install_capability_tests.rs index cd2759ef..9cbc21e6 100644 --- a/src/jupiter/storage/native_metadata_install_capability_tests.rs +++ b/src/jupiter/storage/native_metadata_install_capability_tests.rs @@ -241,6 +241,19 @@ async fn install_capability_freezes_every_identity_member_anchor_and_truncate_pa .await .unwrap() .unwrap(); + let member_count_before = + scalar(&first, "SELECT count(*) FROM mst2_metadata_prepare_page").await; + let seal_count_before = scalar(&first, "SELECT count(*) FROM mst2_metadata_install_seal").await; + let restrict_error = first + .execute_unprepared("TRUNCATE mst2_metadata_prepare_page") + .await + .expect_err("plain truncate must preserve foreign-key protection"); + assert!( + restrict_error + .to_string() + .contains("cannot truncate a table referenced in a foreign key constraint"), + "{restrict_error}" + ); assert_eq!( first.execute_raw(statement( "UPDATE mst2_metadata_prepare SET verification_revision=verification_revision WHERE prepare_id=$1", @@ -311,11 +324,19 @@ async fn install_capability_freezes_every_identity_member_anchor_and_truncate_pa "UPDATE mst2_metadata_install_seal SET members_digest=decode(repeat('ff',32),'hex')".into(), "DELETE FROM mst2_metadata_install_seal".into(), "TRUNCATE mst2_metadata_install_seal".into(), - "TRUNCATE mst2_metadata_prepare_page".into(), + "TRUNCATE mst2_metadata_prepare_page CASCADE".into(), "TRUNCATE mst2_metadata_prepare CASCADE".into(), ] { assert_guard(&first, &sql).await; } + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_prepare_page").await, + member_count_before + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_install_seal").await, + seal_count_before + ); assert_eq!( mst2_metadata_prepare::Entity::find_by_id(intent.prepare_id().to_owned()) .one(&first) From ce988809a05b72d36a086501ea29ea52aa6311a0 Mon Sep 17 00:00:00 2001 From: Xiaoyang Han Date: Thu, 8 Oct 2026 00:36:45 +0800 Subject: [PATCH 19/50] perf(mst2): stream CHUNK maps and read bounded exact ranges (#69) Build verified chunk maps from raw streams with full digest, exact size and EOF validation. Retain inline data through 512 MiB; larger files retain map metadata and read selected chunks by strict current-source ranges. Keep the existing singleflight, attach projection/response credits to actual owners, and admit complete batches before body I/O. Add seventeen stream, budget, storage and HTTP regressions. Native and performance results remain separate; persistent locators and process RSS are not completed by this slice. --- .../router/snapshot_chunks_bounded_tests.rs | 472 ++++++++++++++++++ src/api/router/snapshot_content.rs | 209 ++++++-- src/api/router/snapshot_content_tests.rs | 63 +++ src/api/router/snapshot_router.rs | 40 +- src/ceres/api_service/mod.rs | 25 + .../snapshot/chunk_singleflight_tests.rs | 2 +- src/ceres/snapshot/chunks.rs | 124 ++++- src/ceres/snapshot/chunks_stream.rs | 274 ++++++++++ src/ceres/snapshot/chunks_stream_tests.rs | 391 +++++++++++++++ src/ceres/snapshot/content_budget.rs | 147 ++++++ src/ceres/snapshot/mod.rs | 1 + src/jupiter/service/git_service.rs | 20 + src/jupiter/storage/object_storage.rs | 71 +++ src/orbit/adapter/object.rs | 89 ++++ src/orbit_api/object_storage.rs | 32 ++ 15 files changed, 1904 insertions(+), 56 deletions(-) create mode 100644 src/api/router/snapshot_chunks_bounded_tests.rs create mode 100644 src/ceres/snapshot/chunks_stream.rs create mode 100644 src/ceres/snapshot/chunks_stream_tests.rs create mode 100644 src/ceres/snapshot/content_budget.rs diff --git a/src/api/router/snapshot_chunks_bounded_tests.rs b/src/api/router/snapshot_chunks_bounded_tests.rs new file mode 100644 index 00000000..a32add5c --- /dev/null +++ b/src/api/router/snapshot_chunks_bounded_tests.rs @@ -0,0 +1,472 @@ +use std::io; + +use tokio::{sync::Notify, time::timeout}; + +use super::*; +use crate::orbit_api::error::IoOrbitError; + +#[derive(Clone)] +pub(super) struct ChunkFault { + pub(super) oid: String, + pub(super) size: u64, + pattern: Bytes, + mode: RangeMode, + requests: Arc>>, +} + +#[derive(Clone)] +enum RangeMode { + Good, + WrongMeta, + WrongDigest, + Truncated, + TooLong, + LateError, + Unsupported, + WrongOffset, + Missing, + Held { + entered: Arc, + release: Arc, + drops: Arc, + }, +} + +struct DropCount(Arc); +impl Drop for DropCount { + fn drop(&mut self) { + self.0.fetch_add(1, Ordering::SeqCst); + } +} + +impl ChunkFault { + pub(super) fn full_stream(self) -> ObjectByteStream { + Box::pin(futures::stream::unfold( + (self.pattern, 0u64, self.size), + |(pattern, offset, size)| async move { + if offset == size { + return None; + } + let len = (size - offset).min(CHUNK_SIZE as u64) as usize; + let bytes = pattern.slice(..len); + Some((Ok(bytes), (pattern, offset + len as u64, size))) + }, + )) + } + + pub(super) fn range_stream( + self, + start: u64, + end: u64, + ) -> OrbitResult> { + self.requests + .lock() + .unwrap() + .push((self.oid.clone(), start, end)); + if matches!(self.mode, RangeMode::Unsupported) { + return Ok(None); + } + if matches!(self.mode, RangeMode::WrongOffset) { + return Err(io::Error::new(io::ErrorKind::InvalidData, "wrong backend range").into()); + } + if matches!(self.mode, RangeMode::Missing) { + return Err(IoOrbitError::object_store_not_found("fixed raw missing")); + } + assert!(start < end && end <= self.size && end - start <= CHUNK_SIZE as u64); + assert_eq!(start % CHUNK_SIZE as u64, 0); + let len = (end - start) as usize; + let raw = self.pattern.slice(..len); + let meta = ObjectMeta { + size: if matches!(self.mode, RangeMode::WrongMeta) { + self.size as i64 + 1 + } else { + self.size as i64 + }, + ..Default::default() + }; + let stream: ObjectByteStream = match self.mode { + RangeMode::WrongDigest => { + Box::pin(futures::stream::iter([Ok(Bytes::from(vec![0; len]))])) + } + RangeMode::Truncated => Box::pin(futures::stream::iter([Ok(raw.slice(..len - 1))])), + RangeMode::TooLong => Box::pin(futures::stream::iter([ + Ok(raw), + Ok(Bytes::from_static(b"!")), + ])), + RangeMode::LateError => Box::pin(futures::stream::iter([ + Ok(raw), + Err(io::Error::other("late range error")), + ])), + RangeMode::Held { + entered, + release, + drops, + } => Box::pin(futures::stream::unfold( + (Some(raw), entered, release, DropCount(drops)), + |(raw, entered, release, owner)| async move { + let raw = raw?; + entered.notify_one(); + release.notified().await; + Some((Ok(raw), (None, entered, release, owner))) + }, + )), + _ => Box::pin(futures::stream::iter([Ok(raw)])), + }; + Ok(Some((stream, meta))) + } +} + +async fn oid_for(fixture: &Fixture, path: &str) -> String { + let handler = MonoApiService::from(&fixture.state); + let main = fixture + .state + .storage + .mono_storage() + .get_main_ref("/project") + .await + .unwrap() + .unwrap(); + let tree = handler.get_tree_by_hash(&main.ref_tree_hash).await.unwrap(); + match resolve_abs_metadata(&handler, &tree, path).await.unwrap() { + MetadataWalkOutcome::FoundFile { oid, .. } => oid, + other => panic!("fixture did not resolve: {other:?}"), + } +} + +async fn set_fact(fixture: &Fixture, oid: &str, size: u64, digest: [u8; 32]) { + let db = fixture.state.storage.mono_storage().get_connection(); + let fact = mst2_verified_object::Entity::find() + .filter(mst2_verified_object::Column::GitOid.eq(oid)) + .one(db) + .await + .unwrap() + .unwrap(); + let mut fact = fact.into_active_model(); + fact.size = Set(size as i64); + fact.raw_sha256 = Set(digest.to_vec()); + fact.update(db).await.unwrap(); +} + +fn request(path: &str, digest: [u8; 32], map_id: &str, index: u64) -> Value { + json!({"items":[{"path":path,"expected_digest":format!("sha256:{}",hex_of(&digest)),"map_id":map_id,"chunk_index":index.to_string()}],"encoding":"identity"}) +} + +async fn assert_chunk(response: Response, body: &Value, raw: &[u8], index: u64) { + assert_eq!(response.status(), 200); + let wire = to_bytes(response.into_body(), 2 * 1024 * 1024) + .await + .unwrap(); + let frames = parse_stream(&wire).unwrap(); + let [Frame::Chunk(chunk), Frame::End(end)] = frames.as_slice() else { + panic!("one CHUNK and END required"); + }; + assert_eq!(chunk.chunk_bytes, raw); + assert_eq!(chunk.chunk_index, index); + assert_eq!(end.request_item_count, 1); + assert_eq!(end.unique_unit_count, 1); + assert_eq!(end.logical_bytes, raw.len() as u64); + assert_eq!( + end.request_body_sha256, + <[u8; 32]>::from(Sha256::digest(body.to_string().as_bytes())) + ); +} + +#[tokio::test] +async fn mst2_large_chunk_uses_current_oid_strict_range_faults_cancel_retry_and_lease() { + let fixture = Fixture::new_with_pg_config_directories_and_objects( + false, + 0, + &[("other".to_string(), vec![19; 4096])], + ) + .await; + let mut pattern = vec![51; CHUNK_SIZE as usize]; + let seed = uuid::Uuid::new_v4(); + pattern[..16].copy_from_slice(seed.as_bytes()); + let pattern = Bytes::from(pattern); + let size = 512 * CHUNK_SIZE as u64 + 7; + let mut hash = Sha256::new(); + for _ in 0..512 { + hash.update(&pattern); + } + hash.update(&pattern[..7]); + let digest: [u8; 32] = hash.finalize().into(); + let other = oid_for(&fixture, "/other").await; + let requests = Arc::new(std::sync::Mutex::new(Vec::new())); + for oid in [&fixture.oid, &other] { + set_fact(&fixture, oid, size, digest).await; + fixture + .counts + .chunk_faults + .lock() + .unwrap() + .push(ChunkFault { + oid: oid.clone(), + size, + pattern: pattern.clone(), + mode: RangeMode::Good, + requests: requests.clone(), + }); + } + fixture.counts.reset(); + let map = fixture.map("/file").await; + assert_eq!(map["map"]["file_size"], size.to_string()); + fixture.counts.assert(1, size as usize); + let map_id = map["map"]["map_id"].as_str().unwrap(); + fixture.counts.reset(); + // Equal content map sharing never carries the first source's OID. + let body = request("/other", digest, map_id, 512); + assert_chunk( + fixture + .send("POST", "chunks", Body::from(body.to_string())) + .await, + &body, + &pattern[..7], + 512, + ) + .await; + assert_eq!( + requests.lock().unwrap().as_slice(), + &[(other.clone(), 512 * CHUNK_SIZE as u64, size)] + ); + assert_eq!(fixture.counts.whole.load(Ordering::SeqCst), 0); + assert_eq!(fixture.counts.range.load(Ordering::SeqCst), 1); + assert_eq!(fixture.counts.bytes.load(Ordering::SeqCst), 7); + for (index, encoding) in [(0, "identity"), (511, "zstd")] { + let mut full = request("/other", digest, map_id, index); + full["encoding"] = json!(encoding); + fixture.counts.reset(); + assert_chunk( + fixture + .send("POST", "chunks", Body::from(full.to_string())) + .await, + &full, + &pattern, + index, + ) + .await; + assert_eq!(fixture.counts.whole.load(Ordering::SeqCst), 0); + assert_eq!(fixture.counts.range.load(Ordering::SeqCst), 1); + assert_eq!( + fixture.counts.bytes.load(Ordering::SeqCst), + CHUNK_SIZE as usize + ); + assert_eq!( + requests.lock().unwrap().last().unwrap(), + &( + other.clone(), + index * CHUNK_SIZE as u64, + (index + 1) * CHUNK_SIZE as u64 + ) + ); + } + for (mode, status, code, bytes) in [ + (RangeMode::WrongMeta, 502, "INTEGRITY_ERROR", 0), + (RangeMode::WrongDigest, 502, "INTEGRITY_ERROR", 7), + (RangeMode::Truncated, 502, "INTEGRITY_ERROR", 6), + (RangeMode::TooLong, 502, "INTEGRITY_ERROR", 8), + (RangeMode::LateError, 503, "OBJECT_UNAVAILABLE", 7), + (RangeMode::Unsupported, 400, "RANGE_NOT_SUPPORTED", 0), + (RangeMode::WrongOffset, 502, "INTEGRITY_ERROR", 0), + (RangeMode::Missing, 503, "OBJECT_UNAVAILABLE", 0), + ] { + fixture.counts.chunk_faults.lock().unwrap()[1].mode = mode; + fixture.counts.reset(); + error( + fixture + .send("POST", "chunks", Body::from(body.to_string())) + .await, + status, + code, + false, + ) + .await; + assert_eq!(fixture.counts.whole.load(Ordering::SeqCst), 0); + assert_eq!(fixture.counts.range.load(Ordering::SeqCst), 1); + assert_eq!(fixture.counts.bytes.load(Ordering::SeqCst), bytes); + } + fixture.counts.chunk_faults.lock().unwrap()[1].mode = RangeMode::LateError; + let first = request("/file", digest, map_id, 0); + let batch = json!({"items":[first["items"][0],body["items"][0]],"encoding":"identity"}); + fixture.counts.reset(); + error( + fixture + .send("POST", "chunks", Body::from(batch.to_string())) + .await, + 503, + "OBJECT_UNAVAILABLE", + false, + ) + .await; + assert_eq!(fixture.counts.whole.load(Ordering::SeqCst), 0); + assert_eq!(fixture.counts.range.load(Ordering::SeqCst), 2); + assert_eq!( + fixture.counts.bytes.load(Ordering::SeqCst), + CHUNK_SIZE as usize + 7 + ); + let entered = Arc::new(Notify::new()); + let release = Arc::new(Notify::new()); + let drops = Arc::new(AtomicUsize::new(0)); + fixture.counts.chunk_faults.lock().unwrap()[1].mode = RangeMode::Held { + entered: entered.clone(), + release: release.clone(), + drops: drops.clone(), + }; + let task = tokio::spawn(fixture.app.clone().oneshot(fixture.request( + "POST", + "chunks", + Body::from(body.to_string()), + ))); + timeout(Duration::from_secs(10), entered.notified()) + .await + .unwrap(); + task.abort(); + assert!(task.await.err().unwrap().is_cancelled()); + assert_eq!(drops.load(Ordering::SeqCst), 1); + fixture.counts.chunk_faults.lock().unwrap()[1].mode = RangeMode::Good; + assert_chunk( + fixture + .send("POST", "chunks", Body::from(body.to_string())) + .await, + &body, + &pattern[..7], + 512, + ) + .await; + // Revocation while the range is held is a pre-header 410 JSON response. + fixture.counts.chunk_faults.lock().unwrap()[1].mode = RangeMode::Held { + entered: entered.clone(), + release: release.clone(), + drops: drops.clone(), + }; + let task = tokio::spawn(fixture.app.clone().oneshot(fixture.request( + "POST", + "chunks", + Body::from(body.to_string()), + ))); + timeout(Duration::from_secs(10), entered.notified()) + .await + .unwrap(); + let revoked = fixture + .app + .clone() + .oneshot( + Request::builder() + .method("DELETE") + .uri(format!("/api/v2/snapshots/leases/{}", fixture.lease)) + .header("authorization", format!("Bearer {TOKEN}")) + .body(Body::empty()) + .unwrap(), + ) + .await + .unwrap(); + assert_eq!(revoked.status(), 200); + release.notify_one(); + error( + timeout(Duration::from_secs(10), task) + .await + .unwrap() + .unwrap() + .unwrap(), + 410, + "LEASE_EXPIRED", + false, + ) + .await; + assert_eq!(drops.load(Ordering::SeqCst), 2); +} + +#[tokio::test] +async fn mst2_chunk_batch_live_budget_and_invalid_later_path_reject_before_body_io() { + let fixture = Fixture::new_with_pg_config_directories_and_objects( + false, + 0, + &[ + ("one".to_string(), vec![11; 4096]), + ("two".to_string(), vec![12; 4096]), + ], + ) + .await; + let mut items = Vec::new(); + for (index, path) in ["/file", "/one", "/two"].iter().enumerate() { + let oid = oid_for(&fixture, path).await; + let digest = [index as u8 + 1; 32]; + set_fact(&fixture, &oid, 512 * CHUNK_SIZE as u64, digest).await; + items.push( + request( + path, + digest, + &format!("sha256:{}", hex_of(&[index as u8 + 1; 32])), + 0, + )["items"][0] + .clone(), + ); + } + fixture.counts.reset(); + error( + fixture + .send( + "POST", + "chunks", + Body::from(json!({"items":items,"encoding":"identity"}).to_string()), + ) + .await, + 413, + "LIMIT_EXCEEDED", + false, + ) + .await; + fixture.counts.assert(0, 0); + let fixture = Fixture::new().await; + let mut body = fixture.chunk_body("/file", &format!("sha256:{}", hex_of(&[1; 32])), "0"); + let mut invalid = body["items"][0].clone(); + invalid["path"] = json!("/absent"); + body["items"].as_array_mut().unwrap().push(invalid); + error( + fixture + .send("POST", "chunks", Body::from(body.to_string())) + .await, + 404, + "PATH_NOT_FOUND", + false, + ) + .await; + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn mst2_budgeted_chunk_body_preserves_per_frame_revocation_and_no_end() { + let fixture = Fixture::new().await; + let map = fixture.map("/file").await; + let map_id = map["map"]["map_id"].as_str().unwrap(); + let mut body = fixture.chunk_body("/file", map_id, "0"); + body["items"] + .as_array_mut() + .unwrap() + .push(fixture.chunk_body("/file", map_id, "1")["items"][0].clone()); + let response = fixture + .send("POST", "chunks", Body::from(body.to_string())) + .await; + assert_eq!(response.status(), 200); + let mut data = response.into_body().into_data_stream(); + let first = data.next().await.unwrap().unwrap(); + assert!(matches!( + parse_stream(&first).unwrap().as_slice(), + [Frame::Chunk(_)] + )); + let revoked = fixture + .app + .clone() + .oneshot( + Request::builder() + .method("DELETE") + .uri(format!("/api/v2/snapshots/leases/{}", fixture.lease)) + .header("authorization", format!("Bearer {TOKEN}")) + .body(Body::empty()) + .unwrap(), + ) + .await + .unwrap(); + assert_eq!(revoked.status(), 200); + assert!(data.next().await.unwrap().is_err()); + assert!(data.next().await.is_none()); +} diff --git a/src/api/router/snapshot_content.rs b/src/api/router/snapshot_content.rs index a6d9efc4..ea7d8938 100644 --- a/src/api/router/snapshot_content.rs +++ b/src/api/router/snapshot_content.rs @@ -15,12 +15,14 @@ use serde_json::json; use sha2::{Digest, Sha256}; use super::{ - abs_view_path, guarded_treeframe_response, internal, mst2_error_response, request::Mst2Bytes, + abs_view_path, guarded_treeframe_response, guarded_treeframe_response_with_budget, internal, + mst2_error_response, request::Mst2Bytes, }; use crate::ceres::snapshot::{ - chunks::{ChunkProjection, get_or_project}, + chunks::{ChunkProjection, get_or_project_stream, projection_reservation_bytes}, + content_budget::{PROJECTION_LIVE_BYTES, reserve_range_work, reserve_response}, error::{SnapshotError, SnapshotErrorCode}, - pages::{MetadataWalkOutcome, base64_of, fetch_raw_blob, hex_of, resolve_abs_metadata}, + pages::{MetadataWalkOutcome, base64_of, hex_of, resolve_abs_metadata}, resolver::FsKind, view::validate_scope_relative_path, }; @@ -475,20 +477,24 @@ async fn project_for( expected_digest: Option<&str>, ) -> Result, Response> { let f = resolve_file_metadata(handler, root_tree, scope, path, expected_digest).await?; + project_resolved(handler, &f).await +} + +#[allow(clippy::result_large_err)] +async fn project_resolved( + handler: &T, + f: &ResolvedFileMetadata, +) -> Result, Response> { // The first request for a digest builds the projection from the fixed // Git object; later requests slice the cached representation. A miss // rebuilds, never errors with "missing chunk". let digest = f.digest; let size = f.size; - let projection = get_or_project(digest, || async move { - let raw = fetch_raw_blob(handler, &f.oid).await?; - if raw.len() as u64 != size { - return Err(SnapshotError::new( - SnapshotErrorCode::IntegrityError, - "fixed blob length disagrees with its verified size fact", - )); - } - Ok(raw) + let projection = get_or_project_stream(digest, size, || async move { + handler + .get_raw_blob_stream_by_hash(&f.oid) + .await + .map_err(content_read_error) }) .await .map_err(|error| { @@ -658,9 +664,91 @@ const CHUNKS_TOTAL_MAX: u64 = 128 * 1024 * 1024; struct Planned { projection: std::sync::Arc, + oid: String, index: u64, } +struct ResolvedChunk { + file: ResolvedFileMetadata, + index: u64, + map_id: String, +} + +fn content_read_error(error: crate::common::errors::MegaError) -> SnapshotError { + use crate::common::errors::MegaError; + let code = match &error { + MegaError::ObjStorageNotFound(_) => SnapshotErrorCode::ObjectUnavailable, + MegaError::ObjStorageInconsistent(_) => SnapshotErrorCode::IntegrityError, + MegaError::Io(error) if error.kind() == std::io::ErrorKind::InvalidData => { + SnapshotErrorCode::IntegrityError + } + _ => SnapshotErrorCode::Internal, + }; + tracing::warn!(error = %error, "fixed-view content read failed"); + SnapshotError::new(code, "fixed-view content could not be read") +} + +#[allow(clippy::result_large_err)] +async fn read_chunk_range( + handler: &T, + planned: &Planned, +) -> Result, Response> { + let projection = &planned.projection; + let len = projection.map.chunk_len(planned.index).map_err(|error| { + mst2_error_response(internal(format!("invalid admitted chunk: {error}"))) + })?; + let start = planned + .index + .checked_mul(mst2_codec::chunkmap::CHUNK_SIZE as u64) + .ok_or_else(|| mst2_error_response(internal("chunk offset overflow")))?; + let end = start + .checked_add(len) + .ok_or_else(|| mst2_error_response(internal("chunk end overflow")))?; + let (mut input, meta) = handler + .get_raw_blob_range_stream_exact(&planned.oid, start, end) + .await + .map_err(|error| mst2_error_response(content_read_error(error)))? + .ok_or_else(|| { + mst2_error_response(SnapshotError::new( + SnapshotErrorCode::RangeNotSupported, + "fixed source does not support exact raw ranges", + )) + })?; + if u64::try_from(meta.size).ok() != Some(projection.map.file_size) { + return Err(mst2_error_response(SnapshotError::new( + SnapshotErrorCode::IntegrityError, + "range source size disagrees with the fixed verified fact", + ))); + } + let mut raw = Vec::new(); + raw.try_reserve_exact(len as usize).map_err(|_| { + mst2_error_response(SnapshotError::new( + SnapshotErrorCode::TemporaryUnavailable, + "range allocation could not be admitted", + )) + })?; + while let Some(part) = input.next().await { + let bytes = part.map_err(|error| { + tracing::warn!(error = %error, "fixed-view range stream failed"); + mst2_error_response(SnapshotError::new( + SnapshotErrorCode::ObjectUnavailable, + "fixed-view range stream failed", + )) + })?; + if bytes.len() > len as usize - raw.len() { + return Err(mst2_error_response(SnapshotError::new( + SnapshotErrorCode::IntegrityError, + "range response exceeds its exact requested length", + ))); + } + raw.extend_from_slice(&bytes); + } + projection + .verify_chunk(planned.index, &raw) + .map_err(mst2_error_response)?; + Ok(raw) +} + #[allow(clippy::result_large_err)] pub(super) async fn chunks( state: State, @@ -693,13 +781,16 @@ pub(super) async fn chunks( let root_tree = fixed_root_tree(handler.as_ref(), &ctx.root_tree_oid).await?; let scope = ctx.built.descriptor.scope.clone(); let mut planned: Vec = Vec::new(); + let mut resolved: Vec = Vec::new(); let mut units: Vec<(String, u64)> = Vec::new(); + let mut distinct: std::collections::HashMap<[u8; 32], u64> = std::collections::HashMap::new(); + let mut projection_bytes = 0usize; let mut logical_bytes = 0u64; for item in &req.items { validate_scope_relative_path(&item.path).map_err(mst2_error_response)?; let index = parse_decimal_count(&item.chunk_index, "chunk_index").map_err(mst2_error_response)?; - let proj = project_for( + let file = resolve_file_metadata( handler.as_ref(), &root_tree, &scope, @@ -707,19 +798,13 @@ pub(super) async fn chunks( Some(&item.expected_digest), ) .await?; - let want_map = format!("sha256:{}", hex_of(&proj.map_id)); - if want_map != item.map_id { - return Err(mst2_error_response(SnapshotError::new( - SnapshotErrorCode::DigestMismatch, - format!("{}: map_id does not bind to this file", item.path), - ))); - } - if index >= proj.map.chunk_count { + let chunk_count = file.size.div_ceil(mst2_codec::chunkmap::CHUNK_SIZE as u64); + if index >= chunk_count { return Err(mst2_error_response(SnapshotError::new( SnapshotErrorCode::ScopeInvalid, format!( "{}: chunk_index {index} >= chunk_count {}", - item.path, proj.map.chunk_count + item.path, chunk_count ), ))); } @@ -731,10 +816,8 @@ pub(super) async fn chunks( format!("duplicate chunk unit map_id={} index={index}", item.map_id), ))); } - let len = proj - .map - .chunk_len(index) - .map_err(|e| mst2_error_response(internal(e.to_string())))?; + let start = index * mst2_codec::chunkmap::CHUNK_SIZE as u64; + let len = (file.size - start).min(mst2_codec::chunkmap::CHUNK_SIZE as u64); if logical_bytes + len > CHUNKS_TOTAL_MAX { return Err(mst2_error_response(SnapshotError::new( SnapshotErrorCode::ScopeInvalid, @@ -743,9 +826,53 @@ pub(super) async fn chunks( } logical_bytes += len; units.push(unit); + match distinct.entry(file.digest) { + std::collections::hash_map::Entry::Occupied(entry) => { + if *entry.get() != file.size { + return Err(mst2_error_response(SnapshotError::new( + SnapshotErrorCode::IntegrityError, + "fixed facts disagree for the same content digest", + ))); + } + } + std::collections::hash_map::Entry::Vacant(entry) => { + let bytes = projection_reservation_bytes(file.size).map_err(mst2_error_response)?; + projection_bytes = projection_bytes.checked_add(bytes).ok_or_else(|| { + mst2_error_response(internal("projection batch memory overflow")) + })?; + if projection_bytes > PROJECTION_LIVE_BYTES { + return Err(mst2_error_response(SnapshotError::new( + SnapshotErrorCode::LimitExceeded, + "chunk batch exceeds its live projection memory budget", + ))); + } + entry.insert(file.size); + } + } + resolved.push(ResolvedChunk { + file, + index, + map_id: item.map_id.clone(), + }); + } + let response_bytes = usize::try_from(logical_bytes) + .ok() + .and_then(|bytes| bytes.checked_add(req.items.len() * 1024 + 1024)) + .ok_or_else(|| mst2_error_response(internal("chunk response memory overflow")))?; + let response_memory = reserve_response(response_bytes).map_err(mst2_error_response)?; + for item in resolved { + let proj = project_resolved(handler.as_ref(), &item.file).await?; + let want_map = format!("sha256:{}", hex_of(&proj.map_id)); + if want_map != item.map_id { + return Err(mst2_error_response(SnapshotError::new( + SnapshotErrorCode::DigestMismatch, + "map_id does not bind to the fixed file", + ))); + } planned.push(Planned { projection: proj, - index, + oid: item.file.oid, + index: item.index, }); } @@ -753,17 +880,23 @@ pub(super) async fn chunks( let mut stream = FrameStream::new(1, encoding); let mut out: Vec> = Vec::new(); for p in planned.iter() { - // Re-verified slice (digest + length) from the staged projection. - let bytes = p - .projection - .chunk_bytes(p.index) - .map_err(mst2_error_response)?; + let _work_memory = reserve_range_work().map_err(mst2_error_response)?; + // The source is the current request's fixed OID, never a cached + // handler/backend/credential from a different scope. + let bytes = if p.projection.has_inline_bytes() { + p.projection + .chunk_bytes(p.index) + .map_err(mst2_error_response)? + .to_vec() + } else { + read_chunk_range(handler.as_ref(), p).await? + }; let frame = stream .chunk( p.projection.map_id, p.projection.map.file_content_id, p.index, - bytes.to_vec(), + bytes, ) .map_err(mst2_error_response)?; out.push(frame); @@ -776,7 +909,15 @@ pub(super) async fn chunks( ); out.push(end); - guarded_treeframe_response(&state, &ctx, &snapshot_id, &body, out).map_err(mst2_error_response) + guarded_treeframe_response_with_budget( + &state, + &ctx, + &snapshot_id, + &body, + out, + Some(response_memory), + ) + .map_err(mst2_error_response) } /// Strict decimal-string parse for unsigned counts (spec 04 §1: no leading diff --git a/src/api/router/snapshot_content_tests.rs b/src/api/router/snapshot_content_tests.rs index add40d69..bb58d788 100644 --- a/src/api/router/snapshot_content_tests.rs +++ b/src/api/router/snapshot_content_tests.rs @@ -82,6 +82,9 @@ const TOKEN: &str = "mst2-fixed-content-test"; #[path = "snapshot_objects_bounded_tests.rs"] mod bounded_objects; +#[path = "snapshot_chunks_bounded_tests.rs"] +mod bounded_chunks; + #[path = "snapshot_session_tests.rs"] mod durable_sessions; @@ -103,6 +106,7 @@ struct ReadCounts { range: AtomicUsize, bytes: AtomicUsize, object_fault: std::sync::Mutex>, + chunk_faults: std::sync::Mutex>, } impl ReadCounts { @@ -137,6 +141,31 @@ impl MegaObjectStorage for CountingStorage { async fn get_stream(&self, key: &ObjectKey) -> OrbitResult<(ObjectByteStream, ObjectMeta)> { self.counts.whole.fetch_add(1, Ordering::SeqCst); + let chunk_fault = self + .counts + .chunk_faults + .lock() + .unwrap() + .iter() + .find(|fault| fault.oid == key.key) + .cloned(); + if let Some(fault) = chunk_fault { + let counts = self.counts.clone(); + let size = fault.size; + let stream = fault.full_stream().map(move |part| { + if let Ok(bytes) = &part { + counts.bytes.fetch_add(bytes.len(), Ordering::SeqCst); + } + part + }); + return Ok(( + Box::pin(stream), + ObjectMeta { + size: size as i64, + ..Default::default() + }, + )); + } let (stream, meta) = self.inner.inner.get_stream(key).await?; let fault = self.counts.object_fault.lock().unwrap().clone(); let stream = match fault { @@ -163,6 +192,40 @@ impl MegaObjectStorage for CountingStorage { self.inner.inner.get_range_stream(key, start, end).await } + async fn get_range_stream_exact( + &self, + key: &ObjectKey, + start: u64, + end: u64, + ) -> OrbitResult> { + self.counts.range.fetch_add(1, Ordering::SeqCst); + let fault = self + .counts + .chunk_faults + .lock() + .unwrap() + .iter() + .find(|fault| fault.oid == key.key) + .cloned(); + if let Some(fault) = fault { + let range = fault.range_stream(start, end)?; + return Ok(range.map(|(stream, meta)| { + let counts = self.counts.clone(); + let stream = stream.map(move |part| { + if let Ok(bytes) = &part { + counts.bytes.fetch_add(bytes.len(), Ordering::SeqCst); + } + part + }); + (Box::pin(stream) as ObjectByteStream, meta) + })); + } + self.inner + .inner + .get_range_stream_exact(key, start, end) + .await + } + async fn exists(&self, key: &ObjectKey) -> OrbitResult { self.inner.inner.exists(key).await } diff --git a/src/api/router/snapshot_router.rs b/src/api/router/snapshot_router.rs index 40876942..eed58484 100644 --- a/src/api/router/snapshot_router.rs +++ b/src/api/router/snapshot_router.rs @@ -131,15 +131,51 @@ fn guarded_treeframe_response( request_body: &[u8], frames: Vec>, ) -> Result { + guarded_treeframe_response_with_budget(state, context, snapshot_id, request_body, frames, None) +} + +fn guarded_treeframe_response_with_budget( + state: &MonoApiServiceState, + context: &crate::ceres::snapshot::runtime::SnapshotContext, + snapshot_id: &str, + request_body: &[u8], + frames: Vec>, + memory: Option, +) -> Result { + // Field order also keeps admission credits until frame allocations drop + // on failures before ownership has moved into individual Bytes. + struct FrameAllocation { + frames: Vec>, + memory: Option, + } + let mut allocation = FrameAllocation { frames, memory }; + if let Some(lease) = &allocation.memory { + let allocated = allocation + .frames + .iter() + .try_fold(0usize, |total, frame| total.checked_add(frame.capacity())); + if allocated.is_none_or(|allocated| allocated > lease.bytes) { + return Err(internal("encoded frames exceed their memory reservation")); + } + } let headers = REQUEST_HEADERS.try_with(Clone::clone).map_err(|_| { SnapshotError::new( SnapshotErrorCode::Unauthenticated, "request authentication context missing", ) })?; - let units = frames + let memory = allocation.memory.take().map(std::sync::Arc::new); + let units = std::mem::take(&mut allocation.frames) .into_iter() - .map(Bytes::from) + .map(|bytes| match &memory { + Some(lease) => { + Bytes::from_owner(crate::ceres::snapshot::content_budget::BudgetedFrame { + bytes, + lease: lease.clone(), + }) + } + None => Bytes::from(bytes), + }) .collect::>(); let stream = futures::stream::unfold( (units, state.clone(), context.clone(), headers), diff --git a/src/ceres/api_service/mod.rs b/src/ceres/api_service/mod.rs index 32bbf780..29ec5d74 100644 --- a/src/ceres/api_service/mod.rs +++ b/src/ceres/api_service/mod.rs @@ -191,6 +191,31 @@ pub trait ApiHandler: Send + Sync { } } + async fn get_raw_blob_range_stream_exact( + &self, + hash: &str, + start: u64, + end: u64, + ) -> Result< + Option<( + crate::orbit_api::object_storage::ObjectByteStream, + crate::orbit_api::object_storage::ObjectMeta, + )>, + MegaError, + > { + let storage = self.get_context(); + match storage + .git_service + .get_object_range_stream_exact(hash, start, end) + .await + { + Ok(range) => Ok(range), + Err(error) => Err(storage + .classify_blob_objstorage_not_found(hash, error) + .await), + } + } + /// Preview unified diff for a single file change async fn preview_file_diff( &self, diff --git a/src/ceres/snapshot/chunk_singleflight_tests.rs b/src/ceres/snapshot/chunk_singleflight_tests.rs index 2534acf9..78daff83 100644 --- a/src/ceres/snapshot/chunk_singleflight_tests.rs +++ b/src/ceres/snapshot/chunk_singleflight_tests.rs @@ -70,7 +70,7 @@ fn assert_uncached(content_id: [u8; 32]) { fn evict(content_id: [u8; 32]) { let mut cache = STAGED.get().unwrap().lock().unwrap(); let projection = cache.entries.remove(&content_id).unwrap(); - cache.total_bytes -= projection.raw.len(); + cache.total_bytes -= projection.retained_bytes(); let position = cache.order.iter().position(|id| *id == content_id).unwrap(); cache.order.remove(position); } diff --git a/src/ceres/snapshot/chunks.rs b/src/ceres/snapshot/chunks.rs index e4182c01..8999a45e 100644 --- a/src/ceres/snapshot/chunks.rs +++ b/src/ceres/snapshot/chunks.rs @@ -2,9 +2,9 @@ //! //! Persistent segment/locator storage is T04/T12 work; this identity slice //! builds the MCM2 map and MCL2 leaves from a fixed view's verified blob and -//! stages the raw bytes in a bounded in-process cache, so a CHUNK request -//! slices its range out of an already-verified representation rather than -//! reconstructing the whole Git object per chunk (spec 07 §8). +//! retains inline bytes through 512 MiB and only verified map metadata for +//! larger files. Large CHUNK reads use the current request's strict raw range +//! source. This is a reproducible process cache, not a durable locator. //! //! The cache is an optimization, never the authority: every projected file //! is re-hashed against the `content_id` the fixed view advertised, and a @@ -19,21 +19,33 @@ use std::{ use mst2_codec::chunkmap::{CHUNK_SIZE, CHUNKS_PER_PAGE, ChunkLeaf, ChunkMap}; use sha2::{Digest, Sha256}; +use super::content_budget::{MemoryLease, projection_budget}; use crate::ceres::snapshot::error::{SnapshotError, SnapshotErrorCode}; +#[path = "chunks_stream.rs"] +mod streaming; + +pub use streaming::get_or_project_stream; + +pub(crate) fn projection_reservation_bytes(size: u64) -> Result { + streaming::reserved_bytes(size) +} + /// One file's verified range-readable projection. pub struct ChunkProjection { pub map: ChunkMap, pub map_id: [u8; 32], leaves: Vec, leaf_hashes: Vec<[u8; 32]>, - /// Full verified content. Indexed by fixed 1 MiB chunk boundaries. + /// Empty for range-only projections; empty files have no projection. raw: Vec, + memory: Option, } impl ChunkProjection { /// Build from bytes that the caller already resolved at a fixed path. /// `content_id` is the digest the fixed view advertises. + #[cfg(test)] pub fn build(content_id: [u8; 32], raw: Vec) -> Result { let mut hasher = Sha256::new(); hasher.update(&raw); @@ -95,9 +107,52 @@ impl ChunkProjection { leaves, leaf_hashes, raw, + memory: None, }) } + pub fn has_inline_bytes(&self) -> bool { + !self.raw.is_empty() + } + + fn retained_bytes(&self) -> usize { + self.memory + .as_ref() + .map_or_else(|| self.allocated_bytes(), |lease| lease.bytes) + } + + fn allocated_bytes(&self) -> usize { + std::mem::size_of::() + + self.raw.capacity() + + self.leaves.capacity() * std::mem::size_of::() + + self.leaf_hashes.capacity() * 32 + + self + .leaves + .iter() + .map(|leaf| leaf.chunk_sha256.capacity() * 32) + .sum::() + } + + pub fn verify_chunk(&self, index: u64, bytes: &[u8]) -> Result<(), SnapshotError> { + let want_len = self.map.chunk_len(index).map_err(codec_err)?; + if bytes.len() as u64 != want_len { + return Err(SnapshotError::new( + SnapshotErrorCode::IntegrityError, + "range length disagrees with the verified chunk map", + )); + } + let page = (index / CHUNKS_PER_PAGE as u64) as usize; + let slot = (index % CHUNKS_PER_PAGE as u64) as usize; + let got: [u8; 32] = Sha256::digest(bytes).into(); + if got != self.leaves[page].chunk_sha256[slot] { + return Err(SnapshotError::new( + SnapshotErrorCode::IntegrityError, + "range digest disagrees with the verified chunk map", + )); + } + Ok(()) + } + pub fn page_count(&self) -> u64 { self.map.page_count } @@ -175,6 +230,7 @@ static STAGED: OnceLock> = OnceLock::new(); /// 512 MiB cap for staged file bytes in this identity slice. The real /// persistent RAW_OBJECT locator layout replaces this (spec 10 §4). const STAGED_CAP_BYTES: usize = 512 * 1024 * 1024; +const STAGED_MAX_ENTRIES: usize = 4096; struct ProjectionCache { entries: HashMap<[u8; 32], std::sync::Arc>, @@ -196,26 +252,39 @@ impl ProjectionCache { self.entries.get(&id).cloned() } - fn put(&mut self, proj: std::sync::Arc) { + fn evict_oldest(&mut self) -> Option> { + while let Some(victim) = self.order.pop_front() { + if let Some(projection) = self.entries.remove(&victim) { + self.total_bytes -= projection.retained_bytes(); + return Some(projection); + } + } + None + } + + fn put(&mut self, proj: std::sync::Arc) -> Vec> { + let mut evicted = Vec::new(); let id = proj.map.file_content_id; if self.entries.contains_key(&id) { - return; + return evicted; } // Reclaim oldest entries until the new one fits. A file larger than // the cap alone is not cached between flights but still // served correctly. - while self.total_bytes + proj.raw.len() > STAGED_CAP_BYTES - && let Some(victim) = self.order.pop_front() - { - if let Some(v) = self.entries.remove(&victim) { - self.total_bytes = self.total_bytes.saturating_sub(v.raw.len()); - } + let bytes = proj.retained_bytes(); + if bytes > STAGED_CAP_BYTES { + return evicted; } - if proj.raw.len() <= STAGED_CAP_BYTES { - self.total_bytes += proj.raw.len(); - self.order.push_back(id); - self.entries.insert(id, proj); + while (self.total_bytes + bytes > STAGED_CAP_BYTES + || self.entries.len() >= STAGED_MAX_ENTRIES) + && let Some(victim) = self.evict_oldest() + { + evicted.push(victim); } + self.total_bytes += bytes; + self.order.push_back(id); + self.entries.insert(id, proj); + evicted } } @@ -301,6 +370,7 @@ impl Drop for FlightClaim<'_> { /// Return the cached projection, or build one via `load` (which must resolve /// the file in the fixed view and return its verified bytes). Concurrent /// misses for the same digest share one successful load and projection. +#[cfg(test)] pub async fn get_or_project( content_id: [u8; 32], load: F, @@ -308,6 +378,20 @@ pub async fn get_or_project( where F: FnOnce() -> Fut, Fut: std::future::Future, SnapshotError>>, +{ + get_or_project_with(content_id, || async { + ChunkProjection::build(content_id, load().await?) + }) + .await +} + +async fn get_or_project_with( + content_id: [u8; 32], + build: F, +) -> Result, SnapshotError> +where + F: FnOnce() -> Fut, + Fut: std::future::Future>, { let cache = STAGED.get_or_init(|| Mutex::new(ProjectionCache::new())); if let Some(p) = cache @@ -336,12 +420,14 @@ where if let Some(p) = result.as_ref() { return Ok(p.clone()); } - let raw = load().await?; - let proj = std::sync::Arc::new(ChunkProjection::build(content_id, raw)?); - cache + let proj = Arc::new(build().await?); + let evicted = cache .lock() .map_err(|_| internal("chunk projection cache lock poisoned"))? .put(proj.clone()); + // Final projection/credit drops can be expensive and must not hold the + // process cache mutex. Eviction does not refund other Arc owners. + drop(evicted); *result = Some(proj.clone()); Ok(proj) } diff --git a/src/ceres/snapshot/chunks_stream.rs b/src/ceres/snapshot/chunks_stream.rs new file mode 100644 index 00000000..56f007fd --- /dev/null +++ b/src/ceres/snapshot/chunks_stream.rs @@ -0,0 +1,274 @@ +use futures::StreamExt; +use tokio::sync::Semaphore; + +use super::*; +use crate::orbit_api::object_storage::ObjectByteStream; + +const STREAM_ITEM_MAX_BYTES: usize = 8 * 1024 * 1024; +const MAX_FILE_BYTES: u64 = 8 * 1024 * 1024 * 1024 * 1024; +const MAX_BUILDERS: usize = 4; +const CONSTRUCTION_ALLOWANCE: usize = 64 * 1024; + +pub(super) fn reserved_bytes(size: u64) -> Result { + if size == 0 || size > MAX_FILE_BYTES { + return Err(SnapshotError::new( + if size == 0 { + SnapshotErrorCode::ScopeInvalid + } else { + SnapshotErrorCode::LimitExceeded + }, + "file is outside the chunk projection profile", + )); + } + let chunks = size.div_ceil(CHUNK_SIZE as u64); + let pages = chunks.div_ceil(CHUNKS_PER_PAGE as u64); + let inline = if size <= STAGED_CAP_BYTES as u64 { + size + } else { + 0 + }; + let bytes = chunks + .checked_mul(32) + .and_then(|n| { + pages + .checked_mul((std::mem::size_of::() + 32) as u64) + .and_then(|pages| n.checked_add(pages)) + }) + .and_then(|n| n.checked_add(inline)) + .and_then(|n| n.checked_add(CONSTRUCTION_ALLOWANCE as u64)) + .and_then(|n| usize::try_from(n).ok()) + .ok_or_else(|| internal("chunk projection memory arithmetic overflow"))?; + Ok(bytes) +} + +fn reserve_projection( + size: u64, + budget: &Arc, + cache: &Mutex, +) -> Result { + let bytes = reserved_bytes(size)?; + loop { + match budget.reserve(bytes) { + Ok(lease) => return Ok(lease), + Err(error) => { + if error.code != SnapshotErrorCode::TemporaryUnavailable { + return Err(error); + } + let victim = cache + .lock() + .map_err(|_| internal("chunk projection cache lock poisoned"))? + .evict_oldest(); + let Some(victim) = victim else { + return Err(error); + }; + drop(victim); + } + } + } +} + +pub async fn get_or_project_stream( + content_id: [u8; 32], + size: u64, + open: F, +) -> Result, SnapshotError> +where + F: FnOnce() -> Fut, + Fut: std::future::Future>, +{ + // Profile arithmetic also runs on hits; fixed facts remain per-request. + reserved_bytes(size)?; + get_or_project_with(content_id, || async { + static BUILDERS: OnceLock> = OnceLock::new(); + build_with_resources( + content_id, + size, + open, + projection_budget(), + BUILDERS.get_or_init(|| Arc::new(Semaphore::new(MAX_BUILDERS))), + STAGED.get_or_init(|| Mutex::new(ProjectionCache::new())), + ) + .await + }) + .await +} + +async fn build_with_resources( + content_id: [u8; 32], + size: u64, + open: F, + budget: &Arc, + builders: &Arc, + cache: &Mutex, +) -> Result +where + F: FnOnce() -> Fut, + Fut: std::future::Future>, +{ + let _builder = builders.clone().try_acquire_owned().map_err(|_| { + SnapshotError::new( + SnapshotErrorCode::TemporaryUnavailable, + "chunk projection builders are occupied", + ) + })?; + let lease = reserve_projection(size, budget, cache)?; + // Source body I/O starts only after builder and retained-memory credit. + let input = open().await?; + build_stream(content_id, size, input, lease).await +} + +async fn build_stream( + content_id: [u8; 32], + size: u64, + mut input: ObjectByteStream, + lease: MemoryLease, +) -> Result { + let chunk_count = size.div_ceil(CHUNK_SIZE as u64); + let page_count = chunk_count.div_ceil(CHUNKS_PER_PAGE as u64); + let mut raw = Vec::new(); + let mut leaves = Vec::new(); + let mut leaf_hashes = Vec::new(); + let mut current = Vec::new(); + let inline = size <= STAGED_CAP_BYTES as u64; + if inline { + raw.try_reserve_exact(size as usize) + .map_err(allocation_error)?; + } + leaves + .try_reserve_exact(page_count as usize) + .map_err(allocation_error)?; + leaf_hashes + .try_reserve_exact(page_count as usize) + .map_err(allocation_error)?; + current + .try_reserve_exact(chunk_count.min(CHUNKS_PER_PAGE as u64) as usize) + .map_err(allocation_error)?; + let mut full_hash = Sha256::new(); + let mut chunk_hash = Sha256::new(); + let mut received = 0u64; + let mut chunk_bytes = 0usize; + let mut digests = 0u64; + let mut since_yield = 0usize; + while let Some(part) = input.next().await { + let bytes = part.map_err(|error| { + tracing::warn!(error = %error, "fixed-view chunk projection stream failed"); + SnapshotError::new( + SnapshotErrorCode::ObjectUnavailable, + "content stream failed", + ) + })?; + if bytes.len() as u64 > size - received { + return Err(length_error()); + } + // A visible-item admission limit cannot bound a producer's backing + // allocation, buffers or work surviving cancellation. + if bytes.len() > STREAM_ITEM_MAX_BYTES { + return Err(SnapshotError::new( + SnapshotErrorCode::TemporaryUnavailable, + "content producer item exceeds the consumer processing limit", + )); + } + full_hash.update(&bytes); + if inline { + raw.extend_from_slice(&bytes); + } + received += bytes.len() as u64; + let mut rest = bytes.as_ref(); + while !rest.is_empty() { + let len = rest.len().min(CHUNK_SIZE as usize - chunk_bytes); + chunk_hash.update(&rest[..len]); + chunk_bytes += len; + rest = &rest[len..]; + if chunk_bytes == CHUNK_SIZE as usize { + current.push(chunk_hash.finalize_reset().into()); + digests += 1; + chunk_bytes = 0; + if current.len() == CHUNKS_PER_PAGE { + append_leaf(&mut leaves, &mut leaf_hashes, std::mem::take(&mut current))?; + if digests < chunk_count { + current + .try_reserve_exact( + (chunk_count - digests).min(CHUNKS_PER_PAGE as u64) as usize + ) + .map_err(allocation_error)?; + } + } + } + } + since_yield += bytes.len(); + if since_yield >= STREAM_ITEM_MAX_BYTES { + // A ready stream still cooperates with cancellation and the + // executor while hashing a large file. No detached CPU worker. + tokio::task::yield_now().await; + since_yield = 0; + } + } + if received != size { + return Err(length_error()); + } + let got: [u8; 32] = full_hash.finalize().into(); + if got != content_id { + return Err(SnapshotError::new( + SnapshotErrorCode::DigestMismatch, + "chunk projection input does not hash to content_id", + )); + } + if chunk_bytes != 0 { + current.push(chunk_hash.finalize().into()); + digests += 1; + } + if !current.is_empty() { + append_leaf(&mut leaves, &mut leaf_hashes, current)?; + } + if digests != chunk_count || leaves.len() as u64 != page_count { + return Err(internal( + "streamed chunk boundaries disagree with file size", + )); + } + let pages_root = mst2_codec::chunkmap::merkle_root(&leaf_hashes).map_err(codec_err)?; + let map = ChunkMap::new(content_id, size, pages_root).map_err(codec_err)?; + let projection = ChunkProjection { + map_id: map.map_id(), + map, + leaves, + leaf_hashes, + raw, + memory: Some(lease), + }; + if projection.allocated_bytes() > projection.retained_bytes() { + return Err(internal("projection allocation exceeds its reservation")); + } + Ok(projection) +} + +fn append_leaf( + leaves: &mut Vec, + hashes: &mut Vec<[u8; 32]>, + chunk_sha256: Vec<[u8; 32]>, +) -> Result<(), SnapshotError> { + let leaf = ChunkLeaf { + page_index: leaves.len() as u64, + chunk_sha256, + }; + hashes.push(leaf.leaf_hash().map_err(codec_err)?); + leaves.push(leaf); + Ok(()) +} + +fn allocation_error(_: std::collections::TryReserveError) -> SnapshotError { + SnapshotError::new( + SnapshotErrorCode::TemporaryUnavailable, + "chunk projection allocation could not be admitted", + ) +} + +fn length_error() -> SnapshotError { + SnapshotError::new( + SnapshotErrorCode::IntegrityError, + "fixed blob length disagrees with its verified size fact", + ) +} + +#[cfg(test)] +#[path = "chunks_stream_tests.rs"] +mod tests; diff --git a/src/ceres/snapshot/chunks_stream_tests.rs b/src/ceres/snapshot/chunks_stream_tests.rs new file mode 100644 index 00000000..2ccebae8 --- /dev/null +++ b/src/ceres/snapshot/chunks_stream_tests.rs @@ -0,0 +1,391 @@ +use std::{ + io, + sync::atomic::{AtomicUsize, Ordering}, + time::Duration, +}; + +use bytes::Bytes; +use tokio::{sync::Notify, time::timeout}; + +use super::*; +use crate::ceres::snapshot::content_budget::MemoryBudget; + +fn stream(parts: Vec>) -> ObjectByteStream { + Box::pin(futures::stream::iter(parts)) +} + +async fn project(raw: &[u8], parts: Vec>) -> ChunkProjection { + let budget = MemoryBudget::new(reserved_bytes(raw.len() as u64).unwrap()); + let lease = budget + .reserve(reserved_bytes(raw.len() as u64).unwrap()) + .unwrap(); + build_stream( + Sha256::digest(raw).into(), + raw.len() as u64, + stream(parts), + lease, + ) + .await + .unwrap() +} + +#[tokio::test] +async fn arbitrary_fragmentation_preserves_canonical_map_leaves_and_full_final_chunk() { + for size in [ + 1, + CHUNK_SIZE as usize - 1, + CHUNK_SIZE as usize, + CHUNK_SIZE as usize + 7, + 2 * CHUNK_SIZE as usize, + ] { + let raw: Vec = (0..size).map(|i| (i % 251) as u8).collect(); + let parts = raw + .chunks(997) + .map(|part| Ok(Bytes::copy_from_slice(part))) + .collect(); + let projection = project(&raw, parts).await; + let oracle = ChunkProjection::build(Sha256::digest(&raw).into(), raw.clone()).unwrap(); + assert_eq!(projection.map, oracle.map); + assert_eq!(projection.map_id, oracle.map_id); + assert_eq!(projection.leaves, oracle.leaves); + assert_eq!(projection.leaf_hashes, oracle.leaf_hashes); + assert!(projection.has_inline_bytes()); + for index in 0..projection.map.chunk_count { + assert_eq!( + projection.chunk_bytes(index).unwrap(), + oracle.chunk_bytes(index).unwrap() + ); + let (leaf, proof) = projection + .leaf_and_proof(index / CHUNKS_PER_PAGE as u64) + .unwrap(); + mst2_codec::chunkmap::verify_leaf( + projection.page_count(), + leaf.page_index, + leaf.leaf_hash().unwrap(), + &proof, + projection.map.pages_root, + ) + .unwrap(); + } + } +} + +#[tokio::test] +async fn large_production_threshold_retains_only_map_and_verifies_every_page() { + // Reuse one producer allocation; never create the >512 MiB raw Vec. + let block = Bytes::from(vec![37; CHUNK_SIZE as usize]); + let size = STAGED_CAP_BYTES as u64 + 7; + let mut full = Sha256::new(); + for _ in 0..512 { + full.update(&block); + } + full.update(&block[..7]); + let content_id: [u8; 32] = full.finalize().into(); + let input = Box::pin(futures::stream::iter((0..513).map(move |index| { + Ok(if index == 512 { + block.slice(..7) + } else { + block.clone() + }) + }))); + let weight = reserved_bytes(size).unwrap(); + let budget = MemoryBudget::new(weight); + let lease = budget.reserve(weight).unwrap(); + let projection = build_stream(content_id, size, input, lease).await.unwrap(); + assert!(!projection.has_inline_bytes()); + assert_eq!(projection.raw.capacity(), 0); + assert_eq!(projection.map.chunk_count, 513); + assert_eq!(projection.page_count(), 3); + assert!(projection.allocated_bytes() < 128 * 1024); + assert_eq!(budget.used(), weight); + let full_chunk: [u8; 32] = Sha256::digest(vec![37; CHUNK_SIZE as usize]).into(); + let last_chunk: [u8; 32] = Sha256::digest([37; 7]).into(); + assert_eq!(projection.leaves[0].chunk_sha256, vec![full_chunk; 256]); + assert_eq!(projection.leaves[1].chunk_sha256, vec![full_chunk; 256]); + assert_eq!(projection.leaves[2].chunk_sha256, vec![last_chunk]); + for page in 0..3 { + let (leaf, proof) = projection.leaf_and_proof(page).unwrap(); + mst2_codec::chunkmap::verify_leaf( + 3, + page, + leaf.leaf_hash().unwrap(), + &proof, + projection.map.pages_root, + ) + .unwrap(); + } + projection.verify_chunk(512, &[37; 7]).unwrap(); + assert_eq!( + projection.verify_chunk(512, &[38; 7]).unwrap_err().code, + SnapshotErrorCode::IntegrityError + ); + drop(projection); + assert_eq!(budget.used(), 0); +} + +#[tokio::test] +async fn exact_eof_wrong_hash_growth_and_late_error_never_publish_a_projection() { + let cases = [ + ( + vec![Ok(Bytes::from_static(b"ab"))], + SnapshotErrorCode::IntegrityError, + ), + ( + vec![Ok(Bytes::from_static(b"abc")), Ok(Bytes::from_static(b"d"))], + SnapshotErrorCode::IntegrityError, + ), + ( + vec![Ok(Bytes::from_static(b"abd"))], + SnapshotErrorCode::DigestMismatch, + ), + ( + vec![ + Ok(Bytes::from_static(b"abc")), + Err(io::Error::other("late")), + ], + SnapshotErrorCode::ObjectUnavailable, + ), + ]; + for (parts, code) in cases { + let budget = MemoryBudget::new(reserved_bytes(3).unwrap()); + let lease = budget.reserve(reserved_bytes(3).unwrap()).unwrap(); + let failure = build_stream(Sha256::digest(b"abc").into(), 3, stream(parts), lease) + .await + .err() + .unwrap(); + assert_eq!(failure.code, code); + assert_eq!(budget.used(), 0); + } +} + +#[tokio::test] +async fn overlong_producer_is_rejected_before_hash_copy_and_tail_poll() { + let tail = Arc::new(AtomicUsize::new(0)); + let input = Box::pin(futures::stream::unfold( + (false, tail.clone()), + |(sent, tail)| async move { + if sent { + tail.fetch_add(1, Ordering::SeqCst); + Some((Err(io::Error::other("must not poll tail")), (true, tail))) + } else { + Some((Ok(Bytes::from_static(b"abcd")), (true, tail))) + } + }, + )); + let budget = MemoryBudget::new(reserved_bytes(3).unwrap()); + let lease = budget.reserve(reserved_bytes(3).unwrap()).unwrap(); + assert_eq!( + build_stream(Sha256::digest(b"abc").into(), 3, input, lease) + .await + .err() + .unwrap() + .code, + SnapshotErrorCode::IntegrityError + ); + assert_eq!(tail.load(Ordering::SeqCst), 0); + assert_eq!(budget.used(), 0); +} + +#[tokio::test] +async fn visible_producer_item_limit_does_not_collect_a_large_item() { + let size = STREAM_ITEM_MAX_BYTES + 1; + let budget = MemoryBudget::new(reserved_bytes(size as u64).unwrap()); + let lease = budget + .reserve(reserved_bytes(size as u64).unwrap()) + .unwrap(); + let input = stream(vec![Ok(Bytes::from(vec![1; size]))]); + assert_eq!( + build_stream([0; 32], size as u64, input, lease) + .await + .err() + .unwrap() + .code, + SnapshotErrorCode::TemporaryUnavailable + ); + assert_eq!(budget.used(), 0); +} + +#[tokio::test] +async fn evicted_projection_remains_charged_until_last_reader_drops() { + let raw = b"owned credits"; + let budget = MemoryBudget::new(reserved_bytes(raw.len() as u64).unwrap()); + let lease = budget + .reserve(reserved_bytes(raw.len() as u64).unwrap()) + .unwrap(); + let projection = Arc::new( + build_stream( + Sha256::digest(raw).into(), + raw.len() as u64, + stream(vec![Ok(Bytes::from_static(raw))]), + lease, + ) + .await + .unwrap(), + ); + let weight = projection.retained_bytes(); + let mut cache = ProjectionCache::new(); + assert!(cache.put(projection.clone()).is_empty()); + let reader = projection.clone(); + drop(projection); + drop(cache.evict_oldest()); + assert_eq!(cache.total_bytes, 0); + assert_eq!(budget.used(), weight); + assert!(budget.reserve(1).is_err()); + drop(reader); + assert_eq!(budget.used(), 0); +} + +struct DropCount(Arc); +impl Drop for DropCount { + fn drop(&mut self) { + self.0.fetch_add(1, Ordering::SeqCst); + } +} + +#[tokio::test] +async fn production_admission_rejects_live_credit_and_builder_overload_before_open() { + let weight = reserved_bytes(3).unwrap(); + let cache = Mutex::new(ProjectionCache::new()); + let budget = MemoryBudget::new(weight); + let builders = Arc::new(Semaphore::new(MAX_BUILDERS)); + let calls = AtomicUsize::new(0); + let held = budget.reserve(weight).unwrap(); + let error = build_with_resources( + Sha256::digest(b"abc").into(), + 3, + || async { + calls.fetch_add(1, Ordering::SeqCst); + Ok(stream(vec![Ok(Bytes::from_static(b"abc"))])) + }, + &budget, + &builders, + &cache, + ) + .await + .err() + .unwrap(); + assert_eq!(error.code, SnapshotErrorCode::TemporaryUnavailable); + assert_eq!(calls.load(Ordering::SeqCst), 0); + assert_eq!(builders.available_permits(), MAX_BUILDERS); + drop(held); + let mut workers = Vec::new(); + for _ in 0..MAX_BUILDERS { + workers.push(builders.clone().try_acquire_owned().unwrap()); + } + let error = build_with_resources( + Sha256::digest(b"abc").into(), + 3, + || async { + calls.fetch_add(1, Ordering::SeqCst); + Ok(stream(vec![Ok(Bytes::from_static(b"abc"))])) + }, + &budget, + &builders, + &cache, + ) + .await + .err() + .unwrap(); + assert_eq!(error.code, SnapshotErrorCode::TemporaryUnavailable); + assert_eq!(calls.load(Ordering::SeqCst), 0); + assert_eq!(budget.used(), 0); + drop(workers); + let projection = build_with_resources( + Sha256::digest(b"abc").into(), + 3, + || async { + calls.fetch_add(1, Ordering::SeqCst); + Ok(stream(vec![Ok(Bytes::from_static(b"abc"))])) + }, + &budget, + &builders, + &cache, + ) + .await + .unwrap(); + assert_eq!(calls.load(Ordering::SeqCst), 1); + assert_eq!(budget.used(), weight); + assert_eq!(builders.available_permits(), MAX_BUILDERS); + drop(projection); + assert_eq!(budget.used(), 0); +} + +#[tokio::test] +async fn cancelled_stream_owner_refunds_after_stream_drop_and_same_flight_waiter_takes_over() { + let raw = Bytes::from(uuid::Uuid::new_v4().as_bytes().to_vec()); + let id: [u8; 32] = Sha256::digest(&raw).into(); + let entered = Arc::new(Notify::new()); + let drops = Arc::new(AtomicUsize::new(0)); + let leader = tokio::spawn({ + let entered = entered.clone(); + let drops = drops.clone(); + async move { + get_or_project_stream(id, 16, || async { + let input: ObjectByteStream = Box::pin(futures::stream::unfold( + (entered, DropCount(drops)), + |(entered, owner)| async move { + entered.notify_one(); + std::future::pending::<()>().await; + Some((Ok(Bytes::new()), (entered, owner))) + }, + )); + Ok(input) + }) + .await + } + }); + timeout(Duration::from_secs(5), entered.notified()) + .await + .unwrap(); + let waiter = tokio::spawn(async move { + get_or_project_stream(id, 16, || async { + let input: ObjectByteStream = stream(vec![Ok(raw)]); + Ok(input) + }) + .await + }); + timeout(Duration::from_secs(5), async { + loop { + if FLIGHTS + .get() + .unwrap() + .lock() + .unwrap() + .entries + .get(&id) + .map_or(0, Weak::strong_count) + == 2 + { + break; + } + tokio::task::yield_now().await; + } + }) + .await + .unwrap(); + leader.abort(); + assert!(leader.await.err().unwrap().is_cancelled()); + assert_eq!(drops.load(Ordering::SeqCst), 1); + let projection = timeout(Duration::from_secs(5), waiter) + .await + .unwrap() + .unwrap() + .unwrap(); + assert_eq!(projection.map.file_content_id, id); + assert_eq!(projection.chunk_bytes(0).unwrap().len(), 16); +} + +#[test] +fn maximum_file_metadata_fits_and_protocol_or_quota_rejection_needs_no_source() { + assert!(reserved_bytes(MAX_FILE_BYTES).unwrap() < 300 * 1024 * 1024); + assert_eq!( + reserved_bytes(0).unwrap_err().code, + SnapshotErrorCode::ScopeInvalid + ); + assert_eq!( + reserved_bytes(MAX_FILE_BYTES + 1).unwrap_err().code, + SnapshotErrorCode::LimitExceeded + ); + assert!(reserved_bytes(STAGED_CAP_BYTES as u64).unwrap() > STAGED_CAP_BYTES); + assert!(reserved_bytes(STAGED_CAP_BYTES as u64 + 1).unwrap() < 128 * 1024); +} diff --git a/src/ceres/snapshot/content_budget.rs b/src/ceres/snapshot/content_budget.rs new file mode 100644 index 00000000..79db0289 --- /dev/null +++ b/src/ceres/snapshot/content_budget.rs @@ -0,0 +1,147 @@ +//! Ownership-based consumer memory admission. Backend allocations and allocator +//! overhead are separate; these credits are not a process RSS guarantee. + +use std::sync::{ + Arc, OnceLock, + atomic::{AtomicUsize, Ordering}, +}; + +use super::error::{SnapshotError, SnapshotErrorCode}; + +pub(crate) const PROJECTION_LIVE_BYTES: usize = 1024 * 1024 * 1024; +const RESPONSE_LIVE_BYTES: usize = 512 * 1024 * 1024; +const RANGE_SCRATCH_BYTES: usize = 128 * 1024 * 1024; +pub(crate) const RANGE_WORK_BYTES: usize = 8 * 1024 * 1024; + +pub(crate) struct MemoryBudget { + limit: usize, + used: AtomicUsize, +} + +impl MemoryBudget { + pub(crate) fn new(limit: usize) -> Arc { + Arc::new(Self { + limit, + used: AtomicUsize::new(0), + }) + } + + pub(crate) fn reserve(self: &Arc, bytes: usize) -> Result { + if bytes > self.limit { + return Err(SnapshotError::new( + SnapshotErrorCode::LimitExceeded, + "content operation exceeds its consumer memory budget", + )); + } + let mut used = self.used.load(Ordering::Acquire); + loop { + if bytes > self.limit - used { + return Err(SnapshotError::new( + SnapshotErrorCode::TemporaryUnavailable, + "content consumer memory budget is occupied", + )); + } + match self.used.compare_exchange_weak( + used, + used + bytes, + Ordering::AcqRel, + Ordering::Acquire, + ) { + Ok(_) => { + return Ok(MemoryLease { + budget: self.clone(), + bytes, + }); + } + Err(actual) => used = actual, + } + } + } + + #[cfg(test)] + pub(crate) fn used(&self) -> usize { + self.used.load(Ordering::Acquire) + } +} + +pub(crate) struct MemoryLease { + budget: Arc, + pub(crate) bytes: usize, +} + +impl Drop for MemoryLease { + fn drop(&mut self) { + self.budget.used.fetch_sub(self.bytes, Ordering::AcqRel); + } +} + +pub(crate) fn projection_budget() -> &'static Arc { + static BUDGET: OnceLock> = OnceLock::new(); + BUDGET.get_or_init(|| MemoryBudget::new(PROJECTION_LIVE_BYTES)) +} + +pub(crate) fn reserve_response(bytes: usize) -> Result { + static BUDGET: OnceLock> = OnceLock::new(); + BUDGET + .get_or_init(|| MemoryBudget::new(RESPONSE_LIVE_BYTES)) + .reserve(bytes) +} + +pub(crate) fn reserve_range_work() -> Result { + static BUDGET: OnceLock> = OnceLock::new(); + BUDGET + .get_or_init(|| MemoryBudget::new(RANGE_SCRATCH_BYTES)) + .reserve(RANGE_WORK_BYTES) +} + +/// Every encoded frame owns shared credit, including after the body passes a +/// frame to a transport that retains or clones its Bytes. +pub(crate) struct BudgetedFrame { + pub(crate) bytes: Vec, + pub(crate) lease: Arc, +} + +impl AsRef<[u8]> for BudgetedFrame { + fn as_ref(&self) -> &[u8] { + &self.bytes + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn frames_retain_credit_until_last_transport_clone_drops() { + let budget = MemoryBudget::new(64); + let lease = Arc::new(budget.reserve(64).unwrap()); + let frame = bytes::Bytes::from_owner(BudgetedFrame { + bytes: vec![7; 32], + lease: lease.clone(), + }); + let transport = frame.clone(); + drop(lease); + drop(frame); + assert_eq!(budget.used(), 64); + assert!(budget.reserve(1).is_err()); + drop(transport); + assert_eq!(budget.used(), 0); + assert!(budget.reserve(64).is_ok()); + } + + #[test] + fn quota_rejects_without_waiting_for_a_callers_own_credits() { + let budget = MemoryBudget::new(12); + let first = budget.reserve(8).unwrap(); + assert_eq!( + budget.reserve(5).err().unwrap().code, + SnapshotErrorCode::TemporaryUnavailable + ); + assert_eq!( + budget.reserve(13).err().unwrap().code, + SnapshotErrorCode::LimitExceeded + ); + drop(first); + assert_eq!(budget.used(), 0); + } +} diff --git a/src/ceres/snapshot/mod.rs b/src/ceres/snapshot/mod.rs index 97a1a50b..532d1da6 100644 --- a/src/ceres/snapshot/mod.rs +++ b/src/ceres/snapshot/mod.rs @@ -4,6 +4,7 @@ //! composition seam; identity, attestation and publication integration remain //! separate gates. Fixed-source readers must never look up current refs. pub mod chunks; +pub(crate) mod content_budget; pub mod descriptor; pub mod error; pub mod frame_stream; diff --git a/src/jupiter/service/git_service.rs b/src/jupiter/service/git_service.rs index bf42b5ee..6569e9a0 100644 --- a/src/jupiter/service/git_service.rs +++ b/src/jupiter/service/git_service.rs @@ -119,6 +119,26 @@ impl GitService { Ok(stream) } + pub async fn get_object_range_stream_exact( + &self, + hash: &str, + start: u64, + end: u64, + ) -> Result, MegaError> { + if !is_full_hex_object_id(hash) { + return Err(MegaError::Other("Invalid object ID format".to_string())); + } + let key = ObjectKey { + namespace: ObjectNamespace::Git, + key: hash.to_string(), + }; + Ok(self + .obj_storage + .inner + .get_range_stream_exact(&key, start, end) + .await?) + } + pub fn get_objects_stream(&self, hashes: Vec) -> MultiObjectByteStream<'_> { // Filter out obviously invalid object ids early to avoid spurious backend requests. // Callers that need strict validation should validate up-front and return 4xx. diff --git a/src/jupiter/storage/object_storage.rs b/src/jupiter/storage/object_storage.rs index f77fbc86..506f2974 100644 --- a/src/jupiter/storage/object_storage.rs +++ b/src/jupiter/storage/object_storage.rs @@ -35,6 +35,51 @@ pub async fn build_object_storage( }) } +#[cfg(test)] +mod exact_range_tests { + use super::*; + use crate::orbit_api::object_storage::ObjectNamespace; + + #[tokio::test] + async fn memory_exact_range_has_no_clipped_or_empty_success() { + let storage = mock_object_storage(); + let key = ObjectKey { + namespace: ObjectNamespace::Git, + key: "abcdef1234567890".to_string(), + }; + storage + .inner + .put_stream( + &key, + Box::pin(futures::stream::iter([Ok(Bytes::from_static(b"abcdef"))])), + ObjectMeta { + size: 6, + ..Default::default() + }, + ) + .await + .unwrap(); + let (mut stream, meta) = storage + .inner + .get_range_stream_exact(&key, 2, 5) + .await + .unwrap() + .unwrap(); + assert_eq!(meta.size, 6); + assert_eq!(stream.next().await.unwrap().unwrap(), b"cde".as_slice()); + assert!(stream.next().await.is_none()); + for (start, end) in [(5, 7), (6, 7), (4, 4), (u64::MAX, u64::MAX)] { + assert!( + storage + .inner + .get_range_stream_exact(&key, start, end) + .await + .is_err() + ); + } + } +} + #[derive(Default)] struct InMemoryObjectStorage { objects: Mutex>, @@ -127,6 +172,32 @@ impl MegaObjectStorage for InMemoryObjectStorage { Ok((Self::stream_bytes(bytes.slice(start..end)), meta)) } + async fn get_range_stream_exact( + &self, + key: &ObjectKey, + start: u64, + end: u64, + ) -> OrbitResult> { + let (bytes, meta) = self + .objects + .lock() + .map_err(|_| IoOrbitError::Other("object storage lock poisoned".to_string()))? + .get(key) + .cloned() + .ok_or_else(|| IoOrbitError::object_store_not_found(key.default_sharding()))?; + let start = + usize::try_from(start).map_err(|_| IoOrbitError::object_store("range overflow"))?; + let end = usize::try_from(end).map_err(|_| IoOrbitError::object_store("range overflow"))?; + if start >= end || end > bytes.len() { + return Err(std::io::Error::new( + std::io::ErrorKind::InvalidData, + "exact object range is outside stored bytes", + ) + .into()); + } + Ok(Some((Self::stream_bytes(bytes.slice(start..end)), meta))) + } + async fn exists(&self, key: &ObjectKey) -> OrbitResult { Ok(self .objects diff --git a/src/orbit/adapter/object.rs b/src/orbit/adapter/object.rs index ca1751b3..58ccacc0 100644 --- a/src/orbit/adapter/object.rs +++ b/src/orbit/adapter/object.rs @@ -106,6 +106,43 @@ impl MegaObjectStorage for ObjectStoreAdapter { Ok((Box::pin(stream), meta)) } + async fn get_range_stream_exact( + &self, + key: &ObjectKey, + start: u64, + end: u64, + ) -> OrbitResult> { + if start >= end { + return Err(std::io::Error::new( + std::io::ErrorKind::InvalidInput, + "invalid exact object range", + ) + .into()); + } + let path = Self::checked_path(key)?; + let res = self + .to_store() + .get_opts( + &path, + object_store::GetOptions { + range: Some(object_store::GetRange::Bounded(start..end)), + ..Default::default() + }, + ) + .await + .map_err(IoOrbitError::from)?; + if res.range != (start..end) || end > res.meta.size { + return Err(std::io::Error::new( + std::io::ErrorKind::InvalidData, + "backend returned a different object range", + ) + .into()); + } + let meta = build_object_meta(&res.meta); + let stream = res.into_stream().map_err(std::io::Error::other); + Ok(Some((Box::pin(stream), meta))) + } + async fn signed_url( &self, key: &ObjectKey, @@ -149,3 +186,55 @@ impl MegaObjectStorage for ObjectStoreAdapter { Ok(()) } } + +#[cfg(test)] +mod exact_range_tests { + use super::*; + + #[tokio::test] + async fn local_exact_range_returns_selected_bytes_and_full_object_size_without_clipping() { + let directory = tempfile::tempdir().unwrap(); + let adapter = ObjectStoreAdapter { + store: BackendStore::Local(Arc::new( + LocalFileSystem::new_with_prefix(directory.path()).unwrap(), + )), + upload_strategy: UploadStrategy::SinglePut, + presign_store: None, + }; + let key = ObjectKey { + namespace: ObjectNamespace::Git, + key: "abcdef1234567890".to_string(), + }; + adapter + .put_stream( + &key, + Box::pin(stream::iter([Ok(Bytes::from_static(b"0123456789"))])), + ObjectMeta::default(), + ) + .await + .unwrap(); + let (input, meta) = adapter + .get_range_stream_exact(&key, 3, 7) + .await + .unwrap() + .unwrap(); + assert_eq!(meta.size, 10); + assert_eq!( + ObjectStoreAdapter::buffer_stream(input, 4).await.unwrap(), + b"3456".as_slice() + ); + let (input, meta) = adapter + .get_range_stream_exact(&key, 9, 10) + .await + .unwrap() + .unwrap(); + assert_eq!(meta.size, 10); + assert_eq!( + ObjectStoreAdapter::buffer_stream(input, 1).await.unwrap(), + b"9".as_slice() + ); + assert!(adapter.get_range_stream_exact(&key, 9, 11).await.is_err()); + assert!(adapter.get_range_stream_exact(&key, 5, 5).await.is_err()); + assert!(adapter.get_range_stream_exact(&key, 10, 11).await.is_err()); + } +} diff --git a/src/orbit_api/object_storage.rs b/src/orbit_api/object_storage.rs index 00e47dda..7fe6f47a 100644 --- a/src/orbit_api/object_storage.rs +++ b/src/orbit_api/object_storage.rs @@ -306,6 +306,20 @@ pub trait MegaObjectStorage: Send + Sync { end: Option, ) -> OrbitResult<(ObjectByteStream, ObjectMeta)>; + /// Exact raw byte range, with no full-download fallback. `None` means + /// unsupported and must be returned without source I/O. Implementations + /// must validate the backend's actual start/end before exposing the stream; + /// metadata describes the complete object. Consumers still verify EOF, + /// length and content hashes. This does not promise bounded producer RSS. + async fn get_range_stream_exact( + &self, + _key: &ObjectKey, + _start: u64, + _end: u64, + ) -> OrbitResult> { + Ok(None) + } + /// Check whether an object exists. async fn exists(&self, key: &ObjectKey) -> OrbitResult; @@ -682,4 +696,22 @@ mod tests { .contains("atomic metadata put is not supported") ); } + + #[tokio::test] + async fn exact_range_default_is_unsupported_without_calling_fallback() { + let store = UnsupportedBoundedStore; + let key = ObjectKey { + namespace: ObjectNamespace::Git, + key: "abcdef".to_string(), + }; + // Its general range getter returns an error; the new default must + // return typed absence without invoking that method or full get. + assert!( + store + .get_range_stream_exact(&key, 1, 2) + .await + .unwrap() + .is_none() + ); + } } From fa47e9080bc1f996082a773fb60893b9a101ed04 Mon Sep 17 00:00:00 2001 From: Xiaoyang Han Date: Thu, 8 Oct 2026 01:06:43 +0800 Subject: [PATCH 20/50] test(mst2): retain storage while updating chunk fixture facts (#70) Bind the fixture MonoStorage Arc before borrowing its database connection across async queries. Fix exact E0716 in the new CHUNK test helper. The whole-file delta is this binding only; all test bodies/assertions and production code remain unchanged. Repaired native validation is pending. --- src/api/router/snapshot_chunks_bounded_tests.rs | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/src/api/router/snapshot_chunks_bounded_tests.rs b/src/api/router/snapshot_chunks_bounded_tests.rs index a32add5c..bbbd14c4 100644 --- a/src/api/router/snapshot_chunks_bounded_tests.rs +++ b/src/api/router/snapshot_chunks_bounded_tests.rs @@ -134,7 +134,8 @@ async fn oid_for(fixture: &Fixture, path: &str) -> String { } async fn set_fact(fixture: &Fixture, oid: &str, size: u64, digest: [u8; 32]) { - let db = fixture.state.storage.mono_storage().get_connection(); + let storage = fixture.state.storage.mono_storage(); + let db = storage.get_connection(); let fact = mst2_verified_object::Entity::find() .filter(mst2_verified_object::Column::GitOid.eq(oid)) .one(db) From ebac4f8abd4c0f9dac06bc67f1b5b466483efdec Mon Sep 17 00:00:00 2001 From: Xiaoyang Han Date: Thu, 8 Oct 2026 01:11:19 +0800 Subject: [PATCH 21/50] test(mst2): repair OBJECT error and storage-route fixtures (#71) Repair five observed regressions in exact #67 native run37647717995 (2374 passed/6 failed/2 ignored). Three OBJECT expectations now use the existing formal EXPECTED_DIGEST_MISMATCH wire code. The half-lease fixture reaches actual failed COMMIT with its completeness trigger enabled and deferred, avoiding PostgreSQL pending-event ALTER; failed commit must restore all trigger modes and route/source/root inventories. The temp-shadow fixture uses legal CASE syntax for its exact identity assertion. Production guards, all integrity assertions and the #70 CHUNK lifetime repair remain unchanged. The sixth old FK expectation was repaired by #68. Repaired native validation remains pending. --- .../router/snapshot_objects_bounded_tests.rs | 6 ++-- .../router/snapshot_storage_route_fixture.rs | 32 ++++++++++++++----- .../router/snapshot_storage_route_tests.rs | 2 +- 3 files changed, 28 insertions(+), 12 deletions(-) diff --git a/src/api/router/snapshot_objects_bounded_tests.rs b/src/api/router/snapshot_objects_bounded_tests.rs index 6475c53a..d001de12 100644 --- a/src/api/router/snapshot_objects_bounded_tests.rs +++ b/src/api/router/snapshot_objects_bounded_tests.rs @@ -264,7 +264,7 @@ async fn every_alias_is_admitted_and_exact_oid_body_is_loaded_once() { .send("POST", "objects", Body::from(request.to_string())) .await, 409, - "DIGEST_MISMATCH", + "EXPECTED_DIGEST_MISMATCH", false, ) .await; @@ -314,7 +314,7 @@ async fn conflicting_sizes_reject_before_io_and_distinct_oids_still_verify_each_ .send("POST", "objects", Body::from(request.to_string())) .await, 409, - "DIGEST_MISMATCH", + "EXPECTED_DIGEST_MISMATCH", false, ) .await; @@ -388,7 +388,7 @@ async fn truncated_wrong_sha_and_late_stream_error_never_produce_200() { ( FaultKind::Parts(vec![Bytes::from(vec![0; 8192])]), 409, - "DIGEST_MISMATCH", + "EXPECTED_DIGEST_MISMATCH", 8192, ), ( diff --git a/src/api/router/snapshot_storage_route_fixture.rs b/src/api/router/snapshot_storage_route_fixture.rs index 751502b1..abc3aa48 100644 --- a/src/api/router/snapshot_storage_route_fixture.rs +++ b/src/api/router/snapshot_storage_route_fixture.rs @@ -65,8 +65,8 @@ pub(super) async fn reject_half_lease_commit_for_test( .await .unwrap(); // Simulate an omitted derived write; all statement, identity and deferred - // completeness guards remain active, and the deriving trigger is restored - // before the actual commit attempt. + // completeness guards remain active. Failed commit rolls back the trigger + // change; pending deferred events forbid ALTER TABLE before commit. txn.execute_unprepared( "ALTER TABLE mst2_snapshot_lease DISABLE TRIGGER mst2_route_lease_insert", ) @@ -91,12 +91,28 @@ pub(super) async fn reject_half_lease_commit_for_test( .unwrap(); assert_eq!(half.try_get::("", "actual").unwrap(), 1); assert_eq!(half.try_get::("", "routes").unwrap(), 0); - txn.execute_unprepared( - "ALTER TABLE mst2_snapshot_lease ENABLE TRIGGER mst2_route_lease_insert", - ) - .await - .unwrap(); - assert_eq!(trigger_modes(&txn).await, before); + let paused = before.replacen( + "mst2_snapshot_lease:mst2_route_lease_insert:O", + "mst2_snapshot_lease:mst2_route_lease_insert:D", + 1, + ); + assert_ne!(paused, before); + assert_eq!(trigger_modes(&txn).await, paused); + let complete = txn + .query_one_raw(Statement::from_string( + DbBackend::Postgres, + "SELECT t.tgenabled::text AS mode,t.tgdeferrable AS deferrable,t.tginitdeferred AS deferred + FROM pg_catalog.pg_trigger t WHERE t.tgrelid='mst2_snapshot_lease'::regclass + AND t.tgname='mst2_route_complete' AND NOT t.tgisinternal", + )) + .await + .unwrap() + .unwrap(); + let mode: String = complete.try_get("", "mode").unwrap(); + assert!(matches!(mode.as_str(), "O" | "A")); + assert!(before.contains(&format!("mst2_snapshot_lease:mst2_route_complete:{mode}"))); + assert!(complete.try_get::("", "deferrable").unwrap()); + assert!(complete.try_get::("", "deferred").unwrap()); let rejected = txn.commit().await.unwrap_err(); assert!( rejected diff --git a/src/api/router/snapshot_storage_route_tests.rs b/src/api/router/snapshot_storage_route_tests.rs index bd8df6f9..25d40ec3 100644 --- a/src/api/router/snapshot_storage_route_tests.rs +++ b/src/api/router/snapshot_storage_route_tests.rs @@ -1038,7 +1038,7 @@ async fn mst2_generic_storage_routes_temp_prepare_shadow_rejects_qualified_conte scalar(&txn, "SELECT count(*) FROM mst2_metadata_prepare").await, 0 ); - assert_eq!(scalar(&txn, "SELECT (to_regclass('mst2_metadata_prepare')='pg_temp.mst2_metadata_prepare'::regclass)::bigint").await, 1); + assert_eq!(scalar(&txn, "SELECT CASE WHEN to_regclass('mst2_metadata_prepare')='pg_temp.mst2_metadata_prepare'::regclass THEN 1::bigint ELSE 0::bigint END").await, 1); let rejected = txn.execute_raw(statement(&format!( "INSERT INTO {schema}.mst2_snapshot_context(snapshot_id,canonical_descriptor,instance_id,commit_oid, root_tree_oid,metadata_root,prepare_id,publication_sequence,writer_epoch,certificate_receipt_id,authorization_epoch,state) From b2e42c51a3e9b3a53e5b0d163a360493dfc8db04 Mon Sep 17 00:00:00 2001 From: Xiaoyang Han Date: Thu, 8 Oct 2026 02:01:55 +0800 Subject: [PATCH 22/50] perf(mst2): omit existing metadata payloads in native installation (#72) The actual native session installer classifies each bounded batch using the existing exact-member query, then encodes and submits only missing metadata payloads in the same transaction. Fully reused batches skip both INSERT and byte-comparison statements; mixed batches retain exact comparisons for newly submitted bytes. Primary/schema/state/seal/member, codec/size/current-lifetime/domain/graph/tombstone checks remain. Presence only omits writes: local input hashing/decoding, the complete plan, final stored-byte/canonical/DAG/graph validation and restart recovery remain. COMMITTED replay keeps complete validation and requested-byte comparison and never inserts or repairs. Batch work records encoded/omitted bytes and specified statements; classification_batches is an embedded count, not a SQL trip, and these counters exclude barrier and full-oracle work. Ten PostgreSQL regressions include the real install wrapper, mixed/full reuse INSERT traps, corruption, rollback, replay and retention Lease-root release/GC fences. Fixtures cover installation, not publication/open or the HTTP lease ledger. No new schema/Q serving or wire/profile compatibility. Full proof remains O(total metadata); no measured latency improvement is claimed. --- .../storage/native_metadata_install.rs | 2 +- .../native_metadata_install_capability.rs | 200 +++- ...ative_metadata_install_capability_tests.rs | 3 + .../native_metadata_install_missing_tests.rs | 902 ++++++++++++++++++ .../storage/native_snapshot_session.rs | 44 +- 5 files changed, 1144 insertions(+), 7 deletions(-) create mode 100644 src/jupiter/storage/native_metadata_install_missing_tests.rs diff --git a/src/jupiter/storage/native_metadata_install.rs b/src/jupiter/storage/native_metadata_install.rs index 6a1c8500..aeddd315 100644 --- a/src/jupiter/storage/native_metadata_install.rs +++ b/src/jupiter/storage/native_metadata_install.rs @@ -1056,7 +1056,7 @@ fn unavailable(message: &str) -> SnapshotError { #[path = "native_metadata_install_capability.rs"] mod capability; -pub(crate) use capability::ValidatedLegacyInstallCapability; +pub(crate) use capability::{LegacyPayloadInstallWork, ValidatedLegacyInstallCapability}; #[cfg(test)] #[path = "native_metadata_install_capability_tests.rs"] diff --git a/src/jupiter/storage/native_metadata_install_capability.rs b/src/jupiter/storage/native_metadata_install_capability.rs index 98a08695..7b912095 100644 --- a/src/jupiter/storage/native_metadata_install_capability.rs +++ b/src/jupiter/storage/native_metadata_install_capability.rs @@ -26,6 +26,50 @@ pub(crate) struct ValidatedLegacyInstallCapability { install_seal: [u8; 32], } +/// Payload batch work only, excluding barriers, intent/capability setup, +/// finalize and the complete COMMITTED replay oracle. Classification reuses +/// the requested-member query; its batch count is not an extra SQL trip. +#[derive(Debug, Clone, Default, PartialEq, Eq)] +pub(crate) struct LegacyPayloadInstallWork { + pub requested_pages: u64, + pub payload_pages_omitted: u64, + pub payload_pages_encoded: u64, + pub requested_payload_bytes_validated: u64, + pub payload_bytes_omitted: u64, + pub payload_bytes_encoded: u64, + pub metadata_parameter_bytes: u64, + pub payload_parameter_bytes: u64, + pub transactions: u64, + pub registration_queries: u64, + pub requested_member_queries: u64, + pub classification_batches: u64, + pub insert_statements: u64, + pub byte_comparison_queries: u64, + pub committed_replay_pages: u64, + pub elapsed_micros: u64, +} + +impl LegacyPayloadInstallWork { + pub(crate) fn record(&mut self, batch: Self) { + self.requested_pages += batch.requested_pages; + self.payload_pages_omitted += batch.payload_pages_omitted; + self.payload_pages_encoded += batch.payload_pages_encoded; + self.requested_payload_bytes_validated += batch.requested_payload_bytes_validated; + self.payload_bytes_omitted += batch.payload_bytes_omitted; + self.payload_bytes_encoded += batch.payload_bytes_encoded; + self.metadata_parameter_bytes += batch.metadata_parameter_bytes; + self.payload_parameter_bytes += batch.payload_parameter_bytes; + self.transactions += batch.transactions; + self.registration_queries += batch.registration_queries; + self.requested_member_queries += batch.requested_member_queries; + self.classification_batches += batch.classification_batches; + self.insert_statements += batch.insert_statements; + self.byte_comparison_queries += batch.byte_comparison_queries; + self.committed_replay_pages += batch.committed_replay_pages; + self.elapsed_micros += batch.elapsed_micros; + } +} + impl PostgresMetadataInstallRepository { pub(crate) async fn mint_legacy_install_capability( &self, @@ -173,6 +217,144 @@ impl PostgresMetadataInstallRepository { .await } + /// Classify and install in one existing payload transaction. Presence is + /// only a write hint; finalize still reads and validates every stored byte. + pub(crate) async fn install_missing_pages_validated( + &self, + capability: &ValidatedLegacyInstallCapability, + payloads: &[MetadataPagePayload], + ) -> Result { + let started = std::time::Instant::now(); + if capability.scope != self.storage_scope { + return Err( + integrity("install capability belongs to another captured primary scope").into(), + ); + } + if payloads.is_empty() || payloads.len() > 64 { + return Err(SnapshotError::new( + SnapshotErrorCode::LimitExceeded, + "metadata installation batch must contain 1..=64 pages", + ) + .into()); + } + let mut ids = BTreeSet::new(); + for payload in payloads { + validate_payload(payload)?; + if !ids.insert(payload.id) { + return Err(integrity("duplicate page in metadata installation batch").into()); + } + if capability + .members + .get(&payload.id) + .map(|member| member.0 as u64) + != Some(payload.size) + { + return Err(integrity( + "metadata payload is not a member of its validated installation", + ) + .into()); + } + } + let metadata = serde_json::to_string( + &payloads + .iter() + .map(|page| json!({"page_id":hex::encode(page.id),"size":page.size})) + .collect::>(), + ) + .map_err(internal)?; + let txn = self.transaction().await?; + let result = async { + self.capability_barrier(&txn).await?; + let state = read_registered_prepare(&txn, capability).await?; + let rows = read_requested_members(&txn, capability, &metadata, payloads.len()).await?; + let mut present = BTreeSet::new(); + for row in rows { + if !row.try_get::("", "payload_present").map_err(internal)? { + continue; + } + if row.try_get::>("", "payload_codec").map_err(internal)? + != Some(capability.identity.metadata_codec as i16) + || row.try_get::>("", "payload_size").map_err(internal)? + != row.try_get::>("", "expected_size").map_err(internal)? + { + return Err(integrity("immutable metadata payload profile conflicts with its fixed member")); + } + let page: Vec = row.try_get("", "page_id").map_err(internal)?; + present.insert(<[u8; 32]>::try_from(page.as_slice()).map_err(internal)?); + } + let committed = state == "COMMITTED"; + if committed { + // A committed replay never repairs missing or corrupt bytes. + let stored = require_plan(&txn, &capability.intent).await?; + load_installed_dag(&txn, &stored).await?; + check_payload_coverage(&txn, &stored).await?; + verify_graph(&txn, &stored).await?; + } + let submitted: Vec<_> = payloads + .iter() + .filter(|page| committed || !present.contains(&page.id)) + .collect(); + let requested_bytes = payloads.iter().map(|page| page.size).sum::(); + let submitted_bytes = submitted.iter().map(|page| page.size).sum::(); + let mut work = LegacyPayloadInstallWork { + requested_pages: payloads.len() as u64, + payload_pages_omitted: (payloads.len() - submitted.len()) as u64, + payload_pages_encoded: submitted.len() as u64, + requested_payload_bytes_validated: requested_bytes, + payload_bytes_omitted: requested_bytes - submitted_bytes, + payload_bytes_encoded: submitted_bytes, + metadata_parameter_bytes: metadata.len() as u64, + transactions: 1, + registration_queries: 1, + requested_member_queries: 1, + classification_batches: u64::from(!committed), + committed_replay_pages: if committed { payloads.len() as u64 } else { 0 }, + ..LegacyPayloadInstallWork::default() + }; + if !submitted.is_empty() { + let pages: Vec<_> = submitted.iter().map(|page| json!({ + "page_id":hex::encode(page.id),"size":page.size,"payload":hex::encode(&page.bytes) + })).collect(); + let encoded = serde_json::to_string(&pages).map_err(internal)?; + let encoded_len = encoded.len() as u64; + if !committed { + txn.execute_raw(statement( + "INSERT INTO mst2_metadata_payload(page_id,metadata_codec,byte_size,payload) + SELECT decode(p.page_id,'hex'),$1,p.size,decode(p.payload,'hex') + FROM jsonb_to_recordset($2::jsonb) AS p(page_id text,size integer,payload text) + ON CONFLICT(page_id) DO NOTHING", + [(capability.identity.metadata_codec as i16).into(),encoded.clone().into()], + )).await.map_err(internal)?; + work.insert_statements = 1; + work.payload_parameter_bytes += encoded_len; + } + let bad = txn.query_one_raw(statement( + "SELECT p.page_id FROM jsonb_to_recordset($2::jsonb) AS p(page_id text,size integer,payload text) + LEFT JOIN mst2_metadata_payload b ON b.page_id=decode(p.page_id,'hex') + WHERE b.page_id IS NULL OR b.metadata_codec<>$1 OR b.byte_size<>p.size + OR b.payload<>decode(p.payload,'hex') LIMIT 1", + [(capability.identity.metadata_codec as i16).into(),encoded.into()], + )).await.map_err(internal)?; + work.byte_comparison_queries = 1; + work.payload_parameter_bytes += encoded_len; + if bad.is_some() { + return Err(integrity("immutable metadata payload identity conflicts with stored bytes")); + } + } + Ok(work) + }.await; + let mut work = commit( + txn, + result, + &capability.intent.operation_id, + capability.intent.manifest_digest, + MetadataCommitPhase::Payload, + ) + .await?; + work.elapsed_micros = started.elapsed().as_micros().min(u64::MAX as u128) as u64; + Ok(work) + } + pub(super) async fn capability_barrier( &self, txn: &DatabaseTransaction, @@ -302,9 +484,21 @@ async fn check_requested_members( encoded: &str, expected_count: usize, ) -> Result<(), SnapshotError> { + read_requested_members(txn, capability, encoded, expected_count) + .await + .map(|_| ()) +} + +async fn read_requested_members( + txn: &DatabaseTransaction, + capability: &ValidatedLegacyInstallCapability, + encoded: &str, + expected_count: usize, +) -> Result, SnapshotError> { let rows=txn.query_all_raw(statement( "SELECT decode(p.page_id,'hex') AS page_id,m.expected_size,m.generation AS member_generation, b.page_id IS NOT NULL AS payload_present,b.generation AS payload_generation, + b.metadata_codec AS payload_codec,b.byte_size AS payload_size, c.generation AS current_generation,l.generation AS lifetime_generation,l.graph_domain, l.state AS lifetime_state,l.metadata_codec AS lifetime_codec,l.expected_size AS lifetime_size, n.state AS graph_state,n.kind AS graph_kind,n.bytes AS graph_bytes, @@ -321,7 +515,7 @@ async fn check_requested_members( if rows.len() != expected_count { return Err(integrity("requested install membership is incomplete")); } - for row in rows { + for row in &rows { let page: Vec = row.try_get("", "page_id").map_err(internal)?; let page: [u8; 32] = page.as_slice().try_into().map_err(internal)?; let size = row @@ -336,12 +530,12 @@ async fn check_requested_members( )); } check_physical_member( - &row, + row, capability.identity.metadata_codec as i16, size.ok_or_else(|| integrity("requested member is missing"))?, )?; } - Ok(()) + Ok(rows) } fn check_physical_member(row: &QueryResult, codec: i16, size: i32) -> Result<(), SnapshotError> { diff --git a/src/jupiter/storage/native_metadata_install_capability_tests.rs b/src/jupiter/storage/native_metadata_install_capability_tests.rs index 9cbc21e6..bb760c08 100644 --- a/src/jupiter/storage/native_metadata_install_capability_tests.rs +++ b/src/jupiter/storage/native_metadata_install_capability_tests.rs @@ -961,3 +961,6 @@ async fn install_capability_batch_fault_rolls_back_and_raw_writer_waits_before_r ); install(&repository, &cap, &pages).await; } + +#[path = "native_metadata_install_missing_tests.rs"] +mod missing; diff --git a/src/jupiter/storage/native_metadata_install_missing_tests.rs b/src/jupiter/storage/native_metadata_install_missing_tests.rs new file mode 100644 index 00000000..eb7b7885 --- /dev/null +++ b/src/jupiter/storage/native_metadata_install_missing_tests.rs @@ -0,0 +1,902 @@ +use mst2_codec::descriptor::ServingDescriptor; + +use super::*; +use crate::{ + ceres::snapshot::descriptor::BuiltDescriptor, + jupiter::storage::{ + mst2_retention::GcClaim, native_snapshot_session::PostgresNativeSessionRepository, + }, +}; + +fn descriptor(pages: &PreparedNativeMetadataRetention, view: u8) -> BuiltDescriptor { + let descriptor = ServingDescriptor { + instance_uuid: [7; 16], + namespace_view_id: [view; 32], + scope: pages.scope().into(), + metadata_root: pages.dag().root(), + }; + BuiltDescriptor { + instance_id: uuid::Uuid::from_bytes(descriptor.instance_uuid).to_string(), + snapshot_id: format!("sha256:{}", hex::encode(descriptor.snapshot_id().unwrap())), + metadata_root: format!("sha256:{}", hex::encode(descriptor.metadata_root)), + descriptor, + } +} + +fn assert_preparing_work( + work: &LegacyPayloadInstallWork, + pages: &[MetadataPagePayload], + existing: &BTreeSet<[u8; 32]>, +) { + let omitted: Vec<_> = pages + .iter() + .filter(|page| existing.contains(&page.id)) + .collect(); + let requested_bytes = pages.iter().map(|page| page.size).sum::(); + let omitted_bytes = omitted.iter().map(|page| page.size).sum::(); + assert_eq!(work.requested_pages, pages.len() as u64); + assert_eq!(work.payload_pages_omitted, omitted.len() as u64); + assert_eq!( + work.payload_pages_encoded, + (pages.len() - omitted.len()) as u64 + ); + assert_eq!(work.requested_payload_bytes_validated, requested_bytes); + assert_eq!(work.payload_bytes_omitted, omitted_bytes); + assert_eq!(work.payload_bytes_encoded, requested_bytes - omitted_bytes); + assert_eq!(work.committed_replay_pages, 0); + assert_eq!(work.transactions, pages.chunks(64).len() as u64); + assert_eq!(work.registration_queries, work.transactions); + assert_eq!(work.requested_member_queries, work.transactions); + assert_eq!(work.classification_batches, work.transactions); + assert!(work.metadata_parameter_bytes > 0); + let missing_batches = pages + .chunks(64) + .filter(|batch| batch.iter().any(|page| !existing.contains(&page.id))) + .count() as u64; + assert_eq!(work.insert_statements, missing_batches); + assert_eq!(work.byte_comparison_queries, missing_batches); + if missing_batches == 0 { + assert_eq!(work.payload_parameter_bytes, 0); + } else { + assert!(work.payload_parameter_bytes >= 4 * work.payload_bytes_encoded); + } +} + +async fn install_missing( + repository: &PostgresMetadataInstallRepository, + cap: &ValidatedLegacyInstallCapability, + pages: &PreparedNativeMetadataRetention, +) -> LegacyPayloadInstallWork { + let mut work = LegacyPayloadInstallWork::default(); + for batch in pages.dag().payloads().chunks(64) { + work.record( + repository + .install_missing_pages_validated(cap, batch) + .await + .unwrap(), + ); + } + work +} + +async fn payload_modes(db: &DatabaseConnection) -> Vec<(String, String)> { + db.query_all_raw(statement( + "SELECT tgname,tgenabled::text AS mode FROM pg_catalog.pg_trigger + WHERE tgrelid='mst2_metadata_payload'::regclass AND NOT tgisinternal ORDER BY tgname", + [], + )) + .await + .unwrap() + .into_iter() + .map(|row| { + ( + row.try_get("", "tgname").unwrap(), + row.try_get("", "mode").unwrap(), + ) + }) + .collect() +} + +enum PayloadDamage { + Bytes(Vec), + Remove, + UnbindGeneration, +} + +async fn damage_payload( + db: &DatabaseConnection, + page: &MetadataPagePayload, + damage: PayloadDamage, +) { + let modes = payload_modes(db).await; + assert!(modes.iter().all(|(_, mode)| mode == "O")); + assert!( + modes + .iter() + .any(|(name, _)| name == "mst2_metadata_payload_fenced") + ); + assert!( + modes + .iter() + .any(|(name, _)| name == "mst2_metadata_payload_removed") + ); + assert!( + db.execute_raw(statement( + "UPDATE mst2_metadata_payload SET payload=payload WHERE page_id=$1", + [page.id.to_vec().into()], + )) + .await + .is_err() + ); + let txn = db.begin().await.unwrap(); + txn.execute_raw(statement( + "SELECT pg_advisory_xact_lock($1,hashtext(current_schema()))", + [RETENTION_LOCK_KEY.into()], + )) + .await + .unwrap(); + txn.execute_unprepared( + "ALTER TABLE mst2_metadata_payload DISABLE TRIGGER mst2_metadata_payload_fenced", + ) + .await + .unwrap(); + let removed = matches!(&damage, PayloadDamage::Remove); + if removed { + txn.execute_unprepared( + "ALTER TABLE mst2_metadata_payload DISABLE TRIGGER mst2_metadata_payload_removed", + ) + .await + .unwrap(); + } + let result = match damage { + PayloadDamage::Bytes(bytes) => { + assert_eq!(bytes.len(), page.bytes.len()); + txn.execute_raw(statement( + "UPDATE mst2_metadata_payload SET payload=$2 WHERE page_id=$1", + [page.id.to_vec().into(), bytes.into()], + )) + .await + .unwrap() + } + PayloadDamage::Remove => txn + .execute_raw(statement( + "DELETE FROM mst2_metadata_payload WHERE page_id=$1", + [page.id.to_vec().into()], + )) + .await + .unwrap(), + PayloadDamage::UnbindGeneration => txn + .execute_raw(statement( + "UPDATE mst2_metadata_payload SET generation=NULL WHERE page_id=$1", + [page.id.to_vec().into()], + )) + .await + .unwrap(), + }; + assert_eq!(result.rows_affected(), 1); + if removed { + txn.execute_unprepared( + "ALTER TABLE mst2_metadata_payload ENABLE TRIGGER mst2_metadata_payload_removed", + ) + .await + .unwrap(); + } + txn.execute_unprepared( + "ALTER TABLE mst2_metadata_payload ENABLE TRIGGER mst2_metadata_payload_fenced", + ) + .await + .unwrap(); + txn.commit().await.unwrap(); + assert_eq!(payload_modes(db).await, modes); + assert!( + db.execute_unprepared("UPDATE mst2_metadata_payload SET payload=payload",) + .await + .is_err(), + "fault injection must restore the production UPDATE fence" + ); +} + +fn corrupt(page: &MetadataPagePayload) -> Vec { + let mut bytes = page.bytes.clone(); + let last = bytes.len() - 1; + bytes[last] ^= 1; + bytes +} + +#[tokio::test] +async fn install_missing_actual_session_new_sid_mixed_reuse_and_restart_keep_complete_oracles() { + let (first, second, _schema) = fixture().await; + let sessions = PostgresNativeSessionRepository::new(first.clone()); + let old = prepared(130); + let (old_receipt, cold) = sessions + .install_with_work(&descriptor(&old, 1), &old) + .await + .unwrap(); + assert_preparing_work(&cold, old.dag().payloads(), &BTreeSet::new()); + assert_eq!(old_receipt.payload_bytes(), old.dag().payload_bytes()); + let existing: BTreeSet<_> = old.dag().payloads().iter().map(|page| page.id).collect(); + let new = prepared(131); + assert_ne!(old.dag().root(), new.dag().root()); + assert!( + new.dag() + .payloads() + .iter() + .any(|page| existing.contains(&page.id)) + ); + assert!( + new.dag() + .payloads() + .iter() + .any(|page| !existing.contains(&page.id)) + ); + first.execute_unprepared( + "CREATE FUNCTION test_missing_mixed_existing_trap() RETURNS trigger LANGUAGE plpgsql AS $$ + BEGIN IF EXISTS(SELECT 1 FROM mst2_metadata_payload WHERE page_id=NEW.page_id) THEN + RAISE EXCEPTION 'mixed reuse attempted existing payload INSERT'; END IF; RETURN NEW; END $$; + CREATE TRIGGER test_missing_mixed_existing_trap BEFORE INSERT ON mst2_metadata_payload + FOR EACH ROW EXECUTE FUNCTION test_missing_mixed_existing_trap()", + ).await.unwrap(); + let old_state = domain_state_for_test(&first).await; + assert!(first.execute_unprepared( + "INSERT INTO mst2_metadata_payload(page_id,metadata_codec,byte_size,payload) + SELECT page_id,metadata_codec,byte_size,payload FROM mst2_metadata_payload ORDER BY page_id LIMIT 1 + ON CONFLICT(page_id) DO NOTHING", + ).await.unwrap_err().to_string().contains("mixed reuse attempted existing payload INSERT")); + assert_eq!(domain_state_for_test(&first).await, old_state); + let (new_receipt, mixed) = sessions + .install_with_work(&descriptor(&new, 2), &new) + .await + .unwrap(); + assert_preparing_work(&mixed, new.dag().payloads(), &existing); + assert_ne!( + old_receipt.intent().prepare_id(), + new_receipt.intent().prepare_id() + ); + let union: BTreeSet<_> = existing + .iter() + .copied() + .chain(new.dag().payloads().iter().map(|page| page.id)) + .collect(); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_payload").await, + union.len() as i64 + ); + first + .execute_unprepared( + "DROP TRIGGER test_missing_mixed_existing_trap ON mst2_metadata_payload; + DROP FUNCTION test_missing_mixed_existing_trap()", + ) + .await + .unwrap(); + first + .execute_unprepared( + "CREATE FUNCTION test_missing_no_insert() RETURNS trigger LANGUAGE plpgsql AS $$ + BEGIN RAISE EXCEPTION 'full reuse attempted payload INSERT'; END $$; + CREATE TRIGGER test_missing_no_insert BEFORE INSERT ON mst2_metadata_payload + FOR EACH STATEMENT EXECUTE FUNCTION test_missing_no_insert()", + ) + .await + .unwrap(); + drop(sessions); + let restarted = PostgresNativeSessionRepository::new(second.clone()); + let (reused_receipt, reused) = restarted + .install_with_work(&descriptor(&new, 3), &new) + .await + .unwrap(); + assert_preparing_work(&reused, new.dag().payloads(), &union); + assert_ne!( + new_receipt.intent().prepare_id(), + reused_receipt.intent().prepare_id() + ); + assert_eq!(reused_receipt.metadata_root(), new.dag().root()); + let installer = PostgresMetadataInstallRepository::new(second) + .await + .unwrap(); + installer + .restore_session_dag(old_receipt.intent().prepare_id()) + .await + .unwrap(); + installer + .restore_session_dag(new_receipt.intent().prepare_id()) + .await + .unwrap(); + installer + .restore_session_dag(reused_receipt.intent().prepare_id()) + .await + .unwrap(); + let cap = installer + .mint_legacy_install_capability(reused_receipt.intent()) + .await + .unwrap(); + installer + .install_missing_pages_validated(&cap, &new.dag().payloads()[..64]) + .await + .unwrap(); + let wrapper_receipt = restarted.install(&descriptor(&new, 4), &new).await.unwrap(); + assert_ne!( + wrapper_receipt.intent().prepare_id(), + reused_receipt.intent().prepare_id() + ); + assert_eq!(wrapper_receipt.metadata_root(), new.dag().root()); + installer + .restore_session_dag(wrapper_receipt.intent().prepare_id()) + .await + .unwrap(); + first.execute_unprepared( + "DROP TRIGGER test_missing_no_insert ON mst2_metadata_payload; DROP FUNCTION test_missing_no_insert()", + ).await.unwrap(); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_prepare WHERE state='COMMITTED'" + ) + .await, + 4 + ); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_retention_node WHERE state='LIVE'" + ) + .await, + union.len() as i64 + ); +} + +#[tokio::test] +async fn install_missing_sql_statement_trap_proves_full_reuse_skips_conflict_insert() { + let (first, _second, _schema) = fixture().await; + let repository = PostgresMetadataInstallRepository::new(first.clone()) + .await + .unwrap(); + let pages = prepared(2); + let intent = repository + .begin_intent("reuse-statement", &pages) + .await + .unwrap(); + let cap = repository + .mint_legacy_install_capability(&intent) + .await + .unwrap(); + install(&repository, &cap, &pages).await; + first + .execute_unprepared( + "CREATE FUNCTION test_missing_insert_trap() RETURNS trigger LANGUAGE plpgsql AS $$ + BEGIN RAISE EXCEPTION 'payload statement trap fired'; END $$; + CREATE TRIGGER test_missing_insert_trap BEFORE INSERT ON mst2_metadata_payload + FOR EACH STATEMENT EXECUTE FUNCTION test_missing_insert_trap()", + ) + .await + .unwrap(); + let before = domain_state_for_test(&first).await; + assert!( + repository + .install_pages_validated(&cap, pages.dag().payloads()) + .await + .unwrap_err() + .to_string() + .contains("payload statement trap fired") + ); + assert_eq!(domain_state_for_test(&first).await, before); + let work = install_missing(&repository, &cap, &pages).await; + let existing = pages.dag().payloads().iter().map(|page| page.id).collect(); + assert_preparing_work(&work, pages.dag().payloads(), &existing); + assert_eq!(domain_state_for_test(&first).await, before); + let receipt = repository.finalize(&intent).await.unwrap(); + let replay = repository + .install_missing_pages_validated(&cap, pages.dag().payloads()) + .await + .unwrap(); + assert_eq!( + replay.committed_replay_pages, + pages.dag().payloads().len() as u64 + ); + assert_eq!(replay.payload_pages_omitted, 0); + assert_eq!(replay.payload_pages_encoded, replay.requested_pages); + assert_eq!(replay.payload_bytes_encoded, pages.dag().payload_bytes()); + assert_eq!(replay.payload_bytes_omitted, 0); + assert_eq!(replay.classification_batches, 0); + assert_eq!(replay.insert_statements, 0); + assert_eq!(replay.byte_comparison_queries, 1); + assert!(replay.payload_parameter_bytes >= 2 * replay.payload_bytes_encoded); + assert_eq!(repository.finalize(&intent).await.unwrap(), receipt); + first.execute_unprepared( + "DROP TRIGGER test_missing_insert_trap ON mst2_metadata_payload; DROP FUNCTION test_missing_insert_trap()", + ).await.unwrap(); +} + +#[tokio::test] +async fn install_missing_invalid_batch_and_foreign_scope_cannot_write_any_member() { + let (first, _second, _schema) = fixture().await; + let (alien, _other, _alien_schema) = fixture().await; + let repository = PostgresMetadataInstallRepository::new(first.clone()) + .await + .unwrap(); + let foreign = PostgresMetadataInstallRepository::new(alien.clone()) + .await + .unwrap(); + let pages = prepared(70); + let intent = repository + .begin_intent("missing-bounds", &pages) + .await + .unwrap(); + let cap = repository + .mint_legacy_install_capability(&intent) + .await + .unwrap(); + let member = pages.dag().payloads()[0].clone(); + let mut bad_size = member.clone(); + bad_size.size += 1; + let mut bad_digest = member.clone(); + bad_digest.bytes = corrupt(&member); + let malformed = MetadataPagePayload { + id: page_id(&[0; HEADER_LEN]), + size: HEADER_LEN as u64, + bytes: vec![0; HEADER_LEN], + }; + let too_big = MetadataPagePayload { + id: page_id(&vec![0; PAGE_MAX_BYTES + 1]), + size: (PAGE_MAX_BYTES + 1) as u64, + bytes: vec![0; PAGE_MAX_BYTES + 1], + }; + let other = prepared(71); + let stranger = other + .dag() + .payloads() + .iter() + .find(|page| { + !pages + .dag() + .payloads() + .iter() + .any(|member| member.id == page.id) + }) + .unwrap() + .clone(); + let before = domain_state_for_test(&first).await; + for batch in [ + vec![], + pages.dag().payloads()[..65].to_vec(), + vec![member.clone(), member.clone()], + vec![member.clone(), bad_size], + vec![member.clone(), bad_digest], + vec![member.clone(), malformed], + vec![member.clone(), too_big], + vec![member, stranger], + ] { + assert!( + repository + .install_missing_pages_validated(&cap, &batch) + .await + .is_err() + ); + assert_eq!(domain_state_for_test(&first).await, before); + } + let foreign_before = domain_state_for_test(&alien).await; + assert!( + foreign + .install_missing_pages_validated(&cap, &pages.dag().payloads()[..64]) + .await + .is_err() + ); + assert_eq!(domain_state_for_test(&alien).await, foreign_before); + assert_eq!(domain_state_for_test(&first).await, before); +} + +#[tokio::test] +async fn install_missing_actual_size_conflict_rejects_before_inserting_missing_siblings() { + let (first, _second, _schema) = fixture().await; + let repository = PostgresMetadataInstallRepository::new(first.clone()) + .await + .unwrap(); + let pages = prepared(2); + let intent = repository + .begin_intent("actual-size", &pages) + .await + .unwrap(); + let cap = repository + .mint_legacy_install_capability(&intent) + .await + .unwrap(); + let page = &pages.dag().payloads()[0]; + let mut wrong = page.bytes.clone(); + wrong.push(0); + assert!(wrong.len() <= PAGE_MAX_BYTES); + first.execute_raw(statement( + "INSERT INTO mst2_metadata_payload(page_id,metadata_codec,byte_size,payload) VALUES($1,1,$2,$3)", + [page.id.to_vec().into(), (wrong.len() as i32).into(), wrong.into()], + )).await.unwrap(); + let before = domain_state_for_test(&first).await; + assert!( + repository + .install_missing_pages_validated(&cap, pages.dag().payloads()) + .await + .unwrap_err() + .to_string() + .contains("payload profile conflicts") + ); + assert_eq!(domain_state_for_test(&first).await, before); + assert!(repository.finalize(&intent).await.is_err()); + assert_eq!(domain_state_for_test(&first).await, before); +} + +#[tokio::test] +async fn install_missing_actual_codec_check_and_after_insert_fault_roll_back_entire_batch() { + let (first, _second, _schema) = fixture().await; + let repository = PostgresMetadataInstallRepository::new(first.clone()) + .await + .unwrap(); + let pages = prepared(2); + let intent = repository + .begin_intent("missing-atomic", &pages) + .await + .unwrap(); + let cap = repository + .mint_legacy_install_capability(&intent) + .await + .unwrap(); + repository + .install_pages_validated(&cap, &pages.dag().payloads()[..1]) + .await + .unwrap(); + let original_modes = payload_modes(&first).await; + let before = domain_state_for_test(&first).await; + let failed = hex::encode(pages.dag().payloads()[1].id); + first.execute_unprepared(&format!( + "CREATE FUNCTION test_missing_bad_codec() RETURNS trigger LANGUAGE plpgsql AS $$ + BEGIN IF NEW.page_id=decode('{failed}','hex') THEN NEW.metadata_codec:=2; END IF; RETURN NEW; END $$; + CREATE TRIGGER test_missing_bad_codec BEFORE INSERT ON mst2_metadata_payload + FOR EACH ROW EXECUTE FUNCTION test_missing_bad_codec()" + )).await.unwrap(); + assert!( + repository + .install_missing_pages_validated(&cap, pages.dag().payloads()) + .await + .unwrap_err() + .to_string() + .contains("mst2_metadata_payload_metadata_codec_check") + ); + assert_eq!(domain_state_for_test(&first).await, before); + first.execute_unprepared( + "DROP TRIGGER test_missing_bad_codec ON mst2_metadata_payload; DROP FUNCTION test_missing_bad_codec(); + CREATE FUNCTION test_missing_after_insert_fault() RETURNS trigger LANGUAGE plpgsql AS $$ + BEGIN RAISE EXCEPTION 'missing insert fault after production guards'; END $$; + CREATE TRIGGER test_missing_after_insert_fault AFTER INSERT ON mst2_metadata_payload + FOR EACH STATEMENT EXECUTE FUNCTION test_missing_after_insert_fault()", + ).await.unwrap(); + assert!( + repository + .install_missing_pages_validated(&cap, pages.dag().payloads()) + .await + .unwrap_err() + .to_string() + .contains("missing insert fault after production guards") + ); + assert_eq!(domain_state_for_test(&first).await, before); + first.execute_unprepared( + "DROP TRIGGER test_missing_after_insert_fault ON mst2_metadata_payload; DROP FUNCTION test_missing_after_insert_fault()", + ).await.unwrap(); + assert_eq!(payload_modes(&first).await, original_modes); + let existing = pages.dag().payloads()[..1] + .iter() + .map(|page| page.id) + .collect(); + let work = install_missing(&repository, &cap, &pages).await; + assert_preparing_work(&work, pages.dag().payloads(), &existing); + repository.finalize(&intent).await.unwrap(); +} + +#[tokio::test] +async fn install_missing_presence_hint_never_certifies_metadata_consistent_corrupt_body() { + let (first, second, _schema) = fixture().await; + let repository = PostgresMetadataInstallRepository::new(first.clone()) + .await + .unwrap(); + let pages = prepared(2); + let intent = repository + .begin_intent("hint-not-proof", &pages) + .await + .unwrap(); + let cap = repository + .mint_legacy_install_capability(&intent) + .await + .unwrap(); + let page = &pages.dag().payloads()[0]; + first.execute_raw(statement( + "INSERT INTO mst2_metadata_payload(page_id,metadata_codec,byte_size,payload) VALUES($1,1,$2,$3)", + [page.id.to_vec().into(), (page.size as i32).into(), corrupt(page).into()], + )).await.unwrap(); + let poisoned = domain_state_for_test(&first).await; + assert!( + repository + .install_pages_validated(&cap, pages.dag().payloads()) + .await + .unwrap_err() + .to_string() + .contains("payload identity conflicts with stored bytes") + ); + assert_eq!(domain_state_for_test(&first).await, poisoned); + let work = install_missing(&repository, &cap, &pages).await; + assert_preparing_work(&work, pages.dag().payloads(), &BTreeSet::from([page.id])); + let before = domain_state_for_test(&first).await; + assert!(repository.finalize(&intent).await.is_err()); + let restarted = PostgresMetadataInstallRepository::new(second) + .await + .unwrap(); + assert!(restarted.finalize(&intent).await.is_err()); + assert!( + restarted + .restore_session_dag(intent.prepare_id()) + .await + .is_err() + ); + assert_eq!(domain_state_for_test(&first).await, before); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_prepare WHERE state='COMMITTED'" + ) + .await, + 0 + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_retention_root").await, + 0 + ); +} + +#[tokio::test] +async fn install_missing_committed_replay_rejects_corrupt_and_missing_bytes_without_repair() { + for remove in [false, true] { + let (first, second, _schema) = fixture().await; + let repository = PostgresMetadataInstallRepository::new(first.clone()) + .await + .unwrap(); + let pages = prepared(2); + let intent = repository + .begin_intent("committed-no-repair", &pages) + .await + .unwrap(); + let cap = repository + .mint_legacy_install_capability(&intent) + .await + .unwrap(); + install_missing(&repository, &cap, &pages).await; + repository.finalize(&intent).await.unwrap(); + let page = &pages.dag().payloads()[0]; + let damage = if remove { + PayloadDamage::Remove + } else { + PayloadDamage::Bytes(corrupt(page)) + }; + damage_payload(&first, page, damage).await; + let before = domain_state_for_test(&first).await; + assert!( + repository + .install_missing_pages_validated(&cap, pages.dag().payloads()) + .await + .is_err() + ); + assert!(repository.finalize(&intent).await.is_err()); + let restarted = PostgresMetadataInstallRepository::new(second) + .await + .unwrap(); + assert!( + restarted + .mint_legacy_install_capability(&intent) + .await + .is_err() + ); + assert!( + restarted + .restore_session_dag(intent.prepare_id()) + .await + .is_err() + ); + assert_eq!(domain_state_for_test(&first).await, before); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_retention_root").await, + pages.dag().payloads().len() as i64 + ); + } +} + +#[tokio::test] +async fn install_missing_generic_generation_reuse_requires_exact_live_physical_binding() { + let (first, second, _schema) = fixture().await; + let generic = generations::PostgresMetadataGenerationRepository::new(first.clone()) + .await + .unwrap(); + let repository = PostgresMetadataInstallRepository::new(first.clone()) + .await + .unwrap(); + let pages = prepared(2); + let bound = generic + .begin_intent("missing-bound-generic", &pages) + .await + .unwrap(); + generic + .install_pages(&bound, pages.dag().payloads()) + .await + .unwrap(); + generic.finalize(&bound).await.unwrap(); + let intent = repository + .begin_intent("missing-share-generic", &pages) + .await + .unwrap(); + let cap = repository + .mint_legacy_install_capability(&intent) + .await + .unwrap(); + let existing = pages.dag().payloads().iter().map(|page| page.id).collect(); + let work = install_missing(&repository, &cap, &pages).await; + assert_preparing_work(&work, pages.dag().payloads(), &existing); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_payload WHERE generation IS NOT NULL" + ) + .await, + pages.dag().payloads().len() as i64 + ); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_prepare_page WHERE generation IS NULL" + ) + .await, + pages.dag().payloads().len() as i64 + ); + let page = &pages.dag().payloads()[0]; + damage_payload(&first, page, PayloadDamage::UnbindGeneration).await; + let before = domain_state_for_test(&first).await; + let restarted = PostgresMetadataInstallRepository::new(second) + .await + .unwrap(); + assert!( + restarted + .install_missing_pages_validated(&cap, pages.dag().payloads()) + .await + .is_err() + ); + assert_eq!(domain_state_for_test(&first).await, before); +} + +#[tokio::test] +async fn install_missing_qualified_owner_cannot_be_adopted_by_generic_preparation() { + let (first, _second, _schema) = fixture().await; + let qualified = generations::qualified::PostgresQualifiedMetadataRepository::new(first.clone()) + .await + .unwrap(); + let legacy = PostgresMetadataInstallRepository::new(first.clone()) + .await + .unwrap(); + let pages = prepared(2); + let bound = qualified + .begin_intent("missing-qualified", &pages) + .await + .unwrap(); + qualified + .install_pages(&bound, pages.dag().payloads()) + .await + .unwrap(); + let before = domain_state_for_test(&first).await; + assert!( + legacy + .begin_intent("missing-wrong-domain", &pages) + .await + .unwrap_err() + .to_string() + .contains("qualified mapping cannot cross domain/current/state fence") + ); + let bound_intent = MetadataPrepareIntent { + prepare_id: bound.prepare_id().into(), + operation_id: bound.operation_id().into(), + manifest_digest: bound.manifest_digest(), + }; + assert!( + legacy + .mint_legacy_install_capability(&bound_intent) + .await + .is_err() + ); + assert_eq!(domain_state_for_test(&first).await, before); +} + +#[tokio::test] +async fn install_missing_postclassification_release_gc_and_tombstone_keep_finalization_fenced() { + let (first, second, _schema) = fixture().await; + let repository = PostgresMetadataInstallRepository::new(first.clone()) + .await + .unwrap(); + let graph = PostgresRetentionRepository::new(second); + let pages = prepared(2); + let old = repository + .begin_intent("old-covered", &pages) + .await + .unwrap(); + let old_cap = repository + .mint_legacy_install_capability(&old) + .await + .unwrap(); + install_missing(&repository, &old_cap, &pages).await; + repository.finalize(&old).await.unwrap(); + let next = repository + .begin_intent("after-classification", &pages) + .await + .unwrap(); + let cap = repository + .mint_legacy_install_capability(&next) + .await + .unwrap(); + let work = install_missing(&repository, &cap, &pages).await; + let existing = pages.dag().payloads().iter().map(|page| page.id).collect(); + assert_preparing_work(&work, pages.dag().payloads(), &existing); + let root = node_id(&pages.dag().root()); + let lease = RetentionRoot::Lease("missing-classification-lease-root".into()); + let txn = first.begin().await.unwrap(); + PostgresRetentionRepository::acquire_existing_roots_in_txn( + &txn, + &root, + std::slice::from_ref(&lease), + ) + .await + .unwrap(); + txn.commit().await.unwrap(); + graph + .release_root(&RetentionRoot::Prepare(old.prepare_id().into())) + .await + .unwrap(); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_retention_root WHERE root_kind='lease'" + ) + .await, + 1 + ); + let protected = domain_state_for_test(&first).await; + assert_eq!( + graph + .mark_deleting("missing-gc-protected", &root) + .await + .unwrap(), + GcClaim::Unavailable + ); + assert_eq!(domain_state_for_test(&first).await, protected); + graph.release_root(&lease).await.unwrap(); + assert_eq!( + graph.mark_deleting("missing-gc-root", &root).await.unwrap(), + GcClaim::Marked + ); + let before = domain_state_for_test(&first).await; + assert!(repository.finalize(&next).await.is_err()); + assert!( + repository + .install_missing_pages_validated(&cap, pages.dag().payloads()) + .await + .is_err() + ); + assert_eq!(domain_state_for_test(&first).await, before); + assert_eq!(graph.node(&root).await.unwrap().unwrap().state, "DELETING"); + let completed = graph.complete_gc("missing-gc-root").await.unwrap(); + assert!(!completed.replayed); + assert!(graph.node(&root).await.unwrap().is_none()); + let tombstoned = domain_state_for_test(&first).await; + assert!( + repository + .install_missing_pages_validated(&cap, pages.dag().payloads()) + .await + .is_err() + ); + assert!(repository.finalize(&next).await.is_err()); + assert_eq!(domain_state_for_test(&first).await, tombstoned); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_retention_root").await, + 0 + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_payload").await, + pages.dag().payloads().len() as i64 + ); +} diff --git a/src/jupiter/storage/native_snapshot_session.rs b/src/jupiter/storage/native_snapshot_session.rs index bb374236..f4abe782 100644 --- a/src/jupiter/storage/native_snapshot_session.rs +++ b/src/jupiter/storage/native_snapshot_session.rs @@ -64,6 +64,41 @@ impl PostgresNativeSessionRepository { built: &BuiltDescriptor, prepared: &PreparedNativeMetadataRetention, ) -> Result { + let (receipt, work) = self.install_with_work(built, prepared).await?; + tracing::debug!( + snapshot_id = %built.snapshot_id, + requested_pages = work.requested_pages, + payload_pages_omitted = work.payload_pages_omitted, + payload_pages_encoded = work.payload_pages_encoded, + requested_payload_bytes_validated = work.requested_payload_bytes_validated, + payload_bytes_omitted = work.payload_bytes_omitted, + payload_bytes_encoded = work.payload_bytes_encoded, + metadata_parameter_bytes = work.metadata_parameter_bytes, + payload_parameter_bytes = work.payload_parameter_bytes, + payload_transactions = work.transactions, + registration_queries = work.registration_queries, + requested_member_queries = work.requested_member_queries, + classification_batches = work.classification_batches, + insert_statements = work.insert_statements, + byte_comparison_queries = work.byte_comparison_queries, + committed_replay_pages = work.committed_replay_pages, + payload_batch_elapsed_micros = work.elapsed_micros, + "native metadata payload batch work" + ); + Ok(receipt) + } + + pub(crate) async fn install_with_work( + &self, + built: &BuiltDescriptor, + prepared: &PreparedNativeMetadataRetention, + ) -> Result< + ( + PreparedMetadataReceipt, + super::native_metadata_install::LegacyPayloadInstallWork, + ), + SnapshotError, + > { if prepared.dag().root() != built.descriptor.metadata_root || prepared.scope() != built.descriptor.scope { @@ -80,13 +115,16 @@ impl PostgresNativeSessionRepository { .mint_legacy_install_capability(&intent) .await .map_err(install_error)?; + let mut work = super::native_metadata_install::LegacyPayloadInstallWork::default(); for pages in prepared.dag().payloads().chunks(64) { - installer - .install_pages_validated(&capability, pages) + let batch = installer + .install_missing_pages_validated(&capability, pages) .await .map_err(install_error)?; + work.record(batch); } - installer.finalize(&intent).await.map_err(install_error) + let receipt = installer.finalize(&intent).await.map_err(install_error)?; + Ok((receipt, work)) } /// Projection and payload installation happen before this boundary. From 042acb75229eed799211db271453c0f3b174a29c Mon Sep 17 00:00:00 2001 From: Xiaoyang Han Date: Thu, 8 Oct 2026 02:29:11 +0800 Subject: [PATCH 23/50] test(mst2): decode the first streamed CHUNK as a single frame (#73) The per-frame revocation regression receives one CHUNK DATA before revoking its lease. Parse that DATA with parse_frame, assert exact consumed length and the real map/content IDs, chunk index and bytes. Keep the original successful DELETE, next DATA error and terminal None assertions. Feed only the actual received bytes to the complete stream parser and require its exact missing END/ERROR rejection. No synthetic END, production change or weakened decoder. This repairs the sole failure in the exact #71 full suite (2396 passed, 1 failed, 2 ignored). The original fixture has no isolated credit observer; existing separate ownership regressions remain unchanged. The corrected source has not yet run natively. --- .../router/snapshot_chunks_bounded_tests.rs | 20 +++++++++++++++---- 1 file changed, 16 insertions(+), 4 deletions(-) diff --git a/src/api/router/snapshot_chunks_bounded_tests.rs b/src/api/router/snapshot_chunks_bounded_tests.rs index bbbd14c4..8d22a68d 100644 --- a/src/api/router/snapshot_chunks_bounded_tests.rs +++ b/src/api/router/snapshot_chunks_bounded_tests.rs @@ -450,10 +450,16 @@ async fn mst2_budgeted_chunk_body_preserves_per_frame_revocation_and_no_end() { assert_eq!(response.status(), 200); let mut data = response.into_body().into_data_stream(); let first = data.next().await.unwrap().unwrap(); - assert!(matches!( - parse_stream(&first).unwrap().as_slice(), - [Frame::Chunk(_)] - )); + let (frame, consumed) = mst2_codec::treeframe::parse_frame(&first).unwrap(); + assert_eq!(consumed, first.len()); + let Frame::Chunk(chunk) = frame else { + panic!("first DATA must contain exactly one CHUNK frame"); + }; + assert_eq!(format!("sha256:{}", hex_of(&chunk.map_id)), map_id); + assert_eq!(chunk.file_content_id, fixture.digest); + assert_eq!(chunk.chunk_index, 0); + assert_eq!(chunk.chunk_bytes, fixture.raw[..CHUNK_SIZE as usize]); + let received = first.to_vec(); let revoked = fixture .app .clone() @@ -470,4 +476,10 @@ async fn mst2_budgeted_chunk_body_preserves_per_frame_revocation_and_no_end() { assert_eq!(revoked.status(), 200); assert!(data.next().await.unwrap().is_err()); assert!(data.next().await.is_none()); + assert!(matches!( + parse_stream(&received), + Err(mst2_codec::CodecError::BadOrdering( + "stream missing END/ERROR frame" + )) + )); } From d0e2af158845dfcde0e3b182500cd8153a8c5fea Mon Sep 17 00:00:00 2001 From: Xiaoyang Han Date: Thu, 8 Oct 2026 02:47:15 +0800 Subject: [PATCH 24/50] perf(mst2): serve native META from protected persisted pages (#74) Native HTTP META now walks durable MTP2 pages from the fixed descriptor root instead of loading Git trees and rebuilding directory/radix projections. One captured G READ COMMITTED transaction proves primary and exact original lease routes before and after bounded retention-barrier acquisition, rechecks the complete immutable session, then owns all response bytes before committing. Every touched page retains prepare membership, exact current lifetime/generation/domain, LIVE graph/codec/size/tombstone, outgoing-reference equality and formal page-ID/codec checks. Path/radix partition counts and scope bounds, first-seen deduplication, expected digest, META frame caps and END exact request-body hash remain; cold complete restore and per-frame auth/lease validation are unchanged. Eight real PostgreSQL/HTTP regressions cover canonical identity/zstd/END, old SID after advance/restart with real Git-table lock traps, absence/digest/limits, corruption/missing pages, exact nullable/bound generic lifetimes, graph damage, lock-wait deadline/release rechecks and deterministic reader versus release/GC protection. Page work counters exclude setup/session/lock wait, cold full restore, response encoding and per-frame checks. This removes warm source projection work; it does not claim cold O(route), O(delta), measured latency, commit-update speedup or a Git win. No migration, dependency or Q compatibility path. --- .../snapshot_persisted_metadata_tests.rs | 870 ++++++++++++++++++ src/api/router/snapshot_router.rs | 132 +-- src/api/router/snapshot_session_tests.rs | 3 + .../storage/native_metadata_install.rs | 7 + .../native_snapshot_metadata_routes.rs | 492 ++++++++++ .../storage/native_snapshot_session.rs | 7 + 6 files changed, 1459 insertions(+), 52 deletions(-) create mode 100644 src/api/router/snapshot_persisted_metadata_tests.rs create mode 100644 src/jupiter/storage/native_snapshot_metadata_routes.rs diff --git a/src/api/router/snapshot_persisted_metadata_tests.rs b/src/api/router/snapshot_persisted_metadata_tests.rs new file mode 100644 index 00000000..028f7d35 --- /dev/null +++ b/src/api/router/snapshot_persisted_metadata_tests.rs @@ -0,0 +1,870 @@ +use std::collections::BTreeSet; + +use mst2_codec::metapage::{Page, page_id}; + +use super::*; +use crate::jupiter::storage::native_snapshot_session::MetadataRouteRequest; + +fn metadata_body(items: Value, encoding: &str) -> Vec { + serde_json::to_vec_pretty(&json!({"encoding": encoding, "items": items})).unwrap() +} + +async fn metadata_bytes(fixture: &Fixture, app: Router, body: &[u8]) -> Vec { + let response = app + .oneshot(fixture.request("POST", "metadata/pages", Body::from(body.to_vec()))) + .await + .unwrap(); + assert_eq!(response.status(), 200); + assert_eq!(response.headers()["x-mega-snapshot-id"], fixture.snapshot); + to_bytes(response.into_body(), usize::MAX) + .await + .unwrap() + .to_vec() +} + +fn assert_metadata(bytes: &[u8], body: &[u8], count: u32, expected: &[([u8; 32], Vec)]) { + let frames = parse_stream(bytes).unwrap(); + let mut actual = Vec::new(); + for frame in &frames[..frames.len() - 1] { + let Frame::Meta(meta) = frame else { + panic!("metadata route emitted a non-META frame") + }; + assert!(meta.pages.len() <= mst2_codec::treeframe::META_MAX_PAGES); + assert!( + meta.pages + .iter() + .map(|(_, page)| 36 + page.len()) + .sum::() + <= mst2_codec::treeframe::META_MAX_RAW + ); + for (id, page) in &meta.pages { + assert_eq!(*id, page_id(page)); + Page::decode(page).unwrap(); + } + actual.extend_from_slice(&meta.pages); + } + assert_eq!(actual, expected); + let Frame::End(end) = frames.last().unwrap() else { + panic!("metadata route omitted END") + }; + assert_eq!(end.request_item_count, count); + assert_eq!(end.unique_unit_count, expected.len() as u32); + assert_eq!( + end.logical_bytes, + expected + .iter() + .map(|(_, page)| page.len() as u64) + .sum::() + ); + assert_eq!( + end.request_body_sha256, + <[u8; 32]>::from(Sha256::digest(body)) + ); +} + +#[tokio::test] +async fn mst2_persisted_meta_matches_canonical_routes_after_rebuild_and_advance_without_git_reads() +{ + let fixture = Fixture::new_with_pg_config_and_directories(true, 140).await; + let context = fixture + .state + .storage + .snapshot_context(&fixture.snapshot, &fixture.lease) + .await + .unwrap(); + let handler = fixture + .state + .api_handler(std::path::Path::new("/")) + .await + .unwrap(); + let root_tree = handler + .get_tree_by_hash(&context.root_tree_oid) + .await + .unwrap(); + let built = crate::ceres::snapshot::pages::build_directory_page( + handler.as_ref(), + &root_tree, + "/project", + ) + .await + .unwrap(); + let (Page::Branch { children, .. }, _) = Page::decode(&built.page_bytes).unwrap() else { + panic!("wide fixture must use canonical radix pages") + }; + let label = children + .iter() + .find(|child| child.label == b'w') + .unwrap() + .label; + let specs = [ + ("/", vec![]), + ("/", vec![label]), + ("/wide-139", vec![]), + ("/nested", vec![]), + ("/directory", vec![]), + ("/", vec![label]), + ]; + let mut expected = Vec::new(); + let mut seen = BTreeSet::new(); + let mut items = Vec::new(); + for (path, route) in &specs { + let absolute = if *path == "/" { + "/project".to_owned() + } else { + format!("/project{path}") + }; + let directory = crate::ceres::snapshot::pages::build_directory_page( + handler.as_ref(), + &root_tree, + &absolute, + ) + .await + .unwrap(); + let pages = Page::pages_along_route(&directory.codec_entries, route).unwrap(); + let reached = page_id(pages.last().unwrap()); + items.push(json!({"directory_path": path, "route": route, + "expected_digest": format!("sha256:{}", hex::encode(reached))})); + for page in pages { + let id = page_id(&page); + if seen.insert(id) { + expected.push((id, page)); + } + } + } + advance(&fixture).await; + let state = rebuilt(&fixture).await; + assert!(state.storage.native_snapshot_sessions.get().is_none()); + let mono = fixture.state.storage.mono_storage(); + let held = mono.get_connection().begin().await.unwrap(); + held.execute_unprepared("LOCK TABLE mega_tree,mst2_verified_object IN ACCESS EXCLUSIVE MODE") + .await + .unwrap(); + fixture.counts.reset(); + for encoding in ["identity", "zstd"] { + let body = metadata_body(json!(items), encoding); + let bytes = tokio::time::timeout( + Duration::from_secs(4), + metadata_bytes(&fixture, app(&state), &body), + ) + .await + .expect("persisted META must not touch locked Git tree or verified-object tables"); + assert_metadata(&bytes, &body, specs.len() as u32, &expected); + } + let requests: Vec<_> = specs + .iter() + .map(|(path, route)| MetadataRouteRequest { + directory_path: path, + route, + expected_digest: None, + }) + .collect(); + let batch = state + .storage + .snapshot_sessions() + .await + .metadata_routes(&context, &requests) + .await + .unwrap(); + assert_eq!(batch.pages, expected); + assert_eq!(batch.work.page_queries, batch.work.pages_loaded); + assert!( + batch.work.pages_loaded < 20, + "route work must not scan the 140-directory DAG" + ); + assert!( + batch.work.walk_visits > batch.work.pages_loaded, + "duplicate/path visits reuse request pages" + ); + assert!( + batch.work.payload_bytes >= batch.pages.iter().map(|(_, page)| page.len() as u64).sum() + ); + held.rollback().await.unwrap(); + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn mst2_persisted_meta_absence_scope_digest_limits_and_release_oracles() { + let fixture = Fixture::new_with_pg_config(true).await; + for (items, status, code) in [ + ( + json!([{"directory_path":"/missing"}]), + 404, + "PATH_NOT_FOUND", + ), + (json!([{"directory_path":"/file"}]), 409, "NOT_DIRECTORY"), + ( + json!([{"directory_path":"/link/child"}]), + 409, + "NOT_DIRECTORY", + ), + ( + json!([{"directory_path":"/nested", "route":[0]}]), + 404, + "PATH_NOT_FOUND", + ), + ( + json!([{"directory_path":"/../outside"}]), + 400, + "SCOPE_INVALID", + ), + ( + json!([{"directory_path":format!("/{}", vec!["a"; 256].join("/"))}]), + 400, + "SCOPE_INVALID", + ), + ( + json!([{"directory_path":"/outside"}]), + 404, + "PATH_NOT_FOUND", + ), + ( + json!([{"directory_path":"/", "expected_digest":format!("sha256:{}", "0".repeat(64))}]), + 409, + "EXPECTED_DIGEST_MISMATCH", + ), + (json!([]), 413, "LIMIT_EXCEEDED"), + ( + json!(vec![json!({"directory_path":"/"}); 65]), + 413, + "LIMIT_EXCEEDED", + ), + ] { + error( + fixture + .send( + "POST", + "metadata/pages", + Body::from(metadata_body(items, "identity")), + ) + .await, + status, + code, + false, + ) + .await; + } + let body = metadata_body(json!([{"directory_path":"/"}]), "identity"); + let response = fixture + .send("POST", "metadata/pages", Body::from(body)) + .await; + assert_eq!(response.status(), 200); + let mut stream = response.into_body().into_data_stream(); + let first = stream.next().await.unwrap().unwrap(); + let (Frame::Meta(meta), consumed) = mst2_codec::treeframe::parse_frame(&first).unwrap() else { + panic!("first persisted frame must be META") + }; + assert_eq!(consumed, first.len()); + assert!(!meta.pages.is_empty()); + lease_control(&fixture, &fixture.lease, "DELETE", false).await; + assert!(stream.next().await.unwrap().is_err()); + assert!(stream.next().await.is_none()); + assert!(matches!( + parse_stream(&first), + Err(mst2_codec::CodecError::BadOrdering( + "stream missing END/ERROR frame" + )) + )); + fixture.counts.assert(0, 0); +} + +async fn page_row(fixture: &Fixture, path: &str) -> ([u8; 32], Vec) { + let body = metadata_body(json!([{"directory_path":path}]), "identity"); + let bytes = metadata_bytes(fixture, fixture.app.clone(), &body).await; + let Frame::Meta(meta) = &parse_stream(&bytes).unwrap()[0] else { + panic!("META required") + }; + meta.pages[0].clone() +} + +async fn damage_page(fixture: &Fixture, id: [u8; 32], remove: bool) { + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + let modes = handoff_trigger_modes_for_test(db).await; + let txn = db.begin().await.unwrap(); + txn.execute_unprepared("SELECT pg_advisory_xact_lock(1296717362,hashtext(current_schema()))") + .await + .unwrap(); + txn.execute_unprepared( + "ALTER TABLE mst2_metadata_payload DISABLE TRIGGER mst2_metadata_payload_fenced", + ) + .await + .unwrap(); + if remove { + txn.execute_unprepared( + "ALTER TABLE mst2_metadata_payload DISABLE TRIGGER mst2_metadata_payload_removed", + ) + .await + .unwrap(); + } + let sql = if remove { + "DELETE FROM mst2_metadata_payload WHERE page_id=$1" + } else { + "UPDATE mst2_metadata_payload SET payload=set_byte(payload,0,0) WHERE page_id=$1" + }; + assert_eq!( + txn.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + sql, + [id.to_vec().into()] + )) + .await + .unwrap() + .rows_affected(), + 1 + ); + if remove { + txn.execute_unprepared( + "ALTER TABLE mst2_metadata_payload ENABLE TRIGGER mst2_metadata_payload_removed", + ) + .await + .unwrap(); + } + txn.execute_unprepared( + "ALTER TABLE mst2_metadata_payload ENABLE TRIGGER mst2_metadata_payload_fenced", + ) + .await + .unwrap(); + txn.commit().await.unwrap(); + assert_eq!(handoff_trigger_modes_for_test(db).await, modes); + assert!( + db.execute_unprepared("UPDATE mst2_metadata_payload SET payload=payload") + .await + .is_err() + ); +} + +#[tokio::test] +async fn mst2_persisted_meta_warm_missing_and_corrupt_pages_never_reproject_from_git() { + for remove in [true, false] { + let fixture = Fixture::new_with_pg_config(true).await; + let (id, _) = page_row(&fixture, "/nested").await; + damage_page(&fixture, id, remove).await; + let body = metadata_body(json!([{"directory_path":"/nested"}]), "identity"); + let mono = fixture.state.storage.mono_storage(); + let held = mono.get_connection().begin().await.unwrap(); + held.execute_unprepared( + "LOCK TABLE mega_tree,mst2_verified_object IN ACCESS EXCLUSIVE MODE", + ) + .await + .unwrap(); + let response = tokio::time::timeout( + Duration::from_secs(4), + fixture.send("POST", "metadata/pages", Body::from(body)), + ) + .await + .expect("damaged persisted page must fail before any source read"); + error( + response, + if remove { 503 } else { 502 }, + if remove { + "OBJECT_UNAVAILABLE" + } else { + "INTEGRITY_ERROR" + }, + false, + ) + .await; + held.rollback().await.unwrap(); + fixture.counts.assert(0, 0); + } +} + +async fn wait_retention_waiter(txn: &sea_orm::DatabaseTransaction) { + tokio::time::timeout(Duration::from_secs(4), async { + loop { + txn.execute_unprepared("SELECT pg_stat_clear_snapshot()") + .await + .unwrap(); + if scalar( + txn, + "SELECT count(*) FROM pg_locks l JOIN pg_stat_activity a ON a.pid=l.pid + WHERE l.locktype='advisory' AND NOT l.granted AND a.datname=current_database() + AND a.application_name=current_schema() AND l.classid=1296717362::oid + AND l.objid=hashtext(current_schema())::oid AND l.objsubid=2", + ) + .await + > 0 + { + break; + } + tokio::task::yield_now().await; + } + }) + .await + .expect("META request did not wait for retention authority"); +} + +#[tokio::test] +async fn mst2_persisted_meta_rechecks_deadline_and_release_after_waiting_for_retention_lock() { + for expire in [true, false] { + let fixture = Fixture::new_with_pg_config(true).await; + let mono = fixture.state.storage.mono_storage(); + let held = mono.get_connection().begin().await.unwrap(); + held.execute_unprepared( + "SELECT pg_advisory_xact_lock(1296717362,hashtext(current_schema()))", + ) + .await + .unwrap(); + let pending = { + let application = fixture.app.clone(); + let request = fixture.request( + "POST", + "metadata/pages", + Body::from(metadata_body( + json!([{"directory_path":"/nested"}]), + "identity", + )), + ); + tokio::spawn(async move { application.oneshot(request).await.unwrap() }) + }; + wait_retention_waiter(&held).await; + let sql = if expire { + "UPDATE mst2_snapshot_lease SET expires_at_unix=0 WHERE lease_id=$1" + } else { + "UPDATE mst2_snapshot_lease SET state='RELEASED' WHERE lease_id=$1" + }; + held.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + sql, + [fixture.lease.clone().into()], + )) + .await + .unwrap(); + held.commit().await.unwrap(); + error(pending.await.unwrap(), 410, "LEASE_EXPIRED", false).await; + fixture.counts.assert(0, 0); + } +} + +async fn bound_fixture() -> ( + Fixture, + crate::ceres::snapshot::retention_dag::MetadataPagePayload, +) { + let mut fixture = Fixture::new_with_pg_config(true).await; + advance(&fixture).await; + let mono = fixture.state.storage.mono_storage(); + let head = mono + .read_native_publication_head( + fixture + .state + .storage + .config() + .mst2 + .instance_uuid + .as_deref() + .unwrap(), + ) + .await + .unwrap(); + let handler = fixture + .state + .api_handler(std::path::Path::new("/")) + .await + .unwrap(); + let tree = handler.get_tree_by_hash(&head.root.tree).await.unwrap(); + let prepared = crate::ceres::snapshot::pages::prepare_native_metadata_retention( + handler.as_ref(), + &tree, + "/project", + crate::ceres::snapshot::retention_dag::MetadataDagLimits::default(), + ) + .await + .unwrap(); + assert_eq!(prepared.dag().payloads().len(), 1); + let expected = prepared.dag().payloads()[0].clone(); + let repository = crate::jupiter::storage::native_metadata_install::generations::PostgresMetadataGenerationRepository::new( + mono.get_connection().clone()).await.unwrap(); + let intent = repository + .begin_intent("persisted-meta-bound-seed", &prepared) + .await + .unwrap(); + repository + .install_pages(&intent, prepared.dag().payloads()) + .await + .unwrap(); + repository.finalize(&intent).await.unwrap(); + let resolved = success_json( + fixture + .app + .clone() + .oneshot(resolve_request("/project")) + .await + .unwrap(), + ) + .await; + assert_ne!(resolved["descriptor"]["snapshot_id"], fixture.snapshot); + fixture.snapshot = resolved["descriptor"]["snapshot_id"] + .as_str() + .unwrap() + .to_owned(); + fixture.lease = resolved["lease_id"].as_str().unwrap().to_owned(); + let db = mono.get_connection(); + assert_eq!( + scalar( + db, + "SELECT count(*) FROM mst2_metadata_payload WHERE generation=1" + ) + .await, + 1 + ); + db.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "INSERT INTO mst2_metadata_lifetime(page_id,node_id,generation,state,metadata_codec,expected_size,graph_domain) + VALUES($1,$2,2,'RESERVED',1,$3,'generic-v1')", + [expected.id.to_vec().into(), format!("page:sha256:{}", hex::encode(expected.id)).into(), + (expected.size as i32).into()])).await.unwrap(); + let membership = db.query_one_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "SELECT pp.generation AS member_generation,c.generation AS current_generation + FROM mst2_snapshot_context s JOIN mst2_metadata_prepare_page pp ON pp.prepare_id=s.prepare_id + JOIN mst2_metadata_current c ON c.page_id=pp.page_id WHERE s.snapshot_id=$1", + [fixture.snapshot.clone().into()])).await.unwrap().unwrap(); + assert_eq!( + membership + .try_get::>("", "member_generation") + .unwrap(), + None + ); + assert_eq!( + membership.try_get::("", "current_generation").unwrap(), + 1 + ); + fixture.counts.reset(); + (fixture, expected) +} + +#[tokio::test] +async fn mst2_persisted_meta_serves_bound_generic_pages_with_null_members_and_extra_history() { + let (fixture, expected) = bound_fixture().await; + let body = metadata_body(json!([{"directory_path":"/"}]), "identity"); + for application in [fixture.app.clone(), app(&rebuilt(&fixture).await)] { + let bytes = metadata_bytes(&fixture, application, &body).await; + assert_metadata(&bytes, &body, 1, &[(expected.id, expected.bytes.clone())]); + } + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn mst2_persisted_meta_rejects_touched_graph_damage_and_prepare_membership_loss() { + for case in 0..5 { + let fixture = Fixture::new_with_pg_config(true).await; + let (id, _) = page_row(&fixture, "/nested").await; + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + let modes = handoff_trigger_modes_for_test(db).await; + let txn = db.begin().await.unwrap(); + txn.execute_unprepared( + "SELECT pg_advisory_xact_lock(1296717362,hashtext(current_schema()))", + ) + .await + .unwrap(); + let node = format!("page:sha256:{}", hex::encode(id)); + let (sql, values) = match case { + 0 => ("UPDATE mst2_retention_node SET state='DELETING' WHERE node_id=$1", vec![node.into()]), + 1 => ("UPDATE mst2_retention_node SET bytes=bytes+1 WHERE node_id=$1", vec![node.into()]), + 2 => ("DELETE FROM mst2_retention_edge WHERE child_id=$1", vec![node.into()]), + 3 => ("INSERT INTO mst2_retention_gc_op(operation_id,node_id,operation,state,attempts,created_at) + VALUES('persisted-meta-tombstone',$1,'REMOVE','PENDING',0,now())", vec![node.into()]), + 4 => { + txn.execute_unprepared("ALTER TABLE mst2_metadata_prepare_page DISABLE TRIGGER mst2_install_capability_mapping_guard").await.unwrap(); + ("DELETE FROM mst2_metadata_prepare_page WHERE page_id=$1 AND prepare_id= + (SELECT prepare_id FROM mst2_snapshot_context WHERE snapshot_id=$2)", + vec![id.to_vec().into(), fixture.snapshot.clone().into()]) + } + _ => unreachable!(), + }; + assert!( + txn.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + sql, + values + )) + .await + .unwrap() + .rows_affected() + > 0 + ); + if case == 4 { + txn.execute_unprepared("ALTER TABLE mst2_metadata_prepare_page ENABLE TRIGGER mst2_install_capability_mapping_guard").await.unwrap(); + } + txn.commit().await.unwrap(); + assert_eq!(handoff_trigger_modes_for_test(db).await, modes); + let response = fixture + .send( + "POST", + "metadata/pages", + Body::from(metadata_body( + json!([{"directory_path":"/nested"}]), + "identity", + )), + ) + .await; + error( + response, + if case == 2 { 502 } else { 503 }, + if case == 2 { + "INTEGRITY_ERROR" + } else { + "OBJECT_UNAVAILABLE" + }, + false, + ) + .await; + fixture.counts.assert(0, 0); + } +} + +#[tokio::test] +async fn mst2_persisted_meta_reader_holds_protection_until_all_route_bytes_are_owned() { + use crate::jupiter::storage::native_snapshot_session::with_metadata_read_barriers; + + for release_lease in [false, true] { + let fixture = Fixture::new_with_pg_config(true).await; + let expected = vec![ + page_row(&fixture, "/nested").await, + page_row(&fixture, "/directory").await, + ]; + let context = fixture + .state + .storage + .snapshot_context(&fixture.snapshot, &fixture.lease) + .await + .unwrap(); + let captured = Arc::new(Barrier::new(2)); + let resume = Arc::new(Barrier::new(2)); + let reading = { + let state = fixture.state.clone(); + let context = context.clone(); + let captured = captured.clone(); + let resume = resume.clone(); + tokio::spawn(async move { + let requests = [ + MetadataRouteRequest { + directory_path: "/nested", + route: &[], + expected_digest: None, + }, + MetadataRouteRequest { + directory_path: "/directory", + route: &[], + expected_digest: None, + }, + ]; + with_metadata_read_barriers( + captured, + resume, + state + .storage + .snapshot_sessions() + .await + .metadata_routes(&context, &requests), + ) + .await + }) + }; + tokio::time::timeout(Duration::from_secs(4), captured.wait()) + .await + .unwrap(); + let root = format!("page:{}", context.built.metadata_root); + let competing = { + let state = fixture.state.clone(); + let lease = fixture.lease.clone(); + let root = root.clone(); + tokio::spawn(async move { + if release_lease { + assert!(state.storage.snapshot_release(&lease).await.unwrap()); + None + } else { + Some( + PostgresRetentionRepository::new( + state.storage.mono_storage().get_connection().clone(), + ) + .mark_deleting("persisted-meta-reader-race", &root) + .await + .unwrap(), + ) + } + }) + }; + let observer = fixture + .state + .storage + .mono_storage() + .get_connection() + .begin() + .await + .unwrap(); + wait_retention_waiter(&observer).await; + assert!(!reading.is_finished()); + assert!(!competing.is_finished()); + resume.wait().await; + let batch = reading.await.unwrap().unwrap(); + assert_eq!(batch.pages, expected); + assert_eq!( + batch.work.pages_loaded, 3, + "root and both directories are read before unlock" + ); + let outcome = competing.await.unwrap(); + observer.rollback().await.unwrap(); + if release_lease { + assert!(outcome.is_none()); + } else { + assert_eq!(outcome, Some(GcClaim::Unavailable)); + lease_control(&fixture, &fixture.lease, "DELETE", false).await; + } + let gc = PostgresRetentionRepository::new( + fixture + .state + .storage + .mono_storage() + .get_connection() + .clone(), + ); + assert_eq!( + gc.mark_deleting("persisted-meta-after-read", &root) + .await + .unwrap(), + GcClaim::Marked + ); + gc.complete_gc("persisted-meta-after-read").await.unwrap(); + let requests = [MetadataRouteRequest { + directory_path: "/nested", + route: &[], + expected_digest: None, + }]; + let rejected = fixture + .state + .storage + .snapshot_sessions() + .await + .metadata_routes(&context, &requests) + .await + .err() + .expect("released and collected root must fail before touching pages"); + assert_eq!(rejected.code, SnapshotErrorCode::LeaseExpired); + assert_eq!( + batch.pages, expected, + "collected graph does not invalidate owned route bytes" + ); + fixture.counts.assert(0, 0); + } +} + +async fn fault_binding(fixture: &Fixture, id: [u8; 32], case: usize, restore: bool) { + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + let modes = handoff_trigger_modes_for_test(db).await; + let txn = db.begin().await.unwrap(); + txn.execute_unprepared("SELECT pg_advisory_xact_lock(1296717362,hashtext(current_schema()))") + .await + .unwrap(); + let (table, guard, sql) = match case { + 0 => ( + "mst2_metadata_payload", + "mst2_metadata_payload_fenced", + if restore { + "UPDATE mst2_metadata_payload SET generation=1 WHERE page_id=$1" + } else { + "UPDATE mst2_metadata_payload SET generation=NULL WHERE page_id=$1" + }, + ), + 1 => ( + "mst2_metadata_current", + "mst2_metadata_current_guard", + if restore { + "UPDATE mst2_metadata_current SET generation=1 WHERE page_id=$1" + } else { + "UPDATE mst2_metadata_current SET generation=2 WHERE page_id=$1" + }, + ), + 2 => ( + "mst2_metadata_lifetime", + "mst2_metadata_lifetime_guard", + if restore { + "UPDATE mst2_metadata_lifetime SET state='LIVE' WHERE page_id=$1 AND generation=1" + } else { + "UPDATE mst2_metadata_lifetime SET state='DELETING' WHERE page_id=$1 AND generation=1" + }, + ), + 3 => ( + "mst2_metadata_lifetime", + "mst2_metadata_lifetime_guard", + if restore { + "UPDATE mst2_metadata_lifetime SET graph_domain='generic-v1' WHERE page_id=$1 AND generation=1" + } else { + "UPDATE mst2_metadata_lifetime SET graph_domain='qualified-v1' WHERE page_id=$1 AND generation=1" + }, + ), + 4 => ( + "mst2_metadata_lifetime", + "mst2_metadata_lifetime_guard", + if restore { + "UPDATE mst2_metadata_lifetime SET expected_size=expected_size-1 WHERE page_id=$1 AND generation=1" + } else { + "UPDATE mst2_metadata_lifetime SET expected_size=expected_size+1 WHERE page_id=$1 AND generation=1" + }, + ), + 5 => ( + "mst2_metadata_prepare_page", + "mst2_install_capability_mapping_guard", + if restore { + "UPDATE mst2_metadata_prepare_page SET generation=NULL WHERE page_id=$1 AND prepare_id= + (SELECT prepare_id FROM mst2_snapshot_context WHERE snapshot_id=$2)" + } else { + "UPDATE mst2_metadata_prepare_page SET generation=2 WHERE page_id=$1 AND prepare_id= + (SELECT prepare_id FROM mst2_snapshot_context WHERE snapshot_id=$2)" + }, + ), + _ => unreachable!(), + }; + txn.execute_unprepared(&format!("ALTER TABLE {table} DISABLE TRIGGER {guard}")) + .await + .unwrap(); + let values = if case == 5 { + vec![id.to_vec().into(), fixture.snapshot.clone().into()] + } else { + vec![id.to_vec().into()] + }; + assert_eq!( + txn.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + sql, + values + )) + .await + .unwrap() + .rows_affected(), + 1 + ); + txn.execute_unprepared(&format!("ALTER TABLE {table} ENABLE TRIGGER {guard}")) + .await + .unwrap(); + txn.commit().await.unwrap(); + assert_eq!(handoff_trigger_modes_for_test(db).await, modes); +} + +#[tokio::test] +async fn mst2_persisted_meta_warm_reads_require_exact_generic_current_lifetime_and_membership() { + let (fixture, expected) = bound_fixture().await; + let body = metadata_body(json!([{"directory_path":"/"}]), "identity"); + let bytes = metadata_bytes(&fixture, fixture.app.clone(), &body).await; + assert_metadata(&bytes, &body, 1, &[(expected.id, expected.bytes.clone())]); + for case in 0..6 { + fault_binding(&fixture, expected.id, case, false).await; + error( + fixture + .send("POST", "metadata/pages", Body::from(body.clone())) + .await, + if matches!(case, 0 | 1 | 5) { 502 } else { 503 }, + if matches!(case, 0 | 1 | 5) { + "INTEGRITY_ERROR" + } else { + "OBJECT_UNAVAILABLE" + }, + false, + ) + .await; + fault_binding(&fixture, expected.id, case, true).await; + let bytes = metadata_bytes(&fixture, fixture.app.clone(), &body).await; + assert_metadata(&bytes, &body, 1, &[(expected.id, expected.bytes.clone())]); + } + fixture.counts.assert(0, 0); +} diff --git a/src/api/router/snapshot_router.rs b/src/api/router/snapshot_router.rs index eed58484..f75b7279 100644 --- a/src/api/router/snapshot_router.rs +++ b/src/api/router/snapshot_router.rs @@ -1476,62 +1476,90 @@ async fn metadata_pages( } let ctx = request_context(&state, &snapshot_id).map_err(mst2_error_response)?; - let handler = state - .api_handler(std::path::Path::new("/")) - .await - .map_err(internal)?; - let root_tree = handler - .get_tree_by_hash(&ctx.root_tree_oid) - .await - .map_err(internal)?; - let scope = &ctx.built.descriptor.scope; - - // Unique pages across all items, in first-seen order. - let mut unique: Vec<([u8; 32], Vec)> = Vec::new(); - let mut seen: Vec<[u8; 32]> = Vec::new(); - let mut logical_bytes: u64 = 0; - for item in &req.items { - let abs_path = abs_view_path(scope, &item.directory_path); - let built = build_directory_page(handler.as_ref(), &root_tree, &abs_path) + let unique = if state.storage.config().mst2.publication_enabled { + use crate::jupiter::storage::native_snapshot_session::MetadataRouteRequest; + let items: Vec<_> = req + .items + .iter() + .map(|item| MetadataRouteRequest { + directory_path: &item.directory_path, + route: &item.route, + expected_digest: item.expected_digest.as_deref(), + }) + .collect(); + let batch = state + .storage + .snapshot_sessions() + .await + .metadata_routes(&ctx, &items) .await .map_err(mst2_error_response)?; - let pages = - mst2_codec::metapage::Page::pages_along_route(&built.codec_entries, &item.route) - .map_err(|e| match e { - // A label the fixed view does not have is proven absence. - mst2_codec::CodecError::BadOrdering(m) => SnapshotError::new( - SnapshotErrorCode::PathNotFound, - format!("{}: route does not resolve ({m})", item.directory_path), - ), - other => SnapshotError::new( - SnapshotErrorCode::Internal, - format!("{}: route walk failed ({other})", item.directory_path), + tracing::debug!( + page_queries = batch.work.page_queries, + pages_loaded = batch.work.pages_loaded, + payload_bytes = batch.work.payload_bytes, + walk_visits = batch.work.walk_visits, + edge_references_checked = batch.work.edge_references_checked, + "served persisted generic metadata routes" + ); + batch.pages + } else { + let handler = state + .api_handler(std::path::Path::new("/")) + .await + .map_err(internal)?; + let root_tree = handler + .get_tree_by_hash(&ctx.root_tree_oid) + .await + .map_err(internal)?; + let scope = &ctx.built.descriptor.scope; + let mut unique: Vec<([u8; 32], Vec)> = Vec::new(); + let mut seen: Vec<[u8; 32]> = Vec::new(); + for item in &req.items { + let abs_path = abs_view_path(scope, &item.directory_path); + let built = build_directory_page(handler.as_ref(), &root_tree, &abs_path) + .await + .map_err(mst2_error_response)?; + let pages = + mst2_codec::metapage::Page::pages_along_route(&built.codec_entries, &item.route) + .map_err(|e| match e { + // A label the fixed view does not have is proven absence. + mst2_codec::CodecError::BadOrdering(m) => SnapshotError::new( + SnapshotErrorCode::PathNotFound, + format!("{}: route does not resolve ({m})", item.directory_path), + ), + other => SnapshotError::new( + SnapshotErrorCode::Internal, + format!("{}: route walk failed ({other})", item.directory_path), + ), + })?; + let last = pages + .last() + .expect("pages_along_route returns at least the root page"); + let last_id = mst2_codec::metapage::page_id(last); + if let Some(expected) = &item.expected_digest + && expected != &format!("sha256:{}", hex_of(&last_id)) + { + return Err(mst2_error_response(SnapshotError::new( + SnapshotErrorCode::DigestMismatch, + format!( + "{}: route does not reach expected_digest", + item.directory_path ), - })?; - let last = pages - .last() - .expect("pages_along_route returns at least the root page"); - let last_id = mst2_codec::metapage::page_id(last); - if let Some(expected) = &item.expected_digest - && expected != &format!("sha256:{}", hex_of(&last_id)) - { - return Err(mst2_error_response(SnapshotError::new( - SnapshotErrorCode::DigestMismatch, - format!( - "{}: route does not reach expected_digest", - item.directory_path - ), - ))); - } - for page in pages { - let id = mst2_codec::metapage::page_id(&page); - if !seen.contains(&id) { - logical_bytes += page.len() as u64; - seen.push(id); - unique.push((id, page)); + ))); + } + for page in pages { + let id = mst2_codec::metapage::page_id(&page); + if !seen.contains(&id) { + seen.push(id); + unique.push((id, page)); + } } } - } + unique + }; + let page_count = unique.len(); + let logical_bytes = unique.iter().map(|(_, page)| page.len() as u64).sum(); // Frames hold at most 64 pages and at most 1 MiB of raw payload (spec 06), // so a wide route set becomes several META frames rather than one @@ -1578,7 +1606,7 @@ async fn metadata_pages( let request_body_sha256: [u8; 32] = sha2::Digest::finalize(hasher).into(); let end = stream.end( req.items.len() as u32, - u32::try_from(seen.len()).unwrap_or(u32::MAX), + u32::try_from(page_count).unwrap_or(u32::MAX), logical_bytes, request_body_sha256, ); diff --git a/src/api/router/snapshot_session_tests.rs b/src/api/router/snapshot_session_tests.rs index 17dfcb21..6c49108d 100644 --- a/src/api/router/snapshot_session_tests.rs +++ b/src/api/router/snapshot_session_tests.rs @@ -1213,3 +1213,6 @@ async fn mst2_durable_http_warm_reads_renew_and_frame_delivery_reject_primary_sc #[path = "snapshot_lookup_metadata_tests.rs"] mod metadata_lookup; + +#[path = "snapshot_persisted_metadata_tests.rs"] +mod persisted_metadata; diff --git a/src/jupiter/storage/native_metadata_install.rs b/src/jupiter/storage/native_metadata_install.rs index aeddd315..5b97f663 100644 --- a/src/jupiter/storage/native_metadata_install.rs +++ b/src/jupiter/storage/native_metadata_install.rs @@ -643,6 +643,13 @@ impl PostgresMetadataInstallRepository { .map_err(internal) } + pub(super) async fn metadata_read_barrier( + &self, + txn: &DatabaseTransaction, + ) -> Result<(), SnapshotError> { + self.capability_barrier(txn).await + } + async fn barrier(&self, txn: &DatabaseTransaction) -> Result<(), SnapshotError> { if txn.get_database_backend() != DbBackend::Postgres { return Err(internal("metadata installation requires PostgreSQL")); diff --git a/src/jupiter/storage/native_snapshot_metadata_routes.rs b/src/jupiter/storage/native_snapshot_metadata_routes.rs new file mode 100644 index 00000000..8326fca6 --- /dev/null +++ b/src/jupiter/storage/native_snapshot_metadata_routes.rs @@ -0,0 +1,492 @@ +//! Fixed-root persisted META reads under the original generic lease authority. + +use std::collections::{BTreeSet, HashMap}; + +use mst2_codec::metapage::{HEADER_LEN, PAGE_MAX_BYTES, Page, page_id}; +use sea_orm::{ConnectionTrait, DatabaseTransaction, QueryResult}; + +use super::{ + PostgresNativeSessionRepository, SnapshotContext, SnapshotError, SnapshotErrorCode, + decode_context, expired, finish, integrity, internal, routes, statement, +}; +use crate::ceres::snapshot::{ + retention_dag::MetadataDagLimits, view::validate_scope_relative_path, +}; + +pub(crate) struct MetadataRouteRequest<'a> { + pub directory_path: &'a str, + pub route: &'a [u8], + pub expected_digest: Option<&'a str>, +} + +#[derive(Debug, Default, Clone, Copy)] +pub(crate) struct PersistedMetadataReadWork { + pub page_queries: u64, + pub pages_loaded: u64, + pub payload_bytes: u64, + pub walk_visits: u64, + pub edge_references_checked: u64, +} + +pub(crate) struct PersistedMetadataRouteBatch { + pub pages: Vec<([u8; 32], Vec)>, + pub work: PersistedMetadataReadWork, +} + +#[cfg(test)] +tokio::task_local! { + static METADATA_READ_BARRIERS: (std::sync::Arc, std::sync::Arc); +} + +#[cfg(test)] +pub(crate) async fn with_metadata_read_barriers( + captured: std::sync::Arc, + release: std::sync::Arc, + future: F, +) -> F::Output { + METADATA_READ_BARRIERS + .scope((captured, release), future) + .await +} + +impl PostgresNativeSessionRepository { + pub(crate) async fn metadata_routes( + &self, + context: &SnapshotContext, + requests: &[MetadataRouteRequest<'_>], + ) -> Result { + if requests.is_empty() || requests.len() > 64 { + return Err(limit("items must hold 1..64 entries")); + } + for request in requests { + validate_scope_relative_path(request.directory_path)?; + let scope = &context.built.descriptor.scope; + let absolute = if scope == "/" { + request.directory_path.to_owned() + } else if request.directory_path == "/" { + scope.clone() + } else { + format!("{scope}{}", request.directory_path) + }; + validate_scope_relative_path(&absolute)?; + } + let installer = self.installer().await?; + let txn = self.transaction().await?; + let result = async { + installer.verify_primary_connection(&txn).await?; + if routes::lease(&txn, installer, &context.lease_id) + .await? + .as_deref() + != Some(context.built.snapshot_id.as_str()) + { + return Err(expired()); + } + // This barrier captures the namespace and uses bounded lock acquisition. + // Do not acquire the publication/route writer lock after retention. + installer.metadata_read_barrier(&txn).await?; + if routes::lease(&txn, installer, &context.lease_id) + .await? + .as_deref() + != Some(context.built.snapshot_id.as_str()) + { + return Err(expired()); + } + let row = txn + .query_one_raw(statement( + self.session_sql(installer).await, + [ + context.built.snapshot_id.clone().into(), + context.lease_id.clone().into(), + ], + )) + .await + .map_err(internal)? + .ok_or_else(expired)?; + installer.verify_primary_scope_row(&row)?; + let current = decode_context( + &row, + &context.built.snapshot_id, + &context.lease_id, + &context.built.instance_id, + )?; + if current.built.descriptor != context.built.descriptor + || current.commit_oid != context.commit_oid + || current.root_tree_oid != context.root_tree_oid + || current.authorization_epoch != context.authorization_epoch + { + return Err(integrity("fixed session changed during metadata read")); + } + let prepare_id: String = row.try_get("", "prepare_id").map_err(internal)?; + let mut reader = Reader { + txn: &txn, + prepare_id, + codec: context.built.descriptor.metadata_codec() as i16, + cache: HashMap::new(), + work: PersistedMetadataReadWork::default(), + limits: MetadataDagLimits::default(), + }; + let mut seen = BTreeSet::new(); + let mut pages = Vec::new(); + for request in requests { + let root = reader + .directory( + context.built.descriptor.metadata_root, + request.directory_path, + ) + .await?; + let route = reader.route(root, request.route).await?; + let reached = route + .last() + .ok_or_else(|| integrity("metadata route is empty"))?; + if let Some(expected) = request.expected_digest + && expected != format!("sha256:{}", hex::encode(reached)) + { + return Err(SnapshotError::new( + SnapshotErrorCode::DigestMismatch, + format!( + "{}: route does not reach expected_digest", + request.directory_path + ), + )); + } + for id in route { + if seen.insert(id) { + let page = reader + .cache + .get(&id) + .ok_or_else(|| integrity("read metadata page is missing"))?; + pages.push((id, page.bytes.clone())); + } + } + } + Ok(PersistedMetadataRouteBatch { + pages, + work: reader.work, + }) + } + .await; + // All bytes are owned before releasing protection. Frame delivery keeps + // the existing per-frame authentication and lease revalidation. + finish(txn, result).await + } +} + +struct StoredPage { + bytes: Vec, + page: Page, + entries: u64, +} + +struct Reader<'a> { + txn: &'a DatabaseTransaction, + prepare_id: String, + codec: i16, + cache: HashMap<[u8; 32], StoredPage>, + work: PersistedMetadataReadWork, + limits: MetadataDagLimits, +} + +impl Reader<'_> { + async fn load(&mut self, id: [u8; 32]) -> Result<(), SnapshotError> { + self.work.walk_visits += 1; + if self.work.walk_visits > self.limits.prepare_entry_visits as u64 { + return Err(limit("metadata route work budget exceeded")); + } + if self.cache.contains_key(&id) { + return Ok(()); + } + if self.cache.len() >= self.limits.nodes { + return Err(limit("metadata route page budget exceeded")); + } + self.work.page_queries += 1; + let row = self + .txn + .query_one_raw(statement( + PAGE_SQL, + [self.prepare_id.clone().into(), id.to_vec().into()], + )) + .await + .map_err(internal)? + .ok_or_else(|| unavailable("metadata page is not a prepared member"))?; + let page = decode_page(&row, id, self.codec)?; + self.work.payload_bytes = self + .work + .payload_bytes + .checked_add(page.bytes.len() as u64) + .filter(|bytes| *bytes <= self.limits.payload_bytes) + .ok_or_else(|| limit("metadata route payload budget exceeded"))?; + let expected = page_references(&page.page); + self.work.edge_references_checked += expected.len() as u64; + if self.work.edge_references_checked > self.limits.edges as u64 { + return Err(limit("metadata route edge budget exceeded")); + } + let edges: String = row.try_get("", "outgoing_edges").map_err(internal)?; + let edges: Vec = serde_json::from_str(&edges).map_err(internal)?; + let actual: BTreeSet<_> = edges.iter().cloned().collect(); + if edges.len() != actual.len() || actual != expected { + return Err(integrity( + "metadata graph edges disagree with fixed page references", + )); + } + self.work.pages_loaded += 1; + self.cache.insert(id, page); + #[cfg(test)] + if self.cache.len() == 1 + && let Ok((captured, release)) = METADATA_READ_BARRIERS.try_with(Clone::clone) + { + captured.wait().await; + release.wait().await; + } + Ok(()) + } + + async fn descend(&mut self, parent: [u8; 32], label: u8) -> Result<[u8; 32], SnapshotError> { + self.load(parent).await?; + let (prefix, child) = match &self.cache[&parent].page { + Page::Branch { + prefix, children, .. + } => { + let child = children + .iter() + .find(|child| child.label == label) + .ok_or_else(|| absent("route label is absent in fixed metadata page"))?; + (prefix.clone(), child.clone()) + } + Page::Leaf { .. } => return Err(absent("route descends past a leaf page")), + }; + let id = child.child_page_id; + self.load(id).await?; + let received = &self.cache[&id]; + let mut partition = prefix; + partition.push(label); + let valid_prefix = match &received.page { + Page::Leaf { entries } => entries + .iter() + .all(|entry| entry.name.starts_with(&partition)), + Page::Branch { prefix, .. } => prefix.starts_with(&partition), + }; + if received.entries != child.subtree_entries || !valid_prefix { + return Err(integrity( + "metadata child differs from fixed radix partition", + )); + } + Ok(id) + } + + async fn directory( + &mut self, + mut root: [u8; 32], + path: &str, + ) -> Result<[u8; 32], SnapshotError> { + if path == "/" { + return Ok(root); + } + for name in path[1..].split('/') { + let mut id = root; + loop { + self.load(id).await?; + let entry = match &self.cache[&id].page { + Page::Leaf { entries } => entries + .iter() + .find(|entry| entry.name == name.as_bytes()) + .cloned(), + Page::Branch { + prefix, terminal, .. + } => { + if name.as_bytes() == prefix { + terminal.clone() + } else { + if !name.as_bytes().starts_with(prefix) { + return Err(absent("name is absent in fixed directory")); + } + let label = name + .as_bytes() + .get(prefix.len()) + .copied() + .ok_or_else(|| absent("name is absent in fixed directory"))?; + id = self.descend(id, label).await?; + continue; + } + } + } + .ok_or_else(|| absent("name is absent in fixed directory"))?; + if !entry.is_dir() { + return Err(SnapshotError::new( + SnapshotErrorCode::NotDirectory, + format!("{path} is not a directory"), + )); + } + root = entry.child_root; + break; + } + } + Ok(root) + } + + async fn route( + &mut self, + root: [u8; 32], + labels: &[u8], + ) -> Result, SnapshotError> { + self.load(root).await?; + let mut pages = vec![root]; + let mut current = root; + for label in labels { + current = self.descend(current, *label).await?; + pages.push(current); + } + Ok(pages) + } +} + +const PAGE_SQL: &str = "SELECT m.expected_size,m.generation AS member_generation,p.graph_domain AS prepare_graph_domain, + b.metadata_codec AS payload_codec,b.byte_size AS payload_size,octet_length(b.payload) AS actual_size, + CASE WHEN octet_length(b.payload) BETWEEN 20 AND 16384 THEN b.payload END AS payload, + b.generation AS payload_generation,c.generation AS current_generation, + l.generation AS lifetime_generation,l.graph_domain,l.state AS lifetime_state, + l.metadata_codec AS lifetime_codec,l.expected_size AS lifetime_size, + n.state AS graph_state,n.kind AS graph_kind,n.bytes AS graph_bytes, + EXISTS(SELECT 1 FROM mst2_retention_gc_op g WHERE g.node_id='page:sha256:'||encode(m.page_id,'hex') + AND g.operation='REMOVE' AND g.state IN ('PENDING','APPLIED')) AS tombstone, + COALESCE((SELECT jsonb_agg(e.child_id) FROM + (SELECT child_id FROM mst2_retention_edge WHERE parent_id='page:sha256:'||encode(m.page_id,'hex') + ORDER BY child_id LIMIT 258) e),'[]'::jsonb)::text AS outgoing_edges + FROM mst2_metadata_prepare_page m + JOIN mst2_metadata_prepare p ON p.prepare_id=m.prepare_id + LEFT JOIN mst2_metadata_payload b ON b.page_id=m.page_id + LEFT JOIN mst2_metadata_current c ON c.page_id=m.page_id + LEFT JOIN mst2_metadata_lifetime l ON l.page_id=c.page_id AND l.generation=c.generation + LEFT JOIN mst2_retention_node n ON n.node_id='page:sha256:'||encode(m.page_id,'hex') + WHERE m.prepare_id=$1 AND m.page_id=$2"; + +fn decode_page(row: &QueryResult, id: [u8; 32], codec: i16) -> Result { + if row + .try_get::>("", "prepare_graph_domain") + .map_err(internal)? + .as_deref() + .is_some_and(|domain| domain != "generic-v1") + { + return Err(unavailable( + "fixed metadata preparation is outside the generic namespace", + )); + } + let size: i32 = row.try_get("", "expected_size").map_err(internal)?; + if !(HEADER_LEN..=PAGE_MAX_BYTES).contains(&(size as usize)) + || row + .try_get::>("", "payload_codec") + .map_err(internal)? + != Some(codec) + || row + .try_get::>("", "payload_size") + .map_err(internal)? + != Some(size) + || row + .try_get::>("", "actual_size") + .map_err(internal)? + != Some(size) + { + return Err(unavailable( + "fixed metadata payload profile or size is unavailable", + )); + } + if row.try_get::("", "tombstone").map_err(internal)? + || row + .try_get::>("", "graph_state") + .map_err(internal)? + .as_deref() + != Some("LIVE") + || row + .try_get::>("", "graph_kind") + .map_err(internal)? + .as_deref() + != Some("page") + || row + .try_get::>("", "graph_bytes") + .map_err(internal)? + != Some(size as i64) + { + return Err(unavailable("fixed generic metadata graph is unavailable")); + } + let member: Option = row.try_get("", "member_generation").map_err(internal)?; + let payload: Option = row.try_get("", "payload_generation").map_err(internal)?; + let current: Option = row.try_get("", "current_generation").map_err(internal)?; + let lifetime: Option = row.try_get("", "lifetime_generation").map_err(internal)?; + if payload != current || payload != lifetime || member.is_some() && member != payload { + return Err(integrity( + "fixed generic page differs from its exact physical lifetime", + )); + } + if let Some(generation) = payload + && (generation <= 0 + || row + .try_get::>("", "graph_domain") + .map_err(internal)? + .as_deref() + != Some("generic-v1") + || row + .try_get::>("", "lifetime_state") + .map_err(internal)? + .as_deref() + != Some("LIVE") + || row + .try_get::>("", "lifetime_codec") + .map_err(internal)? + != Some(codec) + || row + .try_get::>("", "lifetime_size") + .map_err(internal)? + != Some(size)) + { + return Err(unavailable( + "fixed metadata lifetime cannot be served in the generic namespace", + )); + } + let bytes: Vec = row + .try_get::>>("", "payload") + .map_err(internal)? + .ok_or_else(|| unavailable("fixed metadata payload is missing"))?; + if page_id(&bytes) != id { + return Err(integrity("fixed metadata payload digest mismatch")); + } + let (page, entries) = Page::decode(&bytes) + .map_err(|_| integrity("fixed metadata payload is not a canonical page"))?; + Ok(StoredPage { + bytes, + page, + entries, + }) +} + +fn page_references(page: &Page) -> BTreeSet { + let mut edges = BTreeSet::new(); + let entries = match page { + Page::Leaf { entries } => entries.as_slice(), + Page::Branch { + terminal, children, .. + } => { + edges.extend( + children + .iter() + .map(|child| format!("page:sha256:{}", hex::encode(child.child_page_id))), + ); + terminal.as_slice() + } + }; + edges.extend( + entries + .iter() + .filter(|entry| entry.is_dir()) + .map(|entry| format!("page:sha256:{}", hex::encode(entry.child_root))), + ); + edges +} + +fn absent(message: &str) -> SnapshotError { + SnapshotError::new(SnapshotErrorCode::PathNotFound, message) +} +fn unavailable(message: &str) -> SnapshotError { + SnapshotError::new(SnapshotErrorCode::ObjectUnavailable, message) +} +fn limit(message: &str) -> SnapshotError { + SnapshotError::new(SnapshotErrorCode::LimitExceeded, message) +} diff --git a/src/jupiter/storage/native_snapshot_session.rs b/src/jupiter/storage/native_snapshot_session.rs index f4abe782..fb8d9134 100644 --- a/src/jupiter/storage/native_snapshot_session.rs +++ b/src/jupiter/storage/native_snapshot_session.rs @@ -363,6 +363,13 @@ impl PostgresNativeSessionRepository { #[path = "native_snapshot_routes.rs"] mod routes; +#[path = "native_snapshot_metadata_routes.rs"] +mod metadata_routes; + +pub(crate) use metadata_routes::MetadataRouteRequest; +#[cfg(test)] +pub(crate) use metadata_routes::with_metadata_read_barriers; + const SESSION_SQL: &str = "SELECT s.snapshot_id,s.canonical_descriptor,s.commit_oid,s.root_tree_oid, (SELECT storage_uuid FROM mst2_metadata_storage_scope WHERE singleton=1) AS authority_storage_uuid, current_database() AS authority_database, From 14eacf31740638c9eb74f2a1860d43f2a8b746e4 Mon Sep 17 00:00:00 2001 From: Xiaoyang Han Date: Thu, 8 Oct 2026 03:17:31 +0800 Subject: [PATCH 25/50] fix(mst2): exercise missing payload installs and await history rejection rollback (#75) The durable HTTP interruption fixture now resolves the deployment root, proves that only its root page is missing, and preserves the real INSERT failure, rollback, old SID readability, and successful retry. The wide-DAG fixture computes expected INSERT statements from the original 64-page batch boundaries and existing immutable IDs, counts actual new rows, and preserves the real page/edge lock trap and cold old-SID read. Native maintenance explicitly awaits rollback when publication history prevents initialization, so its exclusion lock is released before returning; the existing history-loss regression retains all exact history/root/paused-state assertions and adds immediate independent transaction lock reacquisition plus useful failure diagnostics. The old native log did not print the rejected error, so that failure cause is not conclusively established until the new native regression runs. Production missing-only deduplication and history rejection remain unchanged. Local nightly formatting, diff and locked offline no-dependency metadata checks passed; native tests remain unverified for this candidate, and no performance or large Git comparison was run. --- src/api/router/snapshot_session_tests.rs | 123 ++++++++++++++++-- .../service/native_publication_push_tests.rs | 10 +- .../storage/native_publication_storage.rs | 1 + 3 files changed, 122 insertions(+), 12 deletions(-) diff --git a/src/api/router/snapshot_session_tests.rs b/src/api/router/snapshot_session_tests.rs index 6c49108d..822890d0 100644 --- a/src/api/router/snapshot_session_tests.rs +++ b/src/api/router/snapshot_session_tests.rs @@ -31,6 +31,54 @@ async fn scalar(db: &C, sql: &str) -> i64 { .unwrap() } +async fn root_metadata( + fixture: &Fixture, +) -> crate::ceres::snapshot::pages::PreparedNativeMetadataRetention { + let mono = fixture.state.storage.mono_storage(); + let head = mono + .read_native_publication_head( + fixture + .state + .storage + .config() + .mst2 + .instance_uuid + .as_deref() + .unwrap(), + ) + .await + .unwrap(); + let handler = fixture + .state + .api_handler(std::path::Path::new("/")) + .await + .unwrap(); + let tree = handler.get_tree_by_hash(&head.root.tree).await.unwrap(); + crate::ceres::snapshot::pages::prepare_native_metadata_retention( + handler.as_ref(), + &tree, + "/", + crate::ceres::snapshot::retention_dag::MetadataDagLimits::default(), + ) + .await + .unwrap() +} + +async fn stored_metadata_ids(db: &DatabaseConnection) -> std::collections::BTreeSet<[u8; 32]> { + db.query_all_raw(Statement::from_string( + DbBackend::Postgres, + "SELECT page_id FROM mst2_metadata_payload", + )) + .await + .unwrap() + .into_iter() + .map(|row| { + let id: Vec = row.try_get_by_index(0).unwrap(); + id.try_into().unwrap() + }) + .collect() +} + async fn rebuilt(fixture: &Fixture) -> MonoApiServiceState { let config = fixture.state.storage.config(); let connection = crate::jupiter::storage::init::postgres_connection(&config.database) @@ -679,6 +727,16 @@ async fn mst2_durable_http_install_fault_never_hands_off_or_returns_success() { let fixture = Fixture::new_with_pg_config(true).await; let mono = fixture.state.storage.mono_storage(); let db = mono.get_connection(); + let prepared = root_metadata(&fixture).await; + let existing = stored_metadata_ids(db).await; + let missing: Vec<_> = prepared + .dag() + .payloads() + .iter() + .filter(|page| !existing.contains(&page.id)) + .map(|page| page.id) + .collect(); + assert_eq!(missing, [prepared.dag().root()]); db.execute_unprepared( "CREATE FUNCTION reject_http_page() RETURNS trigger LANGUAGE plpgsql AS $$ BEGIN RAISE EXCEPTION 'forced HTTP install interruption'; END $$; @@ -691,7 +749,7 @@ async fn mst2_durable_http_install_fault_never_hands_off_or_returns_success() { fixture .app .clone() - .oneshot(resolve_request("/project/nested")) + .oneshot(resolve_request("/")) .await .unwrap(), 503, @@ -699,6 +757,7 @@ async fn mst2_durable_http_install_fault_never_hands_off_or_returns_success() { true, ) .await; + assert_eq!(stored_metadata_ids(db).await, existing); assert_eq!( scalar(db, "SELECT count(*) FROM mst2_snapshot_context").await, 1 @@ -739,12 +798,21 @@ async fn mst2_durable_http_install_fault_never_hands_off_or_returns_success() { fixture .app .clone() - .oneshot(resolve_request("/project/nested")) + .oneshot(resolve_request("/")) .await .unwrap(), ) .await; - assert_eq!(success["descriptor"]["scope"], "/project/nested"); + assert_eq!(success["descriptor"]["scope"], "/"); + assert_eq!( + stored_metadata_ids(db).await, + prepared + .dag() + .payloads() + .iter() + .map(|page| page.id) + .collect() + ); assert_eq!( scalar( db, @@ -869,14 +937,35 @@ async fn mst2_durable_http_large_dag_batches_handoff_without_scanning_pages_or_e let fixture = Fixture::new_with_pg_config_and_directories(true, 80).await; let mono = fixture.state.storage.mono_storage(); let db = mono.get_connection(); - assert!(scalar(db, "SELECT count(*) FROM mst2_metadata_payload").await > 64); + let projected = root_metadata(&fixture).await; + let existing = stored_metadata_ids(db).await; + assert!(existing.len() > 64); + let missing: Vec<_> = projected + .dag() + .payloads() + .iter() + .filter(|page| !existing.contains(&page.id)) + .map(|page| page.id) + .collect(); + assert_eq!(missing, [projected.dag().root()]); + let missing_batches = projected + .dag() + .payloads() + .chunks(64) + .filter(|batch| batch.iter().any(|page| !existing.contains(&page.id))) + .count() as i64; + assert_eq!(missing_batches, 1); db.execute_unprepared( - "CREATE TABLE http_install_batch_count(singleton integer PRIMARY KEY,batches bigint NOT NULL); - INSERT INTO http_install_batch_count VALUES(1,0); + "CREATE TABLE http_install_batch_count(singleton integer PRIMARY KEY,batches bigint NOT NULL,rows bigint NOT NULL); + INSERT INTO http_install_batch_count VALUES(1,0,0); CREATE FUNCTION count_http_install_batch() RETURNS trigger LANGUAGE plpgsql AS $$ BEGIN UPDATE http_install_batch_count SET batches=batches+1 WHERE singleton=1; RETURN NULL; END $$; CREATE TRIGGER count_http_install_batch AFTER INSERT ON mst2_metadata_payload - FOR EACH STATEMENT EXECUTE FUNCTION count_http_install_batch()", + FOR EACH STATEMENT EXECUTE FUNCTION count_http_install_batch(); + CREATE FUNCTION count_http_install_row() RETURNS trigger LANGUAGE plpgsql AS $$ + BEGIN UPDATE http_install_batch_count SET rows=rows+1 WHERE singleton=1; RETURN NULL; END $$; + CREATE TRIGGER count_http_install_row AFTER INSERT ON mst2_metadata_payload + FOR EACH ROW EXECUTE FUNCTION count_http_install_row()", ).await.unwrap(); let prepared = Arc::new(Barrier::new(2)); let release = Arc::new(Barrier::new(2)); @@ -898,9 +987,23 @@ async fn mst2_durable_http_large_dag_batches_handoff_without_scanning_pages_or_e ) .await; assert!(members > 64); + assert_eq!(members as usize, projected.dag().payloads().len()); + assert_eq!( + scalar(db, "SELECT rows FROM http_install_batch_count").await, + missing.len() as i64 + ); + assert_eq!( + stored_metadata_ids(db).await, + projected + .dag() + .payloads() + .iter() + .map(|page| page.id) + .collect() + ); assert_eq!( scalar(db, "SELECT batches FROM http_install_batch_count").await, - (members + 63) / 64 + missing_batches ); let writer = db.begin().await.unwrap(); assert!( @@ -931,7 +1034,7 @@ async fn mst2_durable_http_large_dag_batches_handoff_without_scanning_pages_or_e assert_eq!(resolved["descriptor"]["scope"], "/"); assert_eq!( scalar(db, "SELECT batches FROM http_install_batch_count").await, - (members + 63) / 64 + missing_batches ); fixture.counts.assert(0, 0); let warm = success_json( @@ -949,7 +1052,7 @@ async fn mst2_durable_http_large_dag_batches_handoff_without_scanning_pages_or_e ); assert_eq!( scalar(db, "SELECT batches FROM http_install_batch_count").await, - (members + 63) / 64 + missing_batches ); advance(&fixture).await; let state = rebuilt(&fixture).await; diff --git a/src/jupiter/service/native_publication_push_tests.rs b/src/jupiter/service/native_publication_push_tests.rs index 4fb89aa1..ce3febbb 100644 --- a/src/jupiter/service/native_publication_push_tests.rs +++ b/src/jupiter/service/native_publication_push_tests.rs @@ -106,7 +106,10 @@ async fn maintenance_cannot_reinitialize_published_history_after_head_loss() { mono.get_connection().execute_unprepared("DELETE FROM mst2_native_head").await.unwrap(); for instance in [NATIVE_INSTANCE, "11111111-2222-4333-8444-555555555555"] { let error = mono.initialize_native_publication_for_maintenance(instance, &head.root).await.unwrap_err(); - assert!(error.to_string().contains("native publication history exists")); + assert!(error.to_string().contains("native publication history exists"), "rejected initialization for {instance}: {error}"); + let released = mono.get_connection().begin().await.unwrap(); + assert!(PushQueueStorage::try_mono_write_lock(&released).await.unwrap(), "rejected initialization must release the mono write lock before returning"); + released.rollback().await.unwrap(); } assert_eq!(mst2_native_head::Entity::find().count(mono.get_connection()).await.unwrap(), 0); assert_eq!(mst2_native_publication::Entity::find().all(mono.get_connection()).await.unwrap(), certificates); @@ -116,7 +119,10 @@ async fn maintenance_cannot_reinitialize_published_history_after_head_loss() { mono.get_connection().execute_unprepared("DELETE FROM mst2_native_publication").await.unwrap(); for instance in [NATIVE_INSTANCE, "11111111-2222-4333-8444-555555555555"] { let error = mono.initialize_native_publication_for_maintenance(instance, &head.root).await.unwrap_err(); - assert!(error.to_string().contains("native publication history exists")); + assert!(error.to_string().contains("native publication history exists"), "rejected initialization for {instance}: {error}"); + let released = mono.get_connection().begin().await.unwrap(); + assert!(PushQueueStorage::try_mono_write_lock(&released).await.unwrap(), "rejected initialization must release the mono write lock before returning"); + released.rollback().await.unwrap(); } assert_eq!(mst2_native_head::Entity::find().count(mono.get_connection()).await.unwrap(), 0); assert_eq!(mst2_native_publication::Entity::find().count(mono.get_connection()).await.unwrap(), 0); diff --git a/src/jupiter/storage/native_publication_storage.rs b/src/jupiter/storage/native_publication_storage.rs index 2dbff900..979c7314 100644 --- a/src/jupiter/storage/native_publication_storage.rs +++ b/src/jupiter/storage/native_publication_storage.rs @@ -351,6 +351,7 @@ impl MonoStorage { .await? .ok_or_else(|| integrity("native history observation missing"))?; if history.try_get::("", "present")? { + txn.rollback().await?; return Err(integrity( "native publication history exists; initialization cannot repair or replace it", )); From bdb30546a561ee571e29bea4c348edb46208bb11 Mon Sep 17 00:00:00 2001 From: Xiaoyang Han Date: Thu, 8 Oct 2026 03:55:50 +0800 Subject: [PATCH 26/50] test(mst2): observe actual retention waits and drain deferred fixture events (#76) The persisted META concurrency oracle now observes the actual advisory lock tuple in its current PostgreSQL database and schema, without relying on application_name from a statistics snapshot. It still requires a real waiting competitor, unfinished reader and competitor, both exact route bytes, root/directory counts, release or GC behavior, successful post-read collection and rejection of the old context. A competitor that returns early reports its actual result and release/GC case instead of being hidden by a wait timeout. The source of the earlier timeout has not yet been established by a new native run. The bound-current corruption fixture keeps its active, deferrable, initially deferred protection trigger, executes its queued check explicitly before ALTER TABLE, restores deferred mode, and reenables the original mutation guard before commit. All six original damage/repair cases, exact HTTP errors, recovered canonical bytes and complete trigger-mode checks remain. This addresses the observed PostgreSQL pending-trigger-events error without disabling completeness protection. Production code and guards are unchanged. Local nightly formatting, diff and locked offline no-dependency metadata pass; new native results and all performance measurements remain pending. --- .../snapshot_persisted_metadata_tests.rs | 40 ++++++++++++++++--- 1 file changed, 35 insertions(+), 5 deletions(-) diff --git a/src/api/router/snapshot_persisted_metadata_tests.rs b/src/api/router/snapshot_persisted_metadata_tests.rs index 028f7d35..f66d58da 100644 --- a/src/api/router/snapshot_persisted_metadata_tests.rs +++ b/src/api/router/snapshot_persisted_metadata_tests.rs @@ -377,9 +377,10 @@ async fn wait_retention_waiter(txn: &sea_orm::DatabaseTransaction) { .unwrap(); if scalar( txn, - "SELECT count(*) FROM pg_locks l JOIN pg_stat_activity a ON a.pid=l.pid - WHERE l.locktype='advisory' AND NOT l.granted AND a.datname=current_database() - AND a.application_name=current_schema() AND l.classid=1296717362::oid + "SELECT count(*) FROM pg_locks l + WHERE l.locktype='advisory' AND NOT l.granted + AND l.database=(SELECT oid FROM pg_database WHERE datname=current_database()) + AND l.pid<>pg_backend_pid() AND l.classid=1296717362::oid AND l.objid=hashtext(current_schema())::oid AND l.objsubid=2", ) .await @@ -664,7 +665,7 @@ async fn mst2_persisted_meta_reader_holds_protection_until_all_route_bytes_are_o .await .unwrap(); let root = format!("page:{}", context.built.metadata_root); - let competing = { + let mut competing = { let state = fixture.state.clone(); let lease = fixture.lease.clone(); let root = root.clone(); @@ -692,7 +693,12 @@ async fn mst2_persisted_meta_reader_holds_protection_until_all_route_bytes_are_o .begin() .await .unwrap(); - wait_retention_waiter(&observer).await; + tokio::select! { + () = wait_retention_waiter(&observer) => {}, + outcome = &mut competing => { + panic!("retention competitor completed before waiting: release_lease={release_lease}, outcome={outcome:?}"); + }, + } assert!(!reading.is_finished()); assert!(!competing.is_finished()); resume.wait().await; @@ -834,6 +840,30 @@ async fn fault_binding(fixture: &Fixture, id: [u8; 32], case: usize, restore: bo .rows_affected(), 1 ); + if case == 1 { + let protection = txn + .query_one_raw(Statement::from_string( + DbBackend::Postgres, + "SELECT t.tgenabled::text AS mode,t.tgdeferrable AS deferrable,t.tginitdeferred AS deferred + FROM pg_trigger t WHERE t.tgrelid='mst2_metadata_current'::regclass + AND t.tgname='mst2_metadata_current_protected' AND NOT t.tgisinternal", + )) + .await + .unwrap() + .unwrap(); + assert!(matches!( + protection.try_get::("", "mode").unwrap().as_str(), + "O" | "A" + )); + assert!(protection.try_get::("", "deferrable").unwrap()); + assert!(protection.try_get::("", "deferred").unwrap()); + txn.execute_unprepared("SET CONSTRAINTS mst2_metadata_current_protected IMMEDIATE") + .await + .unwrap(); + txn.execute_unprepared("SET CONSTRAINTS mst2_metadata_current_protected DEFERRED") + .await + .unwrap(); + } txn.execute_unprepared(&format!("ALTER TABLE {table} ENABLE TRIGGER {guard}")) .await .unwrap(); From 208c306c65192149d652c992ae9f7c909ac590ba Mon Sep 17 00:00:00 2001 From: Xiaoyang Han Date: Thu, 8 Oct 2026 04:48:18 +0800 Subject: [PATCH 27/50] perf(mst2): persist source-bound chunk maps and authenticate selected pages (#77) Actual chunk-map, page and CHUNK routes replace the digest-only process cache with immutable source-bound PostgreSQL indexes and independent object-store receipts minted only after one complete current-OID size/SHA256/chunk/EOF pass. Exact-source cold callers share an installation gate, then each request reads its own backend receipt. Reconstructed warm readers consume no full source body and fetch only selected MCL2 pages and exact Merkle sibling intervals; CHUNK still reads and authenticates the current request OID range. Captured physical primary scope, bounded qualified SQL, full fact tuple, opaque verifier publication, deferred complete atomic installation, mutation-resistant receipts, owned descriptor/page/JSON credits, cancellation and per-transport lease checks retain fail-closed behavior. Old cache/runtime and page alias are removed; MCM2/MCL2 schema_version=2 remains the formal wire codec within v3. PostgreSQL and actual HTTP regression fixtures cover forged ordinary-DML indexes, receipt size/bytes/EOF, corrupted selected proofs, source and primary changes, temp shadows, SQL rollback, concurrency, cancellation and owned-reader release. Source formatting/diff/offline metadata pass. Native compile, Clippy, PG/HTTP tests and measured latency/RSS/Git comparison remain pending. --- .../router/snapshot_chunks_bounded_tests.rs | 29 + src/api/router/snapshot_content.rs | 292 +++- src/api/router/snapshot_content_tests.rs | 161 +- .../router/snapshot_objects_bounded_tests.rs | 27 +- .../snapshot_persisted_chunk_map_tests.rs | 1303 +++++++++++++++++ src/ceres/snapshot/chunk_map_gate.rs | 185 +++ src/ceres/snapshot/chunk_map_index.rs | 171 +++ .../snapshot/chunk_singleflight_tests.rs | 487 ------ src/ceres/snapshot/chunks.rs | 357 ++--- src/ceres/snapshot/chunks_stream.rs | 116 +- src/ceres/snapshot/chunks_stream_tests.rs | 269 ++-- src/ceres/snapshot/mod.rs | 2 + .../m20261008_000100_add_mst2_chunk_maps.rs | 28 + .../migration/m20261008_000100_chunk_maps.sql | 90 ++ src/jupiter/migration/mod.rs | 2 + src/jupiter/storage/mod.rs | 5 + src/jupiter/storage/mono_storage.rs | 43 +- src/jupiter/storage/native_chunk_map.rs | 684 +++++++++ src/jupiter/storage/object_storage.rs | 146 +- src/orbit/adapter/log.rs | 2 + src/orbit/adapter/object.rs | 59 + src/orbit_api/object_storage.rs | 23 + 22 files changed, 3433 insertions(+), 1048 deletions(-) create mode 100644 src/api/router/snapshot_persisted_chunk_map_tests.rs create mode 100644 src/ceres/snapshot/chunk_map_gate.rs create mode 100644 src/ceres/snapshot/chunk_map_index.rs delete mode 100644 src/ceres/snapshot/chunk_singleflight_tests.rs create mode 100644 src/jupiter/migration/m20261008_000100_add_mst2_chunk_maps.rs create mode 100644 src/jupiter/migration/m20261008_000100_chunk_maps.sql create mode 100644 src/jupiter/storage/native_chunk_map.rs diff --git a/src/api/router/snapshot_chunks_bounded_tests.rs b/src/api/router/snapshot_chunks_bounded_tests.rs index 8d22a68d..243d326a 100644 --- a/src/api/router/snapshot_chunks_bounded_tests.rs +++ b/src/api/router/snapshot_chunks_bounded_tests.rs @@ -214,6 +214,14 @@ async fn mst2_large_chunk_uses_current_oid_strict_range_faults_cancel_retry_and_ fixture.counts.assert(1, size as usize); let map_id = map["map"]["map_id"].as_str().unwrap(); fixture.counts.reset(); + super::persisted_chunk_maps::assert_three_page_proofs_and_selected_sibling_faults( + &fixture, map_id, digest, &pattern, + ) + .await; + // Each distinct OID earns its own full-stream receipt before map reuse. + assert_eq!(fixture.map("/other").await["map"], map["map"]); + fixture.counts.assert(1, size as usize); + fixture.counts.reset(); // Equal content map sharing never carries the first source's OID. let body = request("/other", digest, map_id, 512); assert_chunk( @@ -387,6 +395,26 @@ async fn mst2_chunk_batch_live_budget_and_invalid_later_path_reject_before_body_ ], ) .await; + let budget = crate::ceres::snapshot::content_budget::MemoryBudget::new(1024 * 1024); + let repository = crate::jupiter::storage::native_chunk_map::PostgresChunkMapRepository::new( + fixture + .state + .storage + .mono_storage() + .get_connection() + .clone(), + ) + .await + .unwrap() + .with_test_budget(budget.clone()); + assert!( + fixture + .state + .storage + .native_chunk_maps + .set(repository) + .is_ok() + ); let mut items = Vec::new(); for (index, path) in ["/file", "/one", "/two"].iter().enumerate() { let oid = oid_for(&fixture, path).await; @@ -417,6 +445,7 @@ async fn mst2_chunk_batch_live_budget_and_invalid_later_path_reject_before_body_ ) .await; fixture.counts.assert(0, 0); + assert_eq!(budget.used(), 0); let fixture = Fixture::new().await; let mut body = fixture.chunk_body("/file", &format!("sha256:{}", hex_of(&[1; 32])), "0"); let mut invalid = body["items"][0].clone(); diff --git a/src/api/router/snapshot_content.rs b/src/api/router/snapshot_content.rs index ea7d8938..d642434f 100644 --- a/src/api/router/snapshot_content.rs +++ b/src/api/router/snapshot_content.rs @@ -4,10 +4,9 @@ //! WP and `frame_encodings` advertises identity alone. use axum::{ - Json, extract::{Path as AxumPath, Query, State}, http::HeaderMap, - response::{IntoResponse, Response}, + response::Response, }; use futures::stream::StreamExt; use serde::Deserialize; @@ -19,8 +18,8 @@ use super::{ mst2_error_response, request::Mst2Bytes, }; use crate::ceres::snapshot::{ - chunks::{ChunkProjection, get_or_project_stream, projection_reservation_bytes}, - content_budget::{PROJECTION_LIVE_BYTES, reserve_range_work, reserve_response}, + chunks::{ChunkMapSource, VerifiedSourceChunkMap, map_build_reservation_bytes}, + content_budget::{BudgetedFrame, MemoryLease, reserve_range_work, reserve_response}, error::{SnapshotError, SnapshotErrorCode}, pages::{MetadataWalkOutcome, base64_of, hex_of, resolve_abs_metadata}, resolver::FsKind, @@ -32,6 +31,7 @@ pub(super) struct ResolvedFileMetadata { oid: String, pub(super) digest: [u8; 32], pub(super) size: u64, + fact: crate::callisto::mst2_verified_object::Model, } #[allow(clippy::result_large_err)] @@ -154,6 +154,7 @@ pub(super) async fn verified_file_metadata, #[serde(default)] - pub(super) page: Option, - /// Canonical v3 page requests bind the page to the already verified map - /// instead of repeating the file digest. - #[serde(default)] pub(super) map_id: Option, #[serde(default)] pub(super) page_index: Option, @@ -475,7 +472,8 @@ async fn project_for( scope: &str, path: &str, expected_digest: Option<&str>, -) -> Result, Response> { +) -> Result, Response> +{ let f = resolve_file_metadata(handler, root_tree, scope, path, expected_digest).await?; project_resolved(handler, &f).await } @@ -484,36 +482,61 @@ async fn project_for( async fn project_resolved( handler: &T, f: &ResolvedFileMetadata, -) -> Result, Response> { - // The first request for a digest builds the projection from the fixed - // Git object; later requests slice the cached representation. A miss - // rebuilds, never errors with "missing chunk". - let digest = f.digest; - let size = f.size; - let projection = get_or_project_stream(digest, size, || async move { - handler - .get_raw_blob_stream_by_hash(&f.oid) - .await - .map_err(content_read_error) - }) - .await - .map_err(|error| { - mst2_error_response(if error.code == SnapshotErrorCode::DigestMismatch { - SnapshotError::new( - SnapshotErrorCode::IntegrityError, - "fixed blob digest disagrees with its verified fact", - ) - } else { - error - }) - })?; - if projection.map.file_size != size || projection.map.file_content_id != digest { +) -> Result, Response> +{ + if f.size == 0 { return Err(mst2_error_response(SnapshotError::new( - SnapshotErrorCode::IntegrityError, - "cached projection disagrees with the fixed verified fact", + SnapshotErrorCode::ScopeInvalid, + "empty files have no chunk map", ))); } - Ok(projection) + let source = ChunkMapSource::from_fact(f.fact.clone(), &f.oid).map_err(mst2_error_response)?; + let storage = handler.get_context(); + let repository = storage.chunk_maps().await.map_err(mst2_error_response)?; + let objects = &storage.git_service.obj_storage; + if let Some(map) = repository + .read(&source, objects) + .await + .map_err(mst2_error_response)? + { + return Ok(map); + } + let flight = crate::ceres::snapshot::chunk_map_gate::InstallFlight::acquire( + repository + .source_identity(&source) + .map_err(mst2_error_response)?, + ) + .map_err(mst2_error_response)?; + let _gate = flight.lock().await.map_err(mst2_error_response)?; + if let Some(map) = repository + .read(&source, objects) + .await + .map_err(mst2_error_response)? + { + return Ok(map); + } + let verified = + VerifiedSourceChunkMap::verify(handler, source.clone(), repository.memory_budget()) + .await + .map_err(|error| { + mst2_error_response(if error.code == SnapshotErrorCode::DigestMismatch { + SnapshotError::new( + SnapshotErrorCode::IntegrityError, + "fixed blob digest disagrees with its verified fact", + ) + } else { + error + }) + })?; + repository + .install(verified, objects) + .await + .map_err(mst2_error_response)?; + repository + .read(&source, objects) + .await + .map_err(mst2_error_response)? + .ok_or_else(|| mst2_error_response(internal("installed chunk map source is missing"))) } #[allow(clippy::result_large_err)] @@ -539,10 +562,18 @@ pub(super) async fn chunk_map( q.expected_digest.as_deref(), ) .await?; - // Canonical v3 uses a closed top-level envelope and a nested map - // descriptor. Keeping the map under `map` is part of profile selection; - // the client rejects the legacy flat shape once canonical capabilities - // have been advertised. + let wire_bound = q + .path + .len() + .checked_mul(6) + .and_then(|n| n.checked_add(4096)) + .ok_or_else(|| mst2_error_response(internal("chunk map response bound overflow")))?; + let memory = reserve_response( + wire_bound + .checked_mul(2) + .ok_or_else(|| mst2_error_response(internal("chunk map response credit overflow")))?, + ) + .map_err(mst2_error_response)?; let body = json!({ "snapshot_id": snapshot_id, "path": q.path, @@ -557,7 +588,9 @@ pub(super) async fn chunk_map( "map_id": format!("sha256:{}", hex_of(&proj.map_id)), }, }); - Ok(Json(body).into_response()) + guarded_map_json_response(&state, &ctx, &body, memory) + .await + .map_err(mst2_error_response) } #[allow(clippy::result_large_err)] @@ -569,19 +602,13 @@ pub(super) async fn chunk_map_pages( ensure(&state)?; let ctx = super::request_context(&state, &snapshot_id).map_err(mst2_error_response)?; validate_scope_relative_path(&q.path).map_err(mst2_error_response)?; - // Canonical v3 uses `map_id` + `page_index`; retain parsing of the old - // names only while the legacy client is still present in this checkout. - let page_index: u64 = match (q.page_index.as_deref(), q.page.as_deref()) { - (Some(_), Some(_)) => { - return Err(mst2_error_response(SnapshotError::new( - SnapshotErrorCode::ScopeInvalid, - "page_index and page are mutually exclusive", - ))); - } - (Some(s), None) => parse_decimal_count(s, "page_index").map_err(mst2_error_response)?, - (None, Some(s)) => parse_decimal_count(s, "page").map_err(mst2_error_response)?, - (None, None) => 0, - }; + let page_index = q + .page_index + .as_deref() + .map(|s| parse_decimal_count(s, "page_index")) + .transpose() + .map_err(mst2_error_response)? + .unwrap_or(0); if q.page_index.is_some() != q.map_id.is_some() { return Err(mst2_error_response(SnapshotError::new( SnapshotErrorCode::ScopeInvalid, @@ -611,9 +638,19 @@ pub(super) async fn chunk_map_pages( ))); } } - let (leaf, proof) = proj - .leaf_and_proof(page_index) + let storage = handler.get_context(); + let page = storage + .chunk_maps() + .await + .map_err(mst2_error_response)? + .selected_page(&proj, page_index) + .await .map_err(mst2_error_response)?; + // Reserve both JSON values and the encoded wire buffer before either + // allocation. The authenticated leaf/proof retain their own lease. + let memory = reserve_response(64 * 1024).map_err(mst2_error_response)?; + let leaf = &page.leaf; + let proof = &page.proof; let leaf_bytes = leaf .encode() .map_err(|e| mst2_error_response(internal(format!("chunk leaf encode failed: {e}"))))?; @@ -639,7 +676,84 @@ pub(super) async fn chunk_map_pages( "leaf_base64": base64_of(&leaf_bytes), "proof": proof_json, }); - Ok(Json(body).into_response()) + guarded_map_json_response(&state, &ctx, &body, memory) + .await + .map_err(mst2_error_response) +} + +fn map_json_bytes( + value: &serde_json::Value, + memory: MemoryLease, +) -> Result { + let mut bytes = Vec::new(); + bytes + .try_reserve_exact(memory.bytes / 2) + .map_err(|_| internal("chunk map JSON allocation failed"))?; + let limit = memory.bytes / 2; + let mut writer = BoundedMapJsonWriter { + bytes: &mut bytes, + limit, + }; + serde_json::to_writer(&mut writer, value) + .map_err(|_| internal("chunk map JSON encoding exceeds its owned credit"))?; + if bytes.capacity() > memory.bytes { + return Err(internal( + "chunk map JSON allocation exceeds its owned credit", + )); + } + Ok(bytes::Bytes::from_owner(BudgetedFrame { + bytes, + lease: std::sync::Arc::new(memory), + })) +} + +struct BoundedMapJsonWriter<'a> { + bytes: &'a mut Vec, + limit: usize, +} + +impl std::io::Write for BoundedMapJsonWriter<'_> { + fn write(&mut self, input: &[u8]) -> std::io::Result { + if input.len() > self.limit - self.bytes.len() { + return Err(std::io::Error::other( + "chunk map JSON exceeds its wire bound", + )); + } + self.bytes.extend_from_slice(input); + Ok(input.len()) + } + + fn flush(&mut self) -> std::io::Result<()> { + Ok(()) + } +} + +async fn guarded_map_json_response( + state: &crate::api::MonoApiServiceState, + context: &crate::ceres::snapshot::runtime::SnapshotContext, + value: &serde_json::Value, + memory: MemoryLease, +) -> Result { + let bytes = map_json_bytes(value, memory)?; + let headers = super::REQUEST_HEADERS.try_with(Clone::clone).map_err(|_| { + SnapshotError::new( + SnapshotErrorCode::Unauthenticated, + "request authentication context missing", + ) + })?; + super::revalidate_access(state, context, &headers).await?; + let state = state.clone(); + let context = context.clone(); + let stream = futures::stream::once(async move { + super::revalidate_access(&state, &context, &headers).await?; + Ok::<_, SnapshotError>(bytes) + }); + Response::builder() + .header("content-type", "application/json") + .header("cache-control", "private, no-cache, no-transform") + .header("vary", "Authorization, Accept") + .body(axum::body::Body::from_stream(stream)) + .map_err(|_| internal("chunk map JSON response build failed")) } #[derive(Deserialize, Debug)] @@ -663,7 +777,8 @@ const CHUNKS_MAX_ITEMS: usize = 128; const CHUNKS_TOTAL_MAX: u64 = 128 * 1024 * 1024; struct Planned { - projection: std::sync::Arc, + projection: std::sync::Arc, + page: std::sync::Arc, oid: String, index: u64, } @@ -743,8 +858,9 @@ async fn read_chunk_range( } raw.extend_from_slice(&bytes); } - projection - .verify_chunk(planned.index, &raw) + planned + .page + .verify_chunk(&projection.map, planned.index, &raw) .map_err(mst2_error_response)?; Ok(raw) } @@ -784,7 +900,6 @@ pub(super) async fn chunks( let mut resolved: Vec = Vec::new(); let mut units: Vec<(String, u64)> = Vec::new(); let mut distinct: std::collections::HashMap<[u8; 32], u64> = std::collections::HashMap::new(); - let mut projection_bytes = 0usize; let mut logical_bytes = 0u64; for item in &req.items { validate_scope_relative_path(&item.path).map_err(mst2_error_response)?; @@ -836,16 +951,10 @@ pub(super) async fn chunks( } } std::collections::hash_map::Entry::Vacant(entry) => { - let bytes = projection_reservation_bytes(file.size).map_err(mst2_error_response)?; - projection_bytes = projection_bytes.checked_add(bytes).ok_or_else(|| { - mst2_error_response(internal("projection batch memory overflow")) - })?; - if projection_bytes > PROJECTION_LIVE_BYTES { - return Err(mst2_error_response(SnapshotError::new( - SnapshotErrorCode::LimitExceeded, - "chunk batch exceeds its live projection memory budget", - ))); - } + // Check the profile before any body I/O. Cold construction + // owns its credits and is dropped after durable installation; + // warm requests retain only descriptors and selected pages. + map_build_reservation_bytes(file.size).map_err(mst2_error_response)?; entry.insert(file.size); } } @@ -860,8 +969,19 @@ pub(super) async fn chunks( .and_then(|bytes| bytes.checked_add(req.items.len() * 1024 + 1024)) .ok_or_else(|| mst2_error_response(internal("chunk response memory overflow")))?; let response_memory = reserve_response(response_bytes).map_err(mst2_error_response)?; + let mut maps = std::collections::HashMap::new(); + let mut pages = std::collections::HashMap::new(); for item in resolved { - let proj = project_resolved(handler.as_ref(), &item.file).await?; + let source = ChunkMapSource::from_fact(item.file.fact.clone(), &item.file.oid) + .map_err(mst2_error_response)?; + let source_key = source.canonical_bytes().map_err(mst2_error_response)?; + let proj = if let Some(map) = maps.get(&source_key) { + std::sync::Arc::clone(map) + } else { + let map = project_resolved(handler.as_ref(), &item.file).await?; + maps.insert(source_key, std::sync::Arc::clone(&map)); + map + }; let want_map = format!("sha256:{}", hex_of(&proj.map_id)); if want_map != item.map_id { return Err(mst2_error_response(SnapshotError::new( @@ -869,8 +989,27 @@ pub(super) async fn chunks( "map_id does not bind to the fixed file", ))); } + let page_key = ( + proj.source_id(), + item.index / mst2_codec::chunkmap::CHUNKS_PER_PAGE as u64, + ); + let page = if let Some(page) = pages.get(&page_key) { + std::sync::Arc::clone(page) + } else { + let storage = handler.get_context(); + let page = storage + .chunk_maps() + .await + .map_err(mst2_error_response)? + .selected_page(&proj, page_key.1) + .await + .map_err(mst2_error_response)?; + pages.insert(page_key, std::sync::Arc::clone(&page)); + page + }; planned.push(Planned { projection: proj, + page, oid: item.file.oid, index: item.index, }); @@ -883,14 +1022,7 @@ pub(super) async fn chunks( let _work_memory = reserve_range_work().map_err(mst2_error_response)?; // The source is the current request's fixed OID, never a cached // handler/backend/credential from a different scope. - let bytes = if p.projection.has_inline_bytes() { - p.projection - .chunk_bytes(p.index) - .map_err(mst2_error_response)? - .to_vec() - } else { - read_chunk_range(handler.as_ref(), p).await? - }; + let bytes = read_chunk_range(handler.as_ref(), p).await?; let frame = stream .chunk( p.projection.map_id, diff --git a/src/api/router/snapshot_content_tests.rs b/src/api/router/snapshot_content_tests.rs index bb58d788..4399e692 100644 --- a/src/api/router/snapshot_content_tests.rs +++ b/src/api/router/snapshot_content_tests.rs @@ -1,7 +1,7 @@ use std::{ sync::{ Arc, - atomic::{AtomicUsize, Ordering}, + atomic::{AtomicBool, AtomicUsize, Ordering}, }, time::Duration, }; @@ -85,6 +85,9 @@ mod bounded_objects; #[path = "snapshot_chunks_bounded_tests.rs"] mod bounded_chunks; +#[path = "snapshot_persisted_chunk_map_tests.rs"] +mod persisted_chunk_maps; + #[path = "snapshot_session_tests.rs"] mod durable_sessions; @@ -100,6 +103,8 @@ mod generation_qualified_fixture; #[path = "snapshot_install_capability_fixture.rs"] mod install_capability_fixture; +type ReceiptWriteHold = (Arc, Arc); + #[derive(Default)] struct ReadCounts { whole: AtomicUsize, @@ -107,6 +112,14 @@ struct ReadCounts { bytes: AtomicUsize, object_fault: std::sync::Mutex>, chunk_faults: std::sync::Mutex>, + receipt_reads: AtomicUsize, + receipt_writes: AtomicUsize, + receipt_write_fail_after_create: AtomicBool, + receipt_read_failure: AtomicBool, + receipt_read_corruption: std::sync::Mutex>, + receipt_read_meta_size: std::sync::atomic::AtomicI64, + receipt_read_late_error: AtomicBool, + receipt_write_holds: std::sync::Mutex>, } impl ReadCounts { @@ -114,6 +127,8 @@ impl ReadCounts { self.whole.store(0, Ordering::SeqCst); self.range.store(0, Ordering::SeqCst); self.bytes.store(0, Ordering::SeqCst); + self.receipt_reads.store(0, Ordering::SeqCst); + self.receipt_writes.store(0, Ordering::SeqCst); } fn assert(&self, whole: usize, bytes: usize) { @@ -130,6 +145,36 @@ struct CountingStorage { #[async_trait::async_trait] impl MegaObjectStorage for CountingStorage { + async fn put_metadata_atomic_create( + &self, + key: &ObjectKey, + bytes: Bytes, + meta: ObjectMeta, + ) -> OrbitResult<()> { + self.inner + .inner + .put_metadata_atomic_create(key, bytes, meta) + .await?; + if key.namespace == ObjectNamespace::ChunkMapReceipt { + self.counts.receipt_writes.fetch_add(1, Ordering::SeqCst); + let holds = self.counts.receipt_write_holds.lock().unwrap().clone(); + if let Some((entered, release)) = holds { + entered.notify_one(); + release.notified().await; + } + if self + .counts + .receipt_write_fail_after_create + .swap(false, Ordering::SeqCst) + { + return Err(crate::orbit_api::error::IoOrbitError::Other( + "injected post-create receipt failure".into(), + )); + } + } + Ok(()) + } + async fn put_stream( &self, key: &ObjectKey, @@ -140,6 +185,39 @@ impl MegaObjectStorage for CountingStorage { } async fn get_stream(&self, key: &ObjectKey) -> OrbitResult<(ObjectByteStream, ObjectMeta)> { + if key.namespace == ObjectNamespace::ChunkMapReceipt { + self.counts.receipt_reads.fetch_add(1, Ordering::SeqCst); + if self.counts.receipt_read_failure.load(Ordering::SeqCst) { + return Err( + crate::orbit_api::error::IoOrbitError::object_store_not_found( + key.default_sharding(), + ), + ); + } + let bad = self.counts.receipt_read_corruption.lock().unwrap().clone(); + if let Some(bytes) = bad { + let declared = self.counts.receipt_read_meta_size.load(Ordering::SeqCst); + let size = if declared > 0 { + declared + } else { + bytes.len() as i64 + }; + let mut parts = vec![Ok(bytes)]; + if self.counts.receipt_read_late_error.load(Ordering::SeqCst) { + parts.push(Err(std::io::Error::other( + "injected late receipt read error", + ))); + } + return Ok(( + Box::pin(futures::stream::iter(parts)), + ObjectMeta { + size, + ..Default::default() + }, + )); + } + return self.inner.inner.get_stream(key).await; + } self.counts.whole.fetch_add(1, Ordering::SeqCst); let chunk_fault = self .counts @@ -220,10 +298,21 @@ impl MegaObjectStorage for CountingStorage { (Box::pin(stream) as ObjectByteStream, meta) })); } - self.inner + let result = self + .inner .inner .get_range_stream_exact(key, start, end) - .await + .await?; + Ok(result.map(|(stream, meta)| { + let counts = self.counts.clone(); + let stream = stream.map(move |part| { + if let Ok(bytes) = &part { + counts.bytes.fetch_add(bytes.len(), Ordering::SeqCst); + } + part + }); + (Box::pin(stream) as ObjectByteStream, meta) + })) } async fn exists(&self, key: &ObjectKey) -> OrbitResult { @@ -500,11 +589,16 @@ impl Fixture { .save_object_from_raw(Bytes::copy_from_slice(raw)) .await .unwrap(); - project_items.push(item( - TreeItemMode::Blob, - ObjectHash::from_hex_for_kind(HashKind::Sha1, &oid).unwrap(), - name, - )); + let mut oid = ObjectHash::from_hex_for_kind(HashKind::Sha1, &oid).unwrap(); + let mut mode = TreeItemMode::Blob; + let components: Vec<_> = name.split('/').collect(); + for index in (1..components.len()).rev() { + let child = tree(vec![item(mode, oid, components[index])]); + oid = child.id; + mode = TreeItemMode::Tree; + extra_trees.push(child); + } + project_items.push(item(mode, oid, components[0])); } let project = tree(project_items); let old_tip = Commit::from_tree_id_with_kind( @@ -814,7 +908,7 @@ async fn mst2_fixed_head_uses_verified_facts_without_body_reads_and_preserves_ra } #[tokio::test] -async fn mst2_fixed_warm_map_leaf_and_chunk_aliases_skip_body_reads() { +async fn mst2_fixed_warm_map_and_leaf_aliases_skip_body_reads_and_chunks_use_current_ranges() { let fixture = Fixture::new().await; let initial = fixture.map("/file").await; fixture.counts.assert(1, fixture.raw.len()); @@ -823,18 +917,6 @@ async fn mst2_fixed_warm_map_leaf_and_chunk_aliases_skip_body_reads() { assert_eq!(initial["map"]["chunk_count"], "2"); let map_id = initial["map"]["map_id"].as_str().unwrap(); fixture.counts.reset(); - fixture - .state - .storage - .git_service - .obj_storage - .inner - .delete(&ObjectKey { - namespace: ObjectNamespace::Git, - key: fixture.oid.clone(), - }) - .await - .unwrap(); for path in ["/file", "/alias", "/executable", "/nested/file"] { let cached = fixture.map(path).await; assert_eq!(cached["path"], path); @@ -902,8 +984,43 @@ async fn mst2_fixed_warm_map_leaf_and_chunk_aliases_skip_body_reads() { <[u8; 32]>::from(Sha256::digest(body.as_bytes())) ); } - fixture.counts.assert(0, 0); + assert_eq!(fixture.counts.whole.load(Ordering::SeqCst), 0); + assert_eq!(fixture.counts.range.load(Ordering::SeqCst), 2); + assert_eq!( + fixture.counts.bytes.load(Ordering::SeqCst), + fixture.raw.len() + ); + fixture.counts.reset(); } + fixture + .state + .storage + .git_service + .obj_storage + .inner + .delete(&ObjectKey { + namespace: ObjectNamespace::Git, + key: fixture.oid.clone(), + }) + .await + .unwrap(); + assert_eq!(fixture.map("/alias").await["map"], initial["map"]); + fixture.counts.assert(0, 0); + error( + fixture + .send( + "POST", + "chunks", + Body::from(fixture.chunk_body("/alias", map_id, "0").to_string()), + ) + .await, + 503, + "OBJECT_UNAVAILABLE", + false, + ) + .await; + assert_eq!(fixture.counts.whole.load(Ordering::SeqCst), 0); + assert_eq!(fixture.counts.range.load(Ordering::SeqCst), 1); } #[tokio::test] diff --git a/src/api/router/snapshot_objects_bounded_tests.rs b/src/api/router/snapshot_objects_bounded_tests.rs index d001de12..b5e0d303 100644 --- a/src/api/router/snapshot_objects_bounded_tests.rs +++ b/src/api/router/snapshot_objects_bounded_tests.rs @@ -7,11 +7,11 @@ use super::*; #[derive(Clone)] pub(super) struct StreamFault { pub(super) oid: String, - kind: FaultKind, + pub(super) kind: FaultKind, } #[derive(Clone)] -enum FaultKind { +pub(super) enum FaultKind { Parts(Vec), LateError(Bytes), Oversized(Bytes, Arc), @@ -21,6 +21,11 @@ enum FaultKind { release: Arc, drops: Arc, }, + HeldThenError { + entered: Arc, + release: Arc, + drops: Arc, + }, } struct DropCount(Arc); @@ -34,6 +39,24 @@ impl Drop for DropCount { impl StreamFault { pub(super) fn stream(self) -> ObjectByteStream { match self.kind { + FaultKind::HeldThenError { + entered, + release, + drops, + } => Box::pin(futures::stream::unfold( + (true, entered, release, DropCount(drops)), + |(first, entered, release, owner)| async move { + if !first { + return None; + } + entered.notify_one(); + release.notified().await; + Some(( + Err(io::Error::other("held source read failed")), + (false, entered, release, owner), + )) + }, + )), FaultKind::Parts(parts) => Box::pin(futures::stream::iter(parts.into_iter().map(Ok))), FaultKind::LateError(raw) => Box::pin(futures::stream::iter([ Ok(raw), diff --git a/src/api/router/snapshot_persisted_chunk_map_tests.rs b/src/api/router/snapshot_persisted_chunk_map_tests.rs new file mode 100644 index 00000000..6b261115 --- /dev/null +++ b/src/api/router/snapshot_persisted_chunk_map_tests.rs @@ -0,0 +1,1303 @@ +use sea_orm::{DatabaseConnection, DbBackend, IsolationLevel, Statement, TransactionTrait}; +use tokio::{sync::Notify, time::timeout}; + +use super::*; +use crate::{ + ceres::snapshot::{ + chunks::{ChunkMapSource, ChunkProjection}, + content_budget::MemoryBudget, + }, + jupiter::storage::native_chunk_map::PostgresChunkMapRepository, +}; + +fn statement(sql: &str, values: [sea_orm::Value; N]) -> Statement { + Statement::from_sql_and_values(DbBackend::Postgres, sql, values) +} + +async fn count(db: &DatabaseConnection, table: &str) -> i64 { + db.query_one_raw(statement( + &format!("SELECT count(*) AS count FROM {table}"), + [], + )) + .await + .unwrap() + .unwrap() + .try_get("", "count") + .unwrap() +} + +async fn reconstructed(fixture: &Fixture) -> MonoApiServiceState { + let mut config = (*fixture.state.storage.config()).clone(); + config.database.max_connection = 1; + config.database.min_connection = 1; + let config = Arc::new(config); + let connection = crate::jupiter::storage::init::postgres_connection(&config.database) + .await + .unwrap(); + let storage = crate::jupiter::storage::Storage::new_with_connection( + config, + Arc::new(connection), + fixture.state.storage.git_service.obj_storage.clone(), + ) + .await + .unwrap(); + MonoApiServiceState { + storage, + git_object_cache: Arc::new(GitObjectCache { + connection: fixture.state.git_object_cache.connection.clone(), + prefix: uuid::Uuid::new_v4().to_string(), + }), + ..fixture.state.clone() + } +} + +fn router(state: &MonoApiServiceState) -> Router { + Router::new().nest("/api/v2", routers(state.clone()).with_state(state.clone())) +} + +#[tokio::test] +async fn persisted_map_rebuilt_actual_http_uses_canonical_pages_and_only_requested_raw_range() { + let fixture = Fixture::new_with_pg_config(true).await; + let original = fixture.map("/file").await; + fixture.counts.assert(1, fixture.raw.len()); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 1); + let oracle = ChunkProjection::build(fixture.digest, fixture.raw.clone()).unwrap(); + assert_eq!( + original["map"]["map_id"], + format!("sha256:{}", hex_of(&oracle.map_id)) + ); + let state = reconstructed(&fixture).await; + assert!(state.storage.native_chunk_maps.get().is_none()); + let app = router(&state); + fixture.counts.reset(); + let map = success_json( + app.clone() + .oneshot(fixture.request("GET", "chunk-map?path=/alias", Body::empty())) + .await + .unwrap(), + ) + .await; + assert_eq!(map["map"], original["map"]); + let map_id = map["map"]["map_id"].as_str().unwrap(); + let page = success_json( + app.clone() + .oneshot(fixture.request( + "GET", + &format!("chunk-map/pages?path=/alias&map_id={map_id}&page_index=0"), + Body::empty(), + )) + .await + .unwrap(), + ) + .await; + let bytes = STANDARD + .decode(page["leaf_base64"].as_str().unwrap()) + .unwrap(); + let (leaf, proof) = oracle.leaf_and_proof(0).unwrap(); + assert_eq!(bytes, leaf.encode().unwrap()); + assert!(proof.is_empty()); + assert_eq!(page["proof"], json!([])); + fixture.counts.assert(0, 0); + assert!(fixture.counts.receipt_reads.load(Ordering::SeqCst) >= 2); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 0); + let body = fixture.chunk_body("/alias", map_id, "1").to_string(); + let response = app + .oneshot(fixture.request("POST", "chunks", Body::from(body.clone()))) + .await + .unwrap(); + assert_eq!(response.status(), 200); + let wire = to_bytes(response.into_body(), 2 * 1024 * 1024) + .await + .unwrap(); + let frames = parse_stream(&wire).unwrap(); + let [Frame::Chunk(chunk), Frame::End(end)] = frames.as_slice() else { + panic!("expected exact CHUNK and END"); + }; + assert_eq!(chunk.chunk_bytes, fixture.raw[CHUNK_SIZE as usize..]); + assert_eq!(chunk.chunk_index, 1); + assert_eq!(chunk.file_content_id, fixture.digest); + assert_eq!(chunk.map_id, oracle.map_id); + assert_eq!(end.request_item_count, 1); + assert_eq!(end.unique_unit_count, 1); + assert_eq!(end.logical_bytes, 113); + assert_eq!( + end.request_body_sha256, + <[u8; 32]>::from(Sha256::digest(body.as_bytes())) + ); + assert_eq!(fixture.counts.whole.load(Ordering::SeqCst), 0); + assert_eq!(fixture.counts.range.load(Ordering::SeqCst), 1); + assert_eq!(fixture.counts.bytes.load(Ordering::SeqCst), 113); +} + +#[tokio::test] +async fn persisted_map_current_fact_tuple_receipt_and_leaf_corruption_fail_closed_without_fallback() +{ + let fixture = Fixture::new().await; + let map = fixture.map("/file").await; + let original = fixture.fact().await; + let map_id = map["map"]["map_id"].as_str().unwrap(); + fixture.counts.reset(); + for case in 0..3 { + let mut fact = original.clone(); + match case { + 0 => fact.id += 1_000_000, + 1 => fact.created_at += chrono::Duration::seconds(1), + _ => fact.raw_sha256[0] ^= 1, + } + fixture.replace_fact(fact).await; + error( + fixture + .send("GET", "chunk-map?path=/file", Body::empty()) + .await, + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + fixture.counts.assert(0, 0); + } + fixture.replace_fact(original).await; + fixture + .counts + .receipt_read_failure + .store(true, Ordering::SeqCst); + error( + fixture + .send("GET", "chunk-map?path=/file", Body::empty()) + .await, + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + fixture + .counts + .receipt_read_failure + .store(false, Ordering::SeqCst); + *fixture.counts.receipt_read_corruption.lock().unwrap() = + Some(Bytes::from_static(b"forged DB proof")); + error( + fixture + .send("GET", "chunk-map?path=/file", Body::empty()) + .await, + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + *fixture.counts.receipt_read_corruption.lock().unwrap() = None; + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + let leaf: Vec = db + .query_one_raw(statement( + "SELECT payload FROM mst2_chunk_map_leaf WHERE map_id=$1 AND page_index=0", + [hex::decode(map_id.trim_start_matches("sha256:")) + .unwrap() + .into()], + )) + .await + .unwrap() + .unwrap() + .try_get("", "payload") + .unwrap(); + db.execute_unprepared("ALTER TABLE mst2_chunk_map_leaf DISABLE TRIGGER USER") + .await + .unwrap(); + let mut bad = leaf.clone(); + bad[16] ^= 1; + db.execute_raw(statement( + "UPDATE mst2_chunk_map_leaf SET payload=$1", + [bad.into()], + )) + .await + .unwrap(); + db.execute_unprepared("ALTER TABLE mst2_chunk_map_leaf ENABLE TRIGGER USER") + .await + .unwrap(); + for (method, suffix, body) in [ + ( + "GET", + format!("chunk-map/pages?path=/file&map_id={map_id}&page_index=0"), + Body::empty(), + ), + ( + "POST", + "chunks".to_string(), + Body::from(fixture.chunk_body("/file", map_id, "0").to_string()), + ), + ] { + error( + fixture.send(method, &suffix, body).await, + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + } + fixture.counts.assert(0, 0); + db.execute_unprepared("ALTER TABLE mst2_chunk_map_leaf DISABLE TRIGGER USER") + .await + .unwrap(); + db.execute_raw(statement( + "UPDATE mst2_chunk_map_leaf SET payload=$1", + [leaf.into()], + )) + .await + .unwrap(); + db.execute_unprepared("ALTER TABLE mst2_chunk_map_leaf ENABLE TRIGGER USER") + .await + .unwrap(); + assert_eq!(fixture.map("/file").await["map"], map["map"]); + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn ordinary_db_dml_cannot_forge_full_body_admission_or_mutate_admitted_indexes() { + let fixture = Fixture::new().await; + let source = ChunkMapSource::from_fact(fixture.fact().await, &fixture.oid).unwrap(); + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + let repository = PostgresChunkMapRepository::new(db.clone()).await.unwrap(); + let scope = repository.test_primary_scope(); + let source_bytes = source.canonical_bytes().unwrap(); + let leaf = ChunkLeaf { + page_index: 0, + chunk_sha256: vec![[9; 32]; 2], + }; + let root = leaf.leaf_hash().unwrap(); + let map = mst2_codec::chunkmap::ChunkMap::new(fixture.digest, fixture.raw.len() as u64, root) + .unwrap(); + let mut receipt = b"MST2-CHUNK-MAP-RECEIPT\0".to_vec(); + receipt.extend_from_slice(&(scope.len() as u32).to_be_bytes()); + receipt.extend_from_slice(scope); + receipt.extend_from_slice(&(source_bytes.len() as u32).to_be_bytes()); + receipt.extend_from_slice(&source_bytes); + let source_id: [u8; 32] = Sha256::digest(&receipt).into(); + receipt.extend_from_slice(&map.encode()); + let txn = db + .begin_with_config(Some(IsolationLevel::ReadCommitted), None) + .await + .unwrap(); + txn.execute_raw(statement( + "INSERT INTO mst2_chunk_map(map_id,descriptor,page_count,pages_root) VALUES($1,$2,1,$3)", + [ + map.map_id().to_vec().into(), + map.encode().into(), + root.to_vec().into(), + ], + )) + .await + .unwrap(); + txn.execute_raw(statement( + "INSERT INTO mst2_chunk_map_leaf(map_id,page_index,payload) VALUES($1,0,$2)", + [map.map_id().to_vec().into(), leaf.encode().unwrap().into()], + )) + .await + .unwrap(); + txn.execute_raw(statement( + "INSERT INTO mst2_chunk_map_node(map_id,first_page,page_count,digest) VALUES($1,0,1,$2)", + [map.map_id().to_vec().into(), root.to_vec().into()], + )) + .await + .unwrap(); + txn.execute_raw(statement("INSERT INTO mst2_chunk_map_source(storage_domain,git_oid,object_kind,fact_id,source_id,source_bytes,primary_scope,map_id,receipt_digest) VALUES('git',$1,'blob',$2,$3,$4,$5,$6,$7)", [fixture.oid.clone().into(),source.fact().id.into(),source_id.to_vec().into(),source_bytes.into(),scope.to_vec().into(),map.map_id().to_vec().into(),Sha256::digest(&receipt).to_vec().into()])).await.unwrap(); + txn.commit().await.unwrap(); + // All row checks and even a self-computed receipt digest pass. They + // still cannot create the trusted writer's independent object receipt. + fixture.counts.reset(); + error( + fixture + .send("GET", "chunk-map?path=/file", Body::empty()) + .await, + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + fixture.counts.assert(0, 0); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 0); + for table in [ + "mst2_chunk_map", + "mst2_chunk_map_leaf", + "mst2_chunk_map_node", + "mst2_chunk_map_source", + ] { + assert!( + db.execute_unprepared(&format!("DELETE FROM {table}")) + .await + .is_err() + ); + assert!( + db.execute_unprepared(&format!("TRUNCATE {table} CASCADE")) + .await + .is_err() + ); + } + assert!( + db.execute_unprepared("UPDATE mst2_chunk_map_source SET fact_id=fact_id") + .await + .is_err() + ); + assert!(db.execute_raw(statement("INSERT INTO mst2_chunk_map_node(map_id,first_page,page_count,digest) VALUES($1,1,1,$2)", [map.map_id().to_vec().into(),root.to_vec().into()])).await.is_err()); +} + +#[tokio::test] +async fn receipt_orphan_failure_and_cancelled_install_release_owned_credit_and_replay_atomically() { + for cancel in [false, true] { + let fixture = Fixture::new().await; + let mono = fixture.state.storage.mono_storage(); + let budget = MemoryBudget::new(8 * 1024 * 1024); + let repository = PostgresChunkMapRepository::new(mono.get_connection().clone()) + .await + .unwrap() + .with_test_budget(budget.clone()); + assert!( + fixture + .state + .storage + .native_chunk_maps + .set(repository) + .is_ok() + ); + if cancel { + let entered = Arc::new(Notify::new()); + let release = Arc::new(Notify::new()); + *fixture.counts.receipt_write_holds.lock().unwrap() = Some((entered.clone(), release)); + let app = fixture.app.clone(); + let request = fixture.request("GET", "chunk-map?path=/file", Body::empty()); + let task = tokio::spawn(async move { app.oneshot(request).await }); + timeout(Duration::from_secs(10), entered.notified()) + .await + .unwrap(); + assert!(budget.used() > 4 * 1024 * 1024); + assert_eq!( + count(mono.get_connection(), "mst2_chunk_map_source").await, + 0 + ); + task.abort(); + assert!(task.await.err().unwrap().is_cancelled()); + *fixture.counts.receipt_write_holds.lock().unwrap() = None; + } else { + fixture + .counts + .receipt_write_fail_after_create + .store(true, Ordering::SeqCst); + error( + fixture + .send("GET", "chunk-map?path=/file", Body::empty()) + .await, + 503, + "TEMPORARY_UNAVAILABLE", + true, + ) + .await; + fixture + .counts + .receipt_write_fail_after_create + .store(false, Ordering::SeqCst); + } + assert_eq!(budget.used(), 0); + fixture.counts.assert(1, fixture.raw.len()); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 1); + assert_eq!(count(mono.get_connection(), "mst2_chunk_map").await, 0); + fixture.counts.reset(); + fixture.map("/file").await; + fixture.counts.assert(1, fixture.raw.len()); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 1); + assert_eq!(count(mono.get_connection(), "mst2_chunk_map").await, 1); + assert_eq!( + count(mono.get_connection(), "mst2_chunk_map_source").await, + 1 + ); + assert_eq!(count(mono.get_connection(), "mst2_chunk_map_leaf").await, 1); + assert_eq!(count(mono.get_connection(), "mst2_chunk_map_node").await, 1); + fixture.counts.reset(); + fixture.map("/file").await; + fixture.counts.assert(0, 0); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 0); + assert_eq!(budget.used(), 0); + } +} + +#[tokio::test] +async fn map_json_transport_bytes_keep_owned_credit_until_the_last_clone_drops() { + let budget = MemoryBudget::new(4096); + let bytes = super::super::map_json_bytes( + &json!({"map_id":"sha256:owned"}), + budget.reserve(4096).unwrap(), + ) + .unwrap(); + let mut stream = Body::from(bytes).into_data_stream(); + let transport = stream.next().await.unwrap().unwrap(); + let clone = transport.clone(); + drop(transport); + drop(stream); + assert_eq!(budget.used(), 4096); + assert!(budget.reserve(1).is_err()); + drop(clone); + assert_eq!(budget.used(), 0); + assert!(budget.reserve(4096).is_ok()); +} + +#[test] +fn json_wire_limit_rejects_growth_and_refunds_credit() { + let budget = MemoryBudget::new(4096); + let value = json!({"path":"\u{1}".repeat(1000)}); + let error = super::super::map_json_bytes(&value, budget.reserve(4096).unwrap()) + .err() + .unwrap(); + assert_eq!(error.code, SnapshotErrorCode::Internal); + assert_eq!(budget.used(), 0); +} + +async fn observer(fixture: &Fixture) -> crate::ceres::snapshot::chunk_map_gate::InstallFlight { + let repository = fixture.state.storage.chunk_maps().await.unwrap(); + let source = ChunkMapSource::from_fact(fixture.fact().await, &fixture.oid).unwrap(); + crate::ceres::snapshot::chunk_map_gate::InstallFlight::acquire( + repository.source_identity(&source).unwrap(), + ) + .unwrap() +} + +async fn wait_owners( + flight: &crate::ceres::snapshot::chunk_map_gate::InstallFlight, + owners: usize, +) { + timeout(Duration::from_secs(10), async { + loop { + if flight.test_owner_count() == owners { + break; + } + tokio::task::yield_now().await; + } + }) + .await + .expect("actual HTTP callers did not reach the same-source install gate"); +} + +fn held_leader(fixture: &Fixture) -> (Arc, Arc, tokio::task::JoinHandle) { + let entered = Arc::new(Notify::new()); + let release = Arc::new(Notify::new()); + *fixture.counts.receipt_write_holds.lock().unwrap() = Some((entered.clone(), release.clone())); + let app = fixture.app.clone(); + let request = fixture.request("GET", "chunk-map?path=/file", Body::empty()); + let task = tokio::spawn(async move { app.oneshot(request).await.unwrap() }); + (entered, release, task) +} + +#[tokio::test] +async fn same_source_cold_actual_http_callers_share_one_full_pass_and_each_recheck_their_receipt() { + let fixture = Fixture::new().await; + let (entered, release, leader) = held_leader(&fixture); + timeout(Duration::from_secs(10), entered.notified()) + .await + .unwrap(); + let flight = observer(&fixture).await; + let mut joined = Vec::new(); + for _ in 0..6 { + let app = fixture.app.clone(); + let request = fixture.request("GET", "chunk-map?path=/alias", Body::empty()); + joined.push(tokio::spawn( + async move { app.oneshot(request).await.unwrap() }, + )); + } + let other_counts = Arc::new(ReadCounts::default()); + other_counts + .receipt_read_failure + .store(true, Ordering::SeqCst); + let mut other_state = fixture.state.clone(); + other_state.storage.git_service = GitService { + obj_storage: MegaObjectStorageWrapper::new(Arc::new(CountingStorage { + inner: fixture.state.storage.git_service.obj_storage.clone(), + counts: other_counts.clone(), + })), + }; + let other_app = router(&other_state); + let request = fixture.request("GET", "chunk-map?path=/file", Body::empty()); + let rejected = tokio::spawn(async move { other_app.oneshot(request).await.unwrap() }); + wait_owners(&flight, 9).await; // Leader, seven callers and this observer. + fixture.counts.assert(1, fixture.raw.len()); + release.notify_one(); + let expected = success_json(leader.await.unwrap()).await; + for task in joined { + assert_eq!( + success_json(task.await.unwrap()).await["map"], + expected["map"] + ); + } + error(rejected.await.unwrap(), 502, "INTEGRITY_ERROR", false).await; + assert_eq!(other_counts.receipt_reads.load(Ordering::SeqCst), 1); + assert_eq!(other_counts.whole.load(Ordering::SeqCst), 0); + fixture.counts.assert(1, fixture.raw.len()); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 1); + assert_eq!(fixture.counts.receipt_reads.load(Ordering::SeqCst), 8); + wait_owners(&flight, 1).await; +} + +#[tokio::test] +async fn failed_or_cancelled_leader_and_cancelled_waiter_leave_actual_http_retry_capacity() { + for mode in 0..3 { + let fixture = Fixture::new().await; + let budget = MemoryBudget::new(8 * 1024 * 1024); + let repository = PostgresChunkMapRepository::new( + fixture + .state + .storage + .mono_storage() + .get_connection() + .clone(), + ) + .await + .unwrap() + .with_test_budget(budget.clone()); + assert!( + fixture + .state + .storage + .native_chunk_maps + .set(repository) + .is_ok() + ); + let (entered, release, leader) = held_leader(&fixture); + timeout(Duration::from_secs(10), entered.notified()) + .await + .unwrap(); + let flight = observer(&fixture).await; + let app = fixture.app.clone(); + let request = fixture.request("GET", "chunk-map?path=/alias", Body::empty()); + let waiter = tokio::spawn(async move { app.oneshot(request).await.unwrap() }); + wait_owners(&flight, 3).await; + assert!(budget.used() > 4 * 1024 * 1024); + *fixture.counts.receipt_write_holds.lock().unwrap() = None; + match mode { + 0 => { + fixture + .counts + .receipt_write_fail_after_create + .store(true, Ordering::SeqCst); + release.notify_one(); + error(leader.await.unwrap(), 503, "TEMPORARY_UNAVAILABLE", true).await; + success_json(waiter.await.unwrap()).await; + } + 1 => { + leader.abort(); + assert!(leader.await.err().unwrap().is_cancelled()); + success_json(waiter.await.unwrap()).await; + } + _ => { + waiter.abort(); + assert!(waiter.await.err().unwrap().is_cancelled()); + wait_owners(&flight, 2).await; + release.notify_one(); + success_json(leader.await.unwrap()).await; + } + } + wait_owners(&flight, 1).await; + drop(flight); + assert_eq!(budget.used(), 0); + let passes = if mode == 2 { 1 } else { 2 }; + fixture.counts.assert(passes, passes * fixture.raw.len()); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), passes); + fixture.counts.reset(); + fixture.map("/file").await; + fixture.counts.assert(0, 0); + assert_eq!(budget.used(), 0); + } +} + +#[tokio::test] +async fn actual_http_long_escaped_legal_path_has_bounded_owned_json_and_empty_files_have_no_map() { + let component = "\u{1}".repeat(255); + let mut components = vec![component; 15]; + components.push("\u{1}".repeat(239)); + let name = components.join("/"); + let path = format!("/{name}"); + assert_eq!(path.len(), 4080); + crate::ceres::snapshot::view::validate_scope_relative_path(&path).unwrap(); + let fixture = Fixture::new_with_pg_config_directories_and_objects( + false, + 0, + &[(name, b"escaped path body".to_vec())], + ) + .await; + let encoded = url::form_urlencoded::byte_serialize(path.as_bytes()).collect::(); + for whole in [1, 0] { + fixture.counts.reset(); + let response = fixture + .send("GET", &format!("chunk-map?path={encoded}"), Body::empty()) + .await; + assert_eq!(response.status(), 200); + assert_eq!( + response.headers()["cache-control"], + "private, no-cache, no-transform" + ); + let bytes = to_bytes(response.into_body(), 64 * 1024).await.unwrap(); + assert!(bytes.len() > 4 * 1024); + let value: Value = serde_json::from_slice(&bytes).unwrap(); + assert_eq!(value["path"], path); + assert_eq!(value["map"]["file_size"], "17"); + fixture + .counts + .assert(whole, if whole == 0 { 0 } else { 17 }); + } + fixture.counts.reset(); + for suffix in ["chunk-map?path=/empty", "chunk-map/pages?path=/empty"] { + error( + fixture.send("GET", suffix, Body::empty()).await, + 400, + "SCOPE_INVALID", + false, + ) + .await; + } + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn actual_http_and_repository_initialization_ignore_poisoned_temp_fact_scope_and_map_shadows() +{ + let fixture = Fixture::new_with_pg_config(true).await; + let expected = fixture.map("/file").await; + let state = reconstructed(&fixture).await; + assert!(state.storage.native_chunk_maps.get().is_none()); + let mono = state.storage.mono_storage(); + let db = mono.get_connection(); + let schema = fixture + ._schema + .as_ref() + .unwrap() + .schema() + .replace('"', "\"\""); + let tables = [ + "mst2_verified_object", + "mst2_metadata_storage_scope", + "mst2_chunk_map", + "mst2_chunk_map_source", + "mst2_chunk_map_leaf", + "mst2_chunk_map_node", + ]; + for table in tables { + db.execute_unprepared(&format!("CREATE TEMP TABLE {table}(LIKE \"{schema}\".{table}); INSERT INTO pg_temp.{table} SELECT * FROM \"{schema}\".{table}")).await.unwrap(); + } + db.execute_unprepared("UPDATE pg_temp.mst2_verified_object SET raw_sha256=decode(repeat('09',32),'hex'); UPDATE pg_temp.mst2_metadata_storage_scope SET storage_uuid='temp-poison'; UPDATE pg_temp.mst2_chunk_map SET descriptor=decode('00','hex'); UPDATE pg_temp.mst2_chunk_map_leaf SET payload=decode('00','hex'); UPDATE pg_temp.mst2_chunk_map_node SET digest=decode(repeat('09',32),'hex')").await.unwrap(); + fixture.counts.reset(); + let app = router(&state); + let map = success_json( + app.clone() + .oneshot(fixture.request("GET", "chunk-map?path=/file", Body::empty())) + .await + .unwrap(), + ) + .await; + assert_eq!(map, expected); + let map_id = map["map"]["map_id"].as_str().unwrap(); + let page = success_json( + app.clone() + .oneshot(fixture.request( + "GET", + &format!("chunk-map/pages?path=/file&map_id={map_id}&page_index=0"), + Body::empty(), + )) + .await + .unwrap(), + ) + .await; + let oracle = ChunkProjection::build(fixture.digest, fixture.raw.clone()).unwrap(); + assert_eq!( + STANDARD + .decode(page["leaf_base64"].as_str().unwrap()) + .unwrap(), + oracle.leaf_and_proof(0).unwrap().0.encode().unwrap() + ); + let body = fixture.chunk_body("/file", map_id, "1").to_string(); + let response = app + .clone() + .oneshot(fixture.request("POST", "chunks", Body::from(body))) + .await + .unwrap(); + assert_eq!(response.status(), 200); + let wire = to_bytes(response.into_body(), 2 * 1024 * 1024) + .await + .unwrap(); + let frames = parse_stream(&wire).unwrap(); + let [Frame::Chunk(chunk), Frame::End(end)] = frames.as_slice() else { + panic!("expected CHUNK and END"); + }; + assert_eq!(chunk.chunk_bytes, fixture.raw[CHUNK_SIZE as usize..]); + assert_eq!(end.logical_bytes, 113); + assert_eq!(fixture.counts.whole.load(Ordering::SeqCst), 0); + assert_eq!(fixture.counts.range.load(Ordering::SeqCst), 1); + assert_eq!(fixture.counts.bytes.load(Ordering::SeqCst), 113); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 0); + // Real primary changes still fail despite an apparently healthy shadow. + let actual = fixture.state.storage.mono_storage(); + let source = ChunkMapSource::from_fact(fixture.fact().await, &fixture.oid).unwrap(); + let repository = state.storage.chunk_maps().await.unwrap(); + actual.get_connection().execute_unprepared("ALTER TABLE mst2_metadata_storage_scope DISABLE TRIGGER USER; UPDATE mst2_metadata_storage_scope SET storage_uuid='changed-real-primary'; ALTER TABLE mst2_metadata_storage_scope ENABLE TRIGGER USER").await.unwrap(); + assert_eq!( + repository + .read(&source, &state.storage.git_service.obj_storage) + .await + .err() + .unwrap() + .code, + crate::ceres::snapshot::error::SnapshotErrorCode::IntegrityError + ); + error( + app.oneshot(fixture.request("GET", "chunk-map?path=/file", Body::empty())) + .await + .unwrap(), + 502, + "INTEGRITY_ERROR", + false, + ) + .await; +} + +pub(super) async fn assert_three_page_proofs_and_selected_sibling_faults( + fixture: &Fixture, + map_id: &str, + digest: [u8; 32], + pattern: &[u8], +) { + use mst2_codec::chunkmap::{ChunkMap, ProofSide, leaf_proof, merkle_root}; + let full: [u8; 32] = Sha256::digest(pattern).into(); + let final_chunk: [u8; 32] = Sha256::digest(&pattern[..7]).into(); + let leaves = [ + ChunkLeaf { + page_index: 0, + chunk_sha256: vec![full; 256], + }, + ChunkLeaf { + page_index: 1, + chunk_sha256: vec![full; 256], + }, + ChunkLeaf { + page_index: 2, + chunk_sha256: vec![final_chunk], + }, + ]; + let hashes: Vec<_> = leaves + .iter() + .map(|leaf| leaf.leaf_hash().unwrap()) + .collect(); + let root = merkle_root(&hashes).unwrap(); + let oracle = ChunkMap::new(digest, 512 * CHUNK_SIZE as u64 + 7, root).unwrap(); + assert_eq!(map_id, format!("sha256:{}", hex_of(&oracle.map_id()))); + for index in 0..3 { + let page = success_json( + fixture + .send( + "GET", + &format!("chunk-map/pages?path=/file&map_id={map_id}&page_index={index}"), + Body::empty(), + ) + .await, + ) + .await; + assert_eq!( + STANDARD + .decode(page["leaf_base64"].as_str().unwrap()) + .unwrap(), + leaves[index as usize].encode().unwrap() + ); + let proof = leaf_proof(&hashes, index).unwrap(); + let expected: Vec<_> = proof.iter().map(|step| json!({ + "side": if step.side == ProofSide::Left { "left" } else { "right" }, + "sibling_pages": step.sibling_pages.to_string(), "digest": format!("sha256:{}", hex_of(&step.digest)), + })).collect(); + assert_eq!(page["proof"], json!(expected)); + verify_leaf( + 3, + index, + leaves[index as usize].leaf_hash().unwrap(), + &proof, + root, + ) + .unwrap(); + } + fixture.counts.assert(0, 0); + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + let id = oracle.map_id().to_vec(); + for missing in [false, true] { + db.execute_unprepared("ALTER TABLE mst2_chunk_map_node DISABLE TRIGGER USER") + .await + .unwrap(); + if missing { + db.execute_raw(statement( + "DELETE FROM mst2_chunk_map_node WHERE map_id=$1 AND first_page=2 AND page_count=1", + [id.clone().into()], + )) + .await + .unwrap(); + } else { + db.execute_raw(statement("UPDATE mst2_chunk_map_node SET digest=$2 WHERE map_id=$1 AND first_page=2 AND page_count=1", [id.clone().into(), vec![9; 32].into()])).await.unwrap(); + } + db.execute_unprepared("ALTER TABLE mst2_chunk_map_node ENABLE TRIGGER USER") + .await + .unwrap(); + for (method, suffix, body) in [ + ( + "GET", + format!("chunk-map/pages?path=/file&map_id={map_id}&page_index=0"), + Body::empty(), + ), + ( + "POST", + "chunks".to_string(), + Body::from(fixture.chunk_body("/file", map_id, "0").to_string()), + ), + ] { + error( + fixture.send(method, &suffix, body).await, + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + } + fixture.counts.assert(0, 0); + // Page 2 proves itself through [0,2); the unrelated damaged node is + // absent from its selected SQL and does not turn into a full-map scan. + let page = success_json( + fixture + .send( + "GET", + &format!("chunk-map/pages?path=/file&map_id={map_id}&page_index=2"), + Body::empty(), + ) + .await, + ) + .await; + assert_eq!( + STANDARD + .decode(page["leaf_base64"].as_str().unwrap()) + .unwrap(), + leaves[2].encode().unwrap() + ); + db.execute_unprepared("ALTER TABLE mst2_chunk_map_node DISABLE TRIGGER USER") + .await + .unwrap(); + if missing { + db.execute_raw(statement("INSERT INTO mst2_chunk_map_node(map_id,first_page,page_count,digest) VALUES($1,2,1,$2)", [id.clone().into(), hashes[2].to_vec().into()])).await.unwrap(); + } else { + db.execute_raw(statement("UPDATE mst2_chunk_map_node SET digest=$2 WHERE map_id=$1 AND first_page=2 AND page_count=1", [id.clone().into(), hashes[2].to_vec().into()])).await.unwrap(); + } + db.execute_unprepared("ALTER TABLE mst2_chunk_map_node ENABLE TRIGGER USER") + .await + .unwrap(); + } + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn failed_digest_failed_body_and_cancelled_cold_producer_allow_the_joined_current_source_to_retry() + { + use super::bounded_objects::{FaultKind, StreamFault}; + for mode in 0..3 { + let fixture = Fixture::new().await; + let budget = MemoryBudget::new(8 * 1024 * 1024); + let repository = PostgresChunkMapRepository::new( + fixture + .state + .storage + .mono_storage() + .get_connection() + .clone(), + ) + .await + .unwrap() + .with_test_budget(budget.clone()); + assert!( + fixture + .state + .storage + .native_chunk_maps + .set(repository) + .is_ok() + ); + let entered = Arc::new(Notify::new()); + let release = Arc::new(Notify::new()); + let drops = Arc::new(AtomicUsize::new(0)); + let kind = if mode == 1 { + FaultKind::HeldThenError { + entered: entered.clone(), + release: release.clone(), + drops: drops.clone(), + } + } else { + let mut raw = fixture.raw.clone(); + if mode == 0 { + raw[0] ^= 1; + } + FaultKind::Held { + raw: Bytes::from(raw), + entered: entered.clone(), + release: release.clone(), + drops: drops.clone(), + } + }; + *fixture.counts.object_fault.lock().unwrap() = Some(StreamFault { + oid: fixture.oid.clone(), + kind, + }); + let app = fixture.app.clone(); + let request = fixture.request("GET", "chunk-map?path=/file", Body::empty()); + let leader = tokio::spawn(async move { app.oneshot(request).await.unwrap() }); + timeout(Duration::from_secs(10), entered.notified()) + .await + .unwrap(); + let flight = observer(&fixture).await; + let app = fixture.app.clone(); + let request = fixture.request("GET", "chunk-map?path=/alias", Body::empty()); + let waiter = tokio::spawn(async move { app.oneshot(request).await.unwrap() }); + wait_owners(&flight, 3).await; + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 0); + assert!(budget.used() > 4 * 1024 * 1024); + *fixture.counts.object_fault.lock().unwrap() = None; + if mode == 2 { + leader.abort(); + assert!(leader.await.err().unwrap().is_cancelled()); + } else { + release.notify_one(); + error( + leader.await.unwrap(), + if mode == 0 { 502 } else { 503 }, + if mode == 0 { + "INTEGRITY_ERROR" + } else { + "OBJECT_UNAVAILABLE" + }, + false, + ) + .await; + } + let map = success_json(waiter.await.unwrap()).await; + assert_eq!( + map["map"]["file_content_id"], + format!("sha256:{}", hex_of(&fixture.digest)) + ); + assert_eq!(drops.load(Ordering::SeqCst), 1); + assert_eq!(fixture.counts.whole.load(Ordering::SeqCst), 2); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 1); + let failed_bytes = match mode { + 0 => fixture.raw.len(), + 1 => 0, + _ => 1, + }; + assert_eq!( + fixture.counts.bytes.load(Ordering::SeqCst), + fixture.raw.len() + failed_bytes + ); + wait_owners(&flight, 1).await; + drop(flight); + assert_eq!(budget.used(), 0); + fixture.counts.reset(); + fixture.map("/file").await; + fixture.counts.assert(0, 0); + } +} + +#[tokio::test] +async fn different_current_sources_enter_cold_installations_independently() { + let fixture = Fixture::new_with_pg_config_directories_and_objects( + false, + 0, + &[("other".into(), b"independent source".to_vec())], + ) + .await; + let (entered, release, leader) = held_leader(&fixture); + timeout(Duration::from_secs(10), entered.notified()) + .await + .unwrap(); + let app = fixture.app.clone(); + let request = fixture.request("GET", "chunk-map?path=/other", Body::empty()); + let other = tokio::spawn(async move { app.oneshot(request).await.unwrap() }); + timeout(Duration::from_secs(10), entered.notified()) + .await + .unwrap(); + assert_eq!(fixture.counts.whole.load(Ordering::SeqCst), 2); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 2); + release.notify_waiters(); + let map = success_json(leader.await.unwrap()).await; + let second = success_json(other.await.unwrap()).await; + assert_ne!(map["map"]["map_id"], second["map"]["map_id"]); + fixture.counts.assert(2, fixture.raw.len() + 18); +} + +#[tokio::test] +async fn persisted_descriptors_and_authenticated_pages_charge_until_the_last_live_reader_drops() { + let fixture = Fixture::new().await; + fixture.map("/file").await; + fixture.counts.reset(); + let budget = MemoryBudget::new(96 * 1024); + let repository = PostgresChunkMapRepository::new( + fixture + .state + .storage + .mono_storage() + .get_connection() + .clone(), + ) + .await + .unwrap() + .with_test_budget(budget.clone()); + let source = ChunkMapSource::from_fact(fixture.fact().await, &fixture.oid).unwrap(); + let map = repository + .read(&source, &fixture.state.storage.git_service.obj_storage) + .await + .unwrap() + .unwrap(); + let page = repository.selected_page(&map, 0).await.unwrap(); + let other_map = map.clone(); + let other_page = page.clone(); + drop(map); + drop(page); + assert_eq!(budget.used(), 96 * 1024); + assert_eq!( + budget.reserve(1).err().unwrap().code, + SnapshotErrorCode::TemporaryUnavailable + ); + other_page + .verify_chunk(&other_map.map, 1, &fixture.raw[CHUNK_SIZE as usize..]) + .unwrap(); + drop(other_map); + assert_eq!(budget.used(), 64 * 1024); + drop(other_page); + assert_eq!(budget.used(), 0); + let replay = repository + .read(&source, &fixture.state.storage.git_service.obj_storage) + .await + .unwrap() + .unwrap(); + assert_eq!(budget.used(), 32 * 1024); + drop(replay); + assert_eq!(budget.used(), 0); + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn actual_install_sql_failure_rolls_back_every_index_and_exact_retry_earns_admission_again() { + let fixture = Fixture::new().await; + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + let budget = MemoryBudget::new(8 * 1024 * 1024); + let repository = PostgresChunkMapRepository::new(db.clone()) + .await + .unwrap() + .with_test_budget(budget.clone()); + assert!( + fixture + .state + .storage + .native_chunk_maps + .set(repository) + .is_ok() + ); + db.execute_unprepared("CREATE FUNCTION chunk_map_install_test_failure() RETURNS trigger LANGUAGE plpgsql AS $test$ BEGIN RAISE EXCEPTION 'injected source-row failure after complete index insertion'; END $test$; CREATE TRIGGER chunk_map_install_test_failure BEFORE INSERT ON mst2_chunk_map_source FOR EACH ROW EXECUTE FUNCTION chunk_map_install_test_failure()").await.unwrap(); + error( + fixture + .send("GET", "chunk-map?path=/file", Body::empty()) + .await, + 503, + "TEMPORARY_UNAVAILABLE", + true, + ) + .await; + fixture.counts.assert(1, fixture.raw.len()); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 1); + assert_eq!(budget.used(), 0); + for table in [ + "mst2_chunk_map", + "mst2_chunk_map_leaf", + "mst2_chunk_map_node", + "mst2_chunk_map_source", + ] { + assert_eq!(count(db, table).await, 0); + } + db.execute_unprepared("DROP TRIGGER chunk_map_install_test_failure ON mst2_chunk_map_source; DROP FUNCTION chunk_map_install_test_failure()").await.unwrap(); + fixture.counts.reset(); + let map = fixture.map("/file").await; + fixture.counts.assert(1, fixture.raw.len()); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 1); + assert_eq!(budget.used(), 0); + for table in [ + "mst2_chunk_map", + "mst2_chunk_map_leaf", + "mst2_chunk_map_node", + "mst2_chunk_map_source", + ] { + assert_eq!(count(db, table).await, 1); + } + fixture.counts.reset(); + assert_eq!(fixture.map("/alias").await["map"], map["map"]); + fixture.counts.assert(0, 0); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 0); +} + +#[tokio::test] +async fn ordinary_incomplete_source_dml_cannot_commit_an_admitted_partial_map() { + let fixture = Fixture::new().await; + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + let source = ChunkMapSource::from_fact(fixture.fact().await, &fixture.oid).unwrap(); + let repository = PostgresChunkMapRepository::new(db.clone()).await.unwrap(); + let map = ChunkProjection::build(fixture.digest, fixture.raw.clone()).unwrap(); + let scope = repository.test_primary_scope(); + let txn = db + .begin_with_config(Some(IsolationLevel::ReadCommitted), None) + .await + .unwrap(); + txn.execute_raw(statement( + "INSERT INTO mst2_chunk_map(map_id,descriptor,page_count,pages_root) VALUES($1,$2,$3,$4)", + [ + map.map_id.to_vec().into(), + map.map.encode().into(), + (map.map.page_count as i32).into(), + map.map.pages_root.to_vec().into(), + ], + )) + .await + .unwrap(); + txn.execute_raw(statement("INSERT INTO mst2_chunk_map_source(storage_domain,git_oid,object_kind,fact_id,source_id,source_bytes,primary_scope,map_id,receipt_digest) VALUES('git',$1,'blob',$2,$3,$4,$5,$6,$7)", [fixture.oid.clone().into(), source.fact().id.into(), repository.source_identity(&source).unwrap().to_vec().into(), source.canonical_bytes().unwrap().into(), scope.to_vec().into(), map.map_id.to_vec().into(), vec![9;32].into()])).await.unwrap(); + assert!(txn.commit().await.is_err()); + for table in [ + "mst2_chunk_map", + "mst2_chunk_map_leaf", + "mst2_chunk_map_node", + "mst2_chunk_map_source", + ] { + assert_eq!(count(db, table).await, 0); + } + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn actual_map_and_page_json_revalidate_lease_before_the_first_transport_poll() { + for page in [false, true] { + let fixture = Fixture::new().await; + let map = fixture.map("/file").await; + let suffix = if page { + format!( + "chunk-map/pages?path=/file&map_id={}&page_index=0", + map["map"]["map_id"].as_str().unwrap() + ) + } else { + "chunk-map?path=/file".into() + }; + fixture.counts.reset(); + let response = fixture.send("GET", &suffix, Body::empty()).await; + assert_eq!(response.status(), 200); + let revoked = fixture + .app + .clone() + .oneshot( + Request::builder() + .method("DELETE") + .uri(format!("/api/v2/snapshots/leases/{}", fixture.lease)) + .header("authorization", format!("Bearer {TOKEN}")) + .body(Body::empty()) + .unwrap(), + ) + .await + .unwrap(); + assert_eq!(revoked.status(), 200); + let mut stream = response.into_body().into_data_stream(); + assert!(stream.next().await.unwrap().is_err()); + assert!(stream.next().await.is_none()); + fixture.counts.assert(0, 0); + } +} + +#[tokio::test] +async fn oversized_current_fact_digest_fails_before_source_or_receipt_io_and_recovers_canonical_bytes() + { + let fixture = Fixture::new().await; + let expected = fixture.map("/file").await; + let original = fixture.fact().await; + let mono = fixture.state.storage.mono_storage(); + mono.get_connection().execute_raw(statement("UPDATE mst2_verified_object SET raw_sha256=decode(repeat('ab',1048576),'hex') WHERE id=$1", [original.id.into()])).await.unwrap(); + fixture.counts.reset(); + error( + fixture + .send("GET", "chunk-map?path=/file", Body::empty()) + .await, + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + fixture.counts.assert(0, 0); + assert_eq!(fixture.counts.receipt_reads.load(Ordering::SeqCst), 0); + fixture.replace_fact(original).await; + assert_eq!(fixture.map("/file").await, expected); + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn warm_admission_requires_receipt_exact_bytes_size_and_final_eof_without_fallback() { + let fixture = Fixture::new().await; + let map = fixture.map("/file").await; + let repository = fixture.state.storage.chunk_maps().await.unwrap(); + let source = ChunkMapSource::from_fact(fixture.fact().await, &fixture.oid).unwrap(); + let key = ObjectKey { + namespace: ObjectNamespace::ChunkMapReceipt, + key: hex_of(&repository.source_identity(&source).unwrap()), + }; + let (mut stream, meta) = fixture + .state + .storage + .git_service + .obj_storage + .inner + .get_stream(&key) + .await + .unwrap(); + let mut original = Vec::new(); + while let Some(part) = stream.next().await { + original.extend_from_slice(&part.unwrap()); + } + assert_eq!(original.len() as i64, meta.size); + let mut wrong = original.clone(); + wrong[0] ^= 1; + let mut long = original.clone(); + long.push(0); + let cases = [ + original[..original.len() - 1].to_vec(), + long, + wrong, + original.clone(), + ]; + fixture + .counts + .receipt_read_meta_size + .store(meta.size, Ordering::SeqCst); + for (index, bytes) in cases.into_iter().enumerate() { + fixture.counts.reset(); + *fixture.counts.receipt_read_corruption.lock().unwrap() = Some(Bytes::from(bytes)); + fixture + .counts + .receipt_read_late_error + .store(index == 3, Ordering::SeqCst); + error( + fixture + .send("GET", "chunk-map?path=/file", Body::empty()) + .await, + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + fixture.counts.assert(0, 0); + assert_eq!(fixture.counts.receipt_reads.load(Ordering::SeqCst), 1); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 0); + } + *fixture.counts.receipt_read_corruption.lock().unwrap() = None; + fixture + .counts + .receipt_read_late_error + .store(false, Ordering::SeqCst); + assert_eq!(fixture.map("/file").await, map); + fixture.counts.assert(0, 0); +} diff --git a/src/ceres/snapshot/chunk_map_gate.rs b/src/ceres/snapshot/chunk_map_gate.rs new file mode 100644 index 00000000..5c56a504 --- /dev/null +++ b/src/ceres/snapshot/chunk_map_gate.rs @@ -0,0 +1,185 @@ +//! Bounded exact-source install flights. Completed data is always re-read +//! from the requesting source's trusted receipt, never retained in a flight. + +use std::{ + collections::HashMap, + sync::{Arc, Mutex, OnceLock, Weak}, +}; + +use super::error::{SnapshotError, SnapshotErrorCode}; + +const MAX_INSTALL_FLIGHTS: usize = 128; +type SourceGate = tokio::sync::Mutex<()>; + +#[derive(Default)] +struct Registry { + entries: HashMap<[u8; 32], Weak>, +} + +pub(crate) struct InstallFlight { + key: [u8; 32], + gate: Option>, + registry: Arc>, +} + +impl InstallFlight { + pub(crate) fn acquire(key: [u8; 32]) -> Result { + static REGISTRY: OnceLock>> = OnceLock::new(); + Self::from_registry(REGISTRY.get_or_init(|| Arc::default()).clone(), key) + } + + fn from_registry(registry: Arc>, key: [u8; 32]) -> Result { + let gate = { + let mut state = registry.lock().map_err(|_| internal())?; + if let Some(gate) = state.entries.get(&key).and_then(Weak::upgrade) { + gate + } else { + state.entries.remove(&key); + if state.entries.len() >= MAX_INSTALL_FLIGHTS { + return Err(SnapshotError::new( + SnapshotErrorCode::LimitExceeded, + "too many distinct source map installations in flight", + )); + } + let gate = Arc::new(SourceGate::new(())); + state.entries.insert(key, Arc::downgrade(&gate)); + gate + } + }; + Ok(Self { + key, + gate: Some(gate), + registry, + }) + } + + pub(crate) async fn lock(&self) -> Result, SnapshotError> { + Ok(self.gate.as_ref().ok_or_else(internal)?.lock().await) + } + + #[cfg(test)] + pub(crate) fn test_owner_count(&self) -> usize { + self.gate.as_ref().map_or(0, Arc::strong_count) + } +} + +impl Drop for InstallFlight { + fn drop(&mut self) { + let Some(gate) = self.gate.take() else { + return; + }; + if let Ok(mut state) = self.registry.lock() { + if Arc::strong_count(&gate) == 1 + && state + .entries + .get(&self.key) + .is_some_and(|entry| entry.ptr_eq(&Arc::downgrade(&gate))) + { + state.entries.remove(&self.key); + } + // Owners must release their strong reference while the registry + // is locked, so concurrent final drops cannot leave a stale slot. + drop(gate); + } + } +} + +fn internal() -> SnapshotError { + SnapshotError::new( + SnapshotErrorCode::Internal, + "source map flight registry is unavailable", + ) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[tokio::test] + async fn exact_source_gate_blocks_only_same_source_and_returns_capacity_after_last_owner() { + let registry = Arc::new(Mutex::new(Registry::default())); + let first = InstallFlight::from_registry(registry.clone(), [1; 32]).unwrap(); + let lock = first.lock().await.unwrap(); + let same = InstallFlight::from_registry(registry.clone(), [1; 32]).unwrap(); + assert!(same.gate.as_ref().unwrap().try_lock().is_err()); + let other = InstallFlight::from_registry(registry.clone(), [2; 32]).unwrap(); + assert!(other.gate.as_ref().unwrap().try_lock().is_ok()); + drop(lock); + drop(first); + assert!(same.gate.as_ref().unwrap().try_lock().is_ok()); + assert_eq!(registry.lock().unwrap().entries.len(), 2); + drop(same); + drop(other); + assert!(registry.lock().unwrap().entries.is_empty()); + let mut held = Vec::new(); + for i in 0..MAX_INSTALL_FLIGHTS { + let mut key = [0; 32]; + key[..8].copy_from_slice(&(i as u64).to_le_bytes()); + held.push(InstallFlight::from_registry(registry.clone(), key).unwrap()); + } + assert_eq!( + InstallFlight::from_registry(registry.clone(), [255; 32]) + .err() + .unwrap() + .code, + SnapshotErrorCode::LimitExceeded + ); + let joined_at_capacity = InstallFlight::from_registry(registry.clone(), [0; 32]).unwrap(); + assert_eq!(joined_at_capacity.test_owner_count(), 2); + assert_eq!(registry.lock().unwrap().entries.len(), MAX_INSTALL_FLIGHTS); + drop(joined_at_capacity); + drop(held); + assert!(registry.lock().unwrap().entries.is_empty()); + assert!(InstallFlight::from_registry(registry, [255; 32]).is_ok()); + } + + #[test] + fn concurrent_last_owners_release_registry_capacity() { + for _ in 0..64 { + let registry = Arc::new(Mutex::new(Registry::default())); + let first = InstallFlight::from_registry(registry.clone(), [1; 32]).unwrap(); + let second = InstallFlight::from_registry(registry.clone(), [1; 32]).unwrap(); + let barrier = Arc::new(std::sync::Barrier::new(2)); + std::thread::scope(|scope| { + let other = barrier.clone(); + scope.spawn(move || { + other.wait(); + drop(first); + }); + scope.spawn(move || { + barrier.wait(); + drop(second); + }); + }); + assert!(registry.lock().unwrap().entries.is_empty()); + } + } + + #[test] + fn dropping_an_old_claim_does_not_remove_a_replacement_gate() { + let registry = Arc::new(Mutex::new(Registry::default())); + let old = InstallFlight::from_registry(registry.clone(), [1; 32]).unwrap(); + let replacement = Arc::new(SourceGate::new(())); + registry + .lock() + .unwrap() + .entries + .insert([1; 32], Arc::downgrade(&replacement)); + drop(old); + assert_eq!(registry.lock().unwrap().entries.len(), 1); + let current = InstallFlight::from_registry(registry.clone(), [1; 32]).unwrap(); + assert!(Arc::ptr_eq(current.gate.as_ref().unwrap(), &replacement)); + drop(replacement); + drop(current); + assert!(registry.lock().unwrap().entries.is_empty()); + registry + .lock() + .unwrap() + .entries + .insert([1; 32], Weak::new()); + let reclaimed = InstallFlight::from_registry(registry.clone(), [1; 32]).unwrap(); + assert_eq!(registry.lock().unwrap().entries.len(), 1); + drop(reclaimed); + assert!(registry.lock().unwrap().entries.is_empty()); + } +} diff --git a/src/ceres/snapshot/chunk_map_index.rs b/src/ceres/snapshot/chunk_map_index.rs new file mode 100644 index 00000000..7e253f9c --- /dev/null +++ b/src/ceres/snapshot/chunk_map_index.rs @@ -0,0 +1,171 @@ +//! Canonical MCL2 Merkle subtrees addressed by their exact leaf interval. + +use mst2_codec::chunkmap::{ProofSide, ProofStep}; +use sha2::{Digest, Sha256}; + +use super::error::{SnapshotError, SnapshotErrorCode}; + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(crate) struct ChunkMapNode { + pub start: u64, + pub pages: u64, + pub digest: [u8; 32], +} + +pub(crate) fn indexed_nodes(hashes: &[[u8; 32]]) -> Result, SnapshotError> { + if hashes.is_empty() || hashes.len() > 32_768 { + return Err(integrity("chunk map is outside the indexed page profile")); + } + let mut nodes = Vec::new(); + nodes.try_reserve_exact(hashes.len() * 2 - 1).map_err(|_| { + SnapshotError::new( + SnapshotErrorCode::TemporaryUnavailable, + "chunk map index allocation failed", + ) + })?; + build(hashes, 0, &mut nodes); + Ok(nodes) +} + +fn build(hashes: &[[u8; 32]], start: u64, nodes: &mut Vec) -> [u8; 32] { + let pages = hashes.len() as u64; + let digest = if pages == 1 { + hashes[0] + } else { + let split = split(pages); + let left = build(&hashes[..split as usize], start, nodes); + let right = build(&hashes[split as usize..], start + split, nodes); + let mut hash = Sha256::new(); + hash.update(b"mega.mst2.chunkbranch\0"); + hash.update(split.to_le_bytes()); + hash.update(left); + hash.update((pages - split).to_le_bytes()); + hash.update(right); + hash.finalize().into() + }; + nodes.push(ChunkMapNode { + start, + pages, + digest, + }); + digest +} + +fn split(pages: u64) -> u64 { + 1 << (63 - (pages - 1).leading_zeros()) +} + +/// Only these sibling intervals may be fetched for a selected page. +pub(crate) fn proof_intervals( + pages: u64, + index: u64, +) -> Result, SnapshotError> { + if pages == 0 || pages > 32_768 || index >= pages { + return Err(SnapshotError::new( + SnapshotErrorCode::PathNotFound, + "chunk map page does not exist", + )); + } + let (mut start, mut count) = (0, pages); + let mut intervals = Vec::new(); + while count > 1 { + let left = split(count); + if index < start + left { + intervals.push((ProofSide::Right, start + left, count - left)); + count = left; + } else { + intervals.push((ProofSide::Left, start, left)); + start += left; + count -= left; + } + } + intervals.reverse(); + Ok(intervals) +} + +pub(crate) fn selected_proof( + intervals: &[(ProofSide, u64, u64)], + nodes: &[ChunkMapNode], +) -> Result, SnapshotError> { + if nodes.len() != intervals.len() { + return Err(integrity( + "persisted chunk map proof coverage is incomplete", + )); + } + intervals + .iter() + .map(|&(side, start, pages)| { + let mut matching = nodes + .iter() + .filter(|n| n.start == start && n.pages == pages); + let node = matching + .next() + .ok_or_else(|| integrity("persisted chunk map sibling is missing"))?; + if matching.next().is_some() { + return Err(integrity("persisted chunk map sibling is duplicated")); + } + Ok(ProofStep { + side, + sibling_pages: pages, + digest: node.digest, + }) + }) + .collect() +} + +fn integrity(message: &str) -> SnapshotError { + SnapshotError::new(SnapshotErrorCode::IntegrityError, message) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn indexed_selected_proofs_match_independent_codec_for_uneven_trees() { + for count in [1, 2, 3, 5, 9, 17, 257, 32_768] { + let hashes: Vec<[u8; 32]> = (0u64..count) + .map(|n| Sha256::digest(n.to_le_bytes()).into()) + .collect(); + let nodes = indexed_nodes(&hashes).unwrap(); + assert_eq!(nodes.len(), hashes.len() * 2 - 1); + let root = mst2_codec::chunkmap::merkle_root(&hashes).unwrap(); + assert_eq!(nodes.last().unwrap().digest, root); + for page in [0, count / 2, count - 1] { + let intervals = proof_intervals(count, page).unwrap(); + assert!(intervals.len() <= 15); + let selected: Vec<_> = nodes + .iter() + .filter(|n| { + intervals + .iter() + .any(|&(_, s, c)| n.start == s && n.pages == c) + }) + .copied() + .collect(); + let proof = selected_proof(&intervals, &selected).unwrap(); + assert_eq!( + proof, + mst2_codec::chunkmap::leaf_proof(&hashes, page).unwrap() + ); + mst2_codec::chunkmap::verify_leaf(count, page, hashes[page as usize], &proof, root) + .unwrap(); + if !selected.is_empty() { + assert!(selected_proof(&intervals, &selected[1..]).is_err()); + let mut bad = proof.clone(); + bad[0].digest[0] ^= 1; + assert!( + mst2_codec::chunkmap::verify_leaf( + count, + page, + hashes[page as usize], + &bad, + root + ) + .is_err() + ); + } + } + } + } +} diff --git a/src/ceres/snapshot/chunk_singleflight_tests.rs b/src/ceres/snapshot/chunk_singleflight_tests.rs deleted file mode 100644 index 78daff83..00000000 --- a/src/ceres/snapshot/chunk_singleflight_tests.rs +++ /dev/null @@ -1,487 +0,0 @@ -use std::{ - sync::{ - Barrier, - atomic::{AtomicUsize, Ordering}, - }, - time::Duration, -}; - -use tokio::{sync::Notify, time::timeout}; - -use super::*; - -fn unique_data(size: usize) -> ([u8; 32], Arc>) { - let seed = uuid::Uuid::new_v4(); - let raw: Vec = (0..size).map(|i| seed.as_bytes()[i % 16]).collect(); - (Sha256::digest(&raw).into(), Arc::new(raw)) -} - -async fn started(signal: &Notify) { - timeout(Duration::from_secs(5), signal.notified()) - .await - .unwrap(); -} - -async fn participants(content_id: [u8; 32], expected: usize) { - timeout(Duration::from_secs(5), async { - loop { - let count = FLIGHTS - .get() - .unwrap() - .lock() - .unwrap() - .entries - .get(&content_id) - .map_or(0, Weak::strong_count); - if count == expected { - break; - } - tokio::task::yield_now().await; - } - }) - .await - .unwrap(); -} - -fn assert_flight_released(content_id: [u8; 32]) { - assert!( - !FLIGHTS - .get() - .unwrap() - .lock() - .unwrap() - .entries - .contains_key(&content_id) - ); -} - -fn assert_uncached(content_id: [u8; 32]) { - assert!( - STAGED - .get() - .unwrap() - .lock() - .unwrap() - .get(content_id) - .is_none() - ); -} - -fn evict(content_id: [u8; 32]) { - let mut cache = STAGED.get().unwrap().lock().unwrap(); - let projection = cache.entries.remove(&content_id).unwrap(); - cache.total_bytes -= projection.retained_bytes(); - let position = cache.order.iter().position(|id| *id == content_id).unwrap(); - cache.order.remove(position); -} - -#[tokio::test] -async fn cold_same_digest_loads_once_and_warm_cache_skips_loader() { - let (content_id, raw) = unique_data(CHUNK_SIZE as usize + 7); - let calls = Arc::new(AtomicUsize::new(0)); - let entered = Arc::new(Notify::new()); - let release = Arc::new(Notify::new()); - let leader = tokio::spawn({ - let raw = raw.clone(); - let calls = calls.clone(); - let entered = entered.clone(); - let release = release.clone(); - async move { - get_or_project(content_id, || async move { - calls.fetch_add(1, Ordering::SeqCst); - entered.notify_one(); - release.notified().await; - Ok(raw.as_ref().clone()) - }) - .await - } - }); - started(&entered).await; - let mut waiters = Vec::new(); - for _ in 0..7 { - let raw = raw.clone(); - let calls = calls.clone(); - waiters.push(tokio::spawn(async move { - get_or_project(content_id, || async move { - calls.fetch_add(1, Ordering::SeqCst); - Ok(raw.as_ref().clone()) - }) - .await - })); - } - participants(content_id, 8).await; - release.notify_one(); - let projection = leader.await.unwrap().unwrap(); - for waiter in waiters { - let shared = waiter.await.unwrap().unwrap(); - assert!(Arc::ptr_eq(&projection, &shared)); - } - assert_eq!(calls.load(Ordering::SeqCst), 1); - assert_eq!(projection.map.file_content_id, content_id); - assert_eq!(projection.map.chunk_count, 2); - assert_eq!( - projection.chunk_bytes(0).unwrap(), - &raw[..CHUNK_SIZE as usize] - ); - assert_eq!( - projection.chunk_bytes(1).unwrap(), - &raw[CHUNK_SIZE as usize..] - ); - let (leaf, proof) = projection.leaf_and_proof(0).unwrap(); - mst2_codec::chunkmap::verify_leaf( - projection.page_count(), - 0, - leaf.leaf_hash().unwrap(), - &proof, - projection.map.pages_root, - ) - .unwrap(); - let warm = get_or_project(content_id, || async { - panic!("warm projection must not invoke its loader"); - }) - .await - .unwrap(); - assert!(Arc::ptr_eq(&projection, &warm)); - assert_flight_released(content_id); -} - -#[tokio::test] -async fn different_digests_enter_their_loaders_independently() { - let mut requests = Vec::new(); - let mut signals = Vec::new(); - for _ in 0..2 { - let (content_id, raw) = unique_data(97); - let entered = Arc::new(Notify::new()); - let release = Arc::new(Notify::new()); - signals.push((content_id, entered.clone(), release.clone())); - requests.push(tokio::spawn(async move { - get_or_project(content_id, || async move { - entered.notify_one(); - release.notified().await; - Ok(raw.as_ref().clone()) - }) - .await - })); - } - for (_, entered, _) in &signals { - started(entered).await; - } - for (_, _, release) in &signals { - release.notify_one(); - } - for request in requests { - request.await.unwrap().unwrap(); - } - for (content_id, _, _) in signals { - assert_flight_released(content_id); - } -} - -async fn failed_leader_then_waiter(wrong_digest: bool) { - let (content_id, raw) = unique_data(103); - let calls = Arc::new(AtomicUsize::new(0)); - let first_entered = Arc::new(Notify::new()); - let first_release = Arc::new(Notify::new()); - let second_entered = Arc::new(Notify::new()); - let second_release = Arc::new(Notify::new()); - let leader = tokio::spawn({ - let calls = calls.clone(); - let entered = first_entered.clone(); - let release = first_release.clone(); - async move { - get_or_project(content_id, || async move { - calls.fetch_add(1, Ordering::SeqCst); - entered.notify_one(); - release.notified().await; - if wrong_digest { - Ok(vec![0; 103]) - } else { - Err(SnapshotError::new( - SnapshotErrorCode::ObjectUnavailable, - "failed test loader", - )) - } - }) - .await - } - }); - started(&first_entered).await; - let waiter = tokio::spawn({ - let raw = raw.clone(); - let calls = calls.clone(); - let entered = second_entered.clone(); - let release = second_release.clone(); - async move { - get_or_project(content_id, || async move { - calls.fetch_add(1, Ordering::SeqCst); - entered.notify_one(); - release.notified().await; - Ok(raw.as_ref().clone()) - }) - .await - } - }); - participants(content_id, 2).await; - first_release.notify_one(); - let failure = leader.await.unwrap().err().unwrap(); - assert_eq!( - failure.code, - if wrong_digest { - SnapshotErrorCode::DigestMismatch - } else { - SnapshotErrorCode::ObjectUnavailable - } - ); - started(&second_entered).await; - assert_eq!(calls.load(Ordering::SeqCst), 2); - assert_uncached(content_id); - second_release.notify_one(); - let projection = waiter.await.unwrap().unwrap(); - assert_eq!(projection.chunk_bytes(0).unwrap(), raw.as_slice()); - assert_flight_released(content_id); -} - -#[tokio::test] -async fn load_failure_is_unpublished_and_waiter_uses_its_loader() { - failed_leader_then_waiter(false).await; -} - -#[tokio::test] -async fn digest_failure_is_unpublished_and_waiter_uses_its_loader() { - failed_leader_then_waiter(true).await; -} - -#[tokio::test] -async fn cancelled_leader_releases_gate_for_waiting_loader() { - let (content_id, raw) = unique_data(113); - let entered = Arc::new(Notify::new()); - let release = Arc::new(Notify::new()); - let takeover = Arc::new(Notify::new()); - let takeover_release = Arc::new(Notify::new()); - let calls = Arc::new(AtomicUsize::new(0)); - let leader = tokio::spawn({ - let raw = raw.clone(); - let calls = calls.clone(); - let entered = entered.clone(); - let release = release.clone(); - async move { - get_or_project(content_id, || async move { - calls.fetch_add(1, Ordering::SeqCst); - entered.notify_one(); - release.notified().await; - Ok(raw.as_ref().clone()) - }) - .await - } - }); - started(&entered).await; - let waiter = tokio::spawn({ - let raw = raw.clone(); - let calls = calls.clone(); - let takeover = takeover.clone(); - let takeover_release = takeover_release.clone(); - async move { - get_or_project(content_id, || async move { - calls.fetch_add(1, Ordering::SeqCst); - takeover.notify_one(); - takeover_release.notified().await; - Ok(raw.as_ref().clone()) - }) - .await - } - }); - participants(content_id, 2).await; - leader.abort(); - assert!(leader.await.err().unwrap().is_cancelled()); - started(&takeover).await; - assert_eq!(calls.load(Ordering::SeqCst), 2); - assert_uncached(content_id); - takeover_release.notify_one(); - let projection = waiter.await.unwrap().unwrap(); - assert_eq!(projection.chunk_bytes(0).unwrap(), raw.as_slice()); - assert_flight_released(content_id); -} - -#[tokio::test] -async fn cancelled_waiter_does_not_cancel_or_leak_the_leader() { - let (content_id, raw) = unique_data(127); - let entered = Arc::new(Notify::new()); - let release = Arc::new(Notify::new()); - let leader = tokio::spawn({ - let entered = entered.clone(); - let release = release.clone(); - async move { - get_or_project(content_id, || async move { - entered.notify_one(); - release.notified().await; - Ok(raw.as_ref().clone()) - }) - .await - } - }); - started(&entered).await; - let waiter = tokio::spawn(async move { - get_or_project(content_id, || async { - panic!("cancelled waiter must not load while leader is active"); - }) - .await - }); - participants(content_id, 2).await; - waiter.abort(); - assert!(waiter.await.err().unwrap().is_cancelled()); - participants(content_id, 1).await; - release.notify_one(); - let projection = leader.await.unwrap().unwrap(); - assert_eq!(projection.map.file_content_id, content_id); - assert_flight_released(content_id); -} - -#[tokio::test] -async fn evicted_result_survives_leader_drop_until_joined_waiter_finishes() { - let (content_id, raw) = unique_data(131); - let calls = Arc::new(AtomicUsize::new(0)); - let entered = Arc::new(Notify::new()); - let release = Arc::new(Notify::new()); - let leader = tokio::spawn({ - let raw = raw.clone(); - let calls = calls.clone(); - let entered = entered.clone(); - let release = release.clone(); - async move { - get_or_project(content_id, || async move { - calls.fetch_add(1, Ordering::SeqCst); - entered.notify_one(); - release.notified().await; - Ok(raw.as_ref().clone()) - }) - .await - } - }); - started(&entered).await; - let registry = FLIGHTS.get().unwrap(); - let retained = FlightClaim::acquire(registry, content_id).unwrap(); - let result_weak; - { - // Queue a test-only lock before the real waiter, keeping its turn - // blocked until eviction and the leader's return owner are gone. - let queued = retained.flight.as_ref().unwrap().result.lock(); - tokio::pin!(queued); - assert!(futures::poll!(&mut queued).is_pending()); - let waiter = tokio::spawn({ - let raw = raw.clone(); - let calls = calls.clone(); - async move { - get_or_project(content_id, || async move { - calls.fetch_add(1, Ordering::SeqCst); - Ok(raw.as_ref().clone()) - }) - .await - } - }); - participants(content_id, 3).await; - release.notify_one(); - let projection = leader.await.unwrap().unwrap(); - let guard = queued.await; - result_weak = Arc::downgrade(&projection); - evict(content_id); - drop(projection); - assert_uncached(content_id); - assert!(result_weak.upgrade().is_some()); - drop(guard); - let shared = waiter.await.unwrap().unwrap(); - assert!(result_weak.ptr_eq(&Arc::downgrade(&shared))); - assert_eq!(shared.chunk_bytes(0).unwrap(), raw.as_slice()); - assert_eq!(calls.load(Ordering::SeqCst), 1); - drop(shared); - } - assert!(result_weak.upgrade().is_some()); - drop(retained); - assert!(result_weak.upgrade().is_none()); - assert_flight_released(content_id); - assert_uncached(content_id); - let reloaded = get_or_project(content_id, || async { - calls.fetch_add(1, Ordering::SeqCst); - Ok(raw.as_ref().clone()) - }) - .await - .unwrap(); - assert_eq!(calls.load(Ordering::SeqCst), 2); - assert_eq!(reloaded.chunk_bytes(0).unwrap(), raw.as_slice()); - assert_flight_released(content_id); -} - -#[test] -fn distinct_flights_are_bounded_but_existing_digest_can_join_at_capacity() { - let registry = Mutex::new(FlightRegistry::default()); - let mut claims = Vec::new(); - for index in 0..PROJECTION_FLIGHT_CAP { - claims.push(FlightClaim::acquire(®istry, [index as u8; 32]).unwrap()); - } - let joined = FlightClaim::acquire(®istry, [42; 32]).unwrap(); - assert!(Arc::ptr_eq( - claims[42].flight.as_ref().unwrap(), - joined.flight.as_ref().unwrap() - )); - let error = FlightClaim::acquire(®istry, [255; 32]).err().unwrap(); - assert_eq!(error.code, SnapshotErrorCode::LimitExceeded); - drop(claims); - assert_eq!(registry.lock().unwrap().entries.len(), 1); - drop(joined); - assert!(registry.lock().unwrap().entries.is_empty()); - let fresh = FlightClaim::acquire(®istry, [255; 32]).unwrap(); - drop(fresh); - assert!(registry.lock().unwrap().entries.is_empty()); -} - -#[test] -fn last_claim_removes_only_its_exact_registered_gate() { - let registry = Mutex::new(FlightRegistry::default()); - let content_id = [91; 32]; - let old = FlightClaim::acquire(®istry, content_id).unwrap(); - let replacement = Arc::new(ProjectionFlight { - result: tokio::sync::Mutex::new(None), - }); - registry - .lock() - .unwrap() - .entries - .insert(content_id, Arc::downgrade(&replacement)); - drop(old); - assert!(registry.lock().unwrap().entries[&content_id].ptr_eq(&Arc::downgrade(&replacement))); - let joined = FlightClaim::acquire(®istry, content_id).unwrap(); - assert!(Arc::ptr_eq(joined.flight.as_ref().unwrap(), &replacement)); - drop(replacement); - drop(joined); - assert!(registry.lock().unwrap().entries.is_empty()); - registry - .lock() - .unwrap() - .entries - .insert(content_id, Weak::new()); - let reclaimed = FlightClaim::acquire(®istry, content_id).unwrap(); - drop(reclaimed); - assert!(registry.lock().unwrap().entries.is_empty()); -} - -#[test] -fn concurrent_final_claim_drops_release_registry_capacity() { - let registry = Mutex::new(FlightRegistry::default()); - for index in 0..64 { - let content_id = [index; 32]; - let first = FlightClaim::acquire(®istry, content_id).unwrap(); - let second = FlightClaim::acquire(®istry, content_id).unwrap(); - let barrier = Barrier::new(2); - std::thread::scope(|scope| { - let barrier = &barrier; - scope.spawn(move || { - barrier.wait(); - drop(first); - }); - scope.spawn(move || { - barrier.wait(); - drop(second); - }); - }); - assert!(registry.lock().unwrap().entries.is_empty()); - } -} diff --git a/src/ceres/snapshot/chunks.rs b/src/ceres/snapshot/chunks.rs index 8999a45e..8346cec2 100644 --- a/src/ceres/snapshot/chunks.rs +++ b/src/ceres/snapshot/chunks.rs @@ -1,34 +1,132 @@ -//! On-the-fly, range-readable chunk projection for one file (spec 07). -//! -//! Persistent segment/locator storage is T04/T12 work; this identity slice -//! builds the MCM2 map and MCL2 leaves from a fixed view's verified blob and -//! retains inline bytes through 512 MiB and only verified map metadata for -//! larger files. Large CHUNK reads use the current request's strict raw range -//! source. This is a reproducible process cache, not a durable locator. -//! -//! The cache is an optimization, never the authority: every projected file -//! is re-hashed against the `content_id` the fixed view advertised, and a -//! cache miss simply rebuilds from Git. Entries are addressed by content -//! digest, never by request path. +//! Cold full-stream chunk-map verifier. Actual content callers use immutable +//! source receipts and selected persisted pages, never a digest-only cache. -use std::{ - collections::{HashMap, VecDeque}, - sync::{Arc, Mutex, OnceLock, Weak}, -}; +use std::sync::Arc; use mst2_codec::chunkmap::{CHUNK_SIZE, CHUNKS_PER_PAGE, ChunkLeaf, ChunkMap}; use sha2::{Digest, Sha256}; -use super::content_budget::{MemoryLease, projection_budget}; +use super::content_budget::MemoryLease; use crate::ceres::snapshot::error::{SnapshotError, SnapshotErrorCode}; #[path = "chunks_stream.rs"] mod streaming; -pub use streaming::get_or_project_stream; +/// Full-stream admission for one exact source, independent of the digest cache. +/// This opaque value is the only production input to durable map installation. +pub(crate) struct VerifiedSourceChunkMap { + source: ChunkMapSource, + projection: ChunkProjection, +} + +#[derive(Debug, Clone, PartialEq, Eq)] +pub(crate) struct ChunkMapSource { + fact: crate::callisto::mst2_verified_object::Model, +} + +impl ChunkMapSource { + pub(crate) fn from_fact( + fact: crate::callisto::mst2_verified_object::Model, + oid: &str, + ) -> Result { + if fact.id <= 0 + || fact.storage_domain != "git" + || fact.object_kind != "blob" + || fact.git_oid != oid + || ![40, 64].contains(&oid.len()) + || !oid + .bytes() + .all(|b| b.is_ascii_hexdigit() && !b.is_ascii_uppercase()) + || fact.state != "VERIFIED" + || fact.verification_version + != crate::jupiter::storage::mono_storage::MST2_VERIFICATION_VERSION + || fact.raw_sha256.len() != 32 + || fact.size <= 0 + || fact.size as u64 > 8 * 1024 * 1024 * 1024 * 1024 + { + return Err(SnapshotError::new( + SnapshotErrorCode::IntegrityError, + "invalid exact chunk map source fact", + )); + } + Ok(Self { fact }) + } + + pub(crate) fn fact(&self) -> &crate::callisto::mst2_verified_object::Model { + &self.fact + } + + pub(crate) fn canonical_bytes(&self) -> Result, SnapshotError> { + let f = &self.fact; + serde_json::to_vec(&( + f.id, + &f.storage_domain, + &f.git_oid, + &f.object_kind, + &f.raw_sha256, + f.size, + f.verification_version, + &f.state, + f.created_at, + )) + .map_err(|_| internal("chunk map source encoding failed")) + } +} + +impl VerifiedSourceChunkMap { + pub(crate) async fn verify( + handler: &T, + source: ChunkMapSource, + budget: &Arc, + ) -> Result { + let digest: [u8; 32] = source + .fact + .raw_sha256 + .as_slice() + .try_into() + .map_err(|_| internal("chunk map source digest shape changed"))?; + let projection = streaming::build_source_stream( + digest, + source.fact.size as u64, + || async { + handler + .get_raw_blob_stream_by_hash(&source.fact.git_oid) + .await + .map_err(|error| { + let code = match error { + crate::common::errors::MegaError::ObjStorageNotFound(_) => { + SnapshotErrorCode::ObjectUnavailable + } + crate::common::errors::MegaError::ObjStorageInconsistent(_) => { + SnapshotErrorCode::IntegrityError + } + _ => SnapshotErrorCode::Internal, + }; + SnapshotError::new(code, "exact chunk map source could not be read") + }) + }, + budget, + ) + .await?; + Ok(Self { source, projection }) + } + + pub(crate) fn source(&self) -> &ChunkMapSource { + &self.source + } + pub(crate) fn map(&self) -> &ChunkMap { + &self.projection.map + } + pub(crate) fn leaves(&self) -> &[ChunkLeaf] { + &self.projection.leaves + } + pub(crate) fn leaf_hashes(&self) -> &[[u8; 32]] { + &self.projection.leaf_hashes + } +} -pub(crate) fn projection_reservation_bytes(size: u64) -> Result { - streaming::reserved_bytes(size) +pub(crate) fn map_build_reservation_bytes(size: u64) -> Result { + streaming::reserved_map_bytes(size) } /// One file's verified range-readable projection. @@ -111,6 +209,7 @@ impl ChunkProjection { }) } + #[cfg(test)] pub fn has_inline_bytes(&self) -> bool { !self.raw.is_empty() } @@ -133,6 +232,7 @@ impl ChunkProjection { .sum::() } + #[cfg(test)] pub fn verify_chunk(&self, index: u64, bytes: &[u8]) -> Result<(), SnapshotError> { let want_len = self.map.chunk_len(index).map_err(codec_err)?; if bytes.len() as u64 != want_len { @@ -153,11 +253,13 @@ impl ChunkProjection { Ok(()) } + #[cfg(test)] pub fn page_count(&self) -> u64 { self.map.page_count } /// One MCL2 leaf plus its bottom-up proof toward `pages_root`. + #[cfg(test)] pub fn leaf_and_proof( &self, page_index: u64, @@ -179,6 +281,7 @@ impl ChunkProjection { /// Raw bytes of chunk `index`, length-checked against the map (spec 07 /// §4: full 1 MiB chunks except the positive-length remainder). + #[cfg(test)] pub fn chunk_bytes(&self, index: u64) -> Result<&[u8], SnapshotError> { let want_len = self.map.chunk_len(index).map_err(codec_err)?; let start = (index as usize) @@ -222,220 +325,6 @@ fn internal(m: &str) -> SnapshotError { SnapshotError::new(SnapshotErrorCode::Internal, m) } -/// Bounded process-wide cache of verified projections. Eviction is pure -/// memory reclaim; the projection is reproducible from Git, so an evicted -/// entry is rebuilt, never an error to the client. -static STAGED: OnceLock> = OnceLock::new(); - -/// 512 MiB cap for staged file bytes in this identity slice. The real -/// persistent RAW_OBJECT locator layout replaces this (spec 10 §4). -const STAGED_CAP_BYTES: usize = 512 * 1024 * 1024; -const STAGED_MAX_ENTRIES: usize = 4096; - -struct ProjectionCache { - entries: HashMap<[u8; 32], std::sync::Arc>, - /// Insertion/last-used order for FIFO reclaim. - order: VecDeque<[u8; 32]>, - total_bytes: usize, -} - -impl ProjectionCache { - fn new() -> Self { - ProjectionCache { - entries: HashMap::new(), - order: VecDeque::new(), - total_bytes: 0, - } - } - - fn get(&mut self, id: [u8; 32]) -> Option> { - self.entries.get(&id).cloned() - } - - fn evict_oldest(&mut self) -> Option> { - while let Some(victim) = self.order.pop_front() { - if let Some(projection) = self.entries.remove(&victim) { - self.total_bytes -= projection.retained_bytes(); - return Some(projection); - } - } - None - } - - fn put(&mut self, proj: std::sync::Arc) -> Vec> { - let mut evicted = Vec::new(); - let id = proj.map.file_content_id; - if self.entries.contains_key(&id) { - return evicted; - } - // Reclaim oldest entries until the new one fits. A file larger than - // the cap alone is not cached between flights but still - // served correctly. - let bytes = proj.retained_bytes(); - if bytes > STAGED_CAP_BYTES { - return evicted; - } - while (self.total_bytes + bytes > STAGED_CAP_BYTES - || self.entries.len() >= STAGED_MAX_ENTRIES) - && let Some(victim) = self.evict_oldest() - { - evicted.push(victim); - } - self.total_bytes += bytes; - self.order.push_back(id); - self.entries.insert(id, proj); - evicted - } -} - -static FLIGHTS: OnceLock> = OnceLock::new(); -const PROJECTION_FLIGHT_CAP: usize = 128; - -struct ProjectionFlight { - result: tokio::sync::Mutex>>, -} - -#[derive(Default)] -struct FlightRegistry { - entries: HashMap<[u8; 32], Weak>, -} - -struct FlightClaim<'a> { - content_id: [u8; 32], - flight: Option>, - registry: &'a Mutex, -} - -impl<'a> FlightClaim<'a> { - fn acquire( - registry: &'a Mutex, - content_id: [u8; 32], - ) -> Result { - let mut state = registry - .lock() - .map_err(|_| internal("chunk projection flight registry lock poisoned"))?; - if let Some(flight) = state.entries.get(&content_id).and_then(Weak::upgrade) { - return Ok(Self { - content_id, - flight: Some(flight), - registry, - }); - } - state.entries.remove(&content_id); - if state.entries.len() >= PROJECTION_FLIGHT_CAP { - return Err(SnapshotError::new( - SnapshotErrorCode::LimitExceeded, - "too many distinct chunk projections in flight", - )); - } - let flight = Arc::new(ProjectionFlight { - result: tokio::sync::Mutex::new(None), - }); - state.entries.insert(content_id, Arc::downgrade(&flight)); - Ok(Self { - content_id, - flight: Some(flight), - registry, - }) - } -} - -impl Drop for FlightClaim<'_> { - fn drop(&mut self) { - let Some(flight) = self.flight.take() else { - return; - }; - let Ok(mut state) = self.registry.lock() else { - return; - }; - if Arc::strong_count(&flight) == 1 { - if state - .entries - .get(&self.content_id) - .is_some_and(|registered| registered.ptr_eq(&Arc::downgrade(&flight))) - { - state.entries.remove(&self.content_id); - } - drop(state); - drop(flight); - } else { - // Serialize owner release with admission and other final drops. - // Another participant keeps the potentially large result alive. - drop(flight); - drop(state); - } - } -} - -/// Return the cached projection, or build one via `load` (which must resolve -/// the file in the fixed view and return its verified bytes). Concurrent -/// misses for the same digest share one successful load and projection. -#[cfg(test)] -pub async fn get_or_project( - content_id: [u8; 32], - load: F, -) -> Result, SnapshotError> -where - F: FnOnce() -> Fut, - Fut: std::future::Future, SnapshotError>>, -{ - get_or_project_with(content_id, || async { - ChunkProjection::build(content_id, load().await?) - }) - .await -} - -async fn get_or_project_with( - content_id: [u8; 32], - build: F, -) -> Result, SnapshotError> -where - F: FnOnce() -> Fut, - Fut: std::future::Future>, -{ - let cache = STAGED.get_or_init(|| Mutex::new(ProjectionCache::new())); - if let Some(p) = cache - .lock() - .map_err(|_| internal("chunk projection cache lock poisoned"))? - .get(content_id) - { - return Ok(p); - } - - let registry = FLIGHTS.get_or_init(|| Mutex::new(FlightRegistry::default())); - let claim = FlightClaim::acquire(registry, content_id)?; - let flight = claim - .flight - .as_ref() - .ok_or_else(|| internal("chunk projection flight claim released"))?; - let mut result = flight.result.lock().await; - if let Some(p) = cache - .lock() - .map_err(|_| internal("chunk projection cache lock poisoned"))? - .get(content_id) - { - *result = Some(p.clone()); - return Ok(p); - } - if let Some(p) = result.as_ref() { - return Ok(p.clone()); - } - let proj = Arc::new(build().await?); - let evicted = cache - .lock() - .map_err(|_| internal("chunk projection cache lock poisoned"))? - .put(proj.clone()); - // Final projection/credit drops can be expensive and must not hold the - // process cache mutex. Eviction does not refund other Arc owners. - drop(evicted); - *result = Some(proj.clone()); - Ok(proj) -} - -#[cfg(test)] -#[path = "chunk_singleflight_tests.rs"] -mod singleflight_tests; - #[cfg(test)] mod tests { use super::*; diff --git a/src/ceres/snapshot/chunks_stream.rs b/src/ceres/snapshot/chunks_stream.rs index 56f007fd..ab19e5b5 100644 --- a/src/ceres/snapshot/chunks_stream.rs +++ b/src/ceres/snapshot/chunks_stream.rs @@ -1,3 +1,5 @@ +use std::sync::{Arc, OnceLock}; + use futures::StreamExt; use tokio::sync::Semaphore; @@ -7,9 +9,20 @@ use crate::orbit_api::object_storage::ObjectByteStream; const STREAM_ITEM_MAX_BYTES: usize = 8 * 1024 * 1024; const MAX_FILE_BYTES: u64 = 8 * 1024 * 1024 * 1024 * 1024; const MAX_BUILDERS: usize = 4; +#[cfg(test)] +const STAGED_CAP_BYTES: usize = 512 * 1024 * 1024; const CONSTRUCTION_ALLOWANCE: usize = 64 * 1024; +#[cfg(test)] pub(super) fn reserved_bytes(size: u64) -> Result { + reserved_bytes_for(size, size <= STAGED_CAP_BYTES as u64) +} + +pub(super) fn reserved_map_bytes(size: u64) -> Result { + reserved_bytes_for(size, false) +} + +fn reserved_bytes_for(size: u64, inline: bool) -> Result { if size == 0 || size > MAX_FILE_BYTES { return Err(SnapshotError::new( if size == 0 { @@ -22,11 +35,7 @@ pub(super) fn reserved_bytes(size: u64) -> Result { } let chunks = size.div_ceil(CHUNK_SIZE as u64); let pages = chunks.div_ceil(CHUNKS_PER_PAGE as u64); - let inline = if size <= STAGED_CAP_BYTES as u64 { - size - } else { - 0 - }; + let inline = if inline { size } else { 0 }; let bytes = chunks .checked_mul(32) .and_then(|n| { @@ -41,65 +50,50 @@ pub(super) fn reserved_bytes(size: u64) -> Result { Ok(bytes) } -fn reserve_projection( +#[cfg(test)] +async fn build_stream( + content_id: [u8; 32], size: u64, - budget: &Arc, - cache: &Mutex, -) -> Result { - let bytes = reserved_bytes(size)?; - loop { - match budget.reserve(bytes) { - Ok(lease) => return Ok(lease), - Err(error) => { - if error.code != SnapshotErrorCode::TemporaryUnavailable { - return Err(error); - } - let victim = cache - .lock() - .map_err(|_| internal("chunk projection cache lock poisoned"))? - .evict_oldest(); - let Some(victim) = victim else { - return Err(error); - }; - drop(victim); - } - } - } + input: ObjectByteStream, + lease: MemoryLease, +) -> Result { + build_stream_with_inline( + content_id, + size, + input, + lease, + size <= STAGED_CAP_BYTES as u64, + ) + .await } -pub async fn get_or_project_stream( +pub(super) async fn build_source_stream( content_id: [u8; 32], size: u64, open: F, -) -> Result, SnapshotError> + budget: &Arc, +) -> Result where F: FnOnce() -> Fut, Fut: std::future::Future>, { - // Profile arithmetic also runs on hits; fixed facts remain per-request. - reserved_bytes(size)?; - get_or_project_with(content_id, || async { - static BUILDERS: OnceLock> = OnceLock::new(); - build_with_resources( - content_id, - size, - open, - projection_budget(), - BUILDERS.get_or_init(|| Arc::new(Semaphore::new(MAX_BUILDERS))), - STAGED.get_or_init(|| Mutex::new(ProjectionCache::new())), - ) - .await - }) + static BUILDERS: OnceLock> = OnceLock::new(); + build_source_with_resources( + content_id, + size, + open, + budget, + BUILDERS.get_or_init(|| Arc::new(Semaphore::new(MAX_BUILDERS))), + ) .await } -async fn build_with_resources( +async fn build_source_with_resources( content_id: [u8; 32], size: u64, open: F, budget: &Arc, builders: &Arc, - cache: &Mutex, ) -> Result where F: FnOnce() -> Fut, @@ -108,20 +102,39 @@ where let _builder = builders.clone().try_acquire_owned().map_err(|_| { SnapshotError::new( SnapshotErrorCode::TemporaryUnavailable, - "chunk projection builders are occupied", + "chunk map builders are occupied", ) })?; - let lease = reserve_projection(size, budget, cache)?; - // Source body I/O starts only after builder and retained-memory credit. + let lease = budget.reserve(source_reservation_bytes(size)?)?; let input = open().await?; - build_stream(content_id, size, input, lease).await + build_stream_with_inline(content_id, size, input, lease, false).await } -async fn build_stream( +pub(super) fn source_reservation_bytes(size: u64) -> Result { + let map_bytes = reserved_map_bytes(size)?; + let pages = size + .div_ceil(CHUNK_SIZE as u64) + .div_ceil(CHUNKS_PER_PAGE as u64); + let nodes = pages + .checked_mul(2) + .and_then(|n| n.checked_sub(1)) + .and_then(|n| { + n.checked_mul(std::mem::size_of::() as u64) + }) + .and_then(|n| usize::try_from(n).ok()) + .ok_or_else(|| internal("chunk map install reservation overflow"))?; + map_bytes + .checked_add(nodes) + .and_then(|n| n.checked_add(4 * 1024 * 1024)) + .ok_or_else(|| internal("chunk map install reservation overflow")) +} + +async fn build_stream_with_inline( content_id: [u8; 32], size: u64, mut input: ObjectByteStream, lease: MemoryLease, + inline: bool, ) -> Result { let chunk_count = size.div_ceil(CHUNK_SIZE as u64); let page_count = chunk_count.div_ceil(CHUNKS_PER_PAGE as u64); @@ -129,7 +142,6 @@ async fn build_stream( let mut leaves = Vec::new(); let mut leaf_hashes = Vec::new(); let mut current = Vec::new(); - let inline = size <= STAGED_CAP_BYTES as u64; if inline { raw.try_reserve_exact(size as usize) .map_err(allocation_error)?; diff --git a/src/ceres/snapshot/chunks_stream_tests.rs b/src/ceres/snapshot/chunks_stream_tests.rs index 2ccebae8..2cf7fb53 100644 --- a/src/ceres/snapshot/chunks_stream_tests.rs +++ b/src/ceres/snapshot/chunks_stream_tests.rs @@ -14,6 +14,112 @@ fn stream(parts: Vec>) -> ObjectByteStream { Box::pin(futures::stream::iter(parts)) } +#[tokio::test] +async fn exact_source_builder_admits_all_install_workspace_before_open_and_cancellation_releases_it() + { + let raw = Bytes::from_static(b"exact source body"); + let digest: [u8; 32] = Sha256::digest(&raw).into(); + let weight = source_reservation_bytes(raw.len() as u64).unwrap(); + assert!(weight > 4 * 1024 * 1024); + let budget = MemoryBudget::new(weight); + let builders = Arc::new(Semaphore::new(1)); + let opens = Arc::new(AtomicUsize::new(0)); + let held = budget.reserve(weight).unwrap(); + let error = build_source_with_resources( + digest, + raw.len() as u64, + || async { + opens.fetch_add(1, Ordering::SeqCst); + Ok(stream(vec![Ok(raw.clone())])) + }, + &budget, + &builders, + ) + .await + .err() + .unwrap(); + assert_eq!(error.code, SnapshotErrorCode::TemporaryUnavailable); + assert_eq!(opens.load(Ordering::SeqCst), 0); + assert_eq!(builders.available_permits(), 1); + drop(held); + let held_builder = builders.clone().acquire_owned().await.unwrap(); + assert_eq!( + build_source_with_resources( + digest, + raw.len() as u64, + || async { + opens.fetch_add(1, Ordering::SeqCst); + Ok(stream(vec![Ok(raw.clone())])) + }, + &budget, + &builders + ) + .await + .err() + .unwrap() + .code, + SnapshotErrorCode::TemporaryUnavailable + ); + assert_eq!(opens.load(Ordering::SeqCst), 0); + assert_eq!(budget.used(), 0); + drop(held_builder); + let entered = Arc::new(Notify::new()); + let task_budget = budget.clone(); + let task_builders = builders.clone(); + let task_entered = entered.clone(); + let task_opens = opens.clone(); + let drops = Arc::new(AtomicUsize::new(0)); + let producer_owner = DropCount(drops.clone()); + let size = raw.len() as u64; + let task = tokio::spawn(async move { + build_source_with_resources( + digest, + size, + || async { + task_opens.fetch_add(1, Ordering::SeqCst); + task_entered.notify_one(); + Ok(Box::pin(futures::stream::unfold( + producer_owner, + |owner| async move { + futures::future::pending::<()>().await; + Some((Ok(Bytes::new()), owner)) + }, + )) as ObjectByteStream) + }, + &task_budget, + &task_builders, + ) + .await + }); + timeout(Duration::from_secs(5), entered.notified()) + .await + .unwrap(); + assert_eq!(budget.used(), weight); + assert_eq!(builders.available_permits(), 0); + task.abort(); + assert!(task.await.err().unwrap().is_cancelled()); + assert_eq!(budget.used(), 0); + assert_eq!(builders.available_permits(), 1); + assert_eq!(drops.load(Ordering::SeqCst), 1); + let projection = build_source_with_resources( + digest, + size, + || async { Ok(stream(vec![Ok(raw.clone())])) }, + &budget, + &builders, + ) + .await + .unwrap(); + assert!(!projection.has_inline_bytes()); + assert_eq!(projection.map.file_content_id, digest); + assert_eq!(opens.load(Ordering::SeqCst), 2); + assert_eq!(builders.available_permits(), 1); + assert_eq!(budget.used(), weight); + projection.verify_chunk(0, &raw).unwrap(); + drop(projection); + assert_eq!(budget.used(), 0); +} + async fn project(raw: &[u8], parts: Vec>) -> ChunkProjection { let budget = MemoryBudget::new(reserved_bytes(raw.len() as u64).unwrap()); let lease = budget @@ -205,36 +311,6 @@ async fn visible_producer_item_limit_does_not_collect_a_large_item() { assert_eq!(budget.used(), 0); } -#[tokio::test] -async fn evicted_projection_remains_charged_until_last_reader_drops() { - let raw = b"owned credits"; - let budget = MemoryBudget::new(reserved_bytes(raw.len() as u64).unwrap()); - let lease = budget - .reserve(reserved_bytes(raw.len() as u64).unwrap()) - .unwrap(); - let projection = Arc::new( - build_stream( - Sha256::digest(raw).into(), - raw.len() as u64, - stream(vec![Ok(Bytes::from_static(raw))]), - lease, - ) - .await - .unwrap(), - ); - let weight = projection.retained_bytes(); - let mut cache = ProjectionCache::new(); - assert!(cache.put(projection.clone()).is_empty()); - let reader = projection.clone(); - drop(projection); - drop(cache.evict_oldest()); - assert_eq!(cache.total_bytes, 0); - assert_eq!(budget.used(), weight); - assert!(budget.reserve(1).is_err()); - drop(reader); - assert_eq!(budget.used(), 0); -} - struct DropCount(Arc); impl Drop for DropCount { fn drop(&mut self) { @@ -242,139 +318,6 @@ impl Drop for DropCount { } } -#[tokio::test] -async fn production_admission_rejects_live_credit_and_builder_overload_before_open() { - let weight = reserved_bytes(3).unwrap(); - let cache = Mutex::new(ProjectionCache::new()); - let budget = MemoryBudget::new(weight); - let builders = Arc::new(Semaphore::new(MAX_BUILDERS)); - let calls = AtomicUsize::new(0); - let held = budget.reserve(weight).unwrap(); - let error = build_with_resources( - Sha256::digest(b"abc").into(), - 3, - || async { - calls.fetch_add(1, Ordering::SeqCst); - Ok(stream(vec![Ok(Bytes::from_static(b"abc"))])) - }, - &budget, - &builders, - &cache, - ) - .await - .err() - .unwrap(); - assert_eq!(error.code, SnapshotErrorCode::TemporaryUnavailable); - assert_eq!(calls.load(Ordering::SeqCst), 0); - assert_eq!(builders.available_permits(), MAX_BUILDERS); - drop(held); - let mut workers = Vec::new(); - for _ in 0..MAX_BUILDERS { - workers.push(builders.clone().try_acquire_owned().unwrap()); - } - let error = build_with_resources( - Sha256::digest(b"abc").into(), - 3, - || async { - calls.fetch_add(1, Ordering::SeqCst); - Ok(stream(vec![Ok(Bytes::from_static(b"abc"))])) - }, - &budget, - &builders, - &cache, - ) - .await - .err() - .unwrap(); - assert_eq!(error.code, SnapshotErrorCode::TemporaryUnavailable); - assert_eq!(calls.load(Ordering::SeqCst), 0); - assert_eq!(budget.used(), 0); - drop(workers); - let projection = build_with_resources( - Sha256::digest(b"abc").into(), - 3, - || async { - calls.fetch_add(1, Ordering::SeqCst); - Ok(stream(vec![Ok(Bytes::from_static(b"abc"))])) - }, - &budget, - &builders, - &cache, - ) - .await - .unwrap(); - assert_eq!(calls.load(Ordering::SeqCst), 1); - assert_eq!(budget.used(), weight); - assert_eq!(builders.available_permits(), MAX_BUILDERS); - drop(projection); - assert_eq!(budget.used(), 0); -} - -#[tokio::test] -async fn cancelled_stream_owner_refunds_after_stream_drop_and_same_flight_waiter_takes_over() { - let raw = Bytes::from(uuid::Uuid::new_v4().as_bytes().to_vec()); - let id: [u8; 32] = Sha256::digest(&raw).into(); - let entered = Arc::new(Notify::new()); - let drops = Arc::new(AtomicUsize::new(0)); - let leader = tokio::spawn({ - let entered = entered.clone(); - let drops = drops.clone(); - async move { - get_or_project_stream(id, 16, || async { - let input: ObjectByteStream = Box::pin(futures::stream::unfold( - (entered, DropCount(drops)), - |(entered, owner)| async move { - entered.notify_one(); - std::future::pending::<()>().await; - Some((Ok(Bytes::new()), (entered, owner))) - }, - )); - Ok(input) - }) - .await - } - }); - timeout(Duration::from_secs(5), entered.notified()) - .await - .unwrap(); - let waiter = tokio::spawn(async move { - get_or_project_stream(id, 16, || async { - let input: ObjectByteStream = stream(vec![Ok(raw)]); - Ok(input) - }) - .await - }); - timeout(Duration::from_secs(5), async { - loop { - if FLIGHTS - .get() - .unwrap() - .lock() - .unwrap() - .entries - .get(&id) - .map_or(0, Weak::strong_count) - == 2 - { - break; - } - tokio::task::yield_now().await; - } - }) - .await - .unwrap(); - leader.abort(); - assert!(leader.await.err().unwrap().is_cancelled()); - assert_eq!(drops.load(Ordering::SeqCst), 1); - let projection = timeout(Duration::from_secs(5), waiter) - .await - .unwrap() - .unwrap() - .unwrap(); - assert_eq!(projection.map.file_content_id, id); - assert_eq!(projection.chunk_bytes(0).unwrap().len(), 16); -} - #[test] fn maximum_file_metadata_fits_and_protocol_or_quota_rejection_needs_no_source() { assert!(reserved_bytes(MAX_FILE_BYTES).unwrap() < 300 * 1024 * 1024); diff --git a/src/ceres/snapshot/mod.rs b/src/ceres/snapshot/mod.rs index 532d1da6..3fae7f32 100644 --- a/src/ceres/snapshot/mod.rs +++ b/src/ceres/snapshot/mod.rs @@ -3,6 +3,8 @@ //! Serving remains native-only. The independent namespace index is an unwired //! composition seam; identity, attestation and publication integration remain //! separate gates. Fixed-source readers must never look up current refs. +pub(crate) mod chunk_map_gate; +pub(crate) mod chunk_map_index; pub mod chunks; pub(crate) mod content_budget; pub mod descriptor; diff --git a/src/jupiter/migration/m20261008_000100_add_mst2_chunk_maps.rs b/src/jupiter/migration/m20261008_000100_add_mst2_chunk_maps.rs new file mode 100644 index 00000000..cd597a16 --- /dev/null +++ b/src/jupiter/migration/m20261008_000100_add_mst2_chunk_maps.rs @@ -0,0 +1,28 @@ +//! Append-only, independently authenticated chunk-map read indexes. + +use sea_orm::{ConnectionTrait, DbBackend}; +use sea_orm_migration::prelude::*; + +#[derive(DeriveMigrationName)] +pub struct Migration; + +#[async_trait::async_trait] +impl MigrationTrait for Migration { + async fn up(&self, manager: &SchemaManager) -> Result<(), DbErr> { + if manager.get_database_backend() != DbBackend::Postgres { + return Err(DbErr::Custom( + "persisted chunk maps require primary PostgreSQL".into(), + )); + } + manager + .get_connection() + .execute_unprepared(include_str!("m20261008_000100_chunk_maps.sql")) + .await?; + Ok(()) + } + + async fn down(&self, _manager: &SchemaManager) -> Result<(), DbErr> { + // No collector or migration rollback may silently discard receipts. + Ok(()) + } +} diff --git a/src/jupiter/migration/m20261008_000100_chunk_maps.sql b/src/jupiter/migration/m20261008_000100_chunk_maps.sql new file mode 100644 index 00000000..75b452d0 --- /dev/null +++ b/src/jupiter/migration/m20261008_000100_chunk_maps.sql @@ -0,0 +1,90 @@ +CREATE TABLE mst2_chunk_map ( + map_id bytea PRIMARY KEY CHECK (pg_catalog.octet_length(map_id)=32), + descriptor bytea NOT NULL CHECK (pg_catalog.octet_length(descriptor)=100), + page_count integer NOT NULL CHECK (page_count BETWEEN 1 AND 32768), + pages_root bytea NOT NULL CHECK (pg_catalog.octet_length(pages_root)=32) +); +CREATE TABLE mst2_chunk_map_leaf ( + map_id bytea NOT NULL REFERENCES mst2_chunk_map(map_id), + page_index integer NOT NULL CHECK (page_index BETWEEN 0 AND 32767), + payload bytea NOT NULL CHECK (pg_catalog.octet_length(payload) BETWEEN 48 AND 8208), + PRIMARY KEY(map_id,page_index) +); +CREATE TABLE mst2_chunk_map_node ( + map_id bytea NOT NULL REFERENCES mst2_chunk_map(map_id), + first_page integer NOT NULL CHECK (first_page BETWEEN 0 AND 32767), + page_count integer NOT NULL CHECK (page_count BETWEEN 1 AND 32768), + digest bytea NOT NULL CHECK (pg_catalog.octet_length(digest)=32), + PRIMARY KEY(map_id,first_page,page_count), + CHECK (first_page+page_count<=32768) +); +CREATE TABLE mst2_chunk_map_source ( + storage_domain text NOT NULL CHECK (storage_domain='git'), + git_oid text NOT NULL CHECK (git_oid ~ '^([0-9a-f]{40}|[0-9a-f]{64})$'), + object_kind text NOT NULL CHECK (object_kind='blob'), + fact_id bigint NOT NULL CHECK (fact_id>0), + source_id bytea NOT NULL UNIQUE CHECK (pg_catalog.octet_length(source_id)=32), + source_bytes bytea NOT NULL CHECK (pg_catalog.octet_length(source_bytes) BETWEEN 1 AND 2048), + primary_scope bytea NOT NULL CHECK (pg_catalog.octet_length(primary_scope) BETWEEN 1 AND 1024), + map_id bytea NOT NULL REFERENCES mst2_chunk_map(map_id), + receipt_digest bytea NOT NULL CHECK (pg_catalog.octet_length(receipt_digest)=32), + PRIMARY KEY(storage_domain,git_oid,object_kind) +); + +-- These rows are indexes, not proof that any source body was consumed. The +-- trusted object writer alone publishes the independent immutable receipt. +CREATE FUNCTION mst2_chunk_map_immutable() RETURNS trigger LANGUAGE plpgsql AS $$ +BEGIN RAISE EXCEPTION 'chunk map indexes and source receipts are append-only'; END $$; + +CREATE FUNCTION mst2_chunk_map_primary() RETURNS trigger LANGUAGE plpgsql AS $$ +BEGIN + IF pg_catalog.current_schema() IS DISTINCT FROM TG_TABLE_SCHEMA + OR pg_catalog.pg_is_in_recovery() + OR pg_catalog.current_setting('transaction_isolation')<>'read committed' THEN + RAISE EXCEPTION 'chunk map insertion requires its actual primary schema and READ COMMITTED'; + END IF; + RETURN NULL; +END $$; + +CREATE FUNCTION mst2_chunk_map_complete() RETURNS trigger LANGUAGE plpgsql AS $$ +DECLARE m record; leaves bigint; nodes bigint; root bytea; first_leaf integer; last_leaf integer; +BEGIN + EXECUTE pg_catalog.format('SELECT * FROM %I.mst2_chunk_map WHERE map_id=$1',TG_TABLE_SCHEMA) + INTO m USING NEW.map_id; + EXECUTE pg_catalog.format('SELECT pg_catalog.count(*),pg_catalog.min(page_index),pg_catalog.max(page_index) FROM %I.mst2_chunk_map_leaf WHERE map_id=$1',TG_TABLE_SCHEMA) + INTO leaves,first_leaf,last_leaf USING NEW.map_id; + EXECUTE pg_catalog.format('SELECT pg_catalog.count(*) FROM %I.mst2_chunk_map_node WHERE map_id=$1',TG_TABLE_SCHEMA) + INTO nodes USING NEW.map_id; + EXECUTE pg_catalog.format('SELECT digest FROM %I.mst2_chunk_map_node WHERE map_id=$1 AND first_page=0 AND page_count=$2',TG_TABLE_SCHEMA) + INTO root USING NEW.map_id,m.page_count; + IF m.map_id IS NULL OR leaves<>m.page_count OR first_leaf<>0 OR last_leaf<>m.page_count-1 + OR nodes<>2*m.page_count-1 OR root IS DISTINCT FROM m.pages_root + OR pg_catalog.substring(m.descriptor,69,32) IS DISTINCT FROM m.pages_root + OR pg_catalog.sha256(pg_catalog.convert_to('mega.mst2.chunkmap','UTF8')||pg_catalog.decode('00','hex')||m.descriptor) IS DISTINCT FROM m.map_id THEN + RAISE EXCEPTION 'chunk map installation is incomplete or inconsistent'; + END IF; + RETURN NULL; +END $$; + +CREATE FUNCTION mst2_chunk_map_index_insert() RETURNS trigger LANGUAGE plpgsql AS $$ +DECLARE admitted boolean; +BEGIN + EXECUTE pg_catalog.format('SELECT EXISTS(SELECT 1 FROM %I.mst2_chunk_map_source WHERE map_id=$1)',TG_TABLE_SCHEMA) + INTO admitted USING NEW.map_id; + IF admitted THEN RAISE EXCEPTION 'admitted chunk map index cannot acquire more rows'; END IF; + RETURN NEW; +END $$; + +DO $$ DECLARE t text; BEGIN + FOREACH t IN ARRAY ARRAY['mst2_chunk_map','mst2_chunk_map_leaf','mst2_chunk_map_node','mst2_chunk_map_source'] LOOP + EXECUTE pg_catalog.format('CREATE TRIGGER mst2_chunk_map_primary BEFORE INSERT ON %I FOR EACH STATEMENT EXECUTE FUNCTION mst2_chunk_map_primary()',t); + EXECUTE pg_catalog.format('CREATE TRIGGER mst2_chunk_map_immutable BEFORE UPDATE OR DELETE ON %I FOR EACH ROW EXECUTE FUNCTION mst2_chunk_map_immutable()',t); + EXECUTE pg_catalog.format('CREATE TRIGGER mst2_chunk_map_no_truncate BEFORE TRUNCATE ON %I FOR EACH STATEMENT EXECUTE FUNCTION mst2_chunk_map_immutable()',t); + END LOOP; +END $$; +CREATE CONSTRAINT TRIGGER mst2_chunk_map_complete AFTER INSERT ON mst2_chunk_map_source + DEFERRABLE INITIALLY DEFERRED FOR EACH ROW EXECUTE FUNCTION mst2_chunk_map_complete(); +CREATE TRIGGER mst2_chunk_map_index_insert BEFORE INSERT ON mst2_chunk_map_leaf + FOR EACH ROW EXECUTE FUNCTION mst2_chunk_map_index_insert(); +CREATE TRIGGER mst2_chunk_map_index_insert BEFORE INSERT ON mst2_chunk_map_node + FOR EACH ROW EXECUTE FUNCTION mst2_chunk_map_index_insert(); diff --git a/src/jupiter/migration/mod.rs b/src/jupiter/migration/mod.rs index 2d667a39..f5981f02 100644 --- a/src/jupiter/migration/mod.rs +++ b/src/jupiter/migration/mod.rs @@ -151,6 +151,7 @@ mod m20261007_000300_add_mst2_metadata_lifetime_history; mod m20261007_000400_add_mst2_qualified_metadata_gc; mod m20261007_000500_add_mst2_install_capability; mod m20261007_000600_add_mst2_storage_routes; +mod m20261008_000100_add_mst2_chunk_maps; mod runner; pub use m20260905_000100_add_push_queue::ensure_queue_control_seed; pub use runner::apply_migrations; @@ -290,6 +291,7 @@ impl MigratorTrait for Migrator { Box::new(m20261007_000400_add_mst2_qualified_metadata_gc::Migration), Box::new(m20261007_000500_add_mst2_install_capability::Migration), Box::new(m20261007_000600_add_mst2_storage_routes::Migration), + Box::new(m20261008_000100_add_mst2_chunk_maps::Migration), ] } } diff --git a/src/jupiter/storage/mod.rs b/src/jupiter/storage/mod.rs index b130aa5f..2ea787de 100644 --- a/src/jupiter/storage/mod.rs +++ b/src/jupiter/storage/mod.rs @@ -18,6 +18,7 @@ pub mod media_paging_storage; pub mod mono_storage; pub(crate) mod mst2_publication_storage; pub mod mst2_retention; +pub(crate) mod native_chunk_map; pub mod native_metadata_install; pub(crate) mod native_publication_storage; pub(crate) mod native_snapshot_session; @@ -159,6 +160,8 @@ pub struct Storage { /// Derived native projection memoization, scoped to this storage assembly. /// Clones share it; independent databases/backends never share entries. pub(crate) native_projection_cache: Arc, + pub(crate) native_chunk_maps: + Arc>, pub(crate) native_snapshot_sessions: Arc>, pub(crate) projection_observation_sink: @@ -326,6 +329,7 @@ impl Storage { app_service: app_service.into(), native_projection_cache: Arc::default(), native_snapshot_sessions: Arc::default(), + native_chunk_maps: Arc::default(), projection_observation_sink: None, config_handle, config, @@ -704,6 +708,7 @@ impl Storage { app_service, native_projection_cache: Arc::default(), native_snapshot_sessions: Arc::default(), + native_chunk_maps: Arc::default(), projection_observation_sink: None, // app_service: AppService::mock(), cl_service: CLService::mock(), diff --git a/src/jupiter/storage/mono_storage.rs b/src/jupiter/storage/mono_storage.rs index 335869ac..5228c6fc 100644 --- a/src/jupiter/storage/mono_storage.rs +++ b/src/jupiter/storage/mono_storage.rs @@ -1789,12 +1789,43 @@ impl MonoStorage { if oids.is_empty() { return Ok(HashMap::new()); } - let rows = mst2_verified_object::Entity::find() - .filter(mst2_verified_object::Column::StorageDomain.eq("git")) - .filter(mst2_verified_object::Column::ObjectKind.eq("blob")) - .filter(mst2_verified_object::Column::GitOid.is_in(oids)) - .all(self.get_connection()) - .await?; + let connection = self.get_connection(); + let rows = if connection.get_database_backend() == sea_orm::DbBackend::Postgres { + use sea_orm::{DbBackend, FromQueryResult, Statement}; + let scope = connection.query_one_raw(Statement::from_string(DbBackend::Postgres, + "SELECT n.nspname AS schema FROM pg_catalog.pg_namespace n JOIN pg_catalog.pg_class c ON c.relnamespace=n.oid AND c.relname='mst2_verified_object' AND c.relkind='r' WHERE n.nspname=pg_catalog.current_schema() AND n.nspname NOT LIKE 'pg_temp_%'")) + .await?.ok_or_else(|| MegaError::Other("actual verified blob relation is missing".into()))?; + let schema: String = scope.try_get("", "schema")?; + let relation = format!("\"{}\".mst2_verified_object", schema.replace('"', "\"\"")); + let oids = serde_json::to_string(&oids) + .map_err(|error| MegaError::Other(error.to_string()))?; + let sql = format!( + "SELECT v.id,v.storage_domain,CASE WHEN pg_catalog.octet_length(v.git_oid) IN (40,64) THEN v.git_oid ELSE NULL END AS git_oid,v.object_kind,CASE WHEN pg_catalog.octet_length(v.raw_sha256)=32 THEN v.raw_sha256 ELSE NULL END AS raw_sha256,CASE WHEN v.size BETWEEN 0 AND {MST2_MAX_FILE_SIZE} THEN v.size ELSE NULL END AS size,CASE WHEN v.verification_version IN (1,{MST2_VERIFICATION_VERSION}) THEN v.verification_version ELSE NULL END AS verification_version,CASE WHEN v.state='VERIFIED' THEN v.state ELSE NULL END AS state,v.created_at FROM {relation} v WHERE v.storage_domain='git' AND v.object_kind='blob' AND v.git_oid IN (SELECT value FROM pg_catalog.jsonb_array_elements_text($1::jsonb))" + ); + connection + .query_all_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + sql, + [oids.into()], + )) + .await? + .iter() + .map(|row| { + mst2_verified_object::Model::from_query_result(row, "").map_err(|_| { + MegaError::ObjStorageInconsistent( + "invalid bounded MST/2 verified blob record".into(), + ) + }) + }) + .collect::, _>>()? + } else { + mst2_verified_object::Entity::find() + .filter(mst2_verified_object::Column::StorageDomain.eq("git")) + .filter(mst2_verified_object::Column::ObjectKind.eq("blob")) + .filter(mst2_verified_object::Column::GitOid.is_in(oids)) + .all(connection) + .await? + }; for row in &rows { if row.state != "VERIFIED" || ![1, MST2_VERIFICATION_VERSION].contains(&row.verification_version) diff --git a/src/jupiter/storage/native_chunk_map.rs b/src/jupiter/storage/native_chunk_map.rs new file mode 100644 index 00000000..7dcbaa22 --- /dev/null +++ b/src/jupiter/storage/native_chunk_map.rs @@ -0,0 +1,684 @@ +//! Source-bound chunk-map receipts with indexed, authenticated selected pages. +//! +//! The object writer is trusted like the verified-object writer. A database +//! row, its checksum or a map's presence is never a full-source proof. Only +//! the opaque full-stream verifier can install a source receipt. This initial +//! store is append-only; bounded retention and collection remain separate work. + +use std::sync::Arc; + +use bytes::Bytes; +use futures::StreamExt; +use mst2_codec::chunkmap::{CHUNKS_PER_PAGE, ChunkLeaf, ChunkMap, ProofStep, verify_leaf}; +use sea_orm::{ + ConnectionTrait, DatabaseConnection, DatabaseTransaction, DbBackend, FromQueryResult, + IsolationLevel, QueryResult, Statement, TransactionTrait, +}; +use serde_json::json; +use sha2::{Digest, Sha256}; + +use super::object_storage::MegaObjectStorageWrapper; +use crate::{ + callisto::mst2_verified_object, + ceres::snapshot::{ + chunk_map_index::{ChunkMapNode, indexed_nodes, proof_intervals, selected_proof}, + chunks::{ChunkMapSource, VerifiedSourceChunkMap}, + content_budget::{MemoryBudget, MemoryLease, projection_budget}, + error::{SnapshotError, SnapshotErrorCode}, + }, + orbit_api::object_storage::{ObjectKey, ObjectMeta, ObjectNamespace}, +}; + +const DESCRIPTOR_CREDIT: usize = 32 * 1024; +const PAGE_CREDIT: usize = 64 * 1024; +const RECEIPT_DOMAIN: &[u8] = b"MST2-CHUNK-MAP-RECEIPT\0"; + +pub(crate) struct PostgresChunkMapRepository { + connection: DatabaseConnection, + primary_scope: Vec, + budget: Arc, + schema: String, +} + +pub(crate) struct PersistedChunkMap { + pub map: ChunkMap, + pub map_id: [u8; 32], + source: ChunkMapSource, + source_id: [u8; 32], + _memory: MemoryLease, +} + +impl PersistedChunkMap { + pub(crate) fn source_id(&self) -> [u8; 32] { + self.source_id + } +} + +pub(crate) struct AuthenticatedChunkPage { + pub leaf: ChunkLeaf, + pub proof: Vec, + _memory: MemoryLease, +} + +impl AuthenticatedChunkPage { + pub(crate) fn verify_chunk( + &self, + map: &ChunkMap, + index: u64, + bytes: &[u8], + ) -> Result<(), SnapshotError> { + let length = map + .chunk_len(index) + .map_err(|_| integrity("invalid requested chunk index"))?; + if index / CHUNKS_PER_PAGE as u64 != self.leaf.page_index || bytes.len() as u64 != length { + return Err(integrity( + "exact range length or selected chunk page disagrees", + )); + } + let slot = (index % CHUNKS_PER_PAGE as u64) as usize; + let expected = self + .leaf + .chunk_sha256 + .get(slot) + .ok_or_else(|| integrity("selected chunk digest is missing"))?; + let digest: [u8; 32] = Sha256::digest(bytes).into(); + if &digest != expected { + return Err(integrity( + "exact range digest disagrees with the authenticated selected page", + )); + } + Ok(()) + } +} + +impl PostgresChunkMapRepository { + pub(crate) async fn new(connection: DatabaseConnection) -> Result { + let schema = capture_schema(&connection).await?; + let primary_scope = read_scope(&connection, &schema).await?; + Ok(Self { + connection, + primary_scope, + budget: projection_budget().clone(), + schema, + }) + } + + pub(crate) async fn read( + &self, + source: &ChunkMapSource, + objects: &MegaObjectStorageWrapper, + ) -> Result>, SnapshotError> { + let memory = self.budget.reserve(DESCRIPTOR_CREDIT)?; + let txn = self.transaction().await?; + let result = async { + self.require_scope(&txn).await?; + require_current_source(&txn, source, &self.schema).await?; + let Some(row) = source_row(&txn, source, &self.schema).await? else { + return Ok(None); + }; + let (map, source_id) = self.validate_source_row(source, &row)?; + let bytes = receipt(&self.primary_scope, source, &map)?; + read_receipt(objects, &receipt_key(source_id), &bytes).await?; + Ok(Some(Arc::new(PersistedChunkMap { + map_id: map.map_id(), + map, + source: source.clone(), + source_id, + _memory: memory, + }))) + } + .await; + finish(txn, result).await + } + + /// Publication accepts no caller-provided map or "verified" flag. + pub(crate) async fn install( + &self, + verified: VerifiedSourceChunkMap, + objects: &MegaObjectStorageWrapper, + ) -> Result<(), SnapshotError> { + let source = verified.source(); + let map = verified.map(); + // The verifier reserved index and bounded SQL parameter workspace + // before opening the body. It owns those credits through publication. + let nodes = indexed_nodes(verified.leaf_hashes())?; + if nodes.last().map(|n| n.digest) != Some(map.pages_root) { + return Err(integrity( + "verified chunk map index disagrees with canonical root", + )); + } + let source_bytes = source.canonical_bytes()?; + let source_id = source_id(&self.primary_scope, &source_bytes); + let bytes = receipt(&self.primary_scope, source, map)?; + let key = receipt_key(source_id); + let txn = self.transaction().await?; + let result = async { + self.require_scope(&txn).await?; + require_current_source(&txn, source, &self.schema).await?; + if let Some(row) = source_row(&txn, source, &self.schema).await? { + let (stored, _) = self.validate_source_row(source, &row)?; + if stored != *map { return Err(integrity("immutable source map conflicts with reverified content")); } + read_receipt(objects, &key, &bytes).await?; + compare_complete_map(&txn, &verified, &nodes, &self.schema).await?; + return Ok(()); + } + // The complete source pass happened before this atomic create. + // A lost DB commit leaves an orphan trusted receipt, never an + // admitted partial map; exact re-verification can replay it. + objects.inner.put_metadata_atomic_create(&key, Bytes::copy_from_slice(&bytes), ObjectMeta { + size: bytes.len() as i64, ..Default::default() + }).await.map_err(storage_error)?; + read_receipt(objects, &key, &bytes).await?; + txn.execute_raw(stmt(&self.schema, "INSERT INTO mst2_chunk_map(map_id,descriptor,page_count,pages_root) VALUES($1,$2,$3,$4) ON CONFLICT(map_id) DO NOTHING", + [map.map_id().to_vec().into(), map.encode().into(), (map.page_count as i32).into(), map.pages_root.to_vec().into()])).await.map_err(db_error)?; + // Existing indexes must be compared, never repaired or overwritten. + let present = txn.query_one_raw(stmt(&self.schema, "SELECT EXISTS(SELECT 1 FROM mst2_chunk_map_source WHERE map_id=$1) AS sealed", [map.map_id().to_vec().into()])) + .await.map_err(db_error)?.ok_or_else(|| integrity("chunk map seal observation is missing"))? + .try_get::("", "sealed").map_err(db_error)?; + if !present { + for batch in verified.leaves().chunks(64) { + let encoded = encode_leaves(batch)?; + txn.execute_raw(stmt(&self.schema, "INSERT INTO mst2_chunk_map_leaf(map_id,page_index,payload) SELECT $1,p.page_index,pg_catalog.decode(p.payload,'hex') FROM pg_catalog.jsonb_to_recordset($2::jsonb) AS p(page_index integer,payload text) ON CONFLICT(map_id,page_index) DO NOTHING", + [map.map_id().to_vec().into(), encoded.into()])).await.map_err(db_error)?; + } + for batch in nodes.chunks(1024) { + let encoded = encode_nodes(batch)?; + txn.execute_raw(stmt(&self.schema, "INSERT INTO mst2_chunk_map_node(map_id,first_page,page_count,digest) SELECT $1,p.first_page,p.page_count,pg_catalog.decode(p.digest,'hex') FROM pg_catalog.jsonb_to_recordset($2::jsonb) AS p(first_page integer,page_count integer,digest text) ON CONFLICT(map_id,first_page,page_count) DO NOTHING", + [map.map_id().to_vec().into(), encoded.into()])).await.map_err(db_error)?; + } + } + compare_complete_map(&txn, &verified, &nodes, &self.schema).await?; + let fact = source.fact(); + txn.execute_raw(stmt(&self.schema, "INSERT INTO mst2_chunk_map_source(storage_domain,git_oid,object_kind,fact_id,source_id,source_bytes,primary_scope,map_id,receipt_digest) VALUES($1,$2,$3,$4,$5,$6,$7,$8,$9) ON CONFLICT(storage_domain,git_oid,object_kind) DO NOTHING", + [fact.storage_domain.clone().into(), fact.git_oid.clone().into(), fact.object_kind.clone().into(), fact.id.into(), + source_id.to_vec().into(), source_bytes.into(), self.primary_scope.clone().into(), map.map_id().to_vec().into(), Sha256::digest(&bytes).to_vec().into()])).await.map_err(db_error)?; + let row = source_row(&txn, source, &self.schema).await?.ok_or_else(|| integrity("installed source receipt is missing"))?; + let (stored, _) = self.validate_source_row(source, &row)?; + if stored != *map { return Err(integrity("concurrent source installation conflicts")); } + Ok(()) + }.await; + finish(txn, result).await + } + + pub(crate) async fn selected_page( + &self, + map: &PersistedChunkMap, + page_index: u64, + ) -> Result, SnapshotError> { + let intervals = proof_intervals(map.map.page_count, page_index)?; + let memory = self.budget.reserve(PAGE_CREDIT)?; + let txn = self.transaction().await?; + let result = async { + self.require_scope(&txn).await?; + require_current_source(&txn, &map.source, &self.schema).await?; + let row = txn.query_one_raw(stmt(&self.schema, "SELECT pg_catalog.octet_length(payload) AS size,CASE WHEN pg_catalog.octet_length(payload) BETWEEN 48 AND 8208 THEN payload ELSE NULL END AS payload FROM mst2_chunk_map_leaf WHERE map_id=$1 AND page_index=$2", + [map.map_id.to_vec().into(), (page_index as i32).into()])).await.map_err(db_error)? + .ok_or_else(|| integrity("admitted chunk map selected leaf is missing"))?; + let bytes = row.try_get::>>("", "payload").map_err(db_error)? + .ok_or_else(|| integrity("admitted chunk map selected leaf exceeds its byte profile"))?; + let leaf = ChunkLeaf::decode(&bytes).map_err(|_| integrity("admitted chunk map selected leaf is not canonical MCL2"))?; + if leaf.page_index != page_index || leaf.chunk_sha256.len() as u64 != ChunkLeaf::expected_count(map.map.chunk_count, page_index) + || leaf.encode().map_err(|_| integrity("invalid selected leaf"))? != bytes { + return Err(integrity("selected chunk map leaf identity or chunk count disagrees")); + } + let requested: Vec<_> = intervals.iter().map(|&(_, start, pages)| json!({"first_page":start,"page_count":pages})).collect(); + let encoded = serde_json::to_string(&requested).map_err(db_error)?; + let rows = txn.query_all_raw(stmt(&self.schema, "SELECT n.first_page,n.page_count,CASE WHEN pg_catalog.octet_length(n.digest)=32 THEN n.digest ELSE NULL END AS digest FROM pg_catalog.jsonb_to_recordset($2::jsonb) AS p(first_page integer,page_count integer) JOIN mst2_chunk_map_node n ON n.map_id=$1 AND n.first_page=p.first_page AND n.page_count=p.page_count", + [map.map_id.to_vec().into(), encoded.into()])).await.map_err(db_error)?; + let mut nodes = Vec::with_capacity(rows.len()); + for row in rows { + let digest: Option> = row.try_get("", "digest").map_err(db_error)?; + nodes.push(ChunkMapNode { + start: row.try_get::("", "first_page").map_err(db_error)? as u64, + pages: row.try_get::("", "page_count").map_err(db_error)? as u64, + digest: digest.ok_or_else(|| integrity("admitted chunk map sibling has invalid digest size"))?.as_slice().try_into().map_err(|_| integrity("invalid persisted sibling digest"))?, + }); + } + let proof = selected_proof(&intervals, &nodes)?; + verify_leaf(map.map.page_count, page_index, leaf.leaf_hash().map_err(|_| integrity("invalid leaf digest"))?, &proof, map.map.pages_root) + .map_err(|_| integrity("selected chunk map leaf or proof authentication failed"))?; + tracing::debug!(target:"mst2::chunk_map", selected_leaf_bytes = bytes.len(), selected_sibling_rows = nodes.len(), page_count = map.map.page_count, "authenticated persisted chunk map selected page"); + Ok(Arc::new(AuthenticatedChunkPage { leaf, proof, _memory: memory })) + }.await; + finish(txn, result).await + } + + async fn transaction(&self) -> Result { + let txn = self + .connection + .begin_with_config(Some(IsolationLevel::ReadCommitted), None) + .await + .map_err(db_error)?; + let search_path = format!("{},pg_catalog,pg_temp", quoted(&self.schema)); + txn.query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT pg_catalog.set_config('search_path',$1,true)", + [search_path.into()], + )) + .await + .map_err(db_error)?; + txn.execute_unprepared("SET LOCAL statement_timeout='5s'; SET LOCAL lock_timeout='5s'") + .await + .map_err(db_error)?; + Ok(txn) + } + + async fn require_scope(&self, connection: &C) -> Result<(), SnapshotError> { + if read_scope(connection, &self.schema).await? != self.primary_scope { + return Err(integrity( + "chunk map repository no longer targets its captured primary scope", + )); + } + Ok(()) + } + + pub(crate) fn memory_budget(&self) -> &Arc { + &self.budget + } + + pub(crate) fn source_identity( + &self, + source: &ChunkMapSource, + ) -> Result<[u8; 32], SnapshotError> { + Ok(source_id(&self.primary_scope, &source.canonical_bytes()?)) + } + + #[cfg(test)] + pub(crate) fn with_test_budget(mut self, budget: Arc) -> Self { + self.budget = budget; + self + } + + #[cfg(test)] + pub(crate) fn test_primary_scope(&self) -> &[u8] { + &self.primary_scope + } + + fn validate_source_row( + &self, + source: &ChunkMapSource, + row: &QueryResult, + ) -> Result<(ChunkMap, [u8; 32]), SnapshotError> { + let source_bytes = source.canonical_bytes()?; + let source_id = source_id(&self.primary_scope, &source_bytes); + let source_bytes_row = bounded_bytes(row, "source_bytes")?; + let primary_scope_row = bounded_bytes(row, "primary_scope")?; + if row.try_get::("", "fact_id").map_err(db_error)? != source.fact().id + || source_bytes_row != source_bytes + || primary_scope_row != self.primary_scope + || bounded_bytes(row, "source_id")? != source_id + { + return Err(integrity( + "persisted map admission disagrees with the exact current source fact or primary", + )); + } + let bytes = row + .try_get::>>("", "descriptor") + .map_err(db_error)? + .ok_or_else(|| { + integrity("admitted chunk map descriptor is missing or outside its byte profile") + })?; + let map = ChunkMap::decode(&bytes) + .map_err(|_| integrity("admitted chunk map descriptor is invalid MCM2"))?; + let receipt_digest: [u8; 32] = + Sha256::digest(receipt(&self.primary_scope, source, &map)?).into(); + if map.encode() != bytes + || map.map_id().as_slice() != bounded_bytes(row, "map_id")?.as_slice() + || map.file_content_id.as_slice() != source.fact().raw_sha256.as_slice() + || map.file_size != source.fact().size as u64 + || map.page_count != row.try_get::("", "page_count").map_err(db_error)? as u64 + || map.pages_root.as_slice() != bounded_bytes(row, "pages_root")?.as_slice() + || bounded_bytes(row, "receipt_digest")? != receipt_digest + { + return Err(integrity( + "admitted chunk map descriptor or receipt identity disagrees", + )); + } + Ok((map, source_id)) + } +} + +async fn capture_schema(connection: &C) -> Result { + let row = connection.query_one_raw(Statement::from_string(DbBackend::Postgres, + "SELECT n.nspname AS schema FROM pg_catalog.pg_namespace n JOIN pg_catalog.pg_class c ON c.relnamespace=n.oid AND c.relname='mst2_metadata_storage_scope' AND c.relkind='r' WHERE n.nspname=pg_catalog.current_schema() AND n.nspname NOT LIKE 'pg_temp_%'")) + .await.map_err(db_error)?.ok_or_else(|| integrity("chunk map actual primary storage relation is missing"))?; + row.try_get("", "schema").map_err(db_error) +} + +async fn read_scope( + connection: &C, + schema: &str, +) -> Result, SnapshotError> { + if connection.get_database_backend() != DbBackend::Postgres { + return Err(integrity("chunk map repository requires PostgreSQL")); + } + let row = connection.query_one_raw(stmt(schema, "SELECT pg_catalog.pg_is_in_recovery() AS replica,CASE WHEN pg_catalog.octet_length(s.storage_uuid)<=255 THEN s.storage_uuid ELSE NULL END AS storage_uuid,pg_catalog.current_database() AS database,d.oid::bigint AS database_oid,n.nspname AS schema,n.oid::bigint AS schema_oid,pg_catalog.inet_server_addr()::text AS address,pg_catalog.inet_server_port() AS port FROM mst2_metadata_storage_scope s JOIN pg_catalog.pg_database d ON d.datname=pg_catalog.current_database() JOIN pg_catalog.pg_namespace n ON n.nspname=$1 JOIN pg_catalog.pg_class c ON c.relnamespace=n.oid AND c.relname='mst2_metadata_storage_scope' AND c.relkind='r' WHERE s.singleton=1", [schema.into()])) + .await.map_err(db_error)?.ok_or_else(|| integrity("chunk map primary scope is missing"))?; + if row.try_get::("", "replica").map_err(db_error)? { + return Err(integrity("chunk map repository is on a replica")); + } + let scope = serde_json::to_vec(&( + row.try_get::>("", "storage_uuid") + .map_err(db_error)? + .ok_or_else(|| integrity("chunk map primary scope UUID exceeds its bounded profile"))?, + row.try_get::("", "database").map_err(db_error)?, + row.try_get::("", "database_oid").map_err(db_error)?, + row.try_get::("", "schema").map_err(db_error)?, + row.try_get::("", "schema_oid").map_err(db_error)?, + row.try_get::>("", "address") + .map_err(db_error)?, + row.try_get::>("", "port").map_err(db_error)?, + )) + .map_err(db_error)?; + if scope.len() > 1024 { + return Err(integrity( + "chunk map primary scope exceeds its admitted profile", + )); + } + Ok(scope) +} + +async fn require_current_source( + connection: &C, + source: &ChunkMapSource, + schema: &str, +) -> Result<(), SnapshotError> { + let row = connection + .query_one_raw(stmt(schema, + "SELECT id,CASE WHEN storage_domain='git' THEN storage_domain ELSE NULL END AS storage_domain,CASE WHEN git_oid=$2 AND pg_catalog.octet_length(git_oid) IN (40,64) THEN git_oid ELSE NULL END AS git_oid,CASE WHEN object_kind='blob' THEN object_kind ELSE NULL END AS object_kind,CASE WHEN pg_catalog.octet_length(raw_sha256)=32 THEN raw_sha256 ELSE NULL END AS raw_sha256,CASE WHEN size BETWEEN 1 AND 8796093022208 THEN size ELSE NULL END AS size,CASE WHEN verification_version=2 THEN verification_version ELSE NULL END AS verification_version,CASE WHEN state='VERIFIED' THEN state ELSE NULL END AS state,created_at FROM mst2_verified_object WHERE id=$1 FOR SHARE", + [source.fact().id.into(), source.fact().git_oid.clone().into()], + )) + .await + .map_err(db_error)? + .ok_or_else(|| { + SnapshotError::new( + SnapshotErrorCode::MetadataNotReady, + "chunk map source has no current verified fact", + ) + })?; + let current = mst2_verified_object::Model::from_query_result(&row, "").map_err(|_| { + integrity("chunk map current source fact is outside its bounded verified profile") + })?; + if ¤t != source.fact() { + return Err(integrity( + "chunk map source fact changed during request or verification", + )); + } + Ok(()) +} + +async fn source_row( + connection: &C, + source: &ChunkMapSource, + schema: &str, +) -> Result, SnapshotError> { + connection.query_one_raw(stmt(schema, "SELECT s.fact_id,CASE WHEN pg_catalog.octet_length(s.source_id)=32 THEN s.source_id ELSE NULL END AS source_id,CASE WHEN pg_catalog.octet_length(s.source_bytes)<=2048 THEN s.source_bytes ELSE NULL END AS source_bytes,CASE WHEN pg_catalog.octet_length(s.primary_scope)<=1024 THEN s.primary_scope ELSE NULL END AS primary_scope,CASE WHEN pg_catalog.octet_length(s.map_id)=32 THEN s.map_id ELSE NULL END AS map_id,CASE WHEN pg_catalog.octet_length(s.receipt_digest)=32 THEN s.receipt_digest ELSE NULL END AS receipt_digest,CASE WHEN pg_catalog.octet_length(m.descriptor)=100 THEN m.descriptor ELSE NULL END AS descriptor,m.page_count,CASE WHEN pg_catalog.octet_length(m.pages_root)=32 THEN m.pages_root ELSE NULL END AS pages_root FROM mst2_chunk_map_source s LEFT JOIN mst2_chunk_map m ON m.map_id=s.map_id WHERE s.storage_domain=$1 AND s.git_oid=$2 AND s.object_kind=$3", + [source.fact().storage_domain.clone().into(), source.fact().git_oid.clone().into(), source.fact().object_kind.clone().into()])).await.map_err(db_error) +} + +fn source_id(scope: &[u8], source: &[u8]) -> [u8; 32] { + let mut hash = Sha256::new(); + hash.update(RECEIPT_DOMAIN); + hash.update((scope.len() as u32).to_be_bytes()); + hash.update(scope); + hash.update((source.len() as u32).to_be_bytes()); + hash.update(source); + hash.finalize().into() +} + +fn receipt( + scope: &[u8], + source: &ChunkMapSource, + map: &ChunkMap, +) -> Result, SnapshotError> { + let source = source.canonical_bytes()?; + if scope.len() > 1024 || source.len() > 2048 { + return Err(integrity( + "chunk map source receipt exceeds its fixed profile", + )); + } + let mut bytes = Vec::with_capacity(RECEIPT_DOMAIN.len() + 8 + scope.len() + source.len() + 100); + bytes.extend_from_slice(RECEIPT_DOMAIN); + bytes.extend_from_slice(&(scope.len() as u32).to_be_bytes()); + bytes.extend_from_slice(scope); + bytes.extend_from_slice(&(source.len() as u32).to_be_bytes()); + bytes.extend_from_slice(&source); + bytes.extend_from_slice(&map.encode()); + Ok(bytes) +} + +fn receipt_key(source_id: [u8; 32]) -> ObjectKey { + ObjectKey { + namespace: ObjectNamespace::ChunkMapReceipt, + key: hex::encode(source_id), + } +} + +async fn read_receipt( + objects: &MegaObjectStorageWrapper, + key: &ObjectKey, + expected: &[u8], +) -> Result<(), SnapshotError> { + tokio::time::timeout(std::time::Duration::from_secs(5), async { + let (mut stream, meta) = objects + .inner + .get_stream(key) + .await + .map_err(|_| integrity("admitted chunk map trusted receipt is unavailable"))?; + if meta.size != expected.len() as i64 { + return Err(integrity( + "chunk map trusted receipt has invalid total size", + )); + } + let mut offset = 0; + while let Some(part) = stream.next().await { + let bytes = part.map_err(|_| integrity("chunk map trusted receipt stream failed"))?; + if bytes.len() > expected.len() - offset + || &expected[offset..offset + bytes.len()] != bytes.as_ref() + { + return Err(integrity( + "chunk map trusted receipt bytes disagree with source and descriptor", + )); + } + offset += bytes.len(); + } + if offset != expected.len() { + return Err(integrity("chunk map trusted receipt is truncated")); + } + Ok(()) + }) + .await + .map_err(|_| { + SnapshotError::new( + SnapshotErrorCode::TemporaryUnavailable, + "chunk map receipt read timed out", + ) + })? +} + +fn encode_leaves(leaves: &[ChunkLeaf]) -> Result { + let values: Result, SnapshotError> = leaves.iter().map(|l| Ok(json!({"page_index":l.page_index,"payload":hex::encode(l.encode().map_err(|_| integrity("invalid verified leaf"))?)}))).collect(); + serde_json::to_string(&values?).map_err(db_error) +} + +fn encode_nodes(nodes: &[ChunkMapNode]) -> Result { + serde_json::to_string(&nodes.iter().map(|n| json!({"first_page":n.start,"page_count":n.pages,"digest":hex::encode(n.digest)})).collect::>()).map_err(db_error) +} + +async fn compare_complete_map( + txn: &DatabaseTransaction, + verified: &VerifiedSourceChunkMap, + nodes: &[ChunkMapNode], + schema: &str, +) -> Result<(), SnapshotError> { + let map = verified.map(); + let row = txn.query_one_raw(stmt(schema, "SELECT CASE WHEN pg_catalog.octet_length(descriptor)=100 THEN descriptor ELSE NULL END AS descriptor,page_count,CASE WHEN pg_catalog.octet_length(pages_root)=32 THEN pages_root ELSE NULL END AS pages_root,(SELECT pg_catalog.count(*) FROM mst2_chunk_map_leaf WHERE map_id=$1) AS leaves,(SELECT pg_catalog.count(*) FROM mst2_chunk_map_node WHERE map_id=$1) AS nodes FROM mst2_chunk_map WHERE map_id=$1", [map.map_id().to_vec().into()])) + .await.map_err(db_error)?.ok_or_else(|| integrity("chunk map installation descriptor is missing"))?; + if bounded_bytes(&row, "descriptor")? != map.encode() + || row.try_get::("", "page_count").map_err(db_error)? as u64 != map.page_count + || bounded_bytes(&row, "pages_root")? != map.pages_root + || row.try_get::("", "leaves").map_err(db_error)? as u64 != map.page_count + || row.try_get::("", "nodes").map_err(db_error)? as usize != nodes.len() + { + return Err(integrity( + "chunk map installation has conflicting descriptor or incomplete coverage", + )); + } + for batch in verified.leaves().chunks(64) { + let bad = txn.query_one_raw(stmt(schema, "SELECT p.page_index FROM pg_catalog.jsonb_to_recordset($2::jsonb) AS p(page_index integer,payload text) LEFT JOIN mst2_chunk_map_leaf l ON l.map_id=$1 AND l.page_index=p.page_index WHERE l.payload IS DISTINCT FROM pg_catalog.decode(p.payload,'hex') LIMIT 1", + [map.map_id().to_vec().into(), encode_leaves(batch)?.into()])).await.map_err(db_error)?; + if bad.is_some() { + return Err(integrity( + "immutable stored chunk map leaf conflicts with verified source", + )); + } + } + for batch in nodes.chunks(1024) { + let bad = txn.query_one_raw(stmt(schema, "SELECT p.first_page FROM pg_catalog.jsonb_to_recordset($2::jsonb) AS p(first_page integer,page_count integer,digest text) LEFT JOIN mst2_chunk_map_node n ON n.map_id=$1 AND n.first_page=p.first_page AND n.page_count=p.page_count WHERE n.digest IS DISTINCT FROM pg_catalog.decode(p.digest,'hex') LIMIT 1", + [map.map_id().to_vec().into(), encode_nodes(batch)?.into()])).await.map_err(db_error)?; + if bad.is_some() { + return Err(integrity( + "immutable stored chunk map node conflicts with verified source", + )); + } + } + Ok(()) +} + +fn quoted(schema: &str) -> String { + format!("\"{}\"", schema.replace('"', "\"\"")) +} + +/// The static statements use these exact relation tokens. Catalog functions +/// are explicitly qualified; relations never resolve through pg_temp. +fn stmt(schema: &str, sql: &str, values: [sea_orm::Value; N]) -> Statement { + Statement::from_sql_and_values(DbBackend::Postgres, qualify_relations(schema, sql), values) +} + +fn qualify_relations(schema: &str, sql: &str) -> String { + let mut qualified = String::with_capacity(sql.len() + 256); + let prefix = quoted(schema); + let bytes = sql.as_bytes(); + let mut cursor = 0; + while cursor < bytes.len() { + let start = cursor; + let byte = bytes[cursor]; + if byte == b'\'' || byte == b'"' { + cursor += 1; + while cursor < bytes.len() { + if bytes[cursor] == byte { + cursor += 1; + if cursor < bytes.len() && bytes[cursor] == byte { + cursor += 1; + } else { + break; + } + } else { + cursor += 1; + } + } + } else if byte.is_ascii_alphanumeric() || byte == b'_' { + cursor += 1; + while cursor < bytes.len() + && (bytes[cursor].is_ascii_alphanumeric() || bytes[cursor] == b'_') + { + cursor += 1; + } + if [ + "mst2_metadata_storage_scope", + "mst2_verified_object", + "mst2_chunk_map", + "mst2_chunk_map_source", + "mst2_chunk_map_leaf", + "mst2_chunk_map_node", + ] + .contains(&&sql[start..cursor]) + { + qualified.push_str(&prefix); + qualified.push('.'); + } + } else { + cursor += sql[cursor..].chars().next().map_or(1, char::len_utf8); + } + qualified.push_str(&sql[start..cursor]); + } + qualified +} + +#[cfg(test)] +mod sql_tests { + use super::*; + + #[test] + fn authority_relations_are_qualified_without_rewriting_catalog_literals() { + let sql = "SELECT 'mst2_metadata_storage_scope','quoted ''mst2_chunk_map''',\"mst2_chunk_map\" FROM mst2_metadata_storage_scope s JOIN mst2_verified_object f ON s.singleton=1 JOIN mst2_chunk_map m ON true JOIN mst2_chunk_map_leaf l ON true JOIN mst2_chunk_map_node n ON true JOIN mst2_chunk_map_source r ON true WHERE c.relname='mst2_metadata_storage_scope'"; + let qualified = qualify_relations("schema\"name", sql); + assert!(qualified.contains(&format!( + "FROM {}.mst2_metadata_storage_scope", + quoted("schema\"name") + ))); + for name in [ + "mst2_verified_object", + "mst2_chunk_map", + "mst2_chunk_map_leaf", + "mst2_chunk_map_node", + "mst2_chunk_map_source", + ] { + assert!(qualified.contains(&format!("JOIN {}.{name}", quoted("schema\"name")))); + } + assert!(qualified.starts_with( + "SELECT 'mst2_metadata_storage_scope','quoted ''mst2_chunk_map''',\"mst2_chunk_map\"" + )); + assert!(qualified.ends_with("WHERE c.relname='mst2_metadata_storage_scope'")); + } +} +fn bounded_bytes(row: &QueryResult, name: &str) -> Result, SnapshotError> { + row.try_get::>>("", name) + .map_err(db_error)? + .ok_or_else(|| { + integrity("persisted chunk map field is missing or outside its bounded profile") + }) +} +fn integrity(message: &str) -> SnapshotError { + SnapshotError::new(SnapshotErrorCode::IntegrityError, message) +} +fn db_error(error: impl std::fmt::Display) -> SnapshotError { + tracing::warn!(%error, "persisted chunk map storage operation failed"); + SnapshotError::new( + SnapshotErrorCode::TemporaryUnavailable, + "persisted chunk map storage operation failed", + ) +} +fn storage_error(error: impl std::fmt::Display) -> SnapshotError { + tracing::warn!(%error, "trusted chunk map receipt publication failed"); + SnapshotError::new( + SnapshotErrorCode::TemporaryUnavailable, + "trusted chunk map receipt publication failed", + ) +} +async fn finish( + txn: DatabaseTransaction, + result: Result, +) -> Result { + match result { + Ok(value) => { + txn.commit().await.map_err(db_error)?; + Ok(value) + } + Err(error) => { + let _ = txn.rollback().await; + Err(error) + } + } +} + +impl super::Storage { + pub(crate) async fn chunk_maps(&self) -> Result<&PostgresChunkMapRepository, SnapshotError> { + use super::base_storage::StorageConnector; + self.native_chunk_maps + .get_or_try_init(|| async { + PostgresChunkMapRepository::new(self.mono_storage().get_connection().clone()).await + }) + .await + } +} diff --git a/src/jupiter/storage/object_storage.rs b/src/jupiter/storage/object_storage.rs index 506f2974..f65a33ab 100644 --- a/src/jupiter/storage/object_storage.rs +++ b/src/jupiter/storage/object_storage.rs @@ -78,6 +78,110 @@ mod exact_range_tests { ); } } + + #[tokio::test] + async fn memory_chunk_receipts_reject_all_mutation_routes() { + let storage = mock_object_storage(); + super::assert_immutable_chunk_receipt_contract(storage.inner.as_ref()).await; + } +} + +#[cfg(test)] +pub(crate) async fn assert_immutable_chunk_receipt_contract( + storage: &dyn crate::orbit_api::factory::MegaObjectStorageWithLog, +) { + let key = ObjectKey { + namespace: crate::orbit_api::object_storage::ObjectNamespace::ChunkMapReceipt, + key: "a".repeat(64), + }; + let first = Bytes::from_static(b"trusted immutable receipt"); + storage + .put_metadata_atomic_create(&key, first.clone(), ObjectMeta::default()) + .await + .unwrap(); + storage + .put_metadata_atomic_create( + &key, + Bytes::from_static(b"conflicting replay"), + ObjectMeta::default(), + ) + .await + .unwrap(); + let input = || { + Box::pin(futures::stream::iter([Ok(Bytes::from_static( + b"overwrite", + ))])) as ObjectByteStream + }; + assert!( + storage + .put_stream(&key, input(), ObjectMeta::default()) + .await + .is_err() + ); + assert!( + storage + .put_stream_bounded(&key, input(), ObjectMeta::default()) + .await + .is_err() + ); + assert!( + storage + .put_metadata_atomic( + &key, + Bytes::from_static(b"overwrite"), + ObjectMeta::default() + ) + .await + .is_err() + ); + assert!( + storage + .put_metadata_atomic_create( + &key, + Bytes::from(vec![ + 0; + crate::orbit_api::object_storage::MAX_METADATA_ATOMIC_BYTES + + 1 + ]), + ObjectMeta::default() + ) + .await + .is_err() + ); + assert!( + storage + .append(&key, input(), ObjectMeta::default()) + .await + .is_err() + ); + assert!( + storage + .append_concurrently(&key, input(), ObjectMeta::default()) + .await + .is_err() + ); + assert!(storage.delete(&key).await.is_err()); + for method in [Method::PUT, Method::POST, Method::DELETE, Method::PATCH] { + assert!( + storage + .signed_url(&key, method, std::time::Duration::from_secs(60)) + .await + .is_err() + ); + } + assert!( + storage + .signed_url(&key, Method::GET, std::time::Duration::from_secs(60)) + .await + .is_ok() + ); + let (mut stream, meta) = storage.get_stream(&key).await.unwrap(); + assert_eq!(meta.size, first.len() as i64); + let mut observed = Vec::new(); + while let Some(part) = stream.next().await { + observed.extend_from_slice(&part.unwrap()); + } + assert_eq!(observed, first.as_ref()); } #[derive(Default)] @@ -111,6 +215,7 @@ impl MegaObjectStorage for InMemoryObjectStorage { data: ObjectByteStream, mut meta: ObjectMeta, ) -> OrbitResult<()> { + reject_receipt_mutation(key)?; let bytes = Self::read_stream(data).await?; meta.size = bytes.len() as i64; self.objects @@ -126,6 +231,7 @@ impl MegaObjectStorage for InMemoryObjectStorage { bytes: Bytes, mut meta: ObjectMeta, ) -> OrbitResult<()> { + reject_receipt_mutation(key)?; use crate::orbit_api::object_storage::MAX_METADATA_ATOMIC_BYTES; if bytes.len() > MAX_METADATA_ATOMIC_BYTES { return Err(IoOrbitError::Other(format!( @@ -140,6 +246,27 @@ impl MegaObjectStorage for InMemoryObjectStorage { Ok(()) } + async fn put_metadata_atomic_create( + &self, + key: &ObjectKey, + bytes: Bytes, + mut meta: ObjectMeta, + ) -> OrbitResult<()> { + if bytes.len() > crate::orbit_api::object_storage::MAX_METADATA_ATOMIC_BYTES { + return Err(IoOrbitError::Other( + "immutable atomic metadata exceeds its byte limit".into(), + )); + } + key.validate()?; + meta.size = bytes.len() as i64; + self.objects + .lock() + .map_err(|_| IoOrbitError::Other("object storage lock poisoned".into()))? + .entry(key.clone()) + .or_insert((bytes, meta)); + Ok(()) + } + async fn get_stream(&self, key: &ObjectKey) -> OrbitResult<(ObjectByteStream, ObjectMeta)> { let (bytes, meta) = self .objects @@ -208,14 +335,18 @@ impl MegaObjectStorage for InMemoryObjectStorage { async fn signed_url( &self, - _key: &ObjectKey, - _method: Method, + key: &ObjectKey, + method: Method, _expires_in: std::time::Duration, ) -> OrbitResult> { + if method != Method::GET { + reject_receipt_mutation(key)?; + } Ok(None) } async fn delete(&self, key: &ObjectKey) -> OrbitResult<()> { + reject_receipt_mutation(key)?; let deleted = self .objects .lock() @@ -230,6 +361,16 @@ impl MegaObjectStorage for InMemoryObjectStorage { } } +fn reject_receipt_mutation(key: &ObjectKey) -> OrbitResult<()> { + if key.namespace == crate::orbit_api::object_storage::ObjectNamespace::ChunkMapReceipt { + Err(IoOrbitError::Other( + "chunk map receipts require immutable atomic creation".into(), + )) + } else { + Ok(()) + } +} + #[async_trait::async_trait] impl LogStorage for InMemoryObjectStorage { async fn append( @@ -238,6 +379,7 @@ impl LogStorage for InMemoryObjectStorage { data: ObjectByteStream, mut meta: ObjectMeta, ) -> OrbitResult<()> { + reject_receipt_mutation(key)?; let bytes = Self::read_stream(data).await?; let mut objects = self .objects diff --git a/src/orbit/adapter/log.rs b/src/orbit/adapter/log.rs index 4d116744..6dd1d78f 100644 --- a/src/orbit/adapter/log.rs +++ b/src/orbit/adapter/log.rs @@ -8,6 +8,7 @@ impl LogStorage for ObjectStoreAdapter { data: ObjectByteStream, _meta: ObjectMeta, ) -> OrbitResult<()> { + super::object::reject_receipt_mutation(key)?; key.validate()?; // Fast path for single writer/single thread (optimized here): // - No conditional writes, no retries, no cleanup (no concurrent write conflicts) @@ -187,6 +188,7 @@ impl LogStorage for ObjectStoreAdapter { data: ObjectByteStream, _meta: ObjectMeta, ) -> OrbitResult<()> { + super::object::reject_receipt_mutation(key)?; key.validate()?; // The local backend has no conditional-write (CAS) support, so this // method cannot be made safe under contention there (it would fall back diff --git a/src/orbit/adapter/object.rs b/src/orbit/adapter/object.rs index 58ccacc0..87327a39 100644 --- a/src/orbit/adapter/object.rs +++ b/src/orbit/adapter/object.rs @@ -12,6 +12,7 @@ impl MegaObjectStorage for ObjectStoreAdapter { data: ObjectByteStream, _meta: ObjectMeta, ) -> OrbitResult<()> { + reject_receipt_mutation(key)?; let path = Self::checked_path(key)?; // Artifacts are keyed by UUID and must not depend on LFS/Git upload_strategy // (e.g. S3 often uses `Multipart` for LFS while we still need create-if-absent semantics). @@ -35,6 +36,7 @@ impl MegaObjectStorage for ObjectStoreAdapter { data: ObjectByteStream, _meta: ObjectMeta, ) -> OrbitResult<()> { + reject_receipt_mutation(key)?; let path = Self::checked_path(key)?; self.put_multipart(&path, data).await } @@ -45,6 +47,7 @@ impl MegaObjectStorage for ObjectStoreAdapter { bytes: Bytes, _meta: ObjectMeta, ) -> OrbitResult<()> { + reject_receipt_mutation(key)?; use crate::orbit_api::object_storage::MAX_METADATA_ATOMIC_BYTES; if bytes.len() > MAX_METADATA_ATOMIC_BYTES { return Err(IoOrbitError::Other(format!( @@ -61,6 +64,32 @@ impl MegaObjectStorage for ObjectStoreAdapter { Ok(()) } + async fn put_metadata_atomic_create( + &self, + key: &ObjectKey, + bytes: Bytes, + _meta: ObjectMeta, + ) -> OrbitResult<()> { + if bytes.len() > crate::orbit_api::object_storage::MAX_METADATA_ATOMIC_BYTES { + return Err(IoOrbitError::Other( + "immutable atomic metadata exceeds its byte limit".into(), + )); + } + let path = Self::checked_path(key)?; + match self + .to_store() + .put_opts( + &path, + PutPayload::from_bytes(bytes), + PutOptions::from(PutMode::Create), + ) + .await + { + Ok(_) | Err(object_store::Error::AlreadyExists { .. }) => Ok(()), + Err(error) => Err(IoOrbitError::from(error)), + } + } + async fn get_stream(&self, key: &ObjectKey) -> OrbitResult<(ObjectByteStream, ObjectMeta)> { let path = Self::checked_path(key)?; @@ -149,6 +178,11 @@ impl MegaObjectStorage for ObjectStoreAdapter { method: Method, expires_in: Duration, ) -> OrbitResult> { + if key.namespace == ObjectNamespace::ChunkMapReceipt && method != Method::GET { + return Err(IoOrbitError::Other( + "immutable chunk map receipts do not permit presigned mutation".into(), + )); + } let path = Self::checked_path(key)?; let url = match &self.store { @@ -178,6 +212,7 @@ impl MegaObjectStorage for ObjectStoreAdapter { } async fn delete(&self, key: &ObjectKey) -> OrbitResult<()> { + reject_receipt_mutation(key)?; let path = Self::checked_path(key)?; self.to_store() .delete(&path) @@ -187,10 +222,34 @@ impl MegaObjectStorage for ObjectStoreAdapter { } } +pub(super) fn reject_receipt_mutation(key: &ObjectKey) -> OrbitResult<()> { + if key.namespace == ObjectNamespace::ChunkMapReceipt { + Err(IoOrbitError::Other( + "chunk map receipts require immutable atomic creation".into(), + )) + } else { + Ok(()) + } +} + #[cfg(test)] mod exact_range_tests { use super::*; + #[tokio::test] + async fn local_chunk_receipts_reject_all_mutation_routes() { + let directory = tempfile::tempdir().unwrap(); + let adapter = ObjectStoreAdapter { + store: BackendStore::Local(Arc::new( + LocalFileSystem::new_with_prefix(directory.path()).unwrap(), + )), + upload_strategy: UploadStrategy::SinglePut, + presign_store: None, + }; + crate::jupiter::storage::object_storage::assert_immutable_chunk_receipt_contract(&adapter) + .await; + } + #[tokio::test] async fn local_exact_range_returns_selected_bytes_and_full_object_size_without_clipping() { let directory = tempfile::tempdir().unwrap(); diff --git a/src/orbit_api/object_storage.rs b/src/orbit_api/object_storage.rs index 7fe6f47a..90539315 100644 --- a/src/orbit_api/object_storage.rs +++ b/src/orbit_api/object_storage.rs @@ -122,6 +122,8 @@ pub enum ObjectNamespace { Media, /// Agent Capture objects (`docs/refactoring/agent-capture.md`). Agent, + /// Immutable receipts minted by the full-stream chunk-map verifier. + ChunkMapReceipt, } impl ObjectNamespace { @@ -135,6 +137,7 @@ impl ObjectNamespace { ObjectNamespace::Oci => "oci", ObjectNamespace::Media => "media", ObjectNamespace::Agent => "agent", + ObjectNamespace::ChunkMapReceipt => "chunk-map-receipt", } } } @@ -260,6 +263,21 @@ pub trait MegaObjectStorage: Send + Sync { )) } + /// Create a complete <=1 MiB immutable metadata object. An existing key + /// must retain its original bytes. The caller must read and compare the + /// existing object before treating an idempotent replay as success. + async fn put_metadata_atomic_create( + &self, + _key: &ObjectKey, + _bytes: Bytes, + _meta: ObjectMeta, + ) -> OrbitResult<()> { + Err(IoOrbitError::Other( + "immutable atomic metadata creation is not supported by this storage backend" + .to_string(), + )) + } + /// Retrieve a single object from the storage backend. /// /// # Returns @@ -527,6 +545,7 @@ mod tests { ObjectNamespace::Oci, ObjectNamespace::Media, ObjectNamespace::Agent, + ObjectNamespace::ChunkMapReceipt, ] { let key = ObjectKey { namespace: ns, @@ -550,6 +569,10 @@ mod tests { assert_eq!(ObjectNamespace::Oci.to_string(), "oci"); assert_eq!(ObjectNamespace::Media.to_string(), "media"); assert_eq!(ObjectNamespace::Agent.to_string(), "agent"); + assert_eq!( + ObjectNamespace::ChunkMapReceipt.to_string(), + "chunk-map-receipt" + ); } #[test] From 9b7a0b397548b7c0c7b170dca8998bf0f87635a4 Mon Sep 17 00:00:00 2001 From: Xiaoyang Han Date: Thu, 8 Oct 2026 05:11:40 +0800 Subject: [PATCH 28/50] fix(mst2): finish chunk-map admission and isolate native observers (#78) Fix the missing native_chunk_maps field in the shared test Storage constructor and replace the redundant install-flight initializer closure without changing its behavior. The persisted-META reader/GC race now observes the actual lock wait through a separate one-connection pool, preserving the original two-connection service pool and every protection, release/GC, owned-byte and damage assertion; the prior observer itself consumed the connection needed by its competitor. Certificate lifetime checks bind the requested ten-day expiry to the actual issuance request interval rather than time spent later parsing/verifying the certificate, preserving the X509/response expiry equality, key and SAN checks. Exact earlier #76 native result was 2413 passed/2 failed/2 ignored (reader observer and certificate timing); #77 failed all-target Clippy for the missing constructor field and redundant closure. Production trust/lease/retention guards remain. Format/diff/locked offline metadata pass; corrected native tests remain pending. --- .../snapshot_persisted_metadata_tests.rs | 19 +++++++++++-------- src/ceres/snapshot/chunk_map_gate.rs | 2 +- src/contract/vault/pki.rs | 17 ++++++++++------- src/jupiter/tests.rs | 1 + 4 files changed, 23 insertions(+), 16 deletions(-) diff --git a/src/api/router/snapshot_persisted_metadata_tests.rs b/src/api/router/snapshot_persisted_metadata_tests.rs index f66d58da..057d9d26 100644 --- a/src/api/router/snapshot_persisted_metadata_tests.rs +++ b/src/api/router/snapshot_persisted_metadata_tests.rs @@ -685,14 +685,17 @@ async fn mst2_persisted_meta_reader_holds_protection_until_all_route_bytes_are_o } }) }; - let observer = fixture - .state - .storage - .mono_storage() - .get_connection() - .begin() - .await - .unwrap(); + // The two service connections belong to the reader and competitor. + // Observe their lock wait without competing for either connection. + let mut observer_config = fixture.state.storage.config().database.clone(); + assert_eq!(observer_config.max_connection, 2); + observer_config.max_connection = 1; + observer_config.min_connection = 0; + let observer_connection = + crate::jupiter::storage::init::database_connection(&observer_config) + .await + .unwrap(); + let observer = observer_connection.begin().await.unwrap(); tokio::select! { () = wait_retention_waiter(&observer) => {}, outcome = &mut competing => { diff --git a/src/ceres/snapshot/chunk_map_gate.rs b/src/ceres/snapshot/chunk_map_gate.rs index 5c56a504..765e6c3b 100644 --- a/src/ceres/snapshot/chunk_map_gate.rs +++ b/src/ceres/snapshot/chunk_map_gate.rs @@ -25,7 +25,7 @@ pub(crate) struct InstallFlight { impl InstallFlight { pub(crate) fn acquire(key: [u8; 32]) -> Result { static REGISTRY: OnceLock>> = OnceLock::new(); - Self::from_registry(REGISTRY.get_or_init(|| Arc::default()).clone(), key) + Self::from_registry(REGISTRY.get_or_init(Arc::default).clone(), key) } fn from_registry(registry: Arc>, key: [u8; 32]) -> Result { diff --git a/src/contract/vault/pki.rs b/src/contract/vault/pki.rs index cc111aaa..2f262a2c 100644 --- a/src/contract/vault/pki.rs +++ b/src/contract/vault/pki.rs @@ -385,8 +385,16 @@ mod tests_raw { .unwrap() .clone(); + let issuance_started = SystemTime::now() + .duration_since(UNIX_EPOCH) + .unwrap() + .as_secs(); // issue cert let resp = test_write_api(core, "pki/issue/tls/test", true, Some(issue_data)).await; + let issuance_completed = SystemTime::now() + .duration_since(UNIX_EPOCH) + .unwrap() + .as_secs(); assert!(resp.is_ok()); let resp_body = resp.unwrap(); assert!(resp_body.is_some()); @@ -429,15 +437,10 @@ mod tests_raw { let ttl_compare = cert.not_after().compare(&expiration_time); assert!(ttl_compare.is_ok()); assert_eq!(ttl_compare.unwrap(), std::cmp::Ordering::Equal); - let now_timestamp = SystemTime::now() - .duration_since(UNIX_EPOCH) - .unwrap() - .as_secs(); let expiration_ttl = cert_data["expiration"].as_u64().unwrap(); - let ttl = expiration_ttl - now_timestamp; let expect_ttl = 10 * 24 * 60 * 60; - assert!(ttl <= expect_ttl); - assert!((ttl + 10) > expect_ttl); + assert!(expiration_ttl >= issuance_started + expect_ttl); + assert!(expiration_ttl <= issuance_completed + expect_ttl); let authority_key_id = cert.authority_key_id(); assert!(authority_key_id.is_some()); diff --git a/src/jupiter/tests.rs b/src/jupiter/tests.rs index 2b03da07..4f39f1ad 100644 --- a/src/jupiter/tests.rs +++ b/src/jupiter/tests.rs @@ -379,6 +379,7 @@ pub async fn test_storage_with_config(temp_dir: impl AsRef, config: Config app_service: Arc::new(svc), native_projection_cache: Arc::default(), native_snapshot_sessions: Arc::default(), + native_chunk_maps: Arc::default(), projection_observation_sink: None, cl_service: CLService::mock(), push_queue_service: PushQueueService::new( From 6db633297e6cca9ad3a7724cb893beb043937ef0 Mon Sep 17 00:00:00 2001 From: Xiaoyang Han Date: Thu, 8 Oct 2026 05:24:50 +0800 Subject: [PATCH 29/50] perf(mst2): stream authenticated raw blobs with owned chunk credit (#79) Raw GET replaces the materializing whole-blob handler with one sequential current-OID stream, authenticated against an independently earned immutable source receipt and one selected canonical chunk page at a time. Consumer scratch and first response credit are admitted before cold source IO; each at-most-1-MiB transport chunk owns its memory credit and map reader through its last Bytes clone. Physical metadata, exact lengths, authenticated per-chunk hashes, final full SHA256 and real final EOF reject corruption, truncation, growth and late IO errors. First-poll, post-await and per-delivery access validation preserves lease revocation, cancellation drops producer ownership, and empty files require a verified empty digest and actual zero-byte EOF. The formal unsupported Range-header HTTP 400 behavior is retained. Thirteen actual HTTP regression tests cover cold two-pass and warm one-stream read counts, fragmentation, owned clone credit, admission, producer item limits, no proof fallback, physical facts, real stored corruption, lease cancellation and immediate post-await rejection of short/empty fragments and real EOF without a later backend poll. Formatting/diff/offline metadata pass; native build/Clippy/PG tests and measured throughput/latency/RSS remain pending. --- src/api/router/snapshot_content.rs | 12 +- src/api/router/snapshot_content_tests.rs | 12 +- .../router/snapshot_objects_bounded_tests.rs | 48 ++ src/api/router/snapshot_raw_blob.rs | 390 +++++++++ src/api/router/snapshot_raw_blob_tests.rs | 791 ++++++++++++++++++ src/api/router/snapshot_router.rs | 130 +-- src/api/router/snapshot_session_tests.rs | 3 +- src/ceres/api_service/mod.rs | 20 + src/ceres/snapshot/content_budget.rs | 16 +- src/jupiter/service/git_service.rs | 10 +- 10 files changed, 1291 insertions(+), 141 deletions(-) create mode 100644 src/api/router/snapshot_raw_blob.rs create mode 100644 src/api/router/snapshot_raw_blob_tests.rs diff --git a/src/api/router/snapshot_content.rs b/src/api/router/snapshot_content.rs index d642434f..6a1f191f 100644 --- a/src/api/router/snapshot_content.rs +++ b/src/api/router/snapshot_content.rs @@ -27,15 +27,15 @@ use crate::ceres::snapshot::{ }; pub(super) struct ResolvedFileMetadata { - fs_kind: FsKind, - oid: String, + pub(super) fs_kind: FsKind, + pub(super) oid: String, pub(super) digest: [u8; 32], pub(super) size: u64, fact: crate::callisto::mst2_verified_object::Model, } #[allow(clippy::result_large_err)] -async fn fixed_root_tree( +pub(super) async fn fixed_root_tree( handler: &T, oid: &str, ) -> Result { @@ -50,7 +50,7 @@ async fn fixed_root_tree( } #[allow(clippy::result_large_err)] -async fn resolve_file_metadata( +pub(super) async fn resolve_file_metadata( handler: &T, root_tree: &git_internal::internal::object::tree::Tree, scope: &str, @@ -479,7 +479,7 @@ async fn project_for( } #[allow(clippy::result_large_err)] -async fn project_resolved( +pub(super) async fn project_resolved( handler: &T, f: &ResolvedFileMetadata, ) -> Result, Response> @@ -789,7 +789,7 @@ struct ResolvedChunk { map_id: String, } -fn content_read_error(error: crate::common::errors::MegaError) -> SnapshotError { +pub(super) fn content_read_error(error: crate::common::errors::MegaError) -> SnapshotError { use crate::common::errors::MegaError; let code = match &error { MegaError::ObjStorageNotFound(_) => SnapshotErrorCode::ObjectUnavailable, diff --git a/src/api/router/snapshot_content_tests.rs b/src/api/router/snapshot_content_tests.rs index 4399e692..617cf2ea 100644 --- a/src/api/router/snapshot_content_tests.rs +++ b/src/api/router/snapshot_content_tests.rs @@ -88,6 +88,9 @@ mod bounded_chunks; #[path = "snapshot_persisted_chunk_map_tests.rs"] mod persisted_chunk_maps; +#[path = "snapshot_raw_blob_tests.rs"] +mod raw_blob; + #[path = "snapshot_session_tests.rs"] mod durable_sessions; @@ -111,6 +114,7 @@ struct ReadCounts { range: AtomicUsize, bytes: AtomicUsize, object_fault: std::sync::Mutex>, + object_size_override: std::sync::Mutex>, chunk_faults: std::sync::Mutex>, receipt_reads: AtomicUsize, receipt_writes: AtomicUsize, @@ -244,7 +248,10 @@ impl MegaObjectStorage for CountingStorage { }, )); } - let (stream, meta) = self.inner.inner.get_stream(key).await?; + let (stream, mut meta) = self.inner.inner.get_stream(key).await?; + if let Some(size) = *self.counts.object_size_override.lock().unwrap() { + meta.size = size; + } let fault = self.counts.object_fault.lock().unwrap().clone(); let stream = match fault { Some(fault) if fault.oid == key.key => fault.stream(), @@ -904,7 +911,8 @@ async fn mst2_fixed_head_uses_verified_facts_without_body_reads_and_preserves_ra .await .unwrap(); assert_eq!(bytes.as_ref(), fixture.raw); - fixture.counts.assert(1, fixture.raw.len()); + fixture.counts.assert(2, 2 * fixture.raw.len()); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 1); } #[tokio::test] diff --git a/src/api/router/snapshot_objects_bounded_tests.rs b/src/api/router/snapshot_objects_bounded_tests.rs index b5e0d303..e7eb0bfa 100644 --- a/src/api/router/snapshot_objects_bounded_tests.rs +++ b/src/api/router/snapshot_objects_bounded_tests.rs @@ -26,6 +26,14 @@ pub(super) enum FaultKind { release: Arc, drops: Arc, }, + HeldFragment { + prefix: Bytes, + fragment: Option, + entered: Arc, + release: Arc, + tail_polls: Arc, + drops: Arc, + }, } struct DropCount(Arc); @@ -39,6 +47,46 @@ impl Drop for DropCount { impl StreamFault { pub(super) fn stream(self) -> ObjectByteStream { match self.kind { + FaultKind::HeldFragment { + prefix, + fragment, + entered, + release, + tail_polls, + drops, + } => Box::pin(futures::stream::unfold( + ( + prefix, + fragment, + entered, + release, + tail_polls, + DropCount(drops), + 0u8, + ), + |(prefix, fragment, entered, release, tail_polls, owner, turn)| async move { + match turn { + 0 => Some(( + Ok(prefix.clone()), + (prefix, fragment, entered, release, tail_polls, owner, 1), + )), + 1 => { + entered.notify_one(); + release.notified().await; + fragment.clone().map(|part| { + ( + Ok(part), + (prefix, fragment, entered, release, tail_polls, owner, 2), + ) + }) + } + _ => { + tail_polls.fetch_add(1, Ordering::SeqCst); + std::future::pending().await + } + } + }, + )), FaultKind::HeldThenError { entered, release, diff --git a/src/api/router/snapshot_raw_blob.rs b/src/api/router/snapshot_raw_blob.rs new file mode 100644 index 00000000..efda9649 --- /dev/null +++ b/src/api/router/snapshot_raw_blob.rs @@ -0,0 +1,390 @@ +//! Whole-file raw delivery from one current-OID stream, authenticated in chunks. + +use std::sync::Arc; + +use axum::{ + body::Body, + extract::{Path as AxumPath, Query, State}, + http::HeaderMap, + response::Response, +}; +use bytes::Bytes; +use futures::StreamExt; +use mst2_codec::chunkmap::{CHUNK_SIZE, CHUNKS_PER_PAGE}; +use serde::Deserialize; +use sha2::{Digest, Sha256}; + +use super::{ + REQUEST_HEADERS, content, ensure_enabled, internal, mst2_error_response, revalidate_access, + revalidate_request, +}; +use crate::{ + api::MonoApiServiceState, + ceres::snapshot::{ + content_budget::{ + MemoryBudget, MemoryLease, RANGE_WORK_BYTES, range_budget, response_budget, + }, + error::{SnapshotError, SnapshotErrorCode}, + pages::hex_of, + runtime::SnapshotContext, + view::validate_scope_relative_path, + }, + jupiter::storage::{ + Storage, + native_chunk_map::{AuthenticatedChunkPage, PersistedChunkMap}, + }, + orbit_api::object_storage::ObjectByteStream, +}; + +const PRODUCER_ITEM_MAX: usize = 8 * 1024 * 1024; + +#[derive(Deserialize, Debug)] +pub(super) struct BlobQuery { + path: String, + #[serde(default)] + expected_digest: Option, +} + +pub(super) struct BlobBudgets { + pub(super) scratch: Arc, + pub(super) response: Arc, +} + +impl Default for BlobBudgets { + fn default() -> Self { + Self { + scratch: range_budget().clone(), + response: response_budget().clone(), + } + } +} + +pub(super) async fn blob( + state: State, + path: AxumPath, + query: Query, + headers: HeaderMap, +) -> Response { + match blob_with_budgets(state, path, query, headers, BlobBudgets::default()).await { + Ok(response) | Err(response) => response, + } +} + +#[allow(clippy::result_large_err)] +pub(super) async fn blob_with_budgets( + state: State, + AxumPath(snapshot_id): AxumPath, + Query(query): Query, + headers: HeaderMap, + budgets: BlobBudgets, +) -> Result { + ensure_enabled(&state).map_err(mst2_error_response)?; + let context = super::request_context(&state, &snapshot_id).map_err(mst2_error_response)?; + validate_scope_relative_path(&query.path).map_err(mst2_error_response)?; + if headers.contains_key("range") { + return Err(mst2_error_response(SnapshotError::new( + SnapshotErrorCode::RangeNotSupported, + "raw blob reads are whole-file; use chunks for ranges", + ))); + } + let handler = state + .api_handler(std::path::Path::new("/")) + .await + .map_err(internal)?; + let root = content::fixed_root_tree(handler.as_ref(), &context.root_tree_oid).await?; + let file = content::resolve_file_metadata( + handler.as_ref(), + &root, + &context.built.descriptor.scope, + &query.path, + query.expected_digest.as_deref(), + ) + .await?; + let scratch = budgets + .scratch + .reserve(if file.size == 0 { 0 } else { RANGE_WORK_BYTES }) + .map_err(mst2_error_response)?; + let first_credit = budgets + .response + .reserve(file.size.min(CHUNK_SIZE as u64) as usize) + .map_err(mst2_error_response)?; + revalidate_request(&state, &context) + .await + .map_err(mst2_error_response)?; + let map = if file.size == 0 { + if file.digest != <[u8; 32]>::from(Sha256::digest([])) { + return Err(mst2_error_response(integrity( + "empty raw source has a nonempty verified digest", + ))); + } + None + } else { + Some(content::project_resolved(handler.as_ref(), &file).await?) + }; + let storage = handler.get_context(); + let page = if let Some(map) = map.as_ref() { + Some( + storage + .chunk_maps() + .await + .map_err(mst2_error_response)? + .selected_page(map, 0) + .await + .map_err(mst2_error_response)?, + ) + } else { + None + }; + let (input, meta) = handler + .get_raw_blob_stream_with_meta(&file.oid) + .await + .map_err(|error| mst2_error_response(content::content_read_error(error)))?; + if u64::try_from(meta.size).ok() != Some(file.size) { + return Err(mst2_error_response(integrity( + "raw source physical size disagrees with its fixed verified fact", + ))); + } + let headers = REQUEST_HEADERS.try_with(Clone::clone).map_err(|_| { + mst2_error_response(SnapshotError::new( + SnapshotErrorCode::Unauthenticated, + "request authentication context missing", + )) + })?; + revalidate_access(&state, &context, &headers) + .await + .map_err(mst2_error_response)?; + let digest = file.digest; + let size = file.size; + let kind = content::fs_kind_str(file.fs_kind); + let mut reader = RawBlobReader { + state: state.0, + context, + headers, + storage, + map, + page, + input: Some(input), + pending: Bytes::new(), + pending_offset: 0, + received: 0, + index: 0, + size, + digest, + hash: Sha256::new(), + first_credit: Some(first_credit), + response_budget: budgets.response, + _scratch: scratch, + }; + let body = if size == 0 { + reader.require_eof().await.map_err(mst2_error_response)?; + reader.validate().await.map_err(mst2_error_response)?; + Body::empty() + } else { + let stream = futures::stream::try_unfold(reader, |mut reader| async move { + let Some(bytes) = reader.next_chunk().await? else { + return Ok::<_, SnapshotError>(None); + }; + Ok(Some((bytes, reader))) + }); + Body::from_stream(stream) + }; + Response::builder() + .header("etag", format!("\"sha256:{}\"", hex_of(&digest))) + .header("content-length", size.to_string()) + .header("x-mega-content-size", size.to_string()) + .header("x-mega-fs-kind", kind) + .header("cache-control", "private, no-cache, no-transform") + .header("vary", "Authorization, Accept") + .body(body) + .map_err(|_| mst2_error_response(internal("raw blob response build failed"))) +} + +struct RawBlobReader { + state: MonoApiServiceState, + context: SnapshotContext, + headers: HeaderMap, + storage: Storage, + map: Option>, + page: Option>, + input: Option, + pending: Bytes, + pending_offset: usize, + received: u64, + index: u64, + size: u64, + digest: [u8; 32], + hash: Sha256, + first_credit: Option, + response_budget: Arc, + _scratch: MemoryLease, +} + +struct RawBlobChunk { + bytes: Vec, + _credit: MemoryLease, + _map: Arc, +} + +impl AsRef<[u8]> for RawBlobChunk { + fn as_ref(&self) -> &[u8] { + &self.bytes + } +} + +impl RawBlobReader { + async fn validate(&self) -> Result<(), SnapshotError> { + revalidate_access(&self.state, &self.context, &self.headers).await + } + + async fn next_chunk(&mut self) -> Result, SnapshotError> { + self.validate().await?; + let map = self + .map + .as_ref() + .ok_or_else(|| integrity("nonempty raw source has no authenticated map"))?; + if self.index == map.map.chunk_count { + return Ok(None); + } + let length = map + .map + .chunk_len(self.index) + .map_err(|_| integrity("raw source chunk index disagrees with its map"))? + as usize; + let page_index = self.index / CHUNKS_PER_PAGE as u64; + if self.page.as_ref().map(|p| p.leaf.page_index) != Some(page_index) { + drop(self.page.take()); + self.page = Some( + self.storage + .chunk_maps() + .await? + .selected_page(map, page_index) + .await?, + ); + self.validate().await?; + } + let credit = if let Some(credit) = self.first_credit.take() { + credit + } else { + self.response_budget.reserve(length)? + }; + let mut raw = Vec::new(); + raw.try_reserve_exact(length).map_err(|_| { + SnapshotError::new( + SnapshotErrorCode::TemporaryUnavailable, + "raw chunk allocation could not be admitted", + ) + })?; + let mut empty_parts = 0; + while raw.len() < length { + if self.pending_offset == self.pending.len() { + self.pending = Bytes::new(); + self.pending_offset = 0; + let Some(part) = self.poll_source().await? else { + return Err(integrity("raw source is truncated before its final chunk")); + }; + if part.len() as u64 > self.size - self.received { + return Err(integrity("raw source exceeds its fixed verified size")); + } + if part.len() > PRODUCER_ITEM_MAX { + return Err(SnapshotError::new( + SnapshotErrorCode::TemporaryUnavailable, + "raw producer item exceeds its consumer processing limit", + )); + } + self.received += part.len() as u64; + if part.is_empty() { + empty_parts += 1; + if empty_parts == 32 { + tokio::task::yield_now().await; + self.validate().await?; + empty_parts = 0; + } + continue; + } + self.pending = part; + } + let amount = (length - raw.len()).min(self.pending.len() - self.pending_offset); + let end = self.pending_offset + amount; + let bytes = &self.pending[self.pending_offset..end]; + self.hash.update(bytes); + raw.extend_from_slice(bytes); + self.pending_offset = end; + } + let map = self + .map + .as_ref() + .ok_or_else(|| integrity("raw source authenticated map disappeared"))?; + self.page + .as_ref() + .ok_or_else(|| integrity("raw source authenticated page disappeared"))? + .verify_chunk(&map.map, self.index, &raw)?; + self.index += 1; + if self.index == map.map.chunk_count { + self.require_eof().await?; + let actual: [u8; 32] = self.hash.clone().finalize().into(); + if actual != self.digest { + return Err(integrity( + "raw source whole-file digest disagrees with its fixed fact", + )); + } + } + self.validate().await?; + if raw.capacity() > credit.bytes { + return Err(integrity("raw chunk allocation exceeds its owned credit")); + } + Ok(Some(Bytes::from_owner(RawBlobChunk { + bytes: raw, + _credit: credit, + _map: self + .map + .as_ref() + .cloned() + .ok_or_else(|| integrity("raw source authenticated map disappeared"))?, + }))) + } + + async fn poll_source(&mut self) -> Result, SnapshotError> { + let input = self + .input + .as_mut() + .ok_or_else(|| integrity("raw source stream was already closed"))?; + let result = input.next().await; + self.validate().await?; + result.transpose().map_err(|error| { + tracing::warn!(%error, "fixed-view raw stream failed"); + SnapshotError::new( + SnapshotErrorCode::ObjectUnavailable, + "fixed-view raw stream failed", + ) + }) + } + + async fn require_eof(&mut self) -> Result<(), SnapshotError> { + if self.received != self.size || self.pending_offset != self.pending.len() { + return Err(integrity( + "raw source final length disagrees with its fixed fact", + )); + } + let mut empty_parts = 0; + while let Some(part) = self.poll_source().await? { + if !part.is_empty() { + return Err(integrity( + "raw source has trailing bytes after its fixed size", + )); + } + empty_parts += 1; + if empty_parts == 32 { + tokio::task::yield_now().await; + self.validate().await?; + empty_parts = 0; + } + } + drop(self.input.take()); + self.pending = Bytes::new(); + Ok(()) + } +} + +fn integrity(message: &str) -> SnapshotError { + SnapshotError::new(SnapshotErrorCode::IntegrityError, message) +} diff --git a/src/api/router/snapshot_raw_blob_tests.rs b/src/api/router/snapshot_raw_blob_tests.rs new file mode 100644 index 00000000..1c4e70f9 --- /dev/null +++ b/src/api/router/snapshot_raw_blob_tests.rs @@ -0,0 +1,791 @@ +use axum::{ + extract::{Path as AxumPath, Query, State}, + http::HeaderMap, + routing::get, +}; +use tokio::{sync::Notify, time::timeout}; + +use super::*; +use crate::{ + api::router::snapshot_router::{raw_blob as serving, snapshot_auth_middleware}, + ceres::snapshot::content_budget::{MemoryBudget, RANGE_WORK_BYTES}, +}; + +fn budgeted_app( + fixture: &Fixture, + response_budget: &Arc, + scratch_budget: &Arc, +) -> Router { + let response_budget = response_budget.clone(); + let scratch_budget = scratch_budget.clone(); + Router::new() + .route( + "/api/v2/snapshots/{snapshot_id}/blob", + get( + move |state: State, + path: AxumPath, + query: Query, + headers: HeaderMap| { + serving::blob_with_budgets( + state, + path, + query, + headers, + serving::BlobBudgets { + response: response_budget.clone(), + scratch: scratch_budget.clone(), + }, + ) + }, + ), + ) + .route_layer(axum::middleware::from_fn_with_state( + fixture.state.clone(), + snapshot_auth_middleware, + )) + .with_state(fixture.state.clone()) +} + +fn fault(fixture: &Fixture, kind: bounded_objects::FaultKind) { + *fixture.counts.object_fault.lock().unwrap() = Some(bounded_objects::StreamFault { + oid: fixture.oid.clone(), + kind, + }); +} + +async fn path_oid(fixture: &Fixture, path: &str) -> String { + use crate::ceres::snapshot::pages::{MetadataWalkOutcome, resolve_abs_metadata}; + let handler = fixture + .state + .api_handler(std::path::Path::new("/")) + .await + .unwrap(); + let context = fixture + .state + .storage + .snapshot_context(&fixture.snapshot, &fixture.lease) + .await + .unwrap(); + let root = handler + .get_tree_by_hash(&context.root_tree_oid) + .await + .unwrap(); + match resolve_abs_metadata(handler.as_ref(), &root, &format!("/project{path}")) + .await + .unwrap() + { + MetadataWalkOutcome::FoundFile { oid, .. } => oid, + other => panic!("raw fixture path must be a fixed file: {other:?}"), + } +} + +async fn release_lease(fixture: &Fixture) { + let response = fixture + .app + .clone() + .oneshot( + Request::builder() + .method("DELETE") + .uri(format!("/api/v2/snapshots/leases/{}", fixture.lease)) + .header("authorization", format!("Bearer {TOKEN}")) + .body(Body::empty()) + .unwrap(), + ) + .await + .unwrap(); + assert_eq!(response.status(), 200); +} + +#[tokio::test] +async fn cold_raw_get_earns_source_proof_then_serves_one_current_stream_with_exact_headers() { + let fixture = Fixture::new().await; + let response = fixture.send("GET", "blob?path=/file", Body::empty()).await; + assert_eq!(response.status(), 200); + assert_eq!( + response.headers()["content-length"], + fixture.raw.len().to_string() + ); + assert_eq!( + response.headers()["x-mega-content-size"], + fixture.raw.len().to_string() + ); + assert_eq!( + response.headers()["etag"], + format!("\"{}\"", fixture.digest_string()) + ); + assert_eq!(response.headers()["x-mega-fs-kind"], "regular"); + assert_eq!(response.headers()["vary"], "Authorization, Accept"); + assert_eq!( + response.headers()["cache-control"], + "private, no-cache, no-transform" + ); + assert_eq!(fixture.counts.whole.load(Ordering::SeqCst), 2); + assert_eq!( + fixture.counts.bytes.load(Ordering::SeqCst), + fixture.raw.len() + ); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 1); + assert_eq!( + to_bytes(response.into_body(), 2 * CHUNK_SIZE as usize) + .await + .unwrap() + .as_ref(), + fixture.raw + ); + fixture.counts.assert(2, 2 * fixture.raw.len()); + fixture.counts.reset(); + let response = fixture.send("GET", "blob?path=/alias", Body::empty()).await; + assert_eq!(response.status(), 200); + assert_eq!(fixture.counts.bytes.load(Ordering::SeqCst), 0); + assert_eq!( + to_bytes(response.into_body(), 2 * CHUNK_SIZE as usize) + .await + .unwrap() + .as_ref(), + fixture.raw + ); + fixture.counts.assert(1, fixture.raw.len()); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 0); + assert_eq!(fixture.counts.receipt_reads.load(Ordering::SeqCst), 1); +} + +#[tokio::test] +async fn raw_memory_pressure_rejects_before_cold_proof_or_delivery_source_io() { + let fixture = Fixture::new().await; + for occupy_scratch in [false, true] { + fixture.counts.reset(); + let response_budget = MemoryBudget::new(CHUNK_SIZE as usize); + let scratch_budget = MemoryBudget::new(RANGE_WORK_BYTES); + let held = if occupy_scratch { + scratch_budget.reserve(RANGE_WORK_BYTES).unwrap() + } else { + response_budget.reserve(CHUNK_SIZE as usize).unwrap() + }; + let response = budgeted_app(&fixture, &response_budget, &scratch_budget) + .oneshot(fixture.request("GET", "blob?path=/file", Body::empty())) + .await + .unwrap(); + error(response, 503, "TEMPORARY_UNAVAILABLE", true).await; + fixture.counts.assert(0, 0); + assert_eq!(fixture.counts.receipt_reads.load(Ordering::SeqCst), 0); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 0); + assert_eq!( + scratch_budget.used(), + if occupy_scratch { RANGE_WORK_BYTES } else { 0 } + ); + assert_eq!( + response_budget.used(), + if occupy_scratch { + 0 + } else { + CHUNK_SIZE as usize + } + ); + drop(held); + assert_eq!(response_budget.used(), 0); + assert_eq!(scratch_budget.used(), 0); + } +} + +#[tokio::test] +async fn raw_missing_or_forged_receipt_and_real_stored_corruption_fail_without_rebuilding() { + let fixture = Fixture::new().await; + fixture.map("/file").await; + for missing in [true, false] { + fixture.counts.reset(); + fixture + .counts + .receipt_read_failure + .store(missing, Ordering::SeqCst); + *fixture.counts.receipt_read_corruption.lock().unwrap() = + (!missing).then(|| Bytes::from_static(b"forged receipt")); + error( + fixture.send("GET", "blob?path=/file", Body::empty()).await, + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + fixture.counts.assert(0, 0); + assert_eq!(fixture.counts.receipt_reads.load(Ordering::SeqCst), 1); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 0); + } + fixture + .counts + .receipt_read_failure + .store(false, Ordering::SeqCst); + *fixture.counts.receipt_read_corruption.lock().unwrap() = None; + let mut wrong = fixture.raw.clone(); + wrong[0] ^= 1; + fixture.write_raw(wrong).await; + let response = fixture.send("GET", "blob?path=/file", Body::empty()).await; + assert_eq!(response.status(), 200); + let mut stream = response.into_body().into_data_stream(); + assert!(stream.next().await.unwrap().is_err()); + assert!(stream.next().await.is_none()); + assert_eq!(fixture.counts.whole.load(Ordering::SeqCst), 1); + assert_eq!(fixture.counts.range.load(Ordering::SeqCst), 0); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 0); + fixture.write_raw(fixture.raw.clone()).await; + let response = fixture.send("GET", "blob?path=/file", Body::empty()).await; + assert_eq!(response.status(), 200); + assert_eq!( + to_bytes(response.into_body(), 2 * CHUNK_SIZE as usize) + .await + .unwrap() + .as_ref(), + fixture.raw + ); + fixture.counts.assert(1, fixture.raw.len()); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 0); +} + +#[tokio::test] +async fn warm_fragmented_actual_raw_body_polls_one_chunk_and_transport_clones_keep_exact_credit() { + let fixture = Fixture::new().await; + fixture.map("/file").await; + fixture.counts.reset(); + fault( + &fixture, + bounded_objects::FaultKind::Parts(vec![ + Bytes::new(), + Bytes::copy_from_slice(&fixture.raw[..13]), + Bytes::copy_from_slice(&fixture.raw[13..CHUNK_SIZE as usize - 7]), + Bytes::copy_from_slice(&fixture.raw[CHUNK_SIZE as usize - 7..CHUNK_SIZE as usize]), + Bytes::copy_from_slice(&fixture.raw[CHUNK_SIZE as usize..]), + Bytes::new(), + ]), + ); + let response_budget = MemoryBudget::new(fixture.raw.len()); + let scratch_budget = MemoryBudget::new(RANGE_WORK_BYTES); + let response = budgeted_app(&fixture, &response_budget, &scratch_budget) + .oneshot(fixture.request("GET", "blob?path=/file", Body::empty())) + .await + .unwrap(); + assert_eq!(response.status(), 200); + fixture.counts.assert(1, 0); + assert_eq!(response_budget.used(), CHUNK_SIZE as usize); + assert_eq!(scratch_budget.used(), RANGE_WORK_BYTES); + let mut stream = response.into_body().into_data_stream(); + let first = stream.next().await.unwrap().unwrap(); + assert_eq!(first.as_ref(), &fixture.raw[..CHUNK_SIZE as usize]); + fixture.counts.assert(1, CHUNK_SIZE as usize); + let transport = first.clone(); + drop(first); + assert_eq!(response_budget.used(), CHUNK_SIZE as usize); + let final_bytes = stream.next().await.unwrap().unwrap(); + assert_eq!(final_bytes.as_ref(), &fixture.raw[CHUNK_SIZE as usize..]); + assert_eq!(response_budget.used(), fixture.raw.len()); + assert!(stream.next().await.is_none()); + assert_eq!(scratch_budget.used(), 0); + assert_eq!(response_budget.used(), fixture.raw.len()); + drop(stream); + drop(final_bytes); + assert_eq!(response_budget.used(), CHUNK_SIZE as usize); + drop(transport); + assert_eq!(response_budget.used(), 0); + fixture.counts.assert(1, fixture.raw.len()); +} + +#[tokio::test] +async fn raw_transport_quota_rejects_next_chunk_before_more_source_io_and_errors_terminally() { + let fixture = Fixture::new().await; + fixture.map("/file").await; + fixture.counts.reset(); + fault( + &fixture, + bounded_objects::FaultKind::Parts(vec![ + Bytes::copy_from_slice(&fixture.raw[..CHUNK_SIZE as usize]), + Bytes::copy_from_slice(&fixture.raw[CHUNK_SIZE as usize..]), + ]), + ); + let response_budget = MemoryBudget::new(CHUNK_SIZE as usize); + let scratch_budget = MemoryBudget::new(RANGE_WORK_BYTES); + let response = budgeted_app(&fixture, &response_budget, &scratch_budget) + .oneshot(fixture.request("GET", "blob?path=/file", Body::empty())) + .await + .unwrap(); + assert_eq!(response.status(), 200); + let mut stream = response.into_body().into_data_stream(); + let transport = stream.next().await.unwrap().unwrap(); + assert_eq!(transport.as_ref(), &fixture.raw[..CHUNK_SIZE as usize]); + assert!(stream.next().await.unwrap().is_err()); + assert!(stream.next().await.is_none()); + fixture.counts.assert(1, CHUNK_SIZE as usize); + assert_eq!(scratch_budget.used(), 0); + assert_eq!(response_budget.used(), CHUNK_SIZE as usize); + drop(transport); + assert_eq!(response_budget.used(), 0); +} + +#[tokio::test] +async fn warm_raw_corruption_growth_truncation_and_late_error_never_yield_the_last_bytes() { + let fixture = Fixture::new().await; + fixture.map("/file").await; + let mut wrong = fixture.raw.clone(); + wrong[0] ^= 1; + let mut grown = fixture.raw.clone(); + grown.push(0); + let cases = [ + ( + bounded_objects::FaultKind::Parts(vec![Bytes::from(wrong)]), + 0, + ), + ( + bounded_objects::FaultKind::Parts(vec![Bytes::from(grown)]), + 0, + ), + ( + bounded_objects::FaultKind::Parts(vec![Bytes::copy_from_slice( + &fixture.raw[..fixture.raw.len() - 1], + )]), + CHUNK_SIZE as usize, + ), + ( + bounded_objects::FaultKind::LateError(Bytes::copy_from_slice(&fixture.raw)), + CHUNK_SIZE as usize, + ), + ]; + for (kind, delivered) in cases { + fixture.counts.reset(); + fault(&fixture, kind); + let response_budget = MemoryBudget::new(2 * CHUNK_SIZE as usize); + let scratch_budget = MemoryBudget::new(RANGE_WORK_BYTES); + let response = budgeted_app(&fixture, &response_budget, &scratch_budget) + .oneshot(fixture.request("GET", "blob?path=/file", Body::empty())) + .await + .unwrap(); + assert_eq!(response.status(), 200); + let mut stream = response.into_body().into_data_stream(); + let mut bytes = Vec::new(); + loop { + match stream.next().await { + Some(Ok(part)) => bytes.extend_from_slice(&part), + Some(Err(_)) => break, + None => panic!("corrupt current raw source must fail the actual response body"), + } + } + assert_eq!(bytes.len(), delivered); + assert_eq!(bytes, fixture.raw[..delivered]); + assert!(stream.next().await.is_none()); + assert_eq!(response_budget.used(), 0); + assert_eq!(scratch_budget.used(), 0); + assert_eq!(fixture.counts.whole.load(Ordering::SeqCst), 1); + assert_eq!(fixture.counts.range.load(Ordering::SeqCst), 0); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 0); + } + *fixture.counts.object_fault.lock().unwrap() = None; + let response = fixture.send("GET", "blob?path=/file", Body::empty()).await; + assert_eq!( + to_bytes(response.into_body(), 2 * CHUNK_SIZE as usize) + .await + .unwrap() + .as_ref(), + fixture.raw + ); +} + +#[tokio::test] +async fn raw_drop_cancel_and_revocation_during_held_io_drop_producer_and_all_owned_credit() { + for mode in 0..3 { + let fixture = Fixture::new().await; + fixture.map("/file").await; + fixture.counts.reset(); + let entered = Arc::new(Notify::new()); + let release = Arc::new(Notify::new()); + let drops = Arc::new(AtomicUsize::new(0)); + fault( + &fixture, + bounded_objects::FaultKind::Held { + raw: Bytes::copy_from_slice(&fixture.raw), + entered: entered.clone(), + release: release.clone(), + drops: drops.clone(), + }, + ); + let response_budget = MemoryBudget::new(2 * CHUNK_SIZE as usize); + let scratch_budget = MemoryBudget::new(RANGE_WORK_BYTES); + let response = budgeted_app(&fixture, &response_budget, &scratch_budget) + .oneshot(fixture.request("GET", "blob?path=/file", Body::empty())) + .await + .unwrap(); + assert_eq!(response.status(), 200); + fixture.counts.assert(1, 0); + if mode == 0 { + drop(response); + } else { + let mut stream = response.into_body().into_data_stream(); + let task = tokio::spawn(async move { + let first = stream.next().await; + (first, stream) + }); + timeout(Duration::from_secs(10), entered.notified()) + .await + .unwrap(); + fixture.counts.assert(1, 1); + if mode == 1 { + task.abort(); + assert!(task.await.err().unwrap().is_cancelled()); + } else { + release_lease(&fixture).await; + release.notify_one(); + let (first, mut stream) = task.await.unwrap(); + assert!(first.unwrap().is_err()); + assert!(stream.next().await.is_none()); + } + } + assert_eq!(drops.load(Ordering::SeqCst), 1); + assert_eq!(response_budget.used(), 0); + assert_eq!(scratch_budget.used(), 0); + } +} + +#[tokio::test] +async fn raw_first_poll_rechecks_released_lease_before_consuming_body_bytes() { + let fixture = Fixture::new().await; + fixture.map("/file").await; + fixture.counts.reset(); + let response_budget = MemoryBudget::new(2 * CHUNK_SIZE as usize); + let scratch_budget = MemoryBudget::new(RANGE_WORK_BYTES); + let response = budgeted_app(&fixture, &response_budget, &scratch_budget) + .oneshot(fixture.request("GET", "blob?path=/file", Body::empty())) + .await + .unwrap(); + assert_eq!(response.status(), 200); + release_lease(&fixture).await; + let mut stream = response.into_body().into_data_stream(); + assert!(stream.next().await.unwrap().is_err()); + assert!(stream.next().await.is_none()); + fixture.counts.assert(1, 0); + assert_eq!(response_budget.used(), 0); + assert_eq!(scratch_budget.used(), 0); +} + +#[tokio::test] +async fn raw_post_await_revocation_rejects_fragments_and_eof_before_another_backend_poll() { + for case in 0..4 { + let fixture = Fixture::new().await; + fixture.map("/file").await; + fixture.counts.reset(); + let entered = Arc::new(Notify::new()); + let release = Arc::new(Notify::new()); + let tail_polls = Arc::new(AtomicUsize::new(0)); + let drops = Arc::new(AtomicUsize::new(0)); + let prefix = if case < 2 { + Bytes::copy_from_slice(&fixture.raw[..1]) + } else { + Bytes::copy_from_slice(&fixture.raw) + }; + let prefix_length = prefix.len(); + let fragment = match case { + 0 => Some(Bytes::copy_from_slice(&fixture.raw[1..2])), + 1 | 2 => Some(Bytes::new()), + _ => None, + }; + fault( + &fixture, + bounded_objects::FaultKind::HeldFragment { + prefix, + fragment, + entered: entered.clone(), + release: release.clone(), + tail_polls: tail_polls.clone(), + drops: drops.clone(), + }, + ); + let response_budget = MemoryBudget::new(2 * CHUNK_SIZE as usize); + let scratch_budget = MemoryBudget::new(RANGE_WORK_BYTES); + let response = budgeted_app(&fixture, &response_budget, &scratch_budget) + .oneshot(fixture.request("GET", "blob?path=/file", Body::empty())) + .await + .unwrap(); + assert_eq!(response.status(), 200); + let mut stream = response.into_body().into_data_stream(); + let mut task = tokio::spawn(async move { + let mut delivered = Vec::new(); + loop { + match stream.next().await { + Some(Ok(bytes)) => delivered.extend_from_slice(&bytes), + Some(Err(error)) => return (delivered, error, stream), + None => panic!("revoked source must fail the actual body"), + } + } + }); + timeout(Duration::from_secs(10), entered.notified()) + .await + .unwrap(); + fixture.counts.assert(1, prefix_length); + release_lease(&fixture).await; + release.notify_one(); + let finished = timeout(Duration::from_secs(10), &mut task).await; + if finished.is_err() { + task.abort(); + let _ = task.await; + panic!("revoked fragmented source continued into another held backend poll"); + } + let (delivered, _, mut stream) = finished.unwrap().unwrap(); + assert_eq!( + delivered, + fixture.raw[..if case < 2 { 0 } else { CHUNK_SIZE as usize }] + ); + assert!(stream.next().await.is_none()); + assert_eq!(tail_polls.load(Ordering::SeqCst), 0); + assert_eq!(drops.load(Ordering::SeqCst), 1); + assert_eq!(response_budget.used(), 0); + assert_eq!(scratch_budget.used(), 0); + } +} + +#[tokio::test] +async fn empty_raw_post_await_revocation_fails_before_another_eof_poll_or_empty_success() { + for fragment in [Some(Bytes::new()), None] { + let fixture = Fixture::new().await; + let oid = path_oid(&fixture, "/empty").await; + fixture.counts.reset(); + let entered = Arc::new(Notify::new()); + let release = Arc::new(Notify::new()); + let tail_polls = Arc::new(AtomicUsize::new(0)); + let drops = Arc::new(AtomicUsize::new(0)); + *fixture.counts.object_fault.lock().unwrap() = Some(bounded_objects::StreamFault { + oid, + kind: bounded_objects::FaultKind::HeldFragment { + prefix: Bytes::new(), + fragment, + entered: entered.clone(), + release: release.clone(), + tail_polls: tail_polls.clone(), + drops: drops.clone(), + }, + }); + let response_budget = MemoryBudget::new(CHUNK_SIZE as usize); + let scratch_budget = MemoryBudget::new(RANGE_WORK_BYTES); + let app = budgeted_app(&fixture, &response_budget, &scratch_budget); + let request = fixture.request("GET", "blob?path=/empty", Body::empty()); + let mut task = tokio::spawn(async move { app.oneshot(request).await.unwrap() }); + timeout(Duration::from_secs(10), entered.notified()) + .await + .unwrap(); + release_lease(&fixture).await; + release.notify_one(); + let finished = timeout(Duration::from_secs(10), &mut task).await; + if finished.is_err() { + task.abort(); + let _ = task.await; + panic!("revoked empty source continued into another held backend poll"); + } + error(finished.unwrap().unwrap(), 410, "LEASE_EXPIRED", false).await; + assert_eq!(tail_polls.load(Ordering::SeqCst), 0); + assert_eq!(drops.load(Ordering::SeqCst), 1); + assert_eq!(response_budget.used(), 0); + assert_eq!(scratch_budget.used(), 0); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 0); + } +} + +#[tokio::test] +async fn raw_physical_size_and_current_fact_fail_before_body_and_path_errors_keep_formal_statuses() +{ + let fixture = Fixture::new().await; + fixture.map("/file").await; + fixture.counts.reset(); + for (path, status, code) in [ + ("/absent", 404, "PATH_NOT_FOUND"), + ("/outside", 404, "PATH_NOT_FOUND"), + ("/directory", 409, "NOT_DIRECTORY"), + ("/file/child", 409, "NOT_DIRECTORY"), + ("/link/child", 409, "SYMLINK_TRAVERSAL"), + ("/../outside", 400, "SCOPE_INVALID"), + ] { + error( + fixture + .send("GET", &format!("blob?path={path}"), Body::empty()) + .await, + status, + code, + false, + ) + .await; + fixture.counts.assert(0, 0); + } + let mut request = fixture.request("GET", "blob?path=/file", Body::empty()); + request + .headers_mut() + .insert("range", "bytes=0-9".parse().unwrap()); + error( + fixture.app.clone().oneshot(request).await.unwrap(), + 400, + "RANGE_NOT_SUPPORTED", + false, + ) + .await; + error( + fixture + .send( + "GET", + &format!("blob?path=/file&expected_digest=sha256:{}", "0".repeat(64)), + Body::empty(), + ) + .await, + 409, + "EXPECTED_DIGEST_MISMATCH", + false, + ) + .await; + fixture.counts.assert(0, 0); + *fixture.counts.object_size_override.lock().unwrap() = Some(fixture.raw.len() as i64 + 1); + error( + fixture.send("GET", "blob?path=/file", Body::empty()).await, + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + fixture.counts.assert(1, 0); + *fixture.counts.object_size_override.lock().unwrap() = None; + let original = fixture.fact().await; + let mut changed = original.clone(); + changed.created_at += chrono::Duration::seconds(1); + fixture.replace_fact(changed).await; + fixture.counts.reset(); + error( + fixture.send("GET", "blob?path=/file", Body::empty()).await, + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + fixture.counts.assert(0, 0); + fixture.replace_fact(original).await; + for (path, kind, bytes) in [ + ("/empty", "regular", &b""[..]), + ("/link", "symlink", &b"file"[..]), + ("/executable", "executable", fixture.raw.as_slice()), + ] { + let response = fixture + .send("GET", &format!("blob?path={path}"), Body::empty()) + .await; + assert_eq!(response.status(), 200); + assert_eq!(response.headers()["x-mega-fs-kind"], kind); + assert_eq!( + response.headers()["content-length"], + bytes.len().to_string() + ); + assert_eq!( + to_bytes(response.into_body(), 2 * CHUNK_SIZE as usize) + .await + .unwrap() + .as_ref(), + bytes + ); + } +} + +#[tokio::test] +async fn raw_visible_producer_item_cap_rejects_before_copy_hash_or_tail_poll() { + let raw = vec![7; 8 * 1024 * 1024 + 1]; + let fixture = Fixture::new_with_pg_config_directories_and_objects( + false, + 0, + &[("large".into(), raw.clone())], + ) + .await; + let oid = path_oid(&fixture, "/large").await; + *fixture.counts.object_fault.lock().unwrap() = Some(bounded_objects::StreamFault { + oid: oid.clone(), + kind: bounded_objects::FaultKind::Parts( + raw.chunks(CHUNK_SIZE as usize) + .map(Bytes::copy_from_slice) + .collect(), + ), + }); + fixture.map("/large").await; + let tails = Arc::new(AtomicUsize::new(0)); + *fixture.counts.object_fault.lock().unwrap() = Some(bounded_objects::StreamFault { + oid, + kind: bounded_objects::FaultKind::Oversized(Bytes::from(raw), tails.clone()), + }); + fixture.counts.reset(); + let response_budget = MemoryBudget::new(2 * CHUNK_SIZE as usize); + let scratch_budget = MemoryBudget::new(RANGE_WORK_BYTES); + let response = budgeted_app(&fixture, &response_budget, &scratch_budget) + .oneshot(fixture.request("GET", "blob?path=/large", Body::empty())) + .await + .unwrap(); + assert_eq!(response.status(), 200); + let mut stream = response.into_body().into_data_stream(); + assert!(stream.next().await.unwrap().is_err()); + assert!(stream.next().await.is_none()); + assert_eq!(tails.load(Ordering::SeqCst), 0); + assert_eq!(response_budget.used(), 0); + assert_eq!(scratch_budget.used(), 0); + assert_eq!(fixture.counts.whole.load(Ordering::SeqCst), 1); + assert_eq!(fixture.counts.range.load(Ordering::SeqCst), 0); + assert_eq!( + fixture.counts.bytes.load(Ordering::SeqCst), + 8 * 1024 * 1024 + 1 + ); +} + +#[tokio::test] +async fn empty_raw_source_requires_physical_zero_exact_eof_and_current_empty_digest() { + let fixture = Fixture::new().await; + let oid = path_oid(&fixture, "/empty").await; + for (kind, status, code) in [ + ( + bounded_objects::FaultKind::Parts(vec![Bytes::from_static(b"growth")]), + 502, + "INTEGRITY_ERROR", + ), + ( + bounded_objects::FaultKind::LateError(Bytes::new()), + 503, + "OBJECT_UNAVAILABLE", + ), + ] { + *fixture.counts.object_fault.lock().unwrap() = Some(bounded_objects::StreamFault { + oid: oid.clone(), + kind, + }); + fixture.counts.reset(); + error( + fixture.send("GET", "blob?path=/empty", Body::empty()).await, + status, + code, + false, + ) + .await; + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 0); + } + *fixture.counts.object_fault.lock().unwrap() = None; + fixture.counts.reset(); + let mono = fixture.state.storage.mono_storage(); + let original = mono + .get_verified_blobs(vec![oid.clone()]) + .await + .unwrap() + .remove(&oid) + .unwrap(); + let mut changed = original.clone().into_active_model(); + changed.raw_sha256 = Set(vec![9; 32]); + changed.update(mono.get_connection()).await.unwrap(); + error( + fixture.send("GET", "blob?path=/empty", Body::empty()).await, + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + fixture.counts.assert(0, 0); + original + .into_active_model() + .reset_all() + .update(mono.get_connection()) + .await + .unwrap(); + let response = fixture.send("GET", "blob?path=/empty", Body::empty()).await; + assert_eq!(response.status(), 200); + assert_eq!(response.headers()["content-length"], "0"); + assert!(to_bytes(response.into_body(), 1).await.unwrap().is_empty()); + fixture.counts.assert(1, 0); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 0); +} diff --git a/src/api/router/snapshot_router.rs b/src/api/router/snapshot_router.rs index f75b7279..a6838aea 100644 --- a/src/api/router/snapshot_router.rs +++ b/src/api/router/snapshot_router.rs @@ -28,8 +28,8 @@ use crate::{ descriptor::build as build_descriptor, error::{SnapshotError, SnapshotErrorCode}, pages::{ - MetadataWalkOutcome, WalkOutcome, base64_of, build_directory_page, - build_directory_page_with_work, hex_of, proof_pages, resolve_abs, resolve_abs_metadata, + MetadataWalkOutcome, base64_of, build_directory_page, build_directory_page_with_work, + hex_of, proof_pages, resolve_abs_metadata, }, projection_observation::{NativeResolveSource, ResolvedProjection}, runtime::{now_unix, runtime}, @@ -47,7 +47,7 @@ pub fn routers(api_state: MonoApiServiceState) -> Router { .route("/snapshots/{snapshot_id}/directory", get(directory)) .route( "/snapshots/{snapshot_id}/blob", - get(blob).head(content::blob_head), + get(raw_blob::blob).head(content::blob_head), ) .route("/snapshots/{snapshot_id}/lookup", post(lookup)) .route( @@ -79,6 +79,9 @@ pub fn routers(api_state: MonoApiServiceState) -> Router { #[path = "snapshot_content.rs"] mod content; +#[path = "snapshot_raw_blob.rs"] +mod raw_blob; + #[path = "snapshot_request.rs"] mod request; @@ -1165,127 +1168,6 @@ async fn directory( Ok(resp) } -#[derive(Deserialize, Debug)] -struct BlobQuery { - path: String, - #[serde(default)] - expected_digest: Option, -} - -#[allow(clippy::result_large_err, clippy::too_many_lines)] -async fn blob( - state: State, - AxumPath(snapshot_id): AxumPath, - Query(q): Query, - headers: HeaderMap, -) -> Result { - ensure_enabled(&state).map_err(mst2_error_response)?; - let ctx = request_context(&state, &snapshot_id).map_err(mst2_error_response)?; - validate_scope_relative_path(&q.path).map_err(mst2_error_response)?; - if headers.contains_key("range") { - // Spec 04 section 9: raw blob has no Range semantics this profile. - return Err(mst2_error_response(SnapshotError::new( - SnapshotErrorCode::RangeNotSupported, - "raw blob reads are whole-file; use chunks for ranges", - ))); - } - - let handler = state - .api_handler(std::path::Path::new("/")) - .await - .map_err(internal)?; - let root_tree = handler - .get_tree_by_hash(&ctx.root_tree_oid) - .await - .map_err(internal)?; - let abs_path = abs_view_path(&ctx.built.descriptor.scope, &q.path); - - match resolve_abs(handler.as_ref(), &root_tree, &abs_path) - .await - .map_err(mst2_error_response)? - { - WalkOutcome::FoundFile { - fs_kind, - raw, - digest, - .. - } => { - if let Some(expected) = &q.expected_digest - && expected != &format!("sha256:{}", hex_of(&digest)) - { - return Err(mst2_error_response(SnapshotError::new( - SnapshotErrorCode::DigestMismatch, - "content does not match expected_digest", - ))); - } - let fs_kind_str = match fs_kind { - crate::ceres::snapshot::resolver::FsKind::Regular => "regular", - crate::ceres::snapshot::resolver::FsKind::Executable => "executable", - crate::ceres::snapshot::resolver::FsKind::Symlink => "symlink", - crate::ceres::snapshot::resolver::FsKind::Directory => "directory", - }; - revalidate_request(&state, &ctx) - .await - .map_err(mst2_error_response)?; - let headers = REQUEST_HEADERS.try_with(Clone::clone).map_err(|_| { - mst2_error_response(SnapshotError::new( - SnapshotErrorCode::Unauthenticated, - "request authentication context missing", - )) - })?; - let stream_state = state.0.clone(); - let stream_context = ctx.clone(); - let blocks = futures::stream::unfold( - ( - Bytes::from(raw), - 0usize, - stream_state, - stream_context, - headers, - ), - |(raw, offset, state, context, headers)| async move { - if offset >= raw.len() { - return None; - } - if let Err(error) = revalidate_access(&state, &context, &headers).await { - return Some((Err(error), (raw, usize::MAX, state, context, headers))); - } - let end = (offset + 1_048_576).min(raw.len()); - let bytes = raw.slice(offset..end); - Some((Ok(bytes), (raw, end, state, context, headers))) - }, - ); - Response::builder() - .header("etag", format!("\"sha256:{}\"", hex_of(&digest))) - .header("cache-control", "private, no-cache, no-transform") - .header("x-mega-fs-kind", fs_kind_str) - .body(axum::body::Body::from_stream(blocks)) - .map_err(|e| { - mst2_error_response(SnapshotError::new( - SnapshotErrorCode::Internal, - format!("body build failed: {e}"), - )) - }) - } - WalkOutcome::FoundDir => Err(mst2_error_response(SnapshotError::new( - SnapshotErrorCode::NotDirectory, - "path is a directory", - ))), - WalkOutcome::Absent => Err(mst2_error_response(SnapshotError::new( - SnapshotErrorCode::PathNotFound, - "path absent in the fixed view", - ))), - WalkOutcome::NotDirectory { symlink } => Err(mst2_error_response(SnapshotError::new( - if symlink { - SnapshotErrorCode::SymlinkTraversal - } else { - SnapshotErrorCode::NotDirectory - }, - "intermediate component is not a directory", - ))), - } -} - /// Scope-relative request path -> absolute view path (spec 04 section 1). fn abs_view_path(scope: &str, path: &str) -> String { if path == "/" { diff --git a/src/api/router/snapshot_session_tests.rs b/src/api/router/snapshot_session_tests.rs index 822890d0..ae84d3c6 100644 --- a/src/api/router/snapshot_session_tests.rs +++ b/src/api/router/snapshot_session_tests.rs @@ -300,7 +300,8 @@ async fn mst2_durable_http_resolve_installs_complete_dag_and_warm_leases_share_i .as_ref(), fixture.raw.as_slice() ); - fixture.counts.assert(1, fixture.raw.len()); + fixture.counts.assert(2, 2 * fixture.raw.len()); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 1); } #[tokio::test] diff --git a/src/ceres/api_service/mod.rs b/src/ceres/api_service/mod.rs index 29ec5d74..7acbd06a 100644 --- a/src/ceres/api_service/mod.rs +++ b/src/ceres/api_service/mod.rs @@ -191,6 +191,26 @@ pub trait ApiHandler: Send + Sync { } } + /// Raw whole-file stream with the physical object's complete size. + async fn get_raw_blob_stream_with_meta( + &self, + hash: &str, + ) -> Result< + ( + crate::orbit_api::object_storage::ObjectByteStream, + crate::orbit_api::object_storage::ObjectMeta, + ), + MegaError, + > { + let storage = self.get_context(); + match storage.git_service.get_object_stream_with_meta(hash).await { + Ok(value) => Ok(value), + Err(error) => Err(storage + .classify_blob_objstorage_not_found(hash, error) + .await), + } + } + async fn get_raw_blob_range_stream_exact( &self, hash: &str, diff --git a/src/ceres/snapshot/content_budget.rs b/src/ceres/snapshot/content_budget.rs index 79db0289..6bb8d32b 100644 --- a/src/ceres/snapshot/content_budget.rs +++ b/src/ceres/snapshot/content_budget.rs @@ -81,17 +81,21 @@ pub(crate) fn projection_budget() -> &'static Arc { } pub(crate) fn reserve_response(bytes: usize) -> Result { + response_budget().reserve(bytes) +} + +pub(crate) fn response_budget() -> &'static Arc { static BUDGET: OnceLock> = OnceLock::new(); - BUDGET - .get_or_init(|| MemoryBudget::new(RESPONSE_LIVE_BYTES)) - .reserve(bytes) + BUDGET.get_or_init(|| MemoryBudget::new(RESPONSE_LIVE_BYTES)) } pub(crate) fn reserve_range_work() -> Result { + range_budget().reserve(RANGE_WORK_BYTES) +} + +pub(crate) fn range_budget() -> &'static Arc { static BUDGET: OnceLock> = OnceLock::new(); - BUDGET - .get_or_init(|| MemoryBudget::new(RANGE_SCRATCH_BYTES)) - .reserve(RANGE_WORK_BYTES) + BUDGET.get_or_init(|| MemoryBudget::new(RANGE_SCRATCH_BYTES)) } /// Every encoded frame owns shared credit, including after the body passes a diff --git a/src/jupiter/service/git_service.rs b/src/jupiter/service/git_service.rs index 6569e9a0..8652cc97 100644 --- a/src/jupiter/service/git_service.rs +++ b/src/jupiter/service/git_service.rs @@ -108,6 +108,13 @@ impl GitService { } pub async fn get_object_stream(&self, hash: &str) -> Result { + Ok(self.get_object_stream_with_meta(hash).await?.0) + } + + pub async fn get_object_stream_with_meta( + &self, + hash: &str, + ) -> Result<(ObjectByteStream, ObjectMeta), MegaError> { if !is_full_hex_object_id(hash) { return Err(MegaError::Other("Invalid object ID format".to_string())); } @@ -115,8 +122,7 @@ impl GitService { namespace: ObjectNamespace::Git, key: hash.to_string(), }; - let (stream, _) = self.obj_storage.inner.get_stream(&key).await?; - Ok(stream) + Ok(self.obj_storage.inner.get_stream(&key).await?) } pub async fn get_object_range_stream_exact( From d465e33d572206b365f6a710aea5b291a9480f47 Mon Sep 17 00:00:00 2001 From: Xiaoyang Han Date: Thu, 8 Oct 2026 05:29:41 +0800 Subject: [PATCH 30/50] fix(mst2): place chunk-map SQL tests after production items (#80) Move the unchanged cfg(test) SQL qualification regression module to the end of native_chunk_map.rs. This fixes the exact #78 all-target/all-feature Clippy items_after_test_module failure without a lint allow or any production or test assertion change. The raw streaming implementation merged in #79 is retained. Nightly formatting, diff and locked offline metadata checks pass; corrected native build/tests remain pending. --- src/jupiter/storage/native_chunk_map.rs | 55 +++++++++++++------------ 1 file changed, 28 insertions(+), 27 deletions(-) diff --git a/src/jupiter/storage/native_chunk_map.rs b/src/jupiter/storage/native_chunk_map.rs index 7dcbaa22..5ce849a8 100644 --- a/src/jupiter/storage/native_chunk_map.rs +++ b/src/jupiter/storage/native_chunk_map.rs @@ -605,33 +605,6 @@ fn qualify_relations(schema: &str, sql: &str) -> String { qualified } -#[cfg(test)] -mod sql_tests { - use super::*; - - #[test] - fn authority_relations_are_qualified_without_rewriting_catalog_literals() { - let sql = "SELECT 'mst2_metadata_storage_scope','quoted ''mst2_chunk_map''',\"mst2_chunk_map\" FROM mst2_metadata_storage_scope s JOIN mst2_verified_object f ON s.singleton=1 JOIN mst2_chunk_map m ON true JOIN mst2_chunk_map_leaf l ON true JOIN mst2_chunk_map_node n ON true JOIN mst2_chunk_map_source r ON true WHERE c.relname='mst2_metadata_storage_scope'"; - let qualified = qualify_relations("schema\"name", sql); - assert!(qualified.contains(&format!( - "FROM {}.mst2_metadata_storage_scope", - quoted("schema\"name") - ))); - for name in [ - "mst2_verified_object", - "mst2_chunk_map", - "mst2_chunk_map_leaf", - "mst2_chunk_map_node", - "mst2_chunk_map_source", - ] { - assert!(qualified.contains(&format!("JOIN {}.{name}", quoted("schema\"name")))); - } - assert!(qualified.starts_with( - "SELECT 'mst2_metadata_storage_scope','quoted ''mst2_chunk_map''',\"mst2_chunk_map\"" - )); - assert!(qualified.ends_with("WHERE c.relname='mst2_metadata_storage_scope'")); - } -} fn bounded_bytes(row: &QueryResult, name: &str) -> Result, SnapshotError> { row.try_get::>>("", name) .map_err(db_error)? @@ -682,3 +655,31 @@ impl super::Storage { .await } } + +#[cfg(test)] +mod sql_tests { + use super::*; + + #[test] + fn authority_relations_are_qualified_without_rewriting_catalog_literals() { + let sql = "SELECT 'mst2_metadata_storage_scope','quoted ''mst2_chunk_map''',\"mst2_chunk_map\" FROM mst2_metadata_storage_scope s JOIN mst2_verified_object f ON s.singleton=1 JOIN mst2_chunk_map m ON true JOIN mst2_chunk_map_leaf l ON true JOIN mst2_chunk_map_node n ON true JOIN mst2_chunk_map_source r ON true WHERE c.relname='mst2_metadata_storage_scope'"; + let qualified = qualify_relations("schema\"name", sql); + assert!(qualified.contains(&format!( + "FROM {}.mst2_metadata_storage_scope", + quoted("schema\"name") + ))); + for name in [ + "mst2_verified_object", + "mst2_chunk_map", + "mst2_chunk_map_leaf", + "mst2_chunk_map_node", + "mst2_chunk_map_source", + ] { + assert!(qualified.contains(&format!("JOIN {}.{name}", quoted("schema\"name")))); + } + assert!(qualified.starts_with( + "SELECT 'mst2_metadata_storage_scope','quoted ''mst2_chunk_map''',\"mst2_chunk_map\"" + )); + assert!(qualified.ends_with("WHERE c.relname='mst2_metadata_storage_scope'")); + } +} From cd8ab035e4baa7b470ce1b5dcbb5a8c445e4b301 Mon Sep 17 00:00:00 2001 From: Xiaoyang Han Date: Thu, 8 Oct 2026 05:43:00 +0800 Subject: [PATCH 31/50] fix(mst2): validate raw streams through their exclusive owner (#81) Change RawBlobReader access validation to borrow its exclusive owner. The byte producer is Send and deliberately need not be Sync; retaining a shared reference to the whole reader across an await made the actual Axum body future non-Send. This fixes exact #80 native compilation at snapshot_raw_blob.rs189/235 without adding a lock or weakening any first-poll, post-await, EOF or per-delivery access check. All13HTTP regression assertions remain. Formatting, diff and locked offline metadata pass; corrected native execution remains pending. --- src/api/router/snapshot_raw_blob.rs | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/src/api/router/snapshot_raw_blob.rs b/src/api/router/snapshot_raw_blob.rs index efda9649..6aee3b5c 100644 --- a/src/api/router/snapshot_raw_blob.rs +++ b/src/api/router/snapshot_raw_blob.rs @@ -232,7 +232,7 @@ impl AsRef<[u8]> for RawBlobChunk { } impl RawBlobReader { - async fn validate(&self) -> Result<(), SnapshotError> { + async fn validate(&mut self) -> Result<(), SnapshotError> { revalidate_access(&self.state, &self.context, &self.headers).await } From 5a2b62199d58319afb5c982cf2b6c88c563a8ad1 Mon Sep 17 00:00:00 2001 From: Xiaoyang Han Date: Thu, 8 Oct 2026 06:39:31 +0800 Subject: [PATCH 32/50] perf(v3): serve fixed snapshots from rooted delta metadata (#83) Commit publication builds canonical local directory proofs and reuses unchanged fixed-root metadata. Default qualified-family serving reads bounded DIR windows and selected LOOKUP routes with current source revision facts; warm body callers avoid Git-tree projection. Permanent snapshot routes, exact publication and lease identities, source mutation invalidation, root ownership and bounded background recovery/GC remain enforced. The authority catalog validates exact registered physical families, full qualified shape and source revision triggers, including precise treatment of legitimate qualified foreign-key RI triggers. Direct lease lock contention maps to retryable HTTP 503. Publication-disabled formal MST/2 sessions retain fixed-root HEAD/body/OBJECT/CHUNK service. Migration 000200 and actual PostgreSQL/HTTP regression sources are integrated without raising existing projection limits. Independent review covers all 46 paths, all four corrected findings and every changed assertion site. Format, diff and locked offline metadata pass; native PostgreSQL execution and performance measurement remain separate. --- src/api/router/snapshot_content.rs | 170 ++- src/api/router/snapshot_content_tests.rs | 220 +++- .../snapshot_generation_upgrade_tests.rs | 18 +- .../snapshot_persisted_metadata_tests.rs | 23 +- src/api/router/snapshot_raw_blob.rs | 8 +- src/api/router/snapshot_rooted_metadata.rs | 185 +++ .../router/snapshot_rooted_metadata_tests.rs | 784 ++++++++++++ src/api/router/snapshot_router.rs | 229 +++- src/api/router/snapshot_session_tests.rs | 42 +- .../router/snapshot_storage_route_fixture.rs | 12 +- .../router/snapshot_storage_route_tests.rs | 44 +- src/ceres/snapshot/mod.rs | 2 + src/ceres/snapshot/projection_observation.rs | 94 +- src/ceres/snapshot/rooted_metadata_install.rs | 848 +++++++++++++ .../snapshot/rooted_metadata_projection.rs | 1108 +++++++++++++++++ ...000200_add_mst2_rooted_qualified_family.rs | 125 ++ ...0261008_000200_rooted_qualified_family.sql | 161 +++ .../m20261008_000200_rooted_routes.sql | 171 +++ src/jupiter/migration/mod.rs | 22 +- src/jupiter/storage/init.rs | 20 + src/jupiter/storage/mod.rs | 63 +- .../storage/native_metadata_generations.rs | 6 +- .../storage/native_metadata_install.rs | 5 + .../storage/native_metadata_qualified.rs | 113 +- .../storage/native_snapshot_session.rs | 74 +- .../storage/qualified_family_catalog.sql | 57 + .../storage/qualified_family_shape.sql | 65 + .../storage/qualified_metadata_anchors.sql | 426 +++++++ .../storage/qualified_metadata_canonical.sql | 227 ++++ .../qualified_metadata_canonical_tests.rs | 688 ++++++++++ .../qualified_metadata_certificates.sql | 293 +++++ .../storage/qualified_metadata_family.rs | 624 ++++++++++ .../storage/qualified_metadata_family.sql | 491 ++++++++ .../qualified_metadata_family_tests.rs | 855 +++++++++++++ src/jupiter/storage/qualified_metadata_gc.rs | 540 ++++++++ src/jupiter/storage/qualified_metadata_gc.sql | 574 +++++++++ .../storage/qualified_metadata_gc_tests.rs | 456 +++++++ .../storage/qualified_metadata_reader.rs | 961 ++++++++++++++ .../storage/qualified_metadata_rooted.rs | 766 ++++++++++++ .../storage/qualified_metadata_rooted.sql | 358 ++++++ .../storage/qualified_metadata_serving.sql | 581 +++++++++ .../storage/qualified_metadata_session.rs | 291 +++++ .../qualified_metadata_source_read.sql | 132 ++ ...ualified_metadata_source_revision_tests.rs | 329 +++++ .../storage/qualified_source_revision.sql | 124 ++ src/jupiter/tests.rs | 41 + 46 files changed, 13256 insertions(+), 170 deletions(-) create mode 100644 src/api/router/snapshot_rooted_metadata.rs create mode 100644 src/api/router/snapshot_rooted_metadata_tests.rs create mode 100644 src/ceres/snapshot/rooted_metadata_install.rs create mode 100644 src/ceres/snapshot/rooted_metadata_projection.rs create mode 100644 src/jupiter/migration/m20261008_000200_add_mst2_rooted_qualified_family.rs create mode 100644 src/jupiter/migration/m20261008_000200_rooted_qualified_family.sql create mode 100644 src/jupiter/migration/m20261008_000200_rooted_routes.sql create mode 100644 src/jupiter/storage/qualified_family_catalog.sql create mode 100644 src/jupiter/storage/qualified_family_shape.sql create mode 100644 src/jupiter/storage/qualified_metadata_anchors.sql create mode 100644 src/jupiter/storage/qualified_metadata_canonical.sql create mode 100644 src/jupiter/storage/qualified_metadata_canonical_tests.rs create mode 100644 src/jupiter/storage/qualified_metadata_certificates.sql create mode 100644 src/jupiter/storage/qualified_metadata_family.rs create mode 100644 src/jupiter/storage/qualified_metadata_family.sql create mode 100644 src/jupiter/storage/qualified_metadata_family_tests.rs create mode 100644 src/jupiter/storage/qualified_metadata_gc.rs create mode 100644 src/jupiter/storage/qualified_metadata_gc.sql create mode 100644 src/jupiter/storage/qualified_metadata_gc_tests.rs create mode 100644 src/jupiter/storage/qualified_metadata_reader.rs create mode 100644 src/jupiter/storage/qualified_metadata_rooted.rs create mode 100644 src/jupiter/storage/qualified_metadata_rooted.sql create mode 100644 src/jupiter/storage/qualified_metadata_serving.sql create mode 100644 src/jupiter/storage/qualified_metadata_session.rs create mode 100644 src/jupiter/storage/qualified_metadata_source_read.sql create mode 100644 src/jupiter/storage/qualified_metadata_source_revision_tests.rs create mode 100644 src/jupiter/storage/qualified_source_revision.sql diff --git a/src/api/router/snapshot_content.rs b/src/api/router/snapshot_content.rs index 6a1f191f..275f7dd7 100644 --- a/src/api/router/snapshot_content.rs +++ b/src/api/router/snapshot_content.rs @@ -50,7 +50,7 @@ pub(super) async fn fixed_root_tree( +async fn resolve_legacy_file_metadata( handler: &T, root_tree: &git_internal::internal::object::tree::Tree, scope: &str, @@ -89,6 +89,132 @@ pub(super) async fn resolve_file_metadata( + handler: &T, + ctx: &crate::ceres::snapshot::runtime::SnapshotContext, +) -> Result, Response> { + use crate::jupiter::storage::qualified_metadata_family::SnapshotMetadataFamily; + if !handler.get_context().config().mst2.publication_enabled { + return fixed_root_tree(handler, &ctx.root_tree_oid).await.map(Some); + } + match handler + .get_context() + .snapshot_metadata_family(&ctx.lease_id, true) + .await + .map_err(mst2_error_response)? + { + Some(SnapshotMetadataFamily::Rooted) => Ok(None), + Some(SnapshotMetadataFamily::Generic) => { + fixed_root_tree(handler, &ctx.root_tree_oid).await.map(Some) + } + None => Err(mst2_error_response(SnapshotError::new( + SnapshotErrorCode::SnapshotGone, + "fixed content lease has no permanent storage route", + ))), + } +} + +#[allow(clippy::result_large_err)] +async fn resolve_file_metadata( + handler: &T, + root_tree: Option<&git_internal::internal::object::tree::Tree>, + ctx: &crate::ceres::snapshot::runtime::SnapshotContext, + path: &str, + expected_digest: Option<&str>, +) -> Result { + if let Some(root_tree) = root_tree { + return resolve_legacy_file_metadata( + handler, + root_tree, + &ctx.built.descriptor.scope, + path, + expected_digest, + ) + .await; + } + use crate::jupiter::storage::qualified_metadata_family::RootedLookupStatus; + let storage = handler.get_context(); + let repository = storage + .rooted_qualified_metadata_writer() + .await + .map_err(internal)?; + let (entry, git_oid) = match repository + .fixed_path_metadata(ctx, path) + .await + .map_err(mst2_error_response)? + { + RootedLookupStatus::File { entry, git_oid } => (entry, git_oid), + RootedLookupStatus::Directory(_) => { + return Err(mst2_error_response(SnapshotError::new( + SnapshotErrorCode::NotDirectory, + format!("{path} is a directory"), + ))); + } + RootedLookupStatus::Absent => { + return Err(mst2_error_response(SnapshotError::new( + SnapshotErrorCode::PathNotFound, + format!("{path} absent in the fixed view"), + ))); + } + RootedLookupStatus::NotDirectory { symlink } => { + return Err(mst2_error_response(SnapshotError::new( + if symlink { + SnapshotErrorCode::SymlinkTraversal + } else { + SnapshotErrorCode::NotDirectory + }, + format!("{path}: intermediate component is not a directory"), + ))); + } + }; + let fs_kind = match entry.kind { + mst2_codec::metapage::EntryKind::Regular => FsKind::Regular, + mst2_codec::metapage::EntryKind::Executable => FsKind::Executable, + mst2_codec::metapage::EntryKind::Symlink => FsKind::Symlink, + mst2_codec::metapage::EntryKind::Directory => { + return Err(mst2_error_response(internal( + "fixed source file has a directory kind", + ))); + } + }; + let oid = git_oid + .split_once(':') + .ok_or_else(|| mst2_error_response(internal("fixed source OID has no hash kind")))? + .1 + .to_owned(); + let file = verified_file_metadata(handler, fs_kind, oid, path, expected_digest).await?; + if file.size != entry.size || file.digest != entry.content_id { + return Err(mst2_error_response(SnapshotError::new( + SnapshotErrorCode::IntegrityError, + "fixed content metadata changed from its certified current source occurrence", + ))); + } + Ok(file) +} + +#[allow(clippy::result_large_err)] +pub(super) async fn resolve_snapshot_file_metadata( + state: &crate::api::MonoApiServiceState, + ctx: &crate::ceres::snapshot::runtime::SnapshotContext, + path: &str, + expected_digest: Option<&str>, +) -> Result { + let handler = state + .api_handler(std::path::Path::new("/")) + .await + .map_err(internal)?; + let root_tree = fixed_content_root(handler.as_ref(), ctx).await?; + resolve_file_metadata( + handler.as_ref(), + root_tree.as_ref(), + ctx, + path, + expected_digest, + ) + .await +} + #[allow(clippy::result_large_err)] pub(super) async fn verified_file_metadata( handler: &T, @@ -264,11 +390,11 @@ pub(super) async fn blob_head( .api_handler(std::path::Path::new("/")) .await .map_err(internal)?; - let root_tree = fixed_root_tree(handler.as_ref(), &ctx.root_tree_oid).await?; + let root_tree = fixed_content_root(handler.as_ref(), &ctx).await?; let f = resolve_file_metadata( handler.as_ref(), - &root_tree, - &ctx.built.descriptor.scope, + root_tree.as_ref(), + &ctx, &q.path, q.expected_digest.as_deref(), ) @@ -331,20 +457,19 @@ pub(super) async fn objects( .api_handler(std::path::Path::new("/")) .await .map_err(internal)?; - let root_tree = fixed_root_tree(handler.as_ref(), &ctx.root_tree_oid).await?; - let scope = ctx.built.descriptor.scope.clone(); + let root_tree = fixed_content_root(handler.as_ref(), &ctx).await?; // Copied references: each per-item future borrows the shared walk state // without moving it (the stream is an FnMut over owned items). let handler_ref = handler.as_ref(); - let root_ref = &root_tree; - let scope_ref = &scope; + let root_ref = root_tree.as_ref(); + let ctx_ref = &ctx; let resolved: Vec> = futures::stream::iter(req.items.clone()) .map(move |item| async move { validate_scope_relative_path(&item.path).map_err(mst2_error_response)?; let f = resolve_file_metadata( handler_ref, root_ref, - scope_ref, + ctx_ref, &item.path, Some(&item.expected_digest), ) @@ -468,13 +593,13 @@ pub(super) struct ChunkMapQuery { #[allow(clippy::result_large_err)] async fn project_for( handler: &T, - root_tree: &git_internal::internal::object::tree::Tree, - scope: &str, + root_tree: Option<&git_internal::internal::object::tree::Tree>, + ctx: &crate::ceres::snapshot::runtime::SnapshotContext, path: &str, expected_digest: Option<&str>, ) -> Result, Response> { - let f = resolve_file_metadata(handler, root_tree, scope, path, expected_digest).await?; + let f = resolve_file_metadata(handler, root_tree, ctx, path, expected_digest).await?; project_resolved(handler, &f).await } @@ -552,12 +677,11 @@ pub(super) async fn chunk_map( .api_handler(std::path::Path::new("/")) .await .map_err(internal)?; - let root_tree = fixed_root_tree(handler.as_ref(), &ctx.root_tree_oid).await?; - let scope = ctx.built.descriptor.scope.clone(); + let root_tree = fixed_content_root(handler.as_ref(), &ctx).await?; let proj = project_for( handler.as_ref(), - &root_tree, - &scope, + root_tree.as_ref(), + &ctx, &q.path, q.expected_digest.as_deref(), ) @@ -619,12 +743,11 @@ pub(super) async fn chunk_map_pages( .api_handler(std::path::Path::new("/")) .await .map_err(internal)?; - let root_tree = fixed_root_tree(handler.as_ref(), &ctx.root_tree_oid).await?; - let scope = ctx.built.descriptor.scope.clone(); + let root_tree = fixed_content_root(handler.as_ref(), &ctx).await?; let proj = project_for( handler.as_ref(), - &root_tree, - &scope, + root_tree.as_ref(), + &ctx, &q.path, q.expected_digest.as_deref(), ) @@ -894,8 +1017,7 @@ pub(super) async fn chunks( .api_handler(std::path::Path::new("/")) .await .map_err(internal)?; - let root_tree = fixed_root_tree(handler.as_ref(), &ctx.root_tree_oid).await?; - let scope = ctx.built.descriptor.scope.clone(); + let root_tree = fixed_content_root(handler.as_ref(), &ctx).await?; let mut planned: Vec = Vec::new(); let mut resolved: Vec = Vec::new(); let mut units: Vec<(String, u64)> = Vec::new(); @@ -907,8 +1029,8 @@ pub(super) async fn chunks( parse_decimal_count(&item.chunk_index, "chunk_index").map_err(mst2_error_response)?; let file = resolve_file_metadata( handler.as_ref(), - &root_tree, - &scope, + root_tree.as_ref(), + &ctx, &item.path, Some(&item.expected_digest), ) diff --git a/src/api/router/snapshot_content_tests.rs b/src/api/router/snapshot_content_tests.rs index 617cf2ea..65234d8b 100644 --- a/src/api/router/snapshot_content_tests.rs +++ b/src/api/router/snapshot_content_tests.rs @@ -107,6 +107,8 @@ mod generation_qualified_fixture; mod install_capability_fixture; type ReceiptWriteHold = (Arc, Arc); +#[path = "snapshot_rooted_metadata_tests.rs"] +mod rooted_metadata; #[derive(Default)] struct ReadCounts { @@ -393,6 +395,7 @@ impl LogStorage for CountingStorage { struct Fixture { app: Router, state: MonoApiServiceState, + generic_history: bool, snapshot: String, lease: String, oid: String, @@ -503,12 +506,47 @@ impl Fixture { rebuildable: bool, directory_count: usize, objects: &[(String, Vec)], + ) -> Self { + Self::new_in_metadata_family(rebuildable, directory_count, objects, false).await + } + + async fn new_generic_history_with_pg_config(rebuildable: bool) -> Self { + Self::new_generic_history_with_pg_config_and_directories(rebuildable, 0).await + } + + async fn new_generic_history_with_pg_config_and_directories( + rebuildable: bool, + directory_count: usize, + ) -> Self { + Self::new_in_metadata_family(rebuildable, directory_count, &[], true).await + } + + async fn new_in_metadata_family( + rebuildable: bool, + directory_count: usize, + objects: &[(String, Vec)], + generic_history: bool, + ) -> Self { + Self::new_in_publication_mode(rebuildable, directory_count, objects, generic_history, true) + .await + } + + async fn new_without_publication() -> Self { + Self::new_in_publication_mode(true, 0, &[], false, false).await + } + + async fn new_in_publication_mode( + rebuildable: bool, + directory_count: usize, + objects: &[(String, Vec)], + generic_history: bool, + publication_enabled: bool, ) -> Self { let temp = tempfile::tempdir().unwrap(); let mut config = isolated_config(temp.path().join("config")); config.monorepo.push_policy = PushPolicy::Trunk; config.mst2.enabled = true; - config.mst2.publication_enabled = true; + config.mst2.publication_enabled = publication_enabled; config.mst2.instance_uuid = Some(uuid::Uuid::new_v4().to_string()); config.mst2.auth_token = Some(TOKEN.to_string()); let backend = build_object_storage(&config.object_storage).await.unwrap(); @@ -516,22 +554,37 @@ impl Fixture { let (mut storage, schema) = if rebuildable { let (database, schema) = test_db_config(temp.path()).await; config.database = database; - let connection = crate::jupiter::storage::init::database_connection(&config.database) - .await - .unwrap(); - ( + let assembly = async { + let connection = + crate::jupiter::storage::init::database_connection(&config.database) + .await + .unwrap(); crate::jupiter::storage::Storage::new_with_connection( Arc::new(config), Arc::new(connection), backend.clone(), ) .await - .unwrap(), - Some(schema), - ) + .unwrap() + }; + let storage = if generic_history { + crate::jupiter::storage::init::with_generic_history_bootstrap(assembly).await + } else { + assembly.await + }; + (storage, Some(schema)) } else { (test_storage_with_config(temp.path(), config).await, None) }; + if generic_history { + let q_rows:i64=storage.mono_storage().get_connection().query_one_raw(sea_orm::Statement::from_string( + sea_orm::DbBackend::Postgres,"SELECT count(*) FROM mst2_metadata_namespace WHERE graph_domain='qualified-v1'")) + .await.unwrap().unwrap().try_get_by_index(0).unwrap(); + assert_eq!( + q_rows, 0, + "G history fixtures must not create or erase actual Q ownership" + ); + } storage.git_service = GitService { obj_storage: MegaObjectStorageWrapper::new(Arc::new(CountingStorage { inner: backend, @@ -664,18 +717,24 @@ impl Fixture { ) .await .unwrap(); - mono.initialize_native_publication(storage.config().mst2.instance_uuid.as_deref().unwrap()) - .await - .unwrap(); - publish_native_push(&storage, "/project", old_tip.id, &new_tip).await; - let head = mono - .read_native_publication_head(storage.config().mst2.instance_uuid.as_deref().unwrap()) + if publication_enabled { + mono.initialize_native_publication( + storage.config().mst2.instance_uuid.as_deref().unwrap(), + ) .await .unwrap(); - assert_eq!(head.token.sequence, 1); - assert!(head.token.certificate.is_some()); - assert_eq!(head.root.commit, commit.id.to_string()); - assert_eq!(head.root.tree, root.id.to_string()); + publish_native_push(&storage, "/project", old_tip.id, &new_tip).await; + let head = mono + .read_native_publication_head( + storage.config().mst2.instance_uuid.as_deref().unwrap(), + ) + .await + .unwrap(); + assert_eq!(head.token.sequence, 1); + assert!(head.token.certificate.is_some()); + assert_eq!(head.root.commit, commit.id.to_string()); + assert_eq!(head.root.tree, root.id.to_string()); + } let state = MonoApiServiceState { entity_store: storage.entity_store.clone(), storage, @@ -686,7 +745,12 @@ impl Fixture { }), listen_addr: "127.0.0.1:0".to_string(), }; - let app = Router::new().nest("/api/v2", routers(state.clone()).with_state(state.clone())); + let routes = if generic_history { + crate::api::router::snapshot_router::generic_history_routers(state.clone()) + } else { + routers(state.clone()) + }; + let app = Router::new().nest("/api/v2", routes.with_state(state.clone())); let response = app .clone() .oneshot( @@ -724,6 +788,7 @@ impl Fixture { digest: Sha256::digest(&raw).into(), raw, counts, + generic_history, _temp: temp, _schema: schema, } @@ -915,6 +980,123 @@ async fn mst2_fixed_head_uses_verified_facts_without_body_reads_and_preserves_ra assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 1); } +#[tokio::test] +async fn mst2_publication_disabled_resolve_preserves_head_body_object_map_page_and_chunk_content() { + let fixture = Fixture::new_without_publication().await; + assert!(!fixture.state.storage.config().mst2.publication_enabled); + assert_eq!( + fixture + .state + .storage + .snapshot_metadata_family(&fixture.lease, true) + .await + .unwrap(), + None + ); + let routes: i64 = fixture + .state + .storage + .mono_storage() + .get_connection() + .query_one_raw(sea_orm::Statement::from_string( + sea_orm::DbBackend::Postgres, + "SELECT count(*) FROM mst2_snapshot_storage_route", + )) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap(); + assert_eq!( + routes, 0, + "runtime resolve does not create a permanent SID route" + ); + let head = fixture.send("HEAD", "blob?path=/file", Body::empty()).await; + assert_eq!(head.status(), 200); + assert_eq!( + head.headers()["content-length"], + fixture.raw.len().to_string() + ); + assert_eq!( + head.headers()["etag"], + format!("\"{}\"", fixture.digest_string()) + ); + assert!(to_bytes(head.into_body(), 1024).await.unwrap().is_empty()); + fixture.counts.assert(0, 0); + let body = fixture.send("GET", "blob?path=/file", Body::empty()).await; + assert_eq!(body.status(), 200); + assert_eq!( + to_bytes(body.into_body(), 2 * 1024 * 1024) + .await + .unwrap() + .as_ref(), + fixture.raw + ); + let link_digest = digest(b"file"); + let request=json!({"items":[{"path":"/link","expected_digest":format!("sha256:{}",hex_of(&link_digest))}],"encoding":"identity"}).to_string(); + let objects = fixture + .send("POST", "objects", Body::from(request.clone())) + .await; + assert_eq!(objects.status(), 200); + let wire = to_bytes(objects.into_body(), 2 * 1024 * 1024) + .await + .unwrap(); + let frames = parse_stream(&wire).unwrap(); + let [Frame::Object(object), Frame::End(end)] = frames.as_slice() else { + panic!("expected OBJECT and terminal END"); + }; + assert_eq!(object.objects, vec![(link_digest, b"file".to_vec())]); + assert_eq!(end.request_item_count, 1); + assert_eq!(end.logical_bytes, 4); + assert_eq!( + end.request_body_sha256, + <[u8; 32]>::from(Sha256::digest(request.as_bytes())) + ); + let map = fixture.map("/file").await; + let map_id = map["map"]["map_id"].as_str().unwrap(); + assert_eq!(map["map"]["file_content_id"], fixture.digest_string()); + let page = success_json( + fixture + .send( + "GET", + &format!("chunk-map/pages?path=/file&map_id={map_id}&page_index=0"), + Body::empty(), + ) + .await, + ) + .await; + let leaf = ChunkLeaf::decode( + &STANDARD + .decode(page["leaf_base64"].as_str().unwrap()) + .unwrap(), + ) + .unwrap(); + let hashes: Vec<[u8; 32]> = fixture + .raw + .chunks(CHUNK_SIZE as usize) + .map(|bytes| Sha256::digest(bytes).into()) + .collect(); + assert_eq!(leaf.chunk_sha256, hashes); + let chunks = fixture + .send( + "POST", + "chunks", + Body::from(fixture.chunk_body("/file", map_id, "0").to_string()), + ) + .await; + assert_eq!(chunks.status(), 200); + let wire = to_bytes(chunks.into_body(), 2 * 1024 * 1024).await.unwrap(); + let frames = parse_stream(&wire).unwrap(); + let [Frame::Chunk(chunk), Frame::End(end)] = frames.as_slice() else { + panic!("expected CHUNK and terminal END"); + }; + assert_eq!(chunk.chunk_bytes, &fixture.raw[..CHUNK_SIZE as usize]); + assert_eq!(chunk.file_content_id, fixture.digest); + assert_eq!(chunk.chunk_index, 0); + assert_eq!(end.request_item_count, 1); + assert_eq!(end.logical_bytes, CHUNK_SIZE); +} + #[tokio::test] async fn mst2_fixed_warm_map_and_leaf_aliases_skip_body_reads_and_chunks_use_current_ranges() { let fixture = Fixture::new().await; diff --git a/src/api/router/snapshot_generation_upgrade_tests.rs b/src/api/router/snapshot_generation_upgrade_tests.rs index 08684866..28f4a203 100644 --- a/src/api/router/snapshot_generation_upgrade_tests.rs +++ b/src/api/router/snapshot_generation_upgrade_tests.rs @@ -17,7 +17,7 @@ async fn scalar(db: &sea_orm::DatabaseConnection, sql: &str) -> i64 { #[tokio::test] async fn mst2_generation_additive_upgrade_preserves_legacy_v3_sid_lease_and_new_resolve() { - let fixture = Fixture::new_with_pg_config(true).await; + let fixture = Fixture::new_generic_history_with_pg_config(true).await; let mono = fixture.state.storage.mono_storage(); let db = mono.get_connection(); assert_eq!( @@ -102,10 +102,12 @@ async fn mst2_generation_additive_upgrade_preserves_legacy_v3_sid_lease_and_new_ let connection = crate::jupiter::storage::init::postgres_connection(&config.database) .await .unwrap(); - let storage = crate::jupiter::storage::Storage::new_with_connection( - config, - Arc::new(connection), - fixture.state.storage.git_service.obj_storage.clone(), + let storage = crate::jupiter::storage::init::with_generic_history_bootstrap( + crate::jupiter::storage::Storage::new_with_connection( + config, + Arc::new(connection), + fixture.state.storage.git_service.obj_storage.clone(), + ), ) .await .unwrap(); @@ -113,7 +115,11 @@ async fn mst2_generation_additive_upgrade_preserves_legacy_v3_sid_lease_and_new_ storage, ..fixture.state.clone() }; - let app = Router::new().nest("/api/v2", routers(state.clone()).with_state(state)); + let app = Router::new().nest( + "/api/v2", + crate::api::router::snapshot_router::generic_history_routers(state.clone()) + .with_state(state), + ); let restored = success_json( app.clone() .oneshot(fixture.request("GET", "descriptor", Body::empty())) diff --git a/src/api/router/snapshot_persisted_metadata_tests.rs b/src/api/router/snapshot_persisted_metadata_tests.rs index 057d9d26..ba95602d 100644 --- a/src/api/router/snapshot_persisted_metadata_tests.rs +++ b/src/api/router/snapshot_persisted_metadata_tests.rs @@ -65,7 +65,7 @@ fn assert_metadata(bytes: &[u8], body: &[u8], count: u32, expected: &[([u8; 32], #[tokio::test] async fn mst2_persisted_meta_matches_canonical_routes_after_rebuild_and_advance_without_git_reads() { - let fixture = Fixture::new_with_pg_config_and_directories(true, 140).await; + let fixture = Fixture::new_generic_history_with_pg_config_and_directories(true, 140).await; let context = fixture .state .storage @@ -184,7 +184,7 @@ async fn mst2_persisted_meta_matches_canonical_routes_after_rebuild_and_advance_ #[tokio::test] async fn mst2_persisted_meta_absence_scope_digest_limits_and_release_oracles() { - let fixture = Fixture::new_with_pg_config(true).await; + let fixture = Fixture::new_generic_history_with_pg_config(true).await; for (items, status, code) in [ ( json!([{"directory_path":"/missing"}]), @@ -336,7 +336,7 @@ async fn damage_page(fixture: &Fixture, id: [u8; 32], remove: bool) { #[tokio::test] async fn mst2_persisted_meta_warm_missing_and_corrupt_pages_never_reproject_from_git() { for remove in [true, false] { - let fixture = Fixture::new_with_pg_config(true).await; + let fixture = Fixture::new_generic_history_with_pg_config(true).await; let (id, _) = page_row(&fixture, "/nested").await; damage_page(&fixture, id, remove).await; let body = metadata_body(json!([{"directory_path":"/nested"}]), "identity"); @@ -398,7 +398,7 @@ async fn wait_retention_waiter(txn: &sea_orm::DatabaseTransaction) { #[tokio::test] async fn mst2_persisted_meta_rechecks_deadline_and_release_after_waiting_for_retention_lock() { for expire in [true, false] { - let fixture = Fixture::new_with_pg_config(true).await; + let fixture = Fixture::new_generic_history_with_pg_config(true).await; let mono = fixture.state.storage.mono_storage(); let held = mono.get_connection().begin().await.unwrap(); held.execute_unprepared( @@ -441,7 +441,7 @@ async fn bound_fixture() -> ( Fixture, crate::ceres::snapshot::retention_dag::MetadataPagePayload, ) { - let mut fixture = Fixture::new_with_pg_config(true).await; + let mut fixture = Fixture::new_generic_history_with_pg_config(true).await; advance(&fixture).await; let mono = fixture.state.storage.mono_storage(); let head = mono @@ -546,7 +546,7 @@ async fn mst2_persisted_meta_serves_bound_generic_pages_with_null_members_and_ex #[tokio::test] async fn mst2_persisted_meta_rejects_touched_graph_damage_and_prepare_membership_loss() { for case in 0..5 { - let fixture = Fixture::new_with_pg_config(true).await; + let fixture = Fixture::new_generic_history_with_pg_config(true).await; let (id, _) = page_row(&fixture, "/nested").await; let mono = fixture.state.storage.mono_storage(); let db = mono.get_connection(); @@ -618,7 +618,7 @@ async fn mst2_persisted_meta_reader_holds_protection_until_all_route_bytes_are_o use crate::jupiter::storage::native_snapshot_session::with_metadata_read_barriers; for release_lease in [false, true] { - let fixture = Fixture::new_with_pg_config(true).await; + let fixture = Fixture::new_generic_history_with_pg_config(true).await; let expected = vec![ page_row(&fixture, "/nested").await, page_row(&fixture, "/directory").await, @@ -691,10 +691,11 @@ async fn mst2_persisted_meta_reader_holds_protection_until_all_route_bytes_are_o assert_eq!(observer_config.max_connection, 2); observer_config.max_connection = 1; observer_config.min_connection = 0; - let observer_connection = - crate::jupiter::storage::init::database_connection(&observer_config) - .await - .unwrap(); + let observer_connection = crate::jupiter::storage::init::with_generic_history_bootstrap( + crate::jupiter::storage::init::database_connection(&observer_config), + ) + .await + .unwrap(); let observer = observer_connection.begin().await.unwrap(); tokio::select! { () = wait_retention_waiter(&observer) => {}, diff --git a/src/api/router/snapshot_raw_blob.rs b/src/api/router/snapshot_raw_blob.rs index 6aee3b5c..ea438ec1 100644 --- a/src/api/router/snapshot_raw_blob.rs +++ b/src/api/router/snapshot_raw_blob.rs @@ -91,11 +91,9 @@ pub(super) async fn blob_with_budgets( .api_handler(std::path::Path::new("/")) .await .map_err(internal)?; - let root = content::fixed_root_tree(handler.as_ref(), &context.root_tree_oid).await?; - let file = content::resolve_file_metadata( - handler.as_ref(), - &root, - &context.built.descriptor.scope, + let file = content::resolve_snapshot_file_metadata( + &state, + &context, &query.path, query.expected_digest.as_deref(), ) diff --git a/src/api/router/snapshot_rooted_metadata.rs b/src/api/router/snapshot_rooted_metadata.rs new file mode 100644 index 00000000..a35c7bd1 --- /dev/null +++ b/src/api/router/snapshot_rooted_metadata.rs @@ -0,0 +1,185 @@ +//! JSON metadata endpoints use the same certified, protected fixed-root reader. + +use mst2_codec::metapage::{Entry, EntryKind}; + +use super::*; +use crate::{ + ceres::snapshot::runtime::SnapshotContext, + jupiter::storage::qualified_metadata_family::RootedLookupStatus, +}; + +fn entry_json(entry: &Entry) -> Result { + let name = std::str::from_utf8(&entry.name).map_err(internal)?; + let kind = match entry.kind { + EntryKind::Regular => "regular", + EntryKind::Executable => "executable", + EntryKind::Symlink => "symlink", + EntryKind::Directory => "directory", + }; + let mut value = json!({"name":name,"fs_kind":kind}); + if entry.is_dir() { + value["directory_root"] = json!(format!("sha256:{}", hex_of(&entry.child_root))); + value["node_class"] = json!("native_tree"); + value["lifecycle"] = json!("mutable"); + } else { + value["size"] = json!(entry.size.to_string()); + value["content_digest"] = json!(format!("sha256:{}", hex_of(&entry.content_id))); + } + Ok(value) +} + +fn cursor_name(sid: &str, path: &str, query: &DirectoryQuery) -> Result, Response> { + let Some(cursor) = query.cursor.as_deref() else { + return Ok(None); + }; + let invalid = || { + mst2_error_response(SnapshotError::new( + SnapshotErrorCode::CursorInvalid, + "cursor bound to different parameters or invalid", + )) + }; + let (payload, signature) = cursor.rsplit_once('.').ok_or_else(invalid)?; + if runtime().sign_cursor(payload) != signature { + return Err(invalid()); + } + let bytes = base64::engine::general_purpose::STANDARD + .decode(payload) + .map_err(|_| invalid())?; + let value: serde_json::Value = serde_json::from_slice(&bytes).map_err(|_| invalid())?; + if value["s"].as_str() != Some(sid) + || value["p"].as_str() != Some(path) + || value["l"].as_u64() != Some(query.limit as u64) + { + return Err(invalid()); + } + let name = value["a"].as_str().ok_or_else(invalid)?; + if name.is_empty() || name.len() > 255 || name.contains('/') || name.contains('\0') { + return Err(invalid()); + } + Ok(Some(name.into())) +} + +#[allow(clippy::result_large_err)] +pub(super) async fn directory_response( + state: &MonoApiServiceState, + ctx: &SnapshotContext, + sid: &str, + query: &DirectoryQuery, +) -> Result { + let absolute = abs_view_path(&ctx.built.descriptor.scope, &query.path); + let last = cursor_name(sid, &absolute, query)?; + let repository = state + .storage + .rooted_qualified_metadata_writer() + .await + .map_err(internal)?; + let window = repository + .directory_window(ctx, &query.path, last.as_deref(), query.limit as usize) + .await + .map_err(mst2_error_response)?; + let entries = window + .entries + .iter() + .map(entry_json) + .collect::, _>>()?; + let next = if window.has_more { + let name = window.entries.last().ok_or_else(|| { + mst2_error_response(internal( + "qualified directory returned an empty continuation", + )) + })?; + let payload = json!({"s":sid,"p":absolute,"l":query.limit,"a":std::str::from_utf8(&name.name).map_err(internal)?}); + let encoded = base64_of(&serde_json::to_vec(&payload).map_err(internal)?); + Some(format!("{encoded}.{}", runtime().sign_cursor(&encoded))) + } else { + None + }; + let ancestors = if query.ancestors.as_deref() == Some("chain") { + window + .ancestors + .iter() + .filter(|(path, _)| path != &query.path) + .map(|(path, root)| { + json!({"path":path, + "directory_root":format!("sha256:{}",hex_of(root)),"node_class":"native_tree"}) + }) + .collect::>() + } else { + Vec::new() + }; + revalidate_request(state, ctx) + .await + .map_err(mst2_error_response)?; + let body = json!({"snapshot_id":sid,"path":query.path,"metadata_root":ctx.built.metadata_root, + "directory_root":format!("sha256:{}",hex_of(&window.directory_root)),"node_class":"native_tree","lifecycle":"mutable", + "range_start_exclusive":last,"entries":entries,"entry_count":window.entry_count.to_string(),"next_cursor":next, + "proof_pages":window.proof_pages.iter().map(|(root,bytes)|json!({"digest":format!("sha256:{}",hex_of(root)), + "data_base64":base64_of(bytes)})).collect::>(),"ancestor_chain":ancestors}); + let mut response = Json(body).into_response(); + let tag = format!( + "\"{}:{}:{}:{}\"", + &sid[..16.min(sid.len())], + hex_of(&window.directory_root), + query.limit, + query.cursor.as_deref().unwrap_or("") + ); + if let Ok(value) = HeaderValue::from_str(&tag) { + response.headers_mut().insert("etag", value); + } + response.headers_mut().insert( + "cache-control", + HeaderValue::from_static("private, no-cache, no-transform"), + ); + Ok(response) +} + +#[allow(clippy::result_large_err)] +pub(super) async fn lookup_response( + state: &MonoApiServiceState, + ctx: &SnapshotContext, + sid: &str, + request: &LookupRequest, +) -> Result { + let repository = state + .storage + .rooted_qualified_metadata_writer() + .await + .map_err(internal)?; + let batch = repository + .lookup_metadata(ctx, &request.paths) + .await + .map_err(mst2_error_response)?; + let mut results = Vec::with_capacity(batch.results.len()); + for (path, status) in request.paths.iter().zip(batch.results) { + let mut item = json!({"path":path}); + match status { + RootedLookupStatus::Directory(root) => { + let mut node = json!({"fs_kind":"directory","directory_root":format!("sha256:{}",hex_of(&root)), + "node_class":"native_tree","lifecycle":"mutable"}); + if path != "/" { + node["name"] = json!(path.rsplit('/').next()); + } + item["status"] = json!("found"); + item["node"] = node; + } + RootedLookupStatus::File { entry, .. } => { + item["status"] = json!("found"); + item["node"] = entry_json(&entry)?; + } + RootedLookupStatus::Absent => item["status"] = json!("absent"), + RootedLookupStatus::NotDirectory { symlink } => { + item["status"] = json!(if symlink { + "symlink_traversal" + } else { + "not_directory" + }) + } + } + results.push(item); + } + revalidate_request(state, ctx) + .await + .map_err(mst2_error_response)?; + Ok(Json(json!({"snapshot_id":sid,"results":results,"proof_pages":batch.proof_pages.iter().map(|(root,bytes)|json!({ + "digest":format!("sha256:{}",hex_of(root)),"data_base64":base64_of(bytes)})).collect::>()})).into_response()) +} diff --git a/src/api/router/snapshot_rooted_metadata_tests.rs b/src/api/router/snapshot_rooted_metadata_tests.rs new file mode 100644 index 00000000..7c944913 --- /dev/null +++ b/src/api/router/snapshot_rooted_metadata_tests.rs @@ -0,0 +1,784 @@ +use sea_orm::{DbBackend, Statement, TransactionTrait}; +use tokio::sync::Barrier; + +use super::*; +use crate::jupiter::storage::qualified_metadata_family::{ + SnapshotMetadataFamily, with_rooted_reader_barriers, with_rooted_source_fact_barriers, + with_rooted_source_temporary_shadow, +}; + +async fn q_schema(fixture: &Fixture) -> String { + fixture + .state + .storage + .mono_storage() + .get_connection() + .query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT namespace.metadata_schema FROM mst2_snapshot_storage_route route + JOIN mst2_metadata_namespace namespace USING(namespace_uuid) WHERE route.snapshot_id=$1 + AND namespace.graph_domain='qualified-v1'", + [fixture.snapshot.clone().into()], + )) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap() +} + +async fn q_count(fixture: &Fixture, query: &str) -> i64 { + let schema = q_schema(fixture).await; + let query = query.replace("{q}", &format!("\"{}\"", schema.replace('"', "\"\""))); + fixture + .state + .storage + .mono_storage() + .get_connection() + .query_one_raw(Statement::from_string(DbBackend::Postgres, query)) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap() +} + +async fn resolve(fixture: &Fixture) -> Value { + success_json( + fixture + .app + .clone() + .oneshot( + Request::builder() + .method("POST") + .uri("/api/v2/snapshots/resolve") + .header("authorization", format!("Bearer {TOKEN}")) + .header("content-type", "application/json") + .body(Body::from( + json!({"target":{"kind":"latest"},"scope":"/project"}).to_string(), + )) + .unwrap(), + ) + .await + .unwrap(), + ) + .await +} + +async fn release(fixture: &Fixture, lease: &str) -> Value { + success_json( + fixture + .app + .clone() + .oneshot( + Request::builder() + .method("DELETE") + .uri(format!("/api/v2/snapshots/leases/{lease}")) + .header("authorization", format!("Bearer {TOKEN}")) + .body(Body::empty()) + .unwrap(), + ) + .await + .unwrap(), + ) + .await +} + +async fn collect_unowned(fixture: &Fixture) { + let writer = fixture + .state + .storage + .rooted_qualified_metadata_writer() + .await + .unwrap(); + for _ in 0..8 { + let work = writer.maintenance_tick(64).await.unwrap(); + assert!(work.collector_enabled && work.examined <= 64); + if q_count(fixture, "SELECT count(*) FROM {q}.mst2_metadata_payload").await == 0 { + return; + } + } + panic!("bounded maintenance failed to collect the small unowned fixture"); +} + +#[tokio::test] +async fn default_rooted_http_handoff_renew_release_and_fresh_incarnation_keep_exact_ownership() { + let fixture = Fixture::new_with_pg_config(true).await; + assert_eq!( + fixture + .state + .storage + .snapshot_metadata_family(&fixture.lease, true) + .await + .unwrap(), + Some(SnapshotMetadataFamily::Rooted) + ); + assert_eq!( + q_count( + &fixture, + "SELECT count(*) FROM {q}.mst2_qualified_session_incarnation WHERE state='READY'" + ) + .await, + 1 + ); + let first_generation = q_count( + &fixture, + "SELECT root_generation FROM {q}.mst2_qualified_session_incarnation WHERE state='READY'", + ) + .await; + assert_eq!(q_count(&fixture,"SELECT count(*) FROM {q}.mst2_metadata_root_anchor WHERE anchor_kind IN ('PREPARE','REUSE')").await,0); + assert_eq!(q_count(&fixture,"SELECT count(*) FROM {q}.mst2_metadata_prepare WHERE plan_kind='ROOTED' AND state='COMMITTED' AND coverage_retired_at IS NOT NULL").await,1); + let original_payloads = + q_count(&fixture, "SELECT count(*) FROM {q}.mst2_metadata_payload").await; + let next = resolve(&fixture).await; + assert_eq!(next["descriptor"]["snapshot_id"], fixture.snapshot); + let second = next["lease_id"].as_str().unwrap(); + assert_ne!(second, fixture.lease); + assert_eq!( + q_count(&fixture, "SELECT count(*) FROM {q}.mst2_metadata_prepare").await, + 1 + ); + assert_eq!( + q_count(&fixture, "SELECT count(*) FROM {q}.mst2_metadata_payload").await, + original_payloads + ); + let previous_deadline = fixture + .state + .storage + .snapshot_context(&fixture.snapshot, second) + .await + .unwrap() + .lease_expires_at_unix; + let renewed = fixture + .state + .storage + .snapshot_renew(second, 1) + .await + .unwrap(); + assert_eq!(renewed.snapshot_id, fixture.snapshot); + assert!(renewed.expires_at_unix >= previous_deadline); + assert_eq!(release(&fixture, &fixture.lease).await["released"], true); + assert_eq!(release(&fixture, &fixture.lease).await["released"], false); + assert_eq!(q_count(&fixture,"SELECT count(*) FROM {q}.mst2_qualified_lease_binding WHERE state='RELEASED' AND lease_epoch=2").await,1); + let mut request = fixture.request("GET", "descriptor", Body::empty()); + request + .headers_mut() + .insert("x-mega-snapshot-lease", second.parse().unwrap()); + success_json(fixture.app.clone().oneshot(request).await.unwrap()).await; + assert_eq!(release(&fixture, second).await["released"], true); + assert_eq!( + q_count( + &fixture, + "SELECT count(*) FROM {q}.mst2_metadata_root_anchor" + ) + .await, + 0 + ); + assert_eq!( + q_count( + &fixture, + "SELECT count(*) FROM {q}.mst2_qualified_session_incarnation WHERE state='RETIRED'" + ) + .await, + 1 + ); + collect_unowned(&fixture).await; + assert_eq!( + q_count( + &fixture, + "SELECT count(*) FROM {q}.mst2_metadata_graph_node" + ) + .await, + 0 + ); + assert!( + q_count( + &fixture, + "SELECT count(*) FROM {q}.mst2_metadata_page_certificate" + ) + .await + > 0 + ); + let fresh = resolve(&fixture).await; + assert_eq!(fresh["descriptor"]["snapshot_id"], fixture.snapshot); + assert_ne!(fresh["lease_id"], next["lease_id"]); + assert_eq!( + q_count( + &fixture, + "SELECT count(*) FROM {q}.mst2_qualified_session_incarnation" + ) + .await, + 2 + ); + assert_eq!( + q_count( + &fixture, + "SELECT count(*) FROM {q}.mst2_qualified_session_incarnation WHERE state='READY'" + ) + .await, + 1 + ); + assert_eq!( + q_count(&fixture, "SELECT count(*) FROM {q}.mst2_metadata_payload").await, + original_payloads + ); + assert_eq!( + q_count( + &fixture, + "SELECT root_generation FROM {q}.mst2_qualified_session_incarnation WHERE state='READY'" + ) + .await, + first_generation + 1 + ); + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn rooted_wide_directory_windows_and_lookup_survive_rebuild_with_valid_proofs() { + let fixture = Fixture::new_with_pg_config_and_directories(true, 140).await; + let config = fixture.state.storage.config(); + let connection = crate::jupiter::storage::init::postgres_connection(&config.database) + .await + .unwrap(); + let storage = crate::jupiter::storage::Storage::new_with_connection( + config, + Arc::new(connection), + fixture.state.storage.git_service.obj_storage.clone(), + ) + .await + .unwrap(); + let state = MonoApiServiceState { + storage, + ..fixture.state.clone() + }; + let app = Router::new().nest("/api/v2", routers(state.clone()).with_state(state)); + let mut cursor = None; + let mut names = Vec::new(); + for _ in 0..12 { + let mut suffix = "directory?path=/&limit=17".to_string(); + if let Some(current) = &cursor { + suffix.push_str(&format!("&cursor={current}")); + } + let response = app + .clone() + .oneshot(fixture.request("GET", &suffix, Body::empty())) + .await + .unwrap(); + let window = success_json(response).await; + assert_eq!(window["entry_count"], "147"); + let entries = window["entries"].as_array().unwrap(); + assert!(!entries.is_empty() && entries.len() <= 17); + names.extend( + entries + .iter() + .map(|entry| entry["name"].as_str().unwrap().to_owned()), + ); + for proof in window["proof_pages"].as_array().unwrap() { + let bytes = STANDARD + .decode(proof["data_base64"].as_str().unwrap()) + .unwrap(); + mst2_codec::metapage::Page::decode(&bytes).unwrap(); + assert_eq!( + proof["digest"], + format!("sha256:{}", hex_of(&mst2_codec::metapage::page_id(&bytes))) + ); + } + cursor = window["next_cursor"].as_str().map(str::to_owned); + if cursor.is_none() { + break; + } + } + assert!(cursor.is_none()); + assert_eq!(names.len(), 147); + assert!( + names + .windows(2) + .all(|pair| pair[0].as_bytes() < pair[1].as_bytes()) + ); + let lookup=success_json(app.oneshot(fixture.request("POST","lookup", + Body::from(json!({"paths":["/nested/file","/wide-139/file-139","/missing","/file/child","/link/child"]}).to_string()))) + .await.unwrap()).await; + let statuses: Vec<_> = lookup["results"] + .as_array() + .unwrap() + .iter() + .map(|row| row["status"].as_str().unwrap()) + .collect(); + assert_eq!( + statuses, + [ + "found", + "found", + "absent", + "not_directory", + "symlink_traversal" + ] + ); + assert_eq!( + q_count( + &fixture, + "SELECT count(*) FROM {q}.mst2_metadata_reader_operation WHERE state='ACTIVE'" + ) + .await, + 0 + ); + assert_eq!(q_count(&fixture,"SELECT count(*) FROM {q}.mst2_metadata_root_anchor WHERE anchor_kind IN ('REQUEST','READER')").await,0); + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn rooted_reader_release_race_keeps_independent_roots_until_owned_buffers_finish() { + let fixture = Fixture::new_with_pg_config(true).await; + let admitted = Arc::new(Barrier::new(2)); + let resume = Arc::new(Barrier::new(2)); + let request = fixture.request( + "POST", + "metadata/pages", + Body::from( + json!({"encoding":"identity","items":[{"directory_path":"/nested"}]}).to_string(), + ), + ); + let app = fixture.app.clone(); + let pending = tokio::spawn(with_rooted_reader_barriers( + admitted.clone(), + resume.clone(), + async move { app.oneshot(request).await.unwrap() }, + )); + tokio::time::timeout(Duration::from_secs(4), admitted.wait()) + .await + .unwrap(); + assert_eq!(q_count(&fixture,"SELECT count(*) FROM {q}.mst2_metadata_root_anchor WHERE anchor_kind IN ('REQUEST','READER')").await,2); + let work = fixture + .state + .storage + .rooted_qualified_metadata_writer() + .await + .unwrap() + .maintenance_tick(64) + .await + .unwrap(); + assert_eq!(work.payload_pages_removed, 0); + assert_eq!(release(&fixture, &fixture.lease).await["released"], true); + assert_eq!(q_count(&fixture,"SELECT count(*) FROM {q}.mst2_metadata_root_anchor WHERE anchor_kind IN ('REQUEST','READER')").await,2); + let work = fixture + .state + .storage + .rooted_qualified_metadata_writer() + .await + .unwrap() + .maintenance_tick(64) + .await + .unwrap(); + assert_eq!(work.payload_pages_removed, 0); + assert_eq!( + q_count( + &fixture, + "SELECT count(*) FROM {q}.mst2_metadata_root_anchor WHERE anchor_kind='LEASE'" + ) + .await, + 1 + ); + resume.wait().await; + error( + tokio::time::timeout(Duration::from_secs(4), pending) + .await + .unwrap() + .unwrap(), + 410, + "LEASE_EXPIRED", + false, + ) + .await; + assert_eq!( + q_count( + &fixture, + "SELECT count(*) FROM {q}.mst2_metadata_reader_operation WHERE state='FINISHED'" + ) + .await, + 1 + ); + assert_eq!( + q_count( + &fixture, + "SELECT count(*) FROM {q}.mst2_metadata_root_anchor" + ) + .await, + 0 + ); + collect_unowned(&fixture).await; + fixture.counts.assert(0, 0); +} + +async fn source_oid(fixture: &Fixture) -> String { + let schema = q_schema(fixture).await; + let quoted = format!("\"{}\"", schema.replace('"', "\"\"")); + fixture.state.storage.mono_storage().get_connection().query_one_raw(Statement::from_string(DbBackend::Postgres, + format!("SELECT split_part(a.tagged_tree_oid,':',2) FROM {quoted}.mst2_qualified_session_incarnation s + JOIN {quoted}.mst2_metadata_source_root_attestation a ON a.attestation_id=s.attestation_id WHERE s.state='READY'"))) + .await.unwrap().unwrap().try_get_by_index(0).unwrap() +} + +async fn source_revision(fixture: &Fixture, oid: &str) -> String { + fixture + .state + .storage + .mono_storage() + .get_connection() + .query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT revision::text FROM mst2_rooted_source_tree_revision WHERE tree_id=$1", + [oid.into()], + )) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap() +} + +#[tokio::test] +async fn actual_q_body_callers_preserve_unchanged_source_revision_and_ignore_temp_shadow() { + let fixture = Fixture::new_with_pg_config_and_directories(true, 140).await; + fixture.map("/file").await; + fixture.counts.reset(); + let oid = source_oid(&fixture).await; + let revision = source_revision(&fixture, &oid).await; + fixture + .state + .storage + .mono_storage() + .get_connection() + .execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "UPDATE mega_tree SET sub_trees=sub_trees,pack_offset=pack_offset WHERE tree_id=$1", + [oid.clone().into()], + )) + .await + .unwrap(); + assert_eq!(source_revision(&fixture, &oid).await, revision); + let response = + with_rooted_source_temporary_shadow(fixture.send("HEAD", "blob?path=/file", Body::empty())) + .await; + assert_eq!(response.status(), 200); + assert_eq!( + response.headers()["x-mega-content-size"], + fixture.raw.len().to_string() + ); + fixture.counts.assert(0, 0); + let response = + with_rooted_source_temporary_shadow(fixture.send("GET", "blob?path=/file", Body::empty())) + .await; + assert_eq!(response.status(), 200); + assert_eq!( + to_bytes(response.into_body(), fixture.raw.len() + 1) + .await + .unwrap() + .as_ref(), + fixture.raw + ); + assert_eq!( + q_count( + &fixture, + "SELECT count(*) FROM {q}.mst2_metadata_reader_operation WHERE state='ACTIVE'" + ) + .await, + 0 + ); +} + +#[tokio::test] +async fn actual_q_body_callers_reject_changed_or_deleted_source_before_opening_a_body() { + for delete in [false, true] { + let fixture = Fixture::new_with_pg_config(true).await; + let oid = source_oid(&fixture).await; + let revision = source_revision(&fixture, &oid).await; + fixture + .state + .storage + .mono_storage() + .get_connection() + .execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + if delete { + "DELETE FROM mega_tree WHERE tree_id=$1" + } else { + "UPDATE mega_tree SET sub_trees=decode('00','hex') WHERE tree_id=$1" + }, + [oid.clone().into()], + )) + .await + .unwrap(); + assert_ne!(source_revision(&fixture, &oid).await, revision); + let response = fixture.send("GET", "blob?path=/file", Body::empty()).await; + assert!(response.status().is_client_error() || response.status().is_server_error()); + fixture.counts.assert(0, 0); + } +} + +#[tokio::test] +async fn actual_q_body_callers_recheck_only_returned_current_file_facts() { + let fixture = Fixture::new_with_pg_config(true).await; + fixture.state.storage.mono_storage().get_connection().execute_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "UPDATE mst2_verified_object SET verification_version=1 WHERE storage_domain='git' AND object_kind='blob' AND git_oid=$1", + [fixture.oid.clone().into()])).await.unwrap(); + let untouched = fixture + .send("HEAD", "blob?path=/empty", Body::empty()) + .await; + assert_eq!(untouched.status(), 200); + assert_eq!(untouched.headers()["x-mega-content-size"], "0"); + error( + fixture.send("GET", "blob?path=/file", Body::empty()).await, + 503, + "METADATA_NOT_READY", + false, + ) + .await; + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn actual_q_body_caller_source_mutation_after_reader_admission_cannot_serve_stale_content() { + let fixture = Fixture::new_with_pg_config(true).await; + let oid = source_oid(&fixture).await; + let admitted = Arc::new(Barrier::new(2)); + let resume = Arc::new(Barrier::new(2)); + let app = fixture.app.clone(); + let request = fixture.request("GET", "blob?path=/file", Body::empty()); + let pending = tokio::spawn(with_rooted_reader_barriers( + admitted.clone(), + resume.clone(), + async move { app.oneshot(request).await.unwrap() }, + )); + tokio::time::timeout(Duration::from_secs(4), admitted.wait()) + .await + .unwrap(); + fixture + .state + .storage + .mono_storage() + .get_connection() + .execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "UPDATE mega_tree SET sub_trees=decode('00','hex') WHERE tree_id=$1", + [oid.into()], + )) + .await + .unwrap(); + resume.wait().await; + let response = tokio::time::timeout(Duration::from_secs(4), pending) + .await + .unwrap() + .unwrap(); + assert!(response.status().is_client_error() || response.status().is_server_error()); + assert_eq!(q_count(&fixture,"SELECT count(*) FROM {q}.mst2_metadata_root_anchor WHERE anchor_kind IN ('REQUEST','READER')").await,0); + assert_eq!( + q_count( + &fixture, + "SELECT count(*) FROM {q}.mst2_metadata_reader_operation WHERE state='FINISHED'" + ) + .await, + 1 + ); + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn actual_q_body_caller_retries_a_busy_current_file_fact_without_opening_a_body() { + let fixture = Fixture::new_with_pg_config(true).await; + let writer = fixture + .state + .storage + .mono_storage() + .get_connection() + .begin() + .await + .unwrap(); + writer.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "UPDATE mst2_verified_object SET raw_sha256=raw_sha256 WHERE storage_domain='git' AND object_kind='blob' AND git_oid=$1", + [fixture.oid.clone().into()], + )).await.unwrap(); + error( + fixture.send("GET", "blob?path=/file", Body::empty()).await, + 503, + "TEMPORARY_UNAVAILABLE", + true, + ) + .await; + assert_eq!(q_count(&fixture,"SELECT count(*) FROM {q}.mst2_metadata_root_anchor WHERE anchor_kind IN ('REQUEST','READER')").await,0); + fixture.counts.assert(0, 0); + writer.rollback().await.unwrap(); + assert_eq!( + fixture + .send("HEAD", "blob?path=/file", Body::empty()) + .await + .status(), + 200 + ); + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn actual_q_body_caller_holds_the_selected_current_fact_until_the_read_transaction_finishes() +{ + let fixture = Fixture::new_with_pg_config(true).await; + let admitted = Arc::new(Barrier::new(2)); + let resume = Arc::new(Barrier::new(2)); + let app = fixture.app.clone(); + let request = fixture.request("HEAD", "blob?path=/file", Body::empty()); + let pending = tokio::spawn(with_rooted_source_fact_barriers( + admitted.clone(), + resume.clone(), + async move { app.oneshot(request).await.unwrap() }, + )); + tokio::time::timeout(Duration::from_secs(4), admitted.wait()) + .await + .unwrap(); + let core_writer = fixture + .state + .storage + .mono_storage() + .get_connection() + .clone(); + let oid = fixture.oid.clone(); + let (ready, received) = tokio::sync::oneshot::channel(); + let update = tokio::spawn(async move { + let txn = core_writer.begin().await.unwrap(); + let pid: i32 = txn + .query_one_raw(Statement::from_string( + DbBackend::Postgres, + "SELECT pg_backend_pid()", + )) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap(); + ready.send(pid).unwrap(); + txn.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "UPDATE mst2_verified_object SET raw_sha256=raw_sha256 WHERE storage_domain='git' AND object_kind='blob' AND git_oid=$1", + [oid.into()])).await.unwrap(); + txn.commit().await.unwrap(); + }); + let pid = received.await.unwrap(); + tokio::time::timeout(Duration::from_secs(4),async { + loop { + let waiting:bool=fixture.state.storage.mono_storage().get_connection().query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres,"SELECT coalesce(wait_event_type='Lock',false) FROM pg_stat_activity WHERE pid=$1",[pid.into()])) + .await.unwrap().unwrap().try_get_by_index(0).unwrap(); + if waiting {break;} tokio::task::yield_now().await; + } + }).await.unwrap(); + assert!(!update.is_finished()); + resume.wait().await; + let response = tokio::time::timeout(Duration::from_secs(4), pending) + .await + .unwrap() + .unwrap(); + assert_eq!(response.status(), 200); + tokio::time::timeout(Duration::from_secs(4), update) + .await + .unwrap() + .unwrap(); + assert_eq!(q_count(&fixture,"SELECT count(*) FROM {q}.mst2_metadata_root_anchor WHERE anchor_kind IN ('REQUEST','READER')").await,0); + fixture.counts.assert(0, 0); +} + +async fn durable_lease_state(fixture: &Fixture) -> Value { + let schema = q_schema(fixture).await; + let quoted = format!("\"{}\"", schema.replace('"', "\"\"")); + fixture.state.storage.mono_storage().get_connection().query_one_raw(Statement::from_string(DbBackend::Postgres, + format!("SELECT jsonb_build_object( + 'sessions',(SELECT jsonb_agg(to_jsonb(s) ORDER BY s.snapshot_id,s.session_incarnation) FROM {quoted}.mst2_qualified_session_incarnation s), + 'leases',(SELECT jsonb_agg(to_jsonb(l) ORDER BY l.lease_id) FROM {quoted}.mst2_qualified_lease_binding l), + 'anchors',(SELECT jsonb_agg(to_jsonb(a) ORDER BY a.anchor_id) FROM {quoted}.mst2_metadata_root_anchor a), + 'roots',(SELECT jsonb_agg(to_jsonb(r) ORDER BY r.prepare_id,r.page_id,r.generation) FROM {quoted}.mst2_metadata_graph_root r), + 'routes',(SELECT jsonb_agg(to_jsonb(r) ORDER BY r.lease_id) FROM mst2_lease_storage_route r))"))) + .await.unwrap().unwrap().try_get_by_index(0).unwrap() +} + +#[tokio::test] +async fn actual_q_lease_http_and_direct_handoff_retry_source_lock_without_partial_durable_mutation() +{ + let fixture = Fixture::new_with_pg_config(true).await; + let storage = &fixture.state.storage; + let context = storage + .snapshot_context(&fixture.snapshot, &fixture.lease) + .await + .unwrap(); + let config = storage.config(); + let instance = config.mst2.instance_uuid.as_deref().unwrap(); + let head = storage + .mono_storage() + .read_native_publication_head(instance) + .await + .unwrap(); + let repository = storage.rooted_qualified_metadata_writer().await.unwrap(); + let before = durable_lease_state(&fixture).await; + let revision = source_revision(&fixture, &source_oid(&fixture).await).await; + let writer = storage + .mono_storage() + .get_connection() + .begin() + .await + .unwrap(); + let locked=writer.query_one_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "SELECT revision::text FROM mst2_rooted_source_tree_revision WHERE revision=$1::uuid FOR UPDATE", + [revision.clone().into()])).await.unwrap().unwrap(); + assert_eq!(locked.try_get_by_index::(0).unwrap(), revision); + let direct_errors = tokio::time::timeout(Duration::from_secs(4), async { + [ + repository + .open_session(&head, &context.built, None, 600) + .await + .unwrap_err(), + repository + .renew(&fixture.lease, 600, instance) + .await + .unwrap_err(), + repository.release(&fixture.lease).await.unwrap_err(), + ] + }) + .await + .unwrap(); + for error in direct_errors { + assert_eq!(error.code, SnapshotErrorCode::TemporaryUnavailable); + } + assert_eq!(durable_lease_state(&fixture).await, before); + for (method, uri) in [ + ("POST", "/api/v2/snapshots/resolve".to_owned()), + ( + "POST", + format!("/api/v2/snapshots/leases/{}/renew", fixture.lease), + ), + ( + "DELETE", + format!("/api/v2/snapshots/leases/{}", fixture.lease), + ), + ] { + let body = if uri.ends_with("/resolve") { + Body::from(json!({"target":{"kind":"latest"},"scope":"/project"}).to_string()) + } else { + Body::empty() + }; + let request = Request::builder() + .method(method) + .uri(uri) + .header("authorization", format!("Bearer {TOKEN}")) + .header("content-type", "application/json") + .body(body) + .unwrap(); + let response = + tokio::time::timeout(Duration::from_secs(4), fixture.app.clone().oneshot(request)) + .await + .unwrap() + .unwrap(); + error(response, 503, "TEMPORARY_UNAVAILABLE", true).await; + assert_eq!(durable_lease_state(&fixture).await, before); + } + writer.rollback().await.unwrap(); + assert_eq!(durable_lease_state(&fixture).await, before); + let next = resolve(&fixture).await; + assert_ne!(next["lease_id"], fixture.lease); + assert_eq!(release(&fixture, &fixture.lease).await["released"], true); +} diff --git a/src/api/router/snapshot_router.rs b/src/api/router/snapshot_router.rs index a6838aea..4b568aeb 100644 --- a/src/api/router/snapshot_router.rs +++ b/src/api/router/snapshot_router.rs @@ -76,6 +76,19 @@ pub fn routers(api_state: MonoApiServiceState) -> Router { )) } +/// Explicit fixture for the historical G authority and corruption contracts. +/// Production resolve always installs the rooted family for a new SID. +#[cfg(test)] +pub(crate) fn generic_history_routers( + api_state: MonoApiServiceState, +) -> Router { + routers(api_state).layer(axum::middleware::from_fn( + |request: axum::extract::Request, next: axum::middleware::Next| async move { + GENERIC_HISTORY_RESOLVE.scope(true, next.run(request)).await + }, + )) +} + #[path = "snapshot_content.rs"] mod content; @@ -212,6 +225,7 @@ tokio::task_local! { static NATIVE_RESOLVE_BARRIERS: (std::sync::Arc, std::sync::Arc); static NATIVE_HANDOFF_BARRIERS: (std::sync::Arc, std::sync::Arc); static REJECT_NATIVE_OBSERVATION_SOURCE: bool; + static GENERIC_HISTORY_RESOLVE: bool; } #[cfg(test)] @@ -770,48 +784,141 @@ async fn resolve( // The scope root doubles as metadata_root; building it also validates // that the scope exists and is a directory in this view. let projection_started = std::time::Instant::now(); - let (scope_page, projection_work) = - build_directory_page_with_work(handler.as_ref(), &root_tree, &req.scope) + #[cfg(test)] + let prepared_generic = if selected_native_head.is_some() + && GENERIC_HISTORY_RESOLVE + .try_with(|value| *value) + .unwrap_or(false) + { + let (page, work) = build_directory_page_with_work(handler.as_ref(), &root_tree, &req.scope) .await .map_err(mst2_error_response)?; - let projection_elapsed = projection_started.elapsed(); - - let built = build_descriptor(&config.mst2, &view, &req.scope, scope_page.page_id) + let prepared = crate::ceres::snapshot::pages::prepare_native_metadata_retention( + handler.as_ref(), + &root_tree, + &req.scope, + crate::ceres::snapshot::retention_dag::MetadataDagLimits::default(), + ) + .await .map_err(mst2_error_response)?; - let ctx = if let Some(head) = selected_native_head.as_ref() { - let sessions = state.storage.snapshot_sessions().await; - if let Some(context) = sessions - .open(head, &built, None, req.lease_seconds) + Some((page.page_id, work, prepared)) + } else { + None + }; + #[cfg(not(test))] + let prepared_generic: Option<( + [u8; 32], + crate::ceres::snapshot::pages::ProjectionWork, + crate::ceres::snapshot::pages::PreparedNativeMetadataRetention, + )> = None; + let (metadata_root, projection_work, prepared_rooted) = if let Some((root, work, _)) = + prepared_generic.as_ref() + { + (*root, Some(work.clone()), None) + } else if selected_native_head.is_some() { + let repository = state + .storage + .rooted_qualified_metadata_writer() .await - .map_err(mst2_error_response)? - { - context - } else { - let prepared = crate::ceres::snapshot::pages::prepare_native_metadata_retention( + .map_err(internal)?; + let prepared = + crate::ceres::snapshot::rooted_metadata_projection::prepare_rooted_native_metadata( handler.as_ref(), &root_tree, &req.scope, - crate::ceres::snapshot::retention_dag::MetadataDagLimits::default(), + repository, ) .await .map_err(mst2_error_response)?; - let receipt = sessions - .install(&built, &prepared) + (prepared.plan.root, None, Some(prepared)) + } else { + let (page, work) = build_directory_page_with_work(handler.as_ref(), &root_tree, &req.scope) + .await + .map_err(mst2_error_response)?; + (page.page_id, Some(work), None) + }; + let projection_elapsed = projection_started.elapsed(); + + let built = build_descriptor(&config.mst2, &view, &req.scope, metadata_root) + .map_err(mst2_error_response)?; + let ctx = if let Some(head) = selected_native_head.as_ref() { + use crate::jupiter::storage::qualified_metadata_family::SnapshotMetadataFamily; + let family = state + .storage + .snapshot_metadata_family(&built.snapshot_id, false) + .await + .map_err(mst2_error_response)?; + if family == Some(SnapshotMetadataFamily::Generic) || prepared_generic.is_some() { + // A permanent SID route retains its original physical family. + let existing = state + .storage + .snapshot_sessions() + .await + .open(head, &built, None, req.lease_seconds) .await .map_err(mst2_error_response)?; - #[cfg(test)] - if let Ok((prepared, release)) = NATIVE_HANDOFF_BARRIERS.try_with(|value| value.clone()) - { - prepared.wait().await; - release.wait().await; + if let Some(existing) = existing { + existing + } else { + let (_, _, prepared) = prepared_generic.as_ref().ok_or_else(|| { + mst2_error_response(internal("existing generic route has no durable session")) + })?; + let sessions = state.storage.snapshot_sessions().await; + let receipt = sessions + .install(&built, prepared) + .await + .map_err(mst2_error_response)?; + #[cfg(test)] + if let Ok((prepared, release)) = NATIVE_HANDOFF_BARRIERS.try_with(Clone::clone) { + prepared.wait().await; + release.wait().await; + } + sessions + .open(head, &built, Some(&receipt), req.lease_seconds) + .await + .map_err(mst2_error_response)? + .ok_or_else(|| { + mst2_error_response(internal( + "generic history fixture handoff returned no context", + )) + })? } - sessions - .open(head, &built, Some(&receipt), req.lease_seconds) + } else { + let repository = state + .storage + .rooted_qualified_metadata_writer() + .await + .map_err(internal)?; + if let Some(context) = repository + .open_session(head, &built, None, req.lease_seconds) .await .map_err(mst2_error_response)? - .ok_or_else(|| { - mst2_error_response(internal("durable session handoff returned no context")) - })? + { + context + } else { + let prepared = prepared_rooted.as_ref().ok_or_else(|| { + mst2_error_response(internal("rooted resolve has no source projection")) + })?; + let receipt = repository + .install(&built, prepared) + .await + .map_err(crate::jupiter::storage::native_snapshot_session::install_error) + .map_err(mst2_error_response)?; + #[cfg(test)] + if let Ok((prepared, release)) = + NATIVE_HANDOFF_BARRIERS.try_with(|value| value.clone()) + { + prepared.wait().await; + release.wait().await; + } + repository + .open_session(head, &built, Some(&receipt), req.lease_seconds) + .await + .map_err(mst2_error_response)? + .ok_or_else(|| { + mst2_error_response(internal("durable session handoff returned no context")) + })? + } } } else { runtime() @@ -821,20 +928,26 @@ async fn resolve( if let Some(source) = native_source { let request_id = current_request_id(); - match source.observe( - ResolvedProjection { - descriptor: &ctx.built.descriptor, - snapshot_id: &ctx.built.snapshot_id, - metadata_root: &ctx.built.metadata_root, - context_commit: &ctx.commit_oid, - context_root_tree: &ctx.root_tree_oid, - fixed_root_tree: root_tree.id, - requested_scope: &req.scope, - request_id: &request_id, - }, - projection_work, - projection_elapsed, - ) { + let resolved = ResolvedProjection { + descriptor: &ctx.built.descriptor, + snapshot_id: &ctx.built.snapshot_id, + metadata_root: &ctx.built.metadata_root, + context_commit: &ctx.commit_oid, + context_root_tree: &ctx.root_tree_oid, + fixed_root_tree: root_tree.id, + requested_scope: &req.scope, + request_id: &request_id, + }; + let observation = if let Some(prepared) = prepared_rooted { + source.observe_rooted(resolved, prepared.work, projection_elapsed) + } else { + source.observe( + resolved, + projection_work.unwrap_or_default(), + projection_elapsed, + ) + }; + match observation { Ok(observation) => { if let Some(sink) = &state.storage.projection_observation_sink { let _ = sink.enqueue(&observation); @@ -963,6 +1076,9 @@ struct DirectoryQuery { ancestors: Option, } +#[path = "snapshot_rooted_metadata.rs"] +mod rooted_metadata; + fn default_limit() -> u32 { 128 } @@ -986,6 +1102,18 @@ async fn directory( "limit must be 1..256", ))); } + if state.storage.config().mst2.publication_enabled + && state + .storage + .snapshot_metadata_family(&ctx.lease_id, true) + .await + .map_err(mst2_error_response)? + == Some( + crate::jupiter::storage::qualified_metadata_family::SnapshotMetadataFamily::Rooted, + ) + { + return rooted_metadata::directory_response(&state, &ctx, &snapshot_id, &q).await; + } let handler = state .api_handler(std::path::Path::new("/")) @@ -1201,11 +1329,24 @@ async fn lookup( ))); } + let ctx = request_context(&state, &snapshot_id).map_err(mst2_error_response)?; + if state.storage.config().mst2.publication_enabled + && state + .storage + .snapshot_metadata_family(&ctx.lease_id, true) + .await + .map_err(mst2_error_response)? + == Some( + crate::jupiter::storage::qualified_metadata_family::SnapshotMetadataFamily::Rooted, + ) + { + return rooted_metadata::lookup_response(&state, &ctx, &snapshot_id, &req).await; + } + let handler = state .api_handler(std::path::Path::new("/")) .await .map_err(internal)?; - let ctx = request_context(&state, &snapshot_id).map_err(mst2_error_response)?; let root_tree = handler .get_tree_by_hash(&ctx.root_tree_oid) .await @@ -1371,9 +1512,7 @@ async fn metadata_pages( .collect(); let batch = state .storage - .snapshot_sessions() - .await - .metadata_routes(&ctx, &items) + .snapshot_metadata_routes(&ctx, &items) .await .map_err(mst2_error_response)?; tracing::debug!( diff --git a/src/api/router/snapshot_session_tests.rs b/src/api/router/snapshot_session_tests.rs index ae84d3c6..36101a05 100644 --- a/src/api/router/snapshot_session_tests.rs +++ b/src/api/router/snapshot_session_tests.rs @@ -84,12 +84,16 @@ async fn rebuilt(fixture: &Fixture) -> MonoApiServiceState { let connection = crate::jupiter::storage::init::postgres_connection(&config.database) .await .unwrap(); - let storage = crate::jupiter::storage::Storage::new_with_connection( + let assembly = crate::jupiter::storage::Storage::new_with_connection( config, Arc::new(connection), fixture.state.storage.git_service.obj_storage.clone(), - ) - .await + ); + let storage = if fixture.generic_history { + crate::jupiter::storage::init::with_generic_history_bootstrap(assembly).await + } else { + assembly.await + } .unwrap(); MonoApiServiceState { storage, @@ -102,7 +106,11 @@ async fn rebuilt(fixture: &Fixture) -> MonoApiServiceState { } fn app(state: &MonoApiServiceState) -> Router { - Router::new().nest("/api/v2", routers(state.clone()).with_state(state.clone())) + Router::new().nest( + "/api/v2", + crate::api::router::snapshot_router::generic_history_routers(state.clone()) + .with_state(state.clone()), + ) } async fn lease_control(fixture: &Fixture, lease: &str, method: &str, renew: bool) -> Value { @@ -150,7 +158,7 @@ async fn advance(fixture: &Fixture) { #[tokio::test] async fn mst2_durable_http_resolve_installs_complete_dag_and_warm_leases_share_it() { - let fixture = Fixture::new_with_pg_config(true).await; + let fixture = Fixture::new_generic_history_with_pg_config(true).await; let mono = fixture.state.storage.mono_storage(); let db = mono.get_connection(); let pages = scalar(db, "SELECT count(*) FROM mst2_metadata_payload").await; @@ -306,7 +314,7 @@ async fn mst2_durable_http_resolve_installs_complete_dag_and_warm_leases_share_i #[tokio::test] async fn mst2_durable_http_old_sid_and_original_lease_survive_real_publication_and_fresh_service() { - let fixture = Fixture::new_with_pg_config(true).await; + let fixture = Fixture::new_generic_history_with_pg_config(true).await; let old = success_json(fixture.send("GET", "descriptor", Body::empty()).await).await; advance(&fixture).await; let next = success_json( @@ -376,7 +384,7 @@ async fn mst2_durable_http_old_sid_and_original_lease_survive_real_publication_a #[tokio::test] async fn mst2_durable_http_renew_release_expiry_and_last_lease_retire_protection() { - let fixture = Fixture::new_with_pg_config(true).await; + let fixture = Fixture::new_generic_history_with_pg_config(true).await; let another = success_json( fixture .app @@ -491,7 +499,7 @@ async fn mst2_durable_http_renew_release_expiry_and_last_lease_retire_protection #[tokio::test] async fn mst2_durable_http_live_to_deleting_winner_rejects_paused_resolve_without_leak() { - let fixture = Fixture::new_with_pg_config(true).await; + let fixture = Fixture::new_generic_history_with_pg_config(true).await; let captured = Arc::new(Barrier::new(2)); let release = Arc::new(Barrier::new(2)); let resolving = { @@ -542,7 +550,7 @@ async fn mst2_durable_http_live_to_deleting_winner_rejects_paused_resolve_withou #[tokio::test] async fn mst2_durable_http_lease_winner_blocks_gc_until_the_last_release() { - let fixture = Fixture::new_with_pg_config(true).await; + let fixture = Fixture::new_generic_history_with_pg_config(true).await; let another = success_json( fixture .app @@ -595,7 +603,7 @@ async fn mst2_durable_http_lease_winner_blocks_gc_until_the_last_release() { #[tokio::test] async fn mst2_durable_http_renew_waits_for_lock_before_checking_database_deadline() { - let fixture = Fixture::new_with_pg_config(true).await; + let fixture = Fixture::new_generic_history_with_pg_config(true).await; let mono = fixture.state.storage.mono_storage(); let db = mono.get_connection(); let held = db.begin().await.unwrap(); @@ -671,7 +679,7 @@ async fn mst2_durable_http_renew_waits_for_lock_before_checking_database_deadlin #[tokio::test] async fn mst2_durable_http_publication_advance_rejects_mixed_resolve_then_retries_current() { - let fixture = Fixture::new_with_pg_config(true).await; + let fixture = Fixture::new_generic_history_with_pg_config(true).await; let captured = Arc::new(Barrier::new(2)); let release = Arc::new(Barrier::new(2)); let resolving = { @@ -725,7 +733,7 @@ async fn mst2_durable_http_publication_advance_rejects_mixed_resolve_then_retrie #[tokio::test] async fn mst2_durable_http_install_fault_never_hands_off_or_returns_success() { - let fixture = Fixture::new_with_pg_config(true).await; + let fixture = Fixture::new_generic_history_with_pg_config(true).await; let mono = fixture.state.storage.mono_storage(); let db = mono.get_connection(); let prepared = root_metadata(&fixture).await; @@ -828,7 +836,7 @@ async fn mst2_durable_http_install_fault_never_hands_off_or_returns_success() { #[tokio::test] async fn mst2_durable_http_frame_and_raw_delivery_recheck_after_release() { for raw in [false, true] { - let fixture = Fixture::new_with_pg_config(true).await; + let fixture = Fixture::new_generic_history_with_pg_config(true).await; let response = if raw { fixture.send("GET", "blob?path=/file", Body::empty()).await } else { @@ -862,7 +870,7 @@ async fn mst2_durable_http_frame_and_raw_delivery_recheck_after_release() { #[tokio::test] async fn mst2_durable_http_current_state_wrong_lease_and_corrupt_source_reject_warm_reads() { - let fixture = Fixture::new_with_pg_config(true).await; + let fixture = Fixture::new_generic_history_with_pg_config(true).await; let warm = success_json(fixture.send("GET", "descriptor", Body::empty()).await).await; let mono = fixture.state.storage.mono_storage(); let db = mono.get_connection(); @@ -935,7 +943,7 @@ async fn mst2_durable_http_current_state_wrong_lease_and_corrupt_source_reject_w #[tokio::test] async fn mst2_durable_http_large_dag_batches_handoff_without_scanning_pages_or_edges() { - let fixture = Fixture::new_with_pg_config_and_directories(true, 80).await; + let fixture = Fixture::new_generic_history_with_pg_config_and_directories(true, 80).await; let mono = fixture.state.storage.mono_storage(); let db = mono.get_connection(); let projected = root_metadata(&fixture).await; @@ -1083,7 +1091,7 @@ async fn mst2_durable_http_handoff_rejects_changed_plan_or_receipt_summary() { "projection_revision=projection_revision+1", "total_bytes=total_bytes+1", ] { - let fixture = Fixture::new_with_pg_config_and_directories(true, 80).await; + let fixture = Fixture::new_generic_history_with_pg_config_and_directories(true, 80).await; let prepared = Arc::new(Barrier::new(2)); let release = Arc::new(Barrier::new(2)); let resolving = { @@ -1254,7 +1262,7 @@ async fn overwrite_primary_scope_for_test(db: &DatabaseConnection, storage_uuid: #[tokio::test] async fn mst2_durable_http_warm_reads_renew_and_frame_delivery_reject_primary_scope_drift() { - let fixture = Fixture::new_with_pg_config(true).await; + let fixture = Fixture::new_generic_history_with_pg_config(true).await; success_json(fixture.send("GET", "descriptor", Body::empty()).await).await; let response = fixture .send( diff --git a/src/api/router/snapshot_storage_route_fixture.rs b/src/api/router/snapshot_storage_route_fixture.rs index abc3aa48..71fb3204 100644 --- a/src/api/router/snapshot_storage_route_fixture.rs +++ b/src/api/router/snapshot_storage_route_fixture.rs @@ -30,12 +30,19 @@ pub(super) async fn restore_pre_route_schema(db: &DatabaseConnection) { // The original contexts, leases, plans, payloads and protection stay intact. txn.execute_unprepared( "DO $$ DECLARE r record; functions text; BEGIN + IF EXISTS(SELECT 1 FROM mst2_metadata_namespace WHERE graph_domain='qualified-v1') THEN + RAISE EXCEPTION 'route upgrade fixture cannot tear down a provisioned qualified family'; + END IF; + IF EXISTS(SELECT 1 FROM mst2_rooted_source_tree_revision) THEN + RAISE EXCEPTION 'route upgrade fixture cannot erase actual qualified source history'; + END IF; + DROP FUNCTION mst2_metadata_has_generic_overlap(bytea); FOR r IN SELECT c.relname,t.tgname FROM pg_trigger t JOIN pg_class c ON c.oid=t.tgrelid JOIN pg_namespace n ON n.oid=c.relnamespace JOIN pg_proc p ON p.oid=t.tgfoid WHERE n.nspname=current_schema() AND NOT t.tgisinternal AND left(p.proname,11)='mst2_route_' LOOP EXECUTE format('DROP TRIGGER %I ON %I.%I',r.tgname,current_schema(),r.relname); END LOOP; - DROP TABLE mst2_lease_storage_route,mst2_generic_session_storage_binding,mst2_snapshot_storage_route,mst2_metadata_namespace; + DROP TABLE mst2_lease_storage_route,mst2_generic_session_storage_binding,mst2_snapshot_storage_route,mst2_metadata_namespace,mst2_qualified_family_policy,mst2_rooted_source_tree_revision; FOR r IN SELECT conname FROM pg_constraint WHERE conrelid='mst2_snapshot_context'::regclass AND contype='u' AND pg_get_constraintdef(oid)='UNIQUE (snapshot_id, prepare_id, metadata_root)' @@ -46,7 +53,8 @@ pub(super) async fn restore_pre_route_schema(db: &DatabaseConnection) { IF functions IS NULL THEN RAISE EXCEPTION 'route upgrade fixture has no route functions'; END IF; EXECUTE 'DROP FUNCTION '||functions; END $$; - DELETE FROM seaql_migrations WHERE version='m20261007_000600_add_mst2_storage_routes'", + DELETE FROM seaql_migrations WHERE version IN ( + 'm20261007_000600_add_mst2_storage_routes','m20261008_000200_add_mst2_rooted_qualified_family')", ) .await .unwrap(); diff --git a/src/api/router/snapshot_storage_route_tests.rs b/src/api/router/snapshot_storage_route_tests.rs index 25d40ec3..63b59fc6 100644 --- a/src/api/router/snapshot_storage_route_tests.rs +++ b/src/api/router/snapshot_storage_route_tests.rs @@ -176,7 +176,11 @@ async fn lease_control(fixture: &Fixture, lease: &str, renew: bool) -> Response async fn rebuilt(fixture: &Fixture) -> Router { let state = rebuilt_state(fixture).await; - Router::new().nest("/api/v2", routers(state.clone()).with_state(state)) + Router::new().nest( + "/api/v2", + crate::api::router::snapshot_router::generic_history_routers(state.clone()) + .with_state(state), + ) } async fn rebuilt_state(fixture: &Fixture) -> MonoApiServiceState { @@ -187,10 +191,12 @@ async fn rebuilt_state(fixture: &Fixture) -> MonoApiServiceState { let connection = crate::jupiter::storage::init::postgres_connection(&database) .await .unwrap(); - let storage = crate::jupiter::storage::Storage::new_with_connection( - config, - Arc::new(connection), - fixture.state.storage.git_service.obj_storage.clone(), + let storage = crate::jupiter::storage::init::with_generic_history_bootstrap( + crate::jupiter::storage::Storage::new_with_connection( + config, + Arc::new(connection), + fixture.state.storage.git_service.obj_storage.clone(), + ), ) .await .unwrap(); @@ -326,7 +332,7 @@ async fn rejected(db: &DatabaseConnection, sql: &str) { #[tokio::test] async fn mst2_generic_storage_routes_actual_resolve_warm_and_fresh_service_keep_exact_tuple() { - let fixture = Fixture::new_with_pg_config(true).await; + let fixture = Fixture::new_generic_history_with_pg_config(true).await; let mono = fixture.state.storage.mono_storage(); let db = mono.get_connection(); assert_exact_bindings(db, 1).await; @@ -374,7 +380,7 @@ async fn mst2_generic_storage_routes_actual_resolve_warm_and_fresh_service_keep_ #[tokio::test] async fn mst2_generic_storage_routes_every_member_delete_and_truncate_are_immutable() { - let fixture = Fixture::new_with_pg_config(true).await; + let fixture = Fixture::new_generic_history_with_pg_config(true).await; let mono = fixture.state.storage.mono_storage(); let db = mono.get_connection(); let original = routes(db).await; @@ -431,7 +437,7 @@ fn lease_insert(lease: &str) -> String { #[tokio::test] async fn mst2_generic_storage_routes_raw_lease_derives_atomically_and_half_rows_roll_back() { - let fixture = Fixture::new_with_pg_config(true).await; + let fixture = Fixture::new_generic_history_with_pg_config(true).await; let mono = fixture.state.storage.mono_storage(); let db = mono.get_connection(); let original = routes(db).await; @@ -488,7 +494,7 @@ async fn mst2_generic_storage_routes_raw_lease_derives_atomically_and_half_rows_ #[tokio::test] async fn mst2_generic_storage_routes_renew_terminal_and_unknown_release_preserve_route_and_roots() { - let fixture = Fixture::new_with_pg_config(true).await; + let fixture = Fixture::new_generic_history_with_pg_config(true).await; let mono = fixture.state.storage.mono_storage(); let db = mono.get_connection(); let original = routes(db).await; @@ -592,7 +598,7 @@ async fn lock_count(held: &DatabaseTransaction, pid: i64, key: i32) -> i64 { #[tokio::test] async fn mst2_generic_storage_routes_raw_statement_waits_mono_then_route_then_retention() { - let fixture = Fixture::new_with_pg_config(true).await; + let fixture = Fixture::new_generic_history_with_pg_config(true).await; let mono = fixture.state.storage.mono_storage(); let db = mono.get_connection(); for held_key in [MONO_WRITE_LOCK_KEY1, ROUTE_LOCK_KEY, RETENTION_LOCK_KEY] { @@ -692,8 +698,8 @@ async fn mst2_generic_storage_routes_raw_statement_waits_mono_then_route_then_re #[tokio::test] async fn mst2_generic_storage_routes_schema_isolation_wrong_caller_rr_and_temp_shadow_fail_closed() { - let fixture = Fixture::new_with_pg_config(true).await; - let alien = Fixture::new_with_pg_config(true).await; + let fixture = Fixture::new_generic_history_with_pg_config(true).await; + let alien = Fixture::new_generic_history_with_pg_config(true).await; let mono = fixture.state.storage.mono_storage(); let db = mono.get_connection(); let original = routes(db).await; @@ -801,7 +807,7 @@ async fn mst2_generic_storage_routes_schema_isolation_wrong_caller_rr_and_temp_s #[tokio::test] async fn mst2_generic_storage_routes_additive_backfill_keeps_active_terminal_sources_bytes_and_roots() { - let fixture = Fixture::new_with_pg_config(true).await; + let fixture = Fixture::new_generic_history_with_pg_config(true).await; let mono = fixture.state.storage.mono_storage(); let db = mono.get_connection(); let warm = success_json( @@ -862,9 +868,13 @@ async fn mst2_generic_storage_routes_additive_backfill_keeps_active_terminal_sou #[tokio::test] async fn mst2_generic_storage_routes_actual_http_ignores_temp_source_and_ledger_shadows() { - let fixture = Fixture::new_with_pg_config(true).await; + let fixture = Fixture::new_generic_history_with_pg_config(true).await; let state = rebuilt_state(&fixture).await; - let app = Router::new().nest("/api/v2", routers(state.clone()).with_state(state.clone())); + let app = Router::new().nest( + "/api/v2", + crate::api::router::snapshot_router::generic_history_routers(state.clone()) + .with_state(state.clone()), + ); let mono = state.storage.mono_storage(); let db = mono.get_connection(); let original = routes(db).await; @@ -936,7 +946,7 @@ async fn mst2_generic_storage_routes_actual_http_ignores_temp_source_and_ledger_ #[tokio::test] async fn mst2_generic_storage_routes_temp_prepare_shadow_rejects_qualified_context_and_registered_generic_rebind() { - let fixture = Fixture::new_with_pg_config(true).await; + let fixture = Fixture::new_generic_history_with_pg_config(true).await; let mono = fixture.state.storage.mono_storage(); let db = mono.get_connection(); let original_routes = routes(db).await; @@ -1118,7 +1128,7 @@ async fn mst2_generic_storage_routes_temp_prepare_shadow_rejects_qualified_conte #[tokio::test] async fn mst2_generic_storage_routes_corrupt_original_incarnation_rejects_reads_renew_release_without_repair() { - let fixture = Fixture::new_with_pg_config(true).await; + let fixture = Fixture::new_generic_history_with_pg_config(true).await; let mono = fixture.state.storage.mono_storage(); let db = mono.get_connection(); let source = sources(db).await; diff --git a/src/ceres/snapshot/mod.rs b/src/ceres/snapshot/mod.rs index 3fae7f32..267a78bb 100644 --- a/src/ceres/snapshot/mod.rs +++ b/src/ceres/snapshot/mod.rs @@ -19,5 +19,7 @@ pub mod publication; pub mod resolver; pub mod retention; pub mod retention_dag; +pub(crate) mod rooted_metadata_install; +pub(crate) mod rooted_metadata_projection; pub mod runtime; pub mod view; diff --git a/src/ceres/snapshot/projection_observation.rs b/src/ceres/snapshot/projection_observation.rs index 819e0e86..532b3f18 100644 --- a/src/ceres/snapshot/projection_observation.rs +++ b/src/ceres/snapshot/projection_observation.rs @@ -74,6 +74,7 @@ pub(crate) struct NativeProjectionObservation { request_id: String, projection_elapsed_micros: u64, work: ProjectionWork, + rooted_work: Option, } /// Closed wire fields, borrowed only from the already validated observation. @@ -106,7 +107,9 @@ pub(crate) struct ProjectionWireRecord<'a> { page_counter_scope: &'static str, codec_radix_work: &'static str, #[serde(flatten)] - work: &'a ProjectionWork, + work: Option<&'a ProjectionWork>, + #[serde(skip_serializing_if = "Option::is_none")] + rooted_work: Option<&'a super::rooted_metadata_projection::RootedProjectionWork>, message: &'static str, } @@ -190,15 +193,32 @@ impl NativeResolveSource { projection_elapsed_micros: u64::try_from(projection_elapsed.as_micros()) .map_err(|_| invalid())?, work, + rooted_work: None, }) } + + pub(crate) fn observe_rooted( + self, + resolved: ResolvedProjection<'_>, + work: super::rooted_metadata_projection::RootedProjectionWork, + projection_elapsed: Duration, + ) -> Result { + let mut observation = + self.observe(resolved, ProjectionWork::default(), projection_elapsed)?; + observation.rooted_work = Some(work); + Ok(observation) + } } impl NativeProjectionObservation { pub(crate) fn wire_record(&self) -> ProjectionWireRecord<'_> { ProjectionWireRecord { - observation_revision: 1, - phase: "resolve_directory_projection", + observation_revision: if self.rooted_work.is_some() { 2 } else { 1 }, + phase: if self.rooted_work.is_some() { + "resolve_rooted_projection" + } else { + "resolve_directory_projection" + }, source_domain: "native-git", request_id: &self.request_id, instance_id: self.source.instance.to_string(), @@ -219,14 +239,37 @@ impl NativeProjectionObservation { snapshot_id: &self.snapshot_id, metadata_root: &self.metadata_root, projection_elapsed_micros: self.projection_elapsed_micros, - page_counter_scope: "returned-directory-root-pages", - codec_radix_work: "NOT_EXPOSED", - work: &self.work, + page_counter_scope: if self.rooted_work.is_some() { + "physical-delta-and-reuse-boundaries" + } else { + "returned-directory-root-pages" + }, + codec_radix_work: if self.rooted_work.is_some() { + "OPERATION_CALLS_ONLY_NOT_TOTAL_SQL" + } else { + "NOT_EXPOSED" + }, + work: self.rooted_work.is_none().then_some(&self.work), + rooted_work: self.rooted_work.as_ref(), message: "native resolve directory projection succeeded", } } pub(crate) fn emit(self) { + if let Some(work) = &self.rooted_work { + tracing::debug!(target:"mst2::native_projection_observation", + observation_revision=2u16,phase="resolve_rooted_projection",source_domain="native-git", + request_id=%self.request_id,instance_id=%self.source.instance, + root_commit_oid=%self.source.commit.to_tagged_string(),root_tree_oid=%self.source.tree.to_tagged_string(), + native_certificate_receipt_id=self.source.certificate_receipt_id, + native_writer_epoch=self.source.writer_epoch,native_publication_sequence=self.source.publication_sequence, + scope=?self.scope,snapshot_id=%self.snapshot_id,metadata_root=%self.metadata_root, + projection_elapsed_micros=self.projection_elapsed_micros,rooted_work=?work, + "native rooted projection succeeded; counters exclude total SQL body and catalog work"); + #[cfg(test)] + let _ = OBSERVATIONS.try_with(|observations| observations.lock().unwrap().push(self)); + return; + } tracing::debug!( target: "mst2::native_projection_observation", observation_revision = 1u16, @@ -406,6 +449,45 @@ mod tests { assert_eq!(observation.snapshot_id, fixture.snapshot_id); } + #[test] + fn rooted_observation_exposes_delta_boundaries_without_legacy_or_total_sql_counters() { + let fixture = Fixture::new(); + let commit = fixture.commit.to_string(); + let tree = fixture.tree.to_string(); + let work = super::super::rooted_metadata_projection::RootedProjectionWork { + tree_fetches: 1, + delta_pages: 2, + reused_roots: 7, + ..Default::default() + }; + let observation = fixture + .source() + .observe_rooted( + fixture.resolved(&commit, &tree), + work, + Duration::from_micros(456), + ) + .unwrap(); + let wire = serde_json::to_value(observation.wire_record()).unwrap(); + assert_eq!(wire["observation_revision"], 2); + assert_eq!(wire["phase"], "resolve_rooted_projection"); + assert_eq!( + wire["page_counter_scope"], + "physical-delta-and-reuse-boundaries" + ); + assert_eq!( + wire["codec_radix_work"], + "OPERATION_CALLS_ONLY_NOT_TOTAL_SQL" + ); + assert_eq!(wire["rooted_work"]["tree_fetches"], 1); + assert_eq!(wire["rooted_work"]["delta_pages"], 2); + assert_eq!(wire["rooted_work"]["reused_roots"], 7); + assert!(wire.get("directories_rebuilt").is_none()); + assert!(wire.get("directory_root_pages_returned").is_none()); + assert_eq!(wire["snapshot_id"], fixture.snapshot_id); + assert_eq!(wire["projection_elapsed_micros"], 456); + } + #[test] fn observation_rejects_each_changed_binding_or_unbounded_trace_token() { let fixture = Fixture::new(); diff --git a/src/ceres/snapshot/rooted_metadata_install.rs b/src/ceres/snapshot/rooted_metadata_install.rs new file mode 100644 index 00000000..5086bfa5 --- /dev/null +++ b/src/ceres/snapshot/rooted_metadata_install.rs @@ -0,0 +1,848 @@ +//! Bounded native metadata delta and certified reused-root installation identity. +//! Reused roots are opaque boundaries; their database proofs grant no authority here. + +use std::collections::{BTreeMap, BTreeSet}; + +use mst2_codec::metapage::{HEADER_LEN, PAGE_MAX_BYTES}; +use sha2::{Digest, Sha256}; +use uuid::Uuid; + +use super::{ + error::{SnapshotError, SnapshotErrorCode}, + metadata_install::{MAX_PLAN_BYTES, MetadataInstallIdentity}, + projection_observation::NATIVE_PROJECTION_REVISION, + retention_dag::{MetadataDagLimits, MetadataPageId}, +}; + +const DOMAIN: &[u8] = b"mega.mst2.rooted-install.v1\0"; +const ENCODING_VERSION: u16 = 1; + +#[derive(Debug, Clone, PartialEq, Eq)] +pub(crate) struct RootedReuseRoot { + pub(crate) generation: i64, + pub(crate) attestation_id: Uuid, + pub(crate) attestation_digest: [u8; 32], + pub(crate) certificate_digest: [u8; 32], +} + +#[derive(Debug, Clone, PartialEq, Eq)] +pub(crate) struct RootedMetadataInstallPlan { + pub(crate) identity: MetadataInstallIdentity, + pub(crate) root: MetadataPageId, + pub(crate) delta: BTreeMap, + pub(crate) edges: BTreeSet<(MetadataPageId, MetadataPageId)>, + pub(crate) reused: BTreeMap, + pub(crate) source_roots: BTreeMap, +} + +impl RootedMetadataInstallPlan { + pub(crate) fn new( + identity: MetadataInstallIdentity, + root: MetadataPageId, + delta: BTreeMap, + edges: BTreeSet<(MetadataPageId, MetadataPageId)>, + reused: BTreeMap, + source_roots: BTreeMap, + ) -> Result { + let plan = Self { + identity, + root, + delta, + edges, + reused, + source_roots, + }; + plan.validate()?; + Ok(plan) + } + + pub(crate) fn digest(&self) -> Result<[u8; 32], SnapshotError> { + Ok(Sha256::digest(self.encode()?).into()) + } + + pub(crate) fn delta_bytes(&self) -> Result { + self.delta + .values() + .try_fold(0u64, |total, size| total.checked_add(*size)) + .ok_or_else(|| limit("rooted metadata delta byte count overflow")) + } + + pub(crate) fn encode(&self) -> Result, SnapshotError> { + self.validate()?; + let mut bytes = Vec::with_capacity(self.encoded_len()); + bytes.extend_from_slice(DOMAIN); + bytes.extend_from_slice(&ENCODING_VERSION.to_be_bytes()); + for value in [ + &self.identity.source_domain, + &self.identity.tagged_root_tree_oid, + &self.identity.scope, + ] { + write_string(&mut bytes, value); + } + for value in [ + self.identity.schema_version, + self.identity.metadata_codec, + self.identity.materialization_policy, + self.identity.fs_semantics, + self.identity.access_projection, + ] { + bytes.extend_from_slice(&value.to_be_bytes()); + } + bytes.extend_from_slice(&self.identity.verification_revision.to_be_bytes()); + bytes.extend_from_slice(&self.identity.projection_revision.to_be_bytes()); + bytes.extend_from_slice(&self.root); + bytes.extend_from_slice(&(self.delta.len() as u32).to_be_bytes()); + for (page, size) in &self.delta { + bytes.extend_from_slice(page); + bytes.extend_from_slice(&size.to_be_bytes()); + } + bytes.extend_from_slice(&(self.edges.len() as u32).to_be_bytes()); + for (parent, child) in &self.edges { + bytes.extend_from_slice(parent); + bytes.extend_from_slice(child); + } + bytes.extend_from_slice(&(self.reused.len() as u32).to_be_bytes()); + for (page, reused) in &self.reused { + bytes.extend_from_slice(page); + bytes.extend_from_slice(&reused.generation.to_be_bytes()); + bytes.extend_from_slice(reused.attestation_id.as_bytes()); + bytes.extend_from_slice(&reused.attestation_digest); + bytes.extend_from_slice(&reused.certificate_digest); + } + bytes.extend_from_slice(&(self.source_roots.len() as u32).to_be_bytes()); + for (tree_oid, page) in &self.source_roots { + write_string(&mut bytes, tree_oid); + bytes.extend_from_slice(page); + } + Ok(bytes) + } + + pub(crate) fn decode(bytes: &[u8], expected_digest: &[u8; 32]) -> Result { + if bytes.len() > MAX_PLAN_BYTES { + return Err(limit("stored rooted metadata plan exceeds its byte budget")); + } + let digest: [u8; 32] = Sha256::digest(bytes).into(); + if &digest != expected_digest { + return Err(integrity("stored rooted metadata plan digest mismatch")); + } + let mut reader = PlanReader(bytes); + if reader.take(DOMAIN.len())? != DOMAIN || reader.u16()? != ENCODING_VERSION { + return Err(integrity("unsupported rooted metadata plan encoding")); + } + let identity = MetadataInstallIdentity { + source_domain: reader.string(64)?, + tagged_root_tree_oid: reader.string(128)?, + scope: reader.string(4096)?, + schema_version: reader.u16()?, + metadata_codec: reader.u16()?, + materialization_policy: reader.u16()?, + fs_semantics: reader.u16()?, + access_projection: reader.u16()?, + verification_revision: i32::from_be_bytes(reader.array()?), + projection_revision: reader.u16()?, + }; + let root = reader.array()?; + let limits = MetadataDagLimits::default(); + let count = reader.count(limits.nodes)?; + let mut delta = BTreeMap::new(); + let mut last = None; + for _ in 0..count { + let page = reader.array()?; + let size = reader.u64()?; + if last.is_some_and(|previous| previous >= page) { + return Err(integrity("rooted metadata delta is not uniquely ordered")); + } + last = Some(page); + delta.insert(page, size); + } + let count = reader.count(limits.edges)?; + let mut edges = BTreeSet::new(); + let mut last = None; + for _ in 0..count { + let edge = (reader.array()?, reader.array()?); + if last.is_some_and(|previous| previous >= edge) { + return Err(integrity("rooted metadata edges are not uniquely ordered")); + } + last = Some(edge); + edges.insert(edge); + } + let count = reader.count(limits.nodes - delta.len())?; + let mut reused = BTreeMap::new(); + let mut last = None; + for _ in 0..count { + let page = reader.array()?; + let proof = RootedReuseRoot { + generation: i64::from_be_bytes(reader.array()?), + attestation_id: Uuid::from_bytes(reader.array()?), + attestation_digest: reader.array()?, + certificate_digest: reader.array()?, + }; + if last.is_some_and(|previous| previous >= page) { + return Err(integrity( + "rooted metadata reused roots are not uniquely ordered", + )); + } + last = Some(page); + reused.insert(page, proof); + } + let count = reader.count(limits.nodes)?; + let mut source_roots: BTreeMap = BTreeMap::new(); + for _ in 0..count { + let tree_oid = reader.string(128)?; + let page = reader.array()?; + if source_roots + .last_key_value() + .is_some_and(|(previous, _)| previous >= &tree_oid) + { + return Err(integrity( + "rooted metadata source roots are not uniquely ordered", + )); + } + source_roots.insert(tree_oid, page); + } + if !reader.0.is_empty() { + return Err(integrity("rooted metadata plan has trailing bytes")); + } + Self::new(identity, root, delta, edges, reused, source_roots) + } + + pub(crate) fn validate(&self) -> Result<(), SnapshotError> { + self.validate_and_order().map(|_| ()) + } + + /// Child-first delta order. Reused roots have no descendants in this plan. + pub(crate) fn child_first_delta(&self) -> Result, SnapshotError> { + self.validate_and_order() + } + + fn validate_and_order(&self) -> Result, SnapshotError> { + let identity = &self.identity; + if identity.source_domain != "native-git" + || identity.schema_version != mst2_codec::descriptor::SCHEMA_VERSION + || identity.metadata_codec != mst2_codec::descriptor::METADATA_CODEC + || identity.materialization_policy + != mst2_codec::descriptor::MATERIALIZATION_POLICY_GIT_RAW_V1 + || identity.fs_semantics != mst2_codec::descriptor::FS_SEMANTICS_LINUX_CODE_V1 + || identity.access_projection != mst2_codec::descriptor::ACCESS_PROJECTION_EXACT_FULL + || identity.verification_revision + != crate::jupiter::storage::mono_storage::MST2_VERIFICATION_VERSION + || identity.projection_revision != NATIVE_PROJECTION_REVISION + { + return Err(integrity("unsupported rooted native metadata profile")); + } + let hash_kind = tagged_hash_kind(&identity.tagged_root_tree_oid)?; + super::view::validate_scope_relative_path(&identity.scope) + .map_err(|_| integrity("noncanonical rooted metadata scope"))?; + let limits = MetadataDagLimits::default(); + if self.delta.len() > limits.nodes + || self.reused.len() > limits.nodes - self.delta.len() + || self.delta.is_empty() && self.reused.is_empty() + || self.edges.len() > limits.edges + || self.source_roots.is_empty() + || self.source_roots.len() > limits.nodes + { + return Err(limit("rooted metadata plan exceeds its group budget")); + } + if self.delta_bytes()? > limits.payload_bytes { + return Err(limit("rooted metadata delta exceeds its payload budget")); + } + if !self.contains(&self.root) + || self.delta.iter().any(|(page, size)| { + self.reused.contains_key(page) + || !(HEADER_LEN as u64..=PAGE_MAX_BYTES as u64).contains(size) + }) + || self.reused.values().any(|proof| proof.generation <= 0) + || self.delta.is_empty() + && (!self.reused.contains_key(&self.root) || !self.edges.is_empty()) + { + return Err(integrity( + "rooted metadata root, delta or reuse boundary is invalid", + )); + } + for (tree_oid, page) in &self.source_roots { + if tagged_hash_kind(tree_oid)? != hash_kind || !self.contains(page) { + return Err(integrity( + "rooted metadata source binding crossed its profile or boundary", + )); + } + } + if !self.source_roots.values().any(|page| page == &self.root) { + return Err(integrity( + "rooted metadata root lacks a source tree binding", + )); + } + if self.encoded_len() > MAX_PLAN_BYTES { + return Err(limit("rooted metadata plan exceeds its byte budget")); + } + + let mut remaining: BTreeMap<_, usize> = self.delta.keys().map(|page| (*page, 0)).collect(); + let mut parents: BTreeMap<_, Vec<_>> = BTreeMap::new(); + let mut children: BTreeMap<_, Vec<_>> = BTreeMap::new(); + for &(parent, child) in &self.edges { + if parent == child || !self.delta.contains_key(&parent) || !self.contains(&child) { + return Err(integrity( + "rooted metadata edge crossed its delta or reused boundary", + )); + } + children.entry(parent).or_default().push(child); + if self.delta.contains_key(&child) { + *remaining + .get_mut(&parent) + .ok_or_else(|| integrity("rooted metadata delta parent is missing"))? += 1; + parents.entry(child).or_default().push(parent); + } + } + let mut ready: BTreeSet<_> = remaining + .iter() + .filter_map(|(page, count)| (*count == 0).then_some(*page)) + .collect(); + let mut ordered = Vec::with_capacity(self.delta.len()); + while let Some(page) = ready.pop_first() { + ordered.push(page); + for parent in parents.get(&page).into_iter().flatten() { + let count = remaining + .get_mut(parent) + .ok_or_else(|| integrity("rooted metadata delta ancestor is missing"))?; + *count = count + .checked_sub(1) + .ok_or_else(|| integrity("rooted metadata delta edge accounting underflow"))?; + if *count == 0 { + ready.insert(*parent); + } + } + } + if ordered.len() != self.delta.len() { + return Err(integrity("rooted metadata delta contains a cycle")); + } + let mut reachable = BTreeSet::new(); + let mut pending = vec![self.root]; + while let Some(page) = pending.pop() { + if reachable.insert(page) { + pending.extend(children.get(&page).into_iter().flatten()); + } + } + if reachable.len() != self.delta.len() + self.reused.len() { + return Err(integrity( + "rooted metadata plan contains unreachable boundaries", + )); + } + Ok(ordered) + } + + fn contains(&self, page: &MetadataPageId) -> bool { + self.delta.contains_key(page) || self.reused.contains_key(page) + } + + fn encoded_len(&self) -> usize { + DOMAIN.len() + + 2 + + 3 * 4 + + self.identity.source_domain.len() + + self.identity.tagged_root_tree_oid.len() + + self.identity.scope.len() + + 5 * 2 + + 4 + + 2 + + 32 + + 4 * 4 + + self.delta.len() * 40 + + self.edges.len() * 64 + + self.reused.len() * 120 + + self + .source_roots + .keys() + .map(|tree_oid| 4 + tree_oid.len() + 32) + .sum::() + } +} + +fn tagged_hash_kind(oid: &str) -> Result<&str, SnapshotError> { + let (kind, hex) = oid + .split_once(':') + .ok_or_else(|| integrity("untagged rooted metadata source tree"))?; + let length = match kind { + "sha1" => 40, + "sha256" | "blake3" => 64, + _ => return Err(integrity("unsupported rooted metadata source hash kind")), + }; + if hex.len() != length + || !hex + .bytes() + .all(|byte| byte.is_ascii_digit() || (b'a'..=b'f').contains(&byte)) + { + return Err(integrity( + "noncanonical rooted metadata source tree identity", + )); + } + Ok(kind) +} + +fn write_string(bytes: &mut Vec, value: &str) { + bytes.extend_from_slice(&(value.len() as u32).to_be_bytes()); + bytes.extend_from_slice(value.as_bytes()); +} + +struct PlanReader<'a>(&'a [u8]); + +impl<'a> PlanReader<'a> { + fn take(&mut self, count: usize) -> Result<&'a [u8], SnapshotError> { + if count > self.0.len() { + return Err(integrity("truncated rooted metadata plan")); + } + let (bytes, rest) = self.0.split_at(count); + self.0 = rest; + Ok(bytes) + } + + fn array(&mut self) -> Result<[u8; N], SnapshotError> { + self.take(N)? + .try_into() + .map_err(|_| integrity("invalid rooted metadata plan field")) + } + + fn u16(&mut self) -> Result { + Ok(u16::from_be_bytes(self.array()?)) + } + + fn u32(&mut self) -> Result { + Ok(u32::from_be_bytes(self.array()?)) + } + + fn u64(&mut self) -> Result { + Ok(u64::from_be_bytes(self.array()?)) + } + + fn count(&mut self, maximum: usize) -> Result { + let count = self.u32()? as usize; + if count > maximum { + return Err(limit("rooted metadata plan count exceeds its budget")); + } + Ok(count) + } + + fn string(&mut self, maximum: usize) -> Result { + let count = self.count(maximum)?; + String::from_utf8(self.take(count)?.to_vec()) + .map_err(|_| integrity("non-UTF8 rooted metadata plan identity")) + } +} + +fn integrity(message: &str) -> SnapshotError { + SnapshotError::new(SnapshotErrorCode::IntegrityError, message) +} + +fn limit(message: &str) -> SnapshotError { + SnapshotError::new(SnapshotErrorCode::LimitExceeded, message) +} + +#[cfg(test)] +mod tests { + use super::*; + + fn page(value: u32) -> MetadataPageId { + let mut id = [0; 32]; + id[28..].copy_from_slice(&value.to_be_bytes()); + id + } + + fn identity() -> MetadataInstallIdentity { + MetadataInstallIdentity { + source_domain: "native-git".into(), + tagged_root_tree_oid: format!("sha1:{}", "a".repeat(40)), + scope: "/".into(), + schema_version: mst2_codec::descriptor::SCHEMA_VERSION, + metadata_codec: mst2_codec::descriptor::METADATA_CODEC, + materialization_policy: mst2_codec::descriptor::MATERIALIZATION_POLICY_GIT_RAW_V1, + fs_semantics: mst2_codec::descriptor::FS_SEMANTICS_LINUX_CODE_V1, + access_projection: mst2_codec::descriptor::ACCESS_PROJECTION_EXACT_FULL, + verification_revision: crate::jupiter::storage::mono_storage::MST2_VERIFICATION_VERSION, + projection_revision: NATIVE_PROJECTION_REVISION, + } + } + + fn reuse(value: u128) -> RootedReuseRoot { + RootedReuseRoot { + generation: 7, + attestation_id: Uuid::from_u128(value), + attestation_digest: [11; 32], + certificate_digest: [12; 32], + } + } + + fn plan() -> RootedMetadataInstallPlan { + let identity = identity(); + RootedMetadataInstallPlan::new( + identity.clone(), + page(1), + BTreeMap::from([(page(1), 20), (page(2), 57), (page(3), 16384)]), + BTreeSet::from([ + (page(1), page(2)), + (page(1), page(3)), + (page(1), page(5)), + (page(2), page(4)), + (page(3), page(4)), + ]), + BTreeMap::from([(page(4), reuse(1)), (page(5), reuse(2))]), + BTreeMap::from([ + (identity.tagged_root_tree_oid, page(1)), + (format!("sha1:{}", "b".repeat(40)), page(4)), + ]), + ) + .unwrap() + } + + fn decode(bytes: &[u8]) -> Result { + RootedMetadataInstallPlan::decode(bytes, &Sha256::digest(bytes).into()) + } + + fn records(bytes: &[u8]) -> [Vec>; 4] { + let mut reader = PlanReader(bytes); + reader.take(DOMAIN.len() + 2).unwrap(); + for _ in 0..3 { + reader.string(4096).unwrap(); + } + reader.take(16 + 32).unwrap(); + let mut sections: [Vec>; 4] = std::array::from_fn(|_| Vec::new()); + for (section, width) in sections[..3].iter_mut().zip([40, 64, 120]) { + let count = reader.u32().unwrap(); + for _ in 0..count { + let start = bytes.len() - reader.0.len(); + reader.take(width).unwrap(); + section.push(start..start + width); + } + } + let count = reader.u32().unwrap(); + for _ in 0..count { + let start = bytes.len() - reader.0.len(); + reader.string(128).unwrap(); + reader.take(32).unwrap(); + sections[3].push(start..bytes.len() - reader.0.len()); + } + assert!(reader.0.is_empty()); + sections + } + + #[test] + fn rooted_plan_roundtrip_and_shared_reuse_need_only_delta_order() { + let plan = plan(); + let bytes = plan.encode().unwrap(); + assert_eq!(bytes.len(), plan.encoded_len()); + assert_eq!(bytes.len(), 1004); + assert_eq!( + hex::encode(plan.digest().unwrap()), + "e74e98e8737cfcbdfb264d9f12e15ba2126e96b1e5f35cb0cfd89fb24dde808b" + ); + assert_eq!(decode(&bytes).unwrap(), plan); + assert_eq!( + plan.child_first_delta().unwrap(), + [page(2), page(3), page(1)] + ); + assert_eq!(plan.delta_bytes().unwrap(), 20 + 57 + 16384); + for range in &records(&bytes)[2] { + assert_eq!(range.len(), 120); + } + let original = plan.digest().unwrap(); + for field in 0..4 { + let mut changed = plan.clone(); + let proof = changed.reused.get_mut(&page(4)).unwrap(); + match field { + 0 => proof.generation += 1, + 1 => proof.attestation_id = Uuid::from_u128(9), + 2 => proof.attestation_digest[0] ^= 1, + _ => proof.certificate_digest[0] ^= 1, + } + assert_ne!(changed.digest().unwrap(), original); + } + let mut changed = plan.clone(); + changed.identity.scope = "/目录/a\\b".into(); + assert_ne!(changed.digest().unwrap(), original); + assert_eq!(decode(&changed.encode().unwrap()).unwrap(), changed); + changed = plan; + changed + .source_roots + .insert(format!("sha1:{}", "c".repeat(40)), page(3)); + assert_ne!(changed.digest().unwrap(), original); + } + + #[test] + fn rooted_plan_stored_order_duplicates_digest_domain_version_truncation_and_eof_reject() { + let plan = plan(); + let bytes = plan.encode().unwrap(); + let mut wrong_digest = plan.digest().unwrap(); + wrong_digest[0] ^= 1; + assert!(RootedMetadataInstallPlan::decode(&bytes, &wrong_digest).is_err()); + for section in records(&bytes) { + let mut changed = bytes.clone(); + let a = §ion[0]; + let b = §ion[1]; + assert_eq!(a.len(), b.len()); + changed[a.clone()].copy_from_slice(&bytes[b.clone()]); + changed[b.clone()].copy_from_slice(&bytes[a.clone()]); + assert!(decode(&changed).is_err()); + changed[b.clone()].copy_from_slice(&bytes[b.clone()]); + assert!(decode(&changed).is_err()); + } + for length in [0, DOMAIN.len(), bytes.len() - 1] { + assert!(decode(&bytes[..length]).is_err()); + } + let mut changed = bytes.clone(); + changed.push(0); + assert!(decode(&changed).is_err()); + changed = bytes.clone(); + changed[0] ^= 1; + assert!(decode(&changed).is_err()); + changed = bytes; + changed[DOMAIN.len() + 1] = 2; + assert!(decode(&changed).is_err()); + let oversized = vec![0; MAX_PLAN_BYTES + 1]; + assert_eq!( + decode(&oversized).unwrap_err().code, + SnapshotErrorCode::LimitExceeded + ); + } + + #[test] + fn rooted_plan_decode_bounds_declared_counts_strings_and_reused_generations() { + let bytes = plan().encode().unwrap(); + let sections = records(&bytes); + for section in §ions { + let mut changed = bytes.clone(); + let count_at = section[0].start - 4; + changed[count_at..count_at + 4].copy_from_slice(&u32::MAX.to_be_bytes()); + assert_eq!( + decode(&changed).unwrap_err().code, + SnapshotErrorCode::LimitExceeded + ); + } + let mut changed = bytes.clone(); + let source_at = sections[3][0].start; + changed[source_at..source_at + 4].copy_from_slice(&u32::MAX.to_be_bytes()); + assert_eq!( + decode(&changed).unwrap_err().code, + SnapshotErrorCode::LimitExceeded + ); + changed = bytes.clone(); + changed[DOMAIN.len() + 2 + 4] = 0xff; + assert_eq!( + decode(&changed).unwrap_err().code, + SnapshotErrorCode::IntegrityError + ); + changed = bytes; + let generation_at = sections[2][0].start + 32; + changed[generation_at..generation_at + 8].copy_from_slice(&i64::MIN.to_be_bytes()); + assert_eq!( + decode(&changed).unwrap_err().code, + SnapshotErrorCode::IntegrityError + ); + } + + #[test] + fn rooted_plan_rejects_cycles_unknown_endpoints_reused_parents_and_unreachable_members() { + let original = plan(); + for invalid in 0..7 { + let mut changed = original.clone(); + match invalid { + 0 => { + changed.edges.insert((page(2), page(1))); + } + 1 => { + changed.edges.insert((page(4), page(2))); + } + 2 => { + changed.edges.insert((page(2), page(99))); + } + 3 => { + changed.edges.insert((page(99), page(2))); + } + 4 => { + changed.delta.insert(page(6), 20); + } + 5 => { + changed.reused.insert(page(6), reuse(6)); + } + _ => { + changed.reused.insert(page(2), reuse(2)); + } + } + assert_eq!( + changed.encode().unwrap_err().code, + SnapshotErrorCode::IntegrityError + ); + } + } + + #[test] + fn rooted_plan_zero_delta_requires_its_single_reused_root_and_source_binding() { + let identity = identity(); + let plan = RootedMetadataInstallPlan::new( + identity.clone(), + page(4), + BTreeMap::new(), + BTreeSet::new(), + BTreeMap::from([(page(4), reuse(4))]), + BTreeMap::from([(identity.tagged_root_tree_oid, page(4))]), + ) + .unwrap(); + assert!(plan.child_first_delta().unwrap().is_empty()); + assert_eq!(plan.delta_bytes().unwrap(), 0); + assert_eq!(plan.encode().unwrap().len(), 363); + assert_eq!( + hex::encode(plan.digest().unwrap()), + "81b4cbdaaacd7efe8687477d05bd4db190b974f5aa99ce7e8101326857d0e7b1" + ); + assert_eq!(decode(&plan.encode().unwrap()).unwrap(), plan); + for invalid in 0..5 { + let mut changed = plan.clone(); + match invalid { + 0 => { + changed.root = page(9); + } + 1 => { + changed.edges.insert((page(4), page(4))); + } + 2 => { + changed.reused.insert(page(5), reuse(5)); + } + 3 => { + changed.delta.insert(page(5), 20); + } + _ => { + changed.source_roots.clear(); + } + } + assert!(changed.encode().is_err()); + } + } + + #[test] + fn rooted_plan_rejects_source_profile_scope_size_and_generation_mismatches() { + let original = plan(); + for invalid in 0..18 { + let mut changed = original.clone(); + match invalid { + 0 => changed.identity.metadata_codec += 1, + 1 => changed.identity.projection_revision += 1, + 2 => changed.identity.verification_revision += 1, + 3 => changed.identity.scope = "/a/..".into(), + 4 => changed.identity.scope = format!("/{}", "a".repeat(256)), + 5 => changed.identity.tagged_root_tree_oid = format!("sha1:{}", "A".repeat(40)), + 6 => { + changed.delta.insert(page(2), HEADER_LEN as u64 - 1); + } + 7 => { + changed.delta.insert(page(2), PAGE_MAX_BYTES as u64 + 1); + } + 8 => { + changed.reused.get_mut(&page(4)).unwrap().generation = 0; + } + 9 => { + changed + .source_roots + .insert(format!("sha256:{}", "a".repeat(64)), page(3)); + } + 10 => { + changed + .source_roots + .insert(format!("sha1:{}", "d".repeat(40)), page(99)); + } + 11 => { + changed + .source_roots + .retain(|_, page_id| *page_id != page(1)); + } + 12 => { + changed.source_roots.insert("sha1:bad".into(), page(3)); + } + 13 => changed.identity.source_domain = "other".into(), + 14 => changed.identity.schema_version += 1, + 15 => changed.identity.materialization_policy += 1, + 16 => changed.identity.fs_semantics += 1, + _ => changed.identity.access_projection += 1, + } + assert!(changed.encode().is_err()); + } + for kind in ["sha256", "blake3"] { + let mut changed = original.clone(); + changed.identity.tagged_root_tree_oid = format!("{kind}:{}", "a".repeat(64)); + changed.source_roots = + BTreeMap::from([(changed.identity.tagged_root_tree_oid.clone(), page(1))]); + assert_eq!(decode(&changed.encode().unwrap()).unwrap(), changed); + } + } + + #[test] + fn rooted_plan_admits_exact_group_payload_and_source_count_boundaries() { + let identity = identity(); + let limits = MetadataDagLimits::default(); + let delta: BTreeMap<_, _> = (1..=limits.nodes as u32) + .map(|value| (page(value), PAGE_MAX_BYTES as u64)) + .collect(); + let edges = (2..=limits.nodes as u32) + .map(|value| (page(1), page(value))) + .collect(); + let full_plan = RootedMetadataInstallPlan::new( + identity.clone(), + page(1), + delta, + edges, + BTreeMap::new(), + BTreeMap::from([(identity.tagged_root_tree_oid, page(1))]), + ) + .unwrap(); + assert_eq!(full_plan.delta_bytes().unwrap(), limits.payload_bytes); + assert_eq!(decode(&full_plan.encode().unwrap()).unwrap(), full_plan); + let mut changed = full_plan.clone(); + changed + .delta + .insert(page(limits.nodes as u32 + 1), HEADER_LEN as u64); + assert_eq!( + changed.encode().unwrap_err().code, + SnapshotErrorCode::LimitExceeded + ); + changed = full_plan.clone(); + changed + .reused + .insert(page(limits.nodes as u32 + 1), reuse(9)); + assert_eq!( + changed.encode().unwrap_err().code, + SnapshotErrorCode::LimitExceeded + ); + + let mut source_plan = plan(); + source_plan.source_roots = (0..limits.nodes) + .map(|value| (format!("sha1:{value:040x}"), page(1))) + .collect(); + assert_eq!(decode(&source_plan.encode().unwrap()).unwrap(), source_plan); + source_plan + .source_roots + .insert(format!("sha1:{:040x}", limits.nodes), page(1)); + assert_eq!( + source_plan.encode().unwrap_err().code, + SnapshotErrorCode::LimitExceeded + ); + } + + #[test] + fn rooted_plan_admits_exact_edge_boundary_then_rejects_one_more_edge() { + let mut value = plan(); + value.delta = (1..=4096).map(|id| (page(id), HEADER_LEN as u64)).collect(); + value.reused.clear(); + value.source_roots.retain(|_, root| *root == page(1)); + value.edges = (2..=4096).map(|id| (page(1), page(id))).collect(); + 'fill: for parent in 2..=4096 { + for child in parent + 1..=4096 { + if value.edges.len() == MetadataDagLimits::default().edges { + break 'fill; + } + value.edges.insert((page(parent), page(child))); + } + } + assert_eq!(value.edges.len(), MetadataDagLimits::default().edges); + assert_eq!(decode(&value.encode().unwrap()).unwrap(), value); + assert!(value.edges.insert((page(4095), page(4096)))); + assert_eq!( + value.encode().unwrap_err().code, + SnapshotErrorCode::LimitExceeded + ); + } +} diff --git a/src/ceres/snapshot/rooted_metadata_projection.rs b/src/ceres/snapshot/rooted_metadata_projection.rs new file mode 100644 index 00000000..086db637 --- /dev/null +++ b/src/ceres/snapshot/rooted_metadata_projection.rs @@ -0,0 +1,1108 @@ +//! Native source projection that stops at certified reused-directory boundaries. +//! Hints and operation-local memoization are not database installation authority. + +use std::{ + borrow::Cow, + collections::{BTreeMap, BTreeSet, VecDeque}, +}; + +use async_trait::async_trait; +use git_internal::{hash::ObjectHash, internal::object::tree::Tree}; +use mst2_codec::{ + descriptor::{ + ACCESS_PROJECTION_EXACT_FULL, FS_SEMANTICS_LINUX_CODE_V1, + MATERIALIZATION_POLICY_GIT_RAW_V1, METADATA_CODEC, SCHEMA_VERSION, + }, + metapage::{Entry, EntryKind, HEADER_LEN, PAGE_MAX_BYTES, Page, page_id}, +}; +use sea_orm::ActiveValue::Set; +use sha2::{Digest, Sha256}; + +use super::{ + error::{SnapshotError, SnapshotErrorCode}, + metadata_install::MetadataInstallIdentity, + projection_observation::NATIVE_PROJECTION_REVISION, + resolver::{FsKind, direct_entries}, + retention_dag::{MetadataDagLimits, MetadataPageId, MetadataPagePayload}, + rooted_metadata_install::{RootedMetadataInstallPlan, RootedReuseRoot}, + view::validate_scope_relative_path, +}; +use crate::{ + ceres::api_service::ApiHandler, + common::errors::MegaError, + jupiter::storage::{Storage, mono_storage::MST2_VERIFICATION_VERSION}, +}; + +#[async_trait] +pub(crate) trait RootedReuseLookup: Send + Sync { + async fn lookup_reuse( + &self, + tree_oid: &str, + identity: &MetadataInstallIdentity, + ) -> Result, SnapshotError>; +} + +#[derive(Debug, Clone, PartialEq, Eq)] +pub(crate) struct CertifiedReusableDirectory { + pub(crate) page_id: MetadataPageId, + pub(crate) proof: RootedReuseRoot, + pub(crate) relative_path_bytes: usize, + pub(crate) relative_components: usize, + pub(crate) closure_nodes_upper: usize, + pub(crate) closure_edges_upper: usize, + pub(crate) closure_bytes_upper: u64, + pub(crate) closure_entries_upper: usize, +} + +/// API calls and materialized outputs only, not total database or codec work. +#[derive(Debug, Clone, Default, PartialEq, Eq, serde::Serialize)] +pub(crate) struct RootedProjectionWork { + pub(crate) input_tree_entries: u64, + pub(crate) scope_path_entries_examined: u64, + pub(crate) tree_fetches: u64, + pub(crate) reuse_lookups: u64, + pub(crate) directory_memo_hits: u64, + pub(crate) directories_scanned: u64, + pub(crate) direct_entries_scanned: u64, + pub(crate) verified_blob_queries: u64, + pub(crate) verified_blob_facts_loaded: u64, + pub(crate) verified_blob_persistence_batches: u64, + pub(crate) blob_fetches: u64, + pub(crate) raw_bytes_fetched: u64, + pub(crate) raw_bytes_hashed: u64, + pub(crate) directory_root_builds: u64, + pub(crate) radix_route_builds: u64, + pub(crate) codec_input_entries: u64, + pub(crate) delta_pages: u64, + pub(crate) delta_bytes: u64, + pub(crate) reused_roots: u64, + pub(crate) reused_closure_nodes_upper: u64, + pub(crate) reused_closure_edges_upper: u64, + pub(crate) reused_closure_bytes_upper: u64, + pub(crate) reused_closure_entries_upper: u64, + pub(crate) plan_bytes: u64, +} + +#[derive(Debug)] +pub(crate) struct PreparedRootedNativeMetadata { + pub(crate) plan: RootedMetadataInstallPlan, + pub(crate) payloads: Vec, + pub(crate) work: RootedProjectionWork, +} + +pub(crate) async fn prepare_rooted_native_metadata< + T: ApiHandler + ?Sized, + R: RootedReuseLookup + ?Sized, +>( + handler: &T, + root_tree: &Tree, + scope: &str, + reuse: &R, +) -> Result { + validate_scope_relative_path(scope)?; + if !handler.native_snapshot_projection() { + return Err(SnapshotError::new( + SnapshotErrorCode::ScopeInvalid, + "rooted metadata projection requires native Git source semantics", + )); + } + let identity = MetadataInstallIdentity { + source_domain: "native-git".into(), + tagged_root_tree_oid: root_tree.id.to_tagged_string(), + scope: scope.to_owned(), + schema_version: SCHEMA_VERSION, + metadata_codec: METADATA_CODEC, + materialization_policy: MATERIALIZATION_POLICY_GIT_RAW_V1, + fs_semantics: FS_SEMANTICS_LINUX_CODE_V1, + access_projection: ACCESS_PROJECTION_EXACT_FULL, + verification_revision: MST2_VERIFICATION_VERSION, + projection_revision: NATIVE_PROJECTION_REVISION, + }; + let storage = handler.get_context(); + let mut state = ProjectionState::new(identity); + state.work.input_tree_entries = root_tree.tree_items.len() as u64; + let scoped_oid = resolve_scope_oid(handler, root_tree, scope, &mut state.work).await?; + let loaded = (scope == "/").then_some(root_tree); + let root = project_directory( + handler, reuse, &storage, scoped_oid, loaded, scope, &mut state, + ) + .await?; + state.finish(root.page_id) +} + +#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)] +struct PathBounds { + bytes: usize, + components: usize, +} + +impl PathBounds { + fn include(&mut self, name: &str, child: Self) -> Result<(), SnapshotError> { + let bytes = name + .len() + .checked_add(1) + .and_then(|value| value.checked_add(child.bytes)) + .filter(|value| *value <= 4096) + .ok_or_else(path_limit)?; + let components = child + .components + .checked_add(1) + .filter(|value| *value <= 256) + .ok_or_else(path_limit)?; + self.bytes = self.bytes.max(bytes); + self.components = self.components.max(components); + Ok(()) + } + + fn validate_at(self, prefix: &str) -> Result<(), SnapshotError> { + validate_scope_relative_path(prefix)?; + let (bytes, components) = if prefix == "/" { + (0, 0) + } else { + (prefix.len(), prefix[1..].split('/').count()) + }; + bytes + .checked_add(self.bytes) + .filter(|value| *value <= 4096) + .ok_or_else(path_limit)?; + components + .checked_add(self.components) + .filter(|value| *value <= 256) + .ok_or_else(path_limit)?; + Ok(()) + } +} + +#[derive(Debug, Clone, Copy)] +struct DirectorySummary { + page_id: MetadataPageId, + bounds: PathBounds, +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +struct BlobFact { + size: u64, + digest: [u8; 32], +} + +#[derive(Default)] +struct ClosureUpper { + nodes: usize, + edges: usize, + bytes: u64, + entries: usize, +} + +impl ClosureUpper { + fn add(&mut self, directory: &CertifiedReusableDirectory) -> Result<(), SnapshotError> { + self.nodes = self + .nodes + .checked_add(directory.closure_nodes_upper) + .ok_or_else(budget_limit)?; + self.edges = self + .edges + .checked_add(directory.closure_edges_upper) + .ok_or_else(budget_limit)?; + self.bytes = self + .bytes + .checked_add(directory.closure_bytes_upper) + .ok_or_else(budget_limit)?; + self.entries = self + .entries + .checked_add(directory.closure_entries_upper) + .ok_or_else(budget_limit)?; + Ok(()) + } +} + +struct ProjectionState { + identity: MetadataInstallIdentity, + required_trees: BTreeSet, + active_trees: BTreeSet, + directories: BTreeMap, + bounds: BTreeMap, + payloads: BTreeMap, + page_entries: BTreeMap, + edges: BTreeSet<(MetadataPageId, MetadataPageId)>, + reused: BTreeMap, + source_roots: BTreeMap, + blob_facts: BTreeMap, + source_entries: usize, + source_encoded_bytes: u64, + codec_entry_visits: usize, + payload_bytes: u64, + delta_entries: usize, + reuse_upper: ClosureUpper, + work: RootedProjectionWork, +} + +impl ProjectionState { + fn new(identity: MetadataInstallIdentity) -> Self { + Self { + identity, + required_trees: BTreeSet::new(), + active_trees: BTreeSet::new(), + directories: BTreeMap::new(), + bounds: BTreeMap::new(), + payloads: BTreeMap::new(), + page_entries: BTreeMap::new(), + edges: BTreeSet::new(), + reused: BTreeMap::new(), + source_roots: BTreeMap::new(), + blob_facts: BTreeMap::new(), + source_entries: 0, + source_encoded_bytes: 0, + codec_entry_visits: 0, + payload_bytes: 0, + delta_entries: 0, + reuse_upper: ClosureUpper::default(), + work: RootedProjectionWork::default(), + } + } + + fn require_tree(&mut self, tree_oid: &str) -> Result<(), SnapshotError> { + if !self.required_trees.contains(tree_oid) { + if self.required_trees.len() >= MetadataDagLimits::default().nodes { + return Err(budget_limit()); + } + self.required_trees.insert(tree_oid.to_owned()); + } + Ok(()) + } + + fn admit_entries(&mut self, entries: &[(String, FsKind, String)]) -> Result<(), SnapshotError> { + let limits = MetadataDagLimits::default(); + self.source_encoded_bytes = self + .source_encoded_bytes + .checked_add(HEADER_LEN as u64) + .filter(|value| *value <= limits.payload_bytes) + .ok_or_else(budget_limit)?; + let mut previous: Option<&str> = None; + for (name, kind, _) in entries { + if name.is_empty() + || name.len() > 255 + || name == "." + || name == ".." + || name.contains('/') + || name.contains('\0') + || previous.is_some_and(|value| value >= name.as_str()) + { + return Err(integrity( + "source tree has an invalid or duplicate UTF-8 name", + )); + } + previous = Some(name); + let value_bytes = if *kind == FsKind::Directory { 32 } else { 40 }; + self.source_encoded_bytes = self + .source_encoded_bytes + .checked_add((3 + name.len() + value_bytes) as u64) + .filter(|value| *value <= limits.payload_bytes) + .ok_or_else(budget_limit)?; + } + Ok(()) + } + + fn record_bounds( + &mut self, + page: MetadataPageId, + bounds: PathBounds, + ) -> Result<(), SnapshotError> { + if self + .bounds + .get(&page) + .is_some_and(|previous| previous != &bounds) + { + return Err(integrity( + "one canonical directory has inconsistent descendant path bounds", + )); + } + self.bounds.insert(page, bounds); + Ok(()) + } + + fn record_reuse( + &mut self, + tree_oid: &str, + directory: CertifiedReusableDirectory, + path: &str, + ) -> Result { + let limits = MetadataDagLimits::default(); + if directory.page_id == [0; 32] + || directory.proof.generation <= 0 + || directory.closure_nodes_upper == 0 + || directory.closure_nodes_upper > limits.nodes + || directory.closure_edges_upper > limits.edges + || !(HEADER_LEN as u64..=limits.payload_bytes).contains(&directory.closure_bytes_upper) + || directory.closure_entries_upper > limits.entries + || directory.relative_path_bytes > 4096 + || directory.relative_components > 256 + { + return Err(integrity( + "reused directory hint has an invalid certificate summary", + )); + } + let summary = DirectorySummary { + page_id: directory.page_id, + bounds: PathBounds { + bytes: directory.relative_path_bytes, + components: directory.relative_components, + }, + }; + summary.bounds.validate_at(path)?; + self.record_bounds(directory.page_id, summary.bounds)?; + if !self.payloads.contains_key(&directory.page_id) { + if let Some(previous) = self.reused.get(&directory.page_id) { + if previous.proof.generation != directory.proof.generation + || previous.proof.certificate_digest != directory.proof.certificate_digest + || previous.closure_nodes_upper != directory.closure_nodes_upper + || previous.closure_edges_upper != directory.closure_edges_upper + || previous.closure_bytes_upper != directory.closure_bytes_upper + || previous.closure_entries_upper != directory.closure_entries_upper + { + return Err(integrity( + "one reused boundary has inconsistent lifetime or certificate summaries", + )); + } + } else { + self.reuse_upper.add(&directory)?; + self.reused.insert(directory.page_id, directory); + self.check_budget()?; + } + } + self.source_roots + .insert(tree_oid.to_owned(), summary.page_id); + Ok(summary) + } + + fn charge_codec(&mut self, entries: usize, route_length: usize) -> Result<(), SnapshotError> { + let count = entries + .checked_mul(route_length.checked_add(1).ok_or_else(budget_limit)?) + .ok_or_else(budget_limit)?; + self.codec_entry_visits = self + .codec_entry_visits + .checked_add(count) + .filter(|value| *value <= MetadataDagLimits::default().prepare_entry_visits) + .ok_or_else(budget_limit)?; + self.work.codec_input_entries = self + .work + .codec_input_entries + .checked_add(entries as u64) + .ok_or_else(budget_limit)?; + Ok(()) + } + + fn add_edge( + &mut self, + parent: MetadataPageId, + child: MetadataPageId, + ) -> Result<(), SnapshotError> { + if self.edges.insert((parent, child)) { + self.check_budget()?; + } + Ok(()) + } + + fn collect_directory( + &mut self, + entries: &[Entry], + root_bytes: Vec, + ) -> Result { + let root = page_id(&root_bytes); + let mut routes = VecDeque::from([(Vec::::new(), root, Some(root_bytes))]); + while let Some((route, expected, provided)) = routes.pop_front() { + if self.payloads.contains_key(&expected) || self.reused.contains_key(&expected) { + continue; + } + let bytes = if let Some(bytes) = provided { + bytes + } else { + self.charge_codec(entries.len(), route.len())?; + self.work.radix_route_builds += 1; + Page::pages_along_route(entries, &route) + .map_err(codec_error)? + .pop() + .ok_or_else(|| integrity("canonical radix route returned no page"))? + }; + if page_id(&bytes) != expected || !(HEADER_LEN..=PAGE_MAX_BYTES).contains(&bytes.len()) + { + return Err(integrity( + "canonical radix page differs from its parent binding", + )); + } + let (page, _) = Page::decode(&bytes).map_err(codec_error)?; + let page_entries = match page { + Page::Leaf { entries } => { + for entry in entries.iter().filter(|entry| entry.is_dir()) { + self.add_edge(expected, entry.child_root)?; + } + entries.len() + } + Page::Branch { + terminal, children, .. + } => { + if let Some(entry) = terminal.as_ref().filter(|entry| entry.is_dir()) { + self.add_edge(expected, entry.child_root)?; + } + for child in children { + self.add_edge(expected, child.child_page_id)?; + let mut child_route = route.clone(); + child_route.push(child.label); + routes.push_back((child_route, child.child_page_id, None)); + } + usize::from(terminal.is_some()) + } + }; + self.payload_bytes = self + .payload_bytes + .checked_add(bytes.len() as u64) + .ok_or_else(budget_limit)?; + self.delta_entries = self + .delta_entries + .checked_add(page_entries) + .ok_or_else(budget_limit)?; + self.page_entries.insert(expected, page_entries); + self.payloads.insert( + expected, + MetadataPagePayload { + id: expected, + size: bytes.len() as u64, + bytes, + }, + ); + self.check_budget()?; + } + Ok(root) + } + + fn check_budget(&self) -> Result<(), SnapshotError> { + let limits = MetadataDagLimits::default(); + if self + .payloads + .len() + .checked_add(self.reuse_upper.nodes) + .is_none_or(|value| value > limits.nodes) + || self + .edges + .len() + .checked_add(self.reuse_upper.edges) + .is_none_or(|value| value > limits.edges) + || self + .payload_bytes + .checked_add(self.reuse_upper.bytes) + .is_none_or(|value| value > limits.payload_bytes) + || self + .delta_entries + .checked_add(self.reuse_upper.entries) + .is_none_or(|value| value > limits.entries) + { + return Err(budget_limit()); + } + Ok(()) + } + + fn finish( + mut self, + root: MetadataPageId, + ) -> Result { + let mut children: BTreeMap<_, Vec<_>> = BTreeMap::new(); + for &(parent, child) in &self.edges { + children.entry(parent).or_default().push(child); + } + let mut reachable = BTreeSet::new(); + let mut pending = vec![root]; + while let Some(page) = pending.pop() { + if reachable.insert(page) && !self.reused.contains_key(&page) { + pending.extend(children.get(&page).into_iter().flatten()); + } + } + self.payloads.retain(|page, _| reachable.contains(page)); + self.page_entries.retain(|page, _| reachable.contains(page)); + self.reused.retain(|page, _| reachable.contains(page)); + self.source_roots.retain(|_, page| reachable.contains(page)); + self.edges + .retain(|(parent, child)| reachable.contains(parent) && reachable.contains(child)); + self.payload_bytes = self + .payloads + .values() + .try_fold(0u64, |total, page| total.checked_add(page.size)) + .ok_or_else(budget_limit)?; + self.delta_entries = self + .page_entries + .values() + .try_fold(0usize, |total, count| total.checked_add(*count)) + .ok_or_else(budget_limit)?; + self.reuse_upper = ClosureUpper::default(); + for directory in self.reused.values() { + self.reuse_upper.add(directory)?; + } + self.check_budget()?; + let plan = RootedMetadataInstallPlan::new( + self.identity, + root, + self.payloads + .iter() + .map(|(page, payload)| (*page, payload.size)) + .collect(), + self.edges, + self.reused + .into_iter() + .map(|(page, directory)| (page, directory.proof)) + .collect(), + self.source_roots, + )?; + self.work.delta_pages = plan.delta.len() as u64; + self.work.delta_bytes = self.payload_bytes; + self.work.reused_roots = plan.reused.len() as u64; + self.work.reused_closure_nodes_upper = self.reuse_upper.nodes as u64; + self.work.reused_closure_edges_upper = self.reuse_upper.edges as u64; + self.work.reused_closure_bytes_upper = self.reuse_upper.bytes; + self.work.reused_closure_entries_upper = self.reuse_upper.entries as u64; + self.work.plan_bytes = plan.encode()?.len() as u64; + Ok(PreparedRootedNativeMetadata { + plan, + payloads: self.payloads.into_values().collect(), + work: self.work, + }) + } +} + +async fn resolve_scope_oid( + handler: &T, + root: &Tree, + scope: &str, + work: &mut RootedProjectionWork, +) -> Result { + if scope == "/" { + return Ok(root.id); + } + let mut current = Cow::Borrowed(root); + let mut components = scope[1..].split('/').peekable(); + while let Some(component) = components.next() { + let position = current + .tree_items + .iter() + .position(|item| item.name == component); + work.scope_path_entries_examined += + position.map_or(current.tree_items.len(), |index| index + 1) as u64; + let item = position + .map(|index| ¤t.tree_items[index]) + .ok_or_else(|| { + SnapshotError::new( + SnapshotErrorCode::PathNotFound, + "name absent in enumerated parent directory", + ) + })?; + let kind = FsKind::from_git_mode(item.mode).ok_or_else(|| { + SnapshotError::new( + SnapshotErrorCode::UnsupportedEntry, + "gitlink entries are not supported in this profile", + ) + })?; + if kind != FsKind::Directory { + return Err(SnapshotError::new( + SnapshotErrorCode::NotDirectory, + "rooted metadata scope is not a directory", + )); + } + let expected = item.id; + if expected.kind() != root.id.kind() { + return Err(integrity("scope tree crossed its source hash kind")); + } + if components.peek().is_none() { + return Ok(expected); + } + work.tree_fetches += 1; + let fetched = handler + .get_tree_by_hash(&expected.to_string()) + .await + .map_err(source_error)?; + if fetched.id != expected { + return Err(integrity("fetched scope ancestor identity mismatch")); + } + current = Cow::Owned(fetched); + } + Err(integrity("rooted metadata scope walk has no target")) +} + +async fn project_directory( + handler: &T, + reuse: &R, + storage: &Storage, + oid: ObjectHash, + loaded: Option<&Tree>, + path: &str, + state: &mut ProjectionState, +) -> Result { + validate_scope_relative_path(path)?; + let tagged = oid.to_tagged_string(); + if let Some(summary) = state.directories.get(&tagged).copied() { + summary.bounds.validate_at(path)?; + state.work.directory_memo_hits += 1; + return Ok(summary); + } + state.require_tree(&tagged)?; + if !state.active_trees.insert(tagged.clone()) { + return Err(integrity("native source trees contain a cycle")); + } + state.work.reuse_lookups += 1; + if let Some(hint) = reuse.lookup_reuse(&tagged, &state.identity).await? { + let summary = state.record_reuse(&tagged, hint, path)?; + state.active_trees.remove(&tagged); + state.directories.insert(tagged, summary); + return Ok(summary); + } + let fetched; + let tree = if let Some(tree) = loaded { + tree + } else { + state.work.tree_fetches += 1; + fetched = handler + .get_tree_by_hash(&oid.to_string()) + .await + .map_err(source_error)?; + &fetched + }; + if tree.id != oid + || tree + .tree_items + .iter() + .any(|item| item.id.kind() != oid.kind()) + { + return Err(integrity( + "fetched source directory identity or child hash kind mismatch", + )); + } + state.source_entries = state + .source_entries + .checked_add(tree.tree_items.len()) + .filter(|value| *value <= MetadataDagLimits::default().entries) + .ok_or_else(budget_limit)?; + let direct = direct_entries(tree)?; + state.admit_entries(&direct)?; + state.work.directories_scanned += 1; + state.work.direct_entries_scanned += direct.len() as u64; + let requested: Vec<_> = direct + .iter() + .filter(|(_, kind, oid)| *kind != FsKind::Directory && !state.blob_facts.contains_key(oid)) + .map(|(_, _, oid)| oid.clone()) + .collect::>() + .into_iter() + .collect(); + for batch in requested.chunks(64) { + state.work.verified_blob_queries += 1; + let facts = storage + .mono_storage() + .get_verified_blobs(batch.to_vec()) + .await + .map_err(source_error)?; + state.work.verified_blob_facts_loaded += facts.len() as u64; + for (oid, fact) in facts { + state.blob_facts.insert(oid, verified_fact(&fact)?); + } + } + let mut pending_facts = Vec::new(); + let mut bounds = PathBounds::default(); + let mut entries = Vec::with_capacity(direct.len()); + for (name, kind, raw_oid) in direct { + let child_path = if path == "/" { + format!("/{name}") + } else { + format!("{path}/{name}") + }; + validate_scope_relative_path(&child_path)?; + match kind { + FsKind::Directory => { + let child_oid = ObjectHash::from_hex_for_kind(oid.kind(), &raw_oid) + .map_err(|error| integrity(&error.to_string()))?; + let child = Box::pin(project_directory( + handler, + reuse, + storage, + child_oid, + None, + &child_path, + state, + )) + .await?; + bounds.include(&name, child.bounds)?; + entries.push(Entry::dir(name.as_bytes(), child.page_id)); + } + FsKind::Regular | FsKind::Executable | FsKind::Symlink => { + bounds.include(&name, PathBounds::default())?; + let fact = if let Some(fact) = state.blob_facts.get(&raw_oid).copied() { + fact + } else { + state.work.blob_fetches += 1; + let raw = super::pages::fetch_raw_blob(handler, &raw_oid).await?; + state.work.raw_bytes_fetched = state + .work + .raw_bytes_fetched + .checked_add(raw.len() as u64) + .ok_or_else(budget_limit)?; + let size = u64::try_from(raw.len()) + .map_err(|_| integrity("raw file size is not representable"))?; + if size > 8_796_093_022_208 { + return Err(integrity( + "raw file exceeds the native verified-object size profile", + )); + } + let digest: [u8; 32] = Sha256::digest(&raw).into(); + state.work.raw_bytes_hashed = state + .work + .raw_bytes_hashed + .checked_add(size) + .ok_or_else(budget_limit)?; + drop(raw); + let fact = BlobFact { size, digest }; + state.blob_facts.insert(raw_oid.clone(), fact); + pending_facts.push((raw_oid.clone(), fact)); + if pending_facts.len() == 64 { + persist_facts(storage, &pending_facts, &mut state.work).await?; + pending_facts.clear(); + } + fact + }; + let entry_kind = match kind { + FsKind::Regular => EntryKind::Regular, + FsKind::Executable => EntryKind::Executable, + FsKind::Symlink => EntryKind::Symlink, + FsKind::Directory => { + return Err(integrity("file projection received a directory kind")); + } + }; + entries.push(Entry::file( + entry_kind, + name.as_bytes(), + fact.size, + fact.digest, + )); + } + } + } + if !pending_facts.is_empty() { + persist_facts(storage, &pending_facts, &mut state.work).await?; + } + bounds.validate_at(path)?; + state.charge_codec(entries.len(), 0)?; + state.work.directory_root_builds += 1; + let root_bytes = Page::build(&entries).map_err(codec_error)?; + let page_id = state.collect_directory(&entries, root_bytes)?; + state.record_bounds(page_id, bounds)?; + state.source_roots.insert(tagged.clone(), page_id); + let summary = DirectorySummary { page_id, bounds }; + state.active_trees.remove(&tagged); + state.directories.insert(tagged, summary); + Ok(summary) +} + +fn verified_fact( + fact: &crate::callisto::mst2_verified_object::Model, +) -> Result { + if fact.state != "VERIFIED" + || fact.verification_version != MST2_VERIFICATION_VERSION + || !(0..=8_796_093_022_208).contains(&fact.size) + { + return Err(integrity( + "native projection received an invalid current verified blob fact", + )); + } + Ok(BlobFact { + size: u64::try_from(fact.size).map_err(|_| integrity("verified blob size is invalid"))?, + digest: fact + .raw_sha256 + .as_slice() + .try_into() + .map_err(|_| integrity("verified blob digest is invalid"))?, + }) +} + +async fn persist_facts( + storage: &Storage, + facts: &[(String, BlobFact)], + work: &mut RootedProjectionWork, +) -> Result<(), SnapshotError> { + let rows = facts + .iter() + .map( + |(oid, fact)| crate::callisto::mst2_verified_object::ActiveModel { + id: sea_orm::ActiveValue::NotSet, + storage_domain: Set("git".into()), + git_oid: Set(oid.clone()), + object_kind: Set("blob".into()), + raw_sha256: Set(fact.digest.to_vec()), + size: Set(fact.size as i64), + verification_version: Set(MST2_VERIFICATION_VERSION), + state: Set("VERIFIED".into()), + created_at: Set(chrono::Utc::now().fixed_offset()), + }, + ) + .collect(); + work.verified_blob_persistence_batches += 1; + storage + .mono_storage() + .insert_verified_blobs(rows) + .await + .map_err(source_error)?; + work.verified_blob_queries += 1; + let stored = storage + .mono_storage() + .get_verified_blobs(facts.iter().map(|(oid, _)| oid.clone()).collect()) + .await + .map_err(source_error)?; + work.verified_blob_facts_loaded += stored.len() as u64; + for (oid, expected) in facts { + let actual = stored + .get(oid) + .ok_or_else(|| integrity("verified blob persistence lost its current fact"))?; + if verified_fact(actual)? != *expected { + return Err(integrity( + "verified blob persistence conflicts with the hashed raw source", + )); + } + } + Ok(()) +} + +fn source_error(error: MegaError) -> SnapshotError { + let code = match error { + MegaError::ObjStorageNotFound(_) => SnapshotErrorCode::ObjectUnavailable, + MegaError::ObjStorageInconsistent(_) => SnapshotErrorCode::IntegrityError, + _ => SnapshotErrorCode::Internal, + }; + tracing::warn!(error = %error, "rooted native metadata source operation failed"); + SnapshotError::new(code, "rooted native metadata source operation failed") +} + +fn codec_error(error: mst2_codec::CodecError) -> SnapshotError { + integrity(&error.to_string()) +} +fn integrity(message: &str) -> SnapshotError { + SnapshotError::new(SnapshotErrorCode::IntegrityError, message) +} +fn budget_limit() -> SnapshotError { + SnapshotError::new( + SnapshotErrorCode::LimitExceeded, + "rooted metadata projection exceeds its fixed budget", + ) +} +fn path_limit() -> SnapshotError { + SnapshotError::new( + SnapshotErrorCode::ScopeInvalid, + "rooted metadata subtree exceeds its full-path budget", + ) +} + +#[cfg(test)] +mod tests { + use git_internal::hash::HashKind; + use uuid::Uuid; + + use super::*; + + fn state() -> ProjectionState { + ProjectionState::new(MetadataInstallIdentity { + source_domain: "native-git".into(), + tagged_root_tree_oid: ObjectHash::from_hex_for_kind(HashKind::Sha1, &"a".repeat(40)) + .unwrap() + .to_tagged_string(), + scope: "/".into(), + schema_version: SCHEMA_VERSION, + metadata_codec: METADATA_CODEC, + materialization_policy: MATERIALIZATION_POLICY_GIT_RAW_V1, + fs_semantics: FS_SEMANTICS_LINUX_CODE_V1, + access_projection: ACCESS_PROJECTION_EXACT_FULL, + verification_revision: MST2_VERIFICATION_VERSION, + projection_revision: NATIVE_PROJECTION_REVISION, + }) + } + + fn reusable(page: MetadataPageId) -> CertifiedReusableDirectory { + CertifiedReusableDirectory { + page_id: page, + proof: RootedReuseRoot { + generation: 1, + attestation_id: Uuid::from_u128(1), + attestation_digest: [1; 32], + certificate_digest: [2; 32], + }, + relative_path_bytes: 0, + relative_components: 0, + closure_nodes_upper: 1, + closure_edges_upper: 0, + closure_bytes_upper: HEADER_LEN as u64, + closure_entries_upper: 0, + } + } + + #[test] + fn reused_path_bounds_recheck_moved_aliases_and_checked_overflow() { + let bounds = PathBounds { + bytes: 4093, + components: 255, + }; + assert!(bounds.validate_at("/").is_ok()); + assert!(bounds.validate_at("/a").is_ok()); + assert_eq!( + bounds.validate_at("/a/b").unwrap_err().code, + SnapshotErrorCode::ScopeInvalid + ); + assert!( + PathBounds { + bytes: usize::MAX, + components: 0 + } + .validate_at("/a") + .is_err() + ); + let mut parent = PathBounds::default(); + parent + .include( + "a", + PathBounds { + bytes: 3, + components: 1, + }, + ) + .unwrap(); + parent.include("longer", PathBounds::default()).unwrap(); + assert_eq!( + parent, + PathBounds { + bytes: 7, + components: 2 + } + ); + assert!( + parent + .include( + "a", + PathBounds { + bytes: usize::MAX, + components: 0 + } + ) + .is_err() + ); + } + + #[test] + fn canonical_radix_payloads_and_shared_reuse_edges_form_a_delta_only_plan() { + let empty = Page::build(&[]).unwrap(); + let reused_page = page_id(&empty); + let mut state = state(); + state + .record_reuse( + &format!("sha1:{}", "b".repeat(40)), + reusable(reused_page), + "/left", + ) + .unwrap(); + let mut entries: Vec<_> = (0..260) + .map(|index| { + Entry::file( + EntryKind::Regular, + format!("file-{index:03}").as_bytes(), + index, + [3; 32], + ) + }) + .collect(); + entries.push(Entry::dir(b"left", reused_page)); + entries.push(Entry::dir(b"right", reused_page)); + entries.sort_by(|left, right| left.name.cmp(&right.name)); + let bytes = Page::build(&entries).unwrap(); + let root = state.collect_directory(&entries, bytes).unwrap(); + state + .source_roots + .insert(state.identity.tagged_root_tree_oid.clone(), root); + let prepared = state.finish(root).unwrap(); + assert!(prepared.payloads.len() > 1); + assert_eq!(prepared.plan.reused.len(), 1); + assert!(!prepared.plan.delta.contains_key(&reused_page)); + assert_eq!(prepared.work.tree_fetches, 0); + assert_eq!(prepared.work.blob_fetches, 0); + assert_eq!( + prepared.plan.child_first_delta().unwrap().last(), + Some(&root) + ); + let mut decoded_edges = BTreeSet::new(); + for payload in &prepared.payloads { + assert_eq!(page_id(&payload.bytes), payload.id); + match Page::decode(&payload.bytes).unwrap().0 { + Page::Leaf { entries } => { + for entry in entries.iter().filter(|entry| entry.is_dir()) { + decoded_edges.insert((payload.id, entry.child_root)); + } + } + Page::Branch { + terminal, children, .. + } => { + if let Some(entry) = terminal.filter(Entry::is_dir) { + decoded_edges.insert((payload.id, entry.child_root)); + } + for child in children { + decoded_edges.insert((payload.id, child.child_page_id)); + } + } + } + } + assert_eq!(decoded_edges, prepared.plan.edges); + assert_eq!( + RootedMetadataInstallPlan::decode( + &prepared.plan.encode().unwrap(), + &prepared.plan.digest().unwrap() + ) + .unwrap(), + prepared.plan + ); + } + + #[test] + fn reuse_budget_is_deduplicated_conservative_and_never_expands_descendants() { + let empty = Page::build(&[]).unwrap(); + let page = page_id(&empty); + let hint = reusable(page); + let mut state = state(); + state + .record_reuse(&format!("sha1:{}", "b".repeat(40)), hint.clone(), "/one") + .unwrap(); + let mut alias = hint.clone(); + alias.proof.attestation_id = Uuid::from_u128(2); + alias.proof.attestation_digest = [9; 32]; + state + .record_reuse(&format!("sha1:{}", "c".repeat(40)), alias, "/two") + .unwrap(); + assert_eq!(state.reuse_upper.nodes, 1); + assert_eq!(state.reuse_upper.bytes, HEADER_LEN as u64); + assert_eq!(state.reused.get(&page).unwrap().proof, hint.proof); + assert!(state.payloads.is_empty()); + assert!(state.edges.is_empty()); + let mut different_lifetime = hint.clone(); + different_lifetime.proof.generation = 2; + assert!( + state + .record_reuse( + &format!("sha1:{}", "f".repeat(40)), + different_lifetime, + "/lifetime" + ) + .is_err() + ); + let mut bad = hint; + bad.relative_path_bytes = 1; + assert!( + state + .record_reuse(&format!("sha1:{}", "d".repeat(40)), bad, "/three") + .is_err() + ); + let mut huge = reusable([4; 32]); + huge.closure_nodes_upper = MetadataDagLimits::default().nodes; + assert_eq!( + state + .record_reuse(&format!("sha1:{}", "e".repeat(40)), huge, "/four") + .unwrap_err() + .code, + SnapshotErrorCode::LimitExceeded + ); + } +} diff --git a/src/jupiter/migration/m20261008_000200_add_mst2_rooted_qualified_family.rs b/src/jupiter/migration/m20261008_000200_add_mst2_rooted_qualified_family.rs new file mode 100644 index 00000000..fd613dce --- /dev/null +++ b/src/jupiter/migration/m20261008_000200_add_mst2_rooted_qualified_family.rs @@ -0,0 +1,125 @@ +//! A bounded namespace registry; physical Q provisioning occurs at bootstrap. + +use sea_orm::{ConnectionTrait, DbBackend, Statement}; +use sea_orm_migration::prelude::*; + +#[derive(DeriveMigrationName)] +pub struct Migration; + +#[async_trait::async_trait] +impl MigrationTrait for Migration { + async fn up(&self, manager: &SchemaManager) -> Result<(), DbErr> { + let connection = manager.get_connection(); + let row = connection.query_one_raw(Statement::from_string(DbBackend::Postgres, + "SELECT current_schema() AS schema,n.oid::bigint AS oid FROM pg_catalog.pg_namespace n WHERE n.nspname=current_schema()")) + .await?.ok_or_else(|| DbErr::Custom("qualified family core schema is missing".into()))?; + let schema: String = row.try_get("", "schema")?; + let oid: i64 = row.try_get("", "oid")?; + let quoted = format!("\"{}\"", schema.replace('"', "\"\"")); + let literal = format!("'{}'", schema.replace('\'', "''")); + let old = include_str!("m20261007_000600_storage_routes.sql"); + let start = old + .find("CREATE FUNCTION mst2_route_insert_guard()") + .ok_or_else(|| { + DbErr::Custom("generic route insertion guard source is missing".into()) + })?; + let end = old[start..] + .find("CREATE FUNCTION mst2_route_context_insert()") + .ok_or_else(|| { + DbErr::Custom("generic route insertion guard boundary is missing".into()) + })? + + start; + let guard = old[start..end].replace("CREATE FUNCTION", "CREATE OR REPLACE FUNCTION") + .replace("WHERE s.snapshot_id=NEW.snapshot_id AND NEW.canonical_descriptor=s.canonical_descriptor", + "WHERE n.singleton=1 AND n.graph_domain='generic-v1' AND s.snapshot_id=NEW.snapshot_id AND NEW.canonical_descriptor=s.canonical_descriptor"); + let catalog = include_str!("../storage/qualified_family_catalog.sql") + .replace("$CORE_OID$", "c_oid") + .replace("$Q_OID$", "q_oid") + .replace("$EXEMPT_Q_OID$", "exempt_q_oid"); + let shape = include_str!("../storage/qualified_family_shape.sql"); + let implementation = hex::encode( + super::super::storage::qualified_metadata_family::implementation_fingerprint(), + ); + let sql = include_str!("m20261008_000200_rooted_qualified_family.sql") + .replace("$CATALOG_SQL$", &catalog) + .replace("$SHAPE_SQL$", shape) + .replace("$GENERIC_INSERT_GUARD$", &guard) + .replace( + "$SOURCE_REVISION_SQL$", + include_str!("../storage/qualified_source_revision.sql"), + ) + .replace( + "$ROOTED_ROUTES_SQL$", + include_str!("m20261008_000200_rooted_routes.sql"), + ) + .replace("$CORE_SCHEMA$", "ed) + .replace("$CORE_LITERAL$", &literal) + .replace("$IMPLEMENTATION_SHA$", &implementation) + .replace("$CORE_OID$", &oid.to_string()); + connection.execute_unprepared(&sql).await?; + // Build the expected physical shape from trusted source exactly once, + // under this forward migration, without registering a namespace or + // preserving any template relations/history after the transaction. + let template_uuid = uuid::Uuid::new_v4().to_string(); + let template_schema = format!("mst2q_{}", template_uuid.replace('-', "")); + let template_quoted = format!("\"{template_schema}\""); + connection + .execute_unprepared(&format!("CREATE SCHEMA {template_quoted}")) + .await?; + let template_oid: i64 = connection + .query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT oid::bigint AS oid FROM pg_catalog.pg_namespace WHERE nspname=$1", + [template_schema.clone().into()], + )) + .await? + .ok_or_else(|| DbErr::Custom("qualified family template schema is missing".into()))? + .try_get("", "oid")?; + let template_storage = uuid::Uuid::new_v4().to_string(); + connection + .execute_unprepared( + &super::super::storage::qualified_metadata_family::render_family( + &schema, + oid, + &template_schema, + template_oid, + &template_uuid, + &template_storage, + ), + ) + .await?; + let expected_shape: Vec = connection.query_one_raw(Statement::from_sql_and_values(DbBackend::Postgres, + format!("SELECT {quoted}.mst2_route_family_shape($1::bigint::oid,$2::uuid,$3) AS fingerprint"), + [template_oid.into(),template_uuid.into(),template_storage.into()])).await? + .ok_or_else(||DbErr::Custom("qualified family template shape is missing".into()))?.try_get("","fingerprint")?; + connection.execute_unprepared(&format!("SET LOCAL search_path={quoted},pg_catalog,pg_temp; DROP SCHEMA {template_quoted} CASCADE")).await?; + let authority_catalog: Vec = connection + .query_one_raw(Statement::from_string( + DbBackend::Postgres, + format!("SELECT {quoted}.mst2_route_family_catalog({oid},0::oid) AS fingerprint"), + )) + .await? + .ok_or_else(|| { + DbErr::Custom("qualified family core authority catalog is missing".into()) + })? + .try_get("", "fingerprint")?; + connection + .execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + format!("INSERT INTO {quoted}.mst2_qualified_family_policy VALUES(1,$1,$2,$3)"), + [ + hex::decode(implementation) + .map_err(|e| DbErr::Custom(e.to_string()))? + .into(), + expected_shape.into(), + authority_catalog.into(), + ], + )) + .await?; + Ok(()) + } + + async fn down(&self, _manager: &SchemaManager) -> Result<(), DbErr> { + Ok(()) + } +} diff --git a/src/jupiter/migration/m20261008_000200_rooted_qualified_family.sql b/src/jupiter/migration/m20261008_000200_rooted_qualified_family.sql new file mode 100644 index 00000000..22038ef2 --- /dev/null +++ b/src/jupiter/migration/m20261008_000200_rooted_qualified_family.sql @@ -0,0 +1,161 @@ +SELECT mst2_route_enter(current_schema()); +SET LOCAL search_path=$CORE_SCHEMA$,pg_catalog,pg_temp; +LOCK TABLE mst2_metadata_namespace IN ACCESS EXCLUSIVE MODE; + +ALTER TABLE mst2_metadata_namespace + DROP CONSTRAINT mst2_metadata_namespace_pkey, + ALTER COLUMN singleton DROP NOT NULL, + DROP CONSTRAINT mst2_metadata_namespace_family_identity_check, + DROP CONSTRAINT mst2_metadata_namespace_graph_domain_check, + DROP CONSTRAINT mst2_metadata_namespace_admission_state_check, + DROP CONSTRAINT mst2_metadata_namespace_collector_state_check, + DROP CONSTRAINT mst2_metadata_namespace_check, + ADD COLUMN metadata_storage_uuid text, + ADD COLUMN implementation_fingerprint bytea, + ADD COLUMN catalog_fingerprint bytea, + ADD CONSTRAINT mst2_namespace_family_pair CHECK (( + (singleton=1 AND family_identity='v3-generic-session-1' AND graph_domain='generic-v1' + AND admission_state='G_ADMITTED_Q_CLOSED' AND collector_state='CLOSED' AND core_schema=metadata_schema + AND core_schema_oid=metadata_schema_oid AND metadata_storage_uuid IS NULL + AND implementation_fingerprint IS NULL AND catalog_fingerprint IS NULL) + OR (singleton IS NULL AND family_identity='v3-rooted-qualified-1' AND graph_domain='qualified-v1' + AND admission_state='ROOTED_Q_ADMITTED' AND collector_state='ENABLED' AND core_schema<>metadata_schema + AND core_schema_oid<>metadata_schema_oid AND metadata_storage_uuid IS NOT NULL + AND octet_length(implementation_fingerprint)=32 AND octet_length(catalog_fingerprint)=32) + ) IS TRUE); +CREATE UNIQUE INDEX idx_mst2_namespace_generic_slot ON mst2_metadata_namespace(singleton) + WHERE singleton IS NOT NULL; +CREATE UNIQUE INDEX idx_mst2_namespace_domain ON mst2_metadata_namespace(graph_domain); +CREATE UNIQUE INDEX idx_mst2_namespace_metadata_schema ON mst2_metadata_namespace(metadata_schema_oid); + +CREATE OR REPLACE FUNCTION mst2_route_enter(caller_schema text) RETURNS uuid LANGUAGE plpgsql VOLATILE +SET search_path=$CORE_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE n mst2_metadata_namespace%ROWTYPE; g mst2_metadata_namespace%ROWTYPE; mono_held boolean; route_held boolean; +BEGIN + IF NOT mst2_route_scope_valid(caller_schema) OR (SELECT count(*) FROM mst2_metadata_namespace)>2 THEN + RAISE EXCEPTION 'storage route captured primary scope is unavailable'; + END IF; + SELECT * INTO STRICT g FROM mst2_metadata_namespace WHERE singleton=1; + mono_held:=mst2_route_lock_held(1297043024,g.mono_lock_key2); + route_held:=mst2_route_lock_held(1296718001,pg_catalog.hashtext(g.core_schema)); + IF mst2_route_lock_held(1296718001,pg_catalog.hashtext(g.core_schema),false) AND NOT mono_held THEN + RAISE EXCEPTION 'storage route lock was acquired before core mono'; + END IF; + IF EXISTS(SELECT 1 FROM mst2_metadata_namespace x + WHERE mst2_route_lock_held(1296717362,pg_catalog.hashtext(x.metadata_schema),false)) + AND NOT (mono_held AND route_held) THEN + RAISE EXCEPTION 'storage route cannot acquire core locks after retention'; + END IF; + PERFORM pg_catalog.set_config('lock_timeout','5000ms',true); + PERFORM pg_catalog.pg_advisory_xact_lock(1297043024,g.mono_lock_key2); + PERFORM pg_catalog.pg_advisory_xact_lock(1296718001,pg_catalog.hashtext(g.core_schema)); + FOR n IN SELECT * FROM mst2_metadata_namespace ORDER BY namespace_uuid LOOP + PERFORM pg_catalog.pg_advisory_xact_lock(1296717362,pg_catalog.hashtext(n.metadata_schema)); + END LOOP; + RETURN g.namespace_uuid; +END $$; + +-- The new namespace is included before acquiring any retention lock. +CREATE FUNCTION mst2_route_family_candidate_enter(caller_schema text,candidate uuid,candidate_schema text) +RETURNS void LANGUAGE plpgsql VOLATILE SET search_path=$CORE_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE g mst2_metadata_namespace%ROWTYPE; r record; +BEGIN + IF NOT mst2_route_scope_valid(caller_schema) OR candidate IS NULL + OR substr(candidate::text,15,1)<>'4' OR substr(candidate::text,20,1) NOT IN ('8','9','a','b') + OR candidate_schema IS DISTINCT FROM 'mst2q_'||replace(candidate::text,'-','') THEN + RAISE EXCEPTION 'qualified provisioning candidate is not a fresh server namespace'; + END IF; + IF EXISTS(SELECT 1 FROM pg_catalog.pg_locks WHERE locktype='advisory' AND pid=pg_catalog.pg_backend_pid() + AND database=(SELECT oid FROM pg_catalog.pg_database WHERE datname=pg_catalog.current_database()) + AND classid=1296717362::oid AND objsubid=2 AND granted) THEN + RAISE EXCEPTION 'qualified provisioning cannot extend a previously locked retention set'; + END IF; + SELECT * INTO STRICT g FROM mst2_metadata_namespace WHERE singleton=1; + IF mst2_route_lock_held(1296718001,pg_catalog.hashtext(g.core_schema),false) + AND NOT mst2_route_lock_held(1297043024,g.mono_lock_key2) THEN + RAISE EXCEPTION 'storage route lock was acquired before core mono'; + END IF; + PERFORM pg_catalog.set_config('lock_timeout','5000ms',true); + PERFORM pg_catalog.pg_advisory_xact_lock(1297043024,g.mono_lock_key2); + PERFORM pg_catalog.pg_advisory_xact_lock(1296718001,pg_catalog.hashtext(g.core_schema)); + -- A racing successful bootstrap can register while this caller waits. + IF EXISTS(SELECT 1 FROM mst2_metadata_namespace WHERE graph_domain='qualified-v1') THEN + RAISE EXCEPTION 'qualified provisioning candidate lost bootstrap serialization'; + END IF; + FOR r IN SELECT namespace_uuid,metadata_schema FROM mst2_metadata_namespace + UNION ALL SELECT candidate,candidate_schema ORDER BY namespace_uuid LOOP + PERFORM pg_catalog.pg_advisory_xact_lock(1296717362,pg_catalog.hashtext(r.metadata_schema)); + END LOOP; +END $$; + +CREATE FUNCTION mst2_route_family_catalog(c_oid oid,q_oid oid,exempt_q_oid oid DEFAULT 0::oid) RETURNS bytea LANGUAGE sql VOLATILE +SET search_path=pg_catalog,pg_temp AS $catalog$ $CATALOG_SQL$ $catalog$; +CREATE FUNCTION mst2_route_family_shape(q_oid oid,n_uuid uuid,s_uuid text) RETURNS bytea LANGUAGE sql VOLATILE +SET search_path=pg_catalog,pg_temp AS $shape$ $SHAPE_SQL$ $shape$; +CREATE TABLE mst2_qualified_family_policy ( + singleton smallint PRIMARY KEY CHECK(singleton=1),implementation_fingerprint bytea NOT NULL CHECK(octet_length(implementation_fingerprint)=32), + expected_shape bytea NOT NULL CHECK(octet_length(expected_shape)=32), + authority_catalog bytea NOT NULL CHECK(octet_length(authority_catalog)=32) +); +CREATE TRIGGER mst2_route_family_policy_immutable BEFORE UPDATE OR DELETE ON mst2_qualified_family_policy + FOR EACH ROW EXECUTE FUNCTION mst2_route_immutable(); +CREATE TRIGGER mst2_route_family_policy_truncate_guard BEFORE TRUNCATE ON mst2_qualified_family_policy + FOR EACH STATEMENT EXECUTE FUNCTION mst2_route_immutable(); + +CREATE FUNCTION mst2_route_family_registration_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$CORE_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE g mst2_metadata_namespace%ROWTYPE; stamp record; +BEGIN + SELECT * INTO STRICT g FROM mst2_metadata_namespace WHERE singleton=1; + IF (SELECT count(*) FROM mst2_metadata_namespace)<>1 OR NEW.singleton IS NOT NULL + OR substr(NEW.namespace_uuid::text,15,1)<>'4' OR substr(NEW.namespace_uuid::text,20,1) NOT IN ('8','9','a','b') + OR NEW.graph_domain<>'qualified-v1' OR NEW.family_identity<>'v3-rooted-qualified-1' + OR NEW.admission_state<>'ROOTED_Q_ADMITTED' OR NEW.collector_state<>'ENABLED' + OR ROW(NEW.core_schema,NEW.core_schema_oid,NEW.database_name,NEW.database_oid,NEW.storage_uuid, + NEW.server_address,NEW.server_port,NEW.mono_lock_key2) IS DISTINCT FROM + ROW(g.core_schema,g.core_schema_oid,g.database_name,g.database_oid,g.storage_uuid, + g.server_address,g.server_port,g.mono_lock_key2) + OR NEW.metadata_schema IS DISTINCT FROM 'mst2q_'||replace(NEW.namespace_uuid::text,'-','') + OR NEW.metadata_storage_uuid IS NOT DISTINCT FROM g.storage_uuid + OR NEW.metadata_storage_uuid IS DISTINCT FROM (NEW.metadata_storage_uuid::uuid)::text + OR substr(NEW.metadata_storage_uuid,15,1)<>'4' OR substr(NEW.metadata_storage_uuid,20,1) NOT IN ('8','9','a','b') + OR NOT EXISTS(SELECT 1 FROM pg_catalog.pg_namespace n + WHERE n.oid=NEW.metadata_schema_oid AND n.nspname=NEW.metadata_schema) + OR NEW.implementation_fingerprint IS DISTINCT FROM decode('$IMPLEMENTATION_SHA$','hex') + OR NOT mst2_route_lock_held(1297043024,g.mono_lock_key2) + OR NOT mst2_route_lock_held(1296718001,pg_catalog.hashtext(g.core_schema)) + OR NOT mst2_route_lock_held(1296717362,pg_catalog.hashtext(NEW.metadata_schema)) + OR NOT mst2_route_lock_held(1296717362,pg_catalog.hashtext(g.metadata_schema)) THEN + RAISE EXCEPTION 'qualified namespace registration has no exact admitted rooted family scope and lock set'; + END IF; + EXECUTE pg_catalog.format('SELECT * FROM %I.mst2_metadata_family_identity WHERE singleton=1',NEW.metadata_schema) + INTO STRICT stamp; + IF ROW(stamp.namespace_uuid,stamp.storage_uuid,stamp.core_schema_oid,stamp.metadata_schema_oid, + stamp.family_identity,stamp.implementation_fingerprint) IS DISTINCT FROM + ROW(NEW.namespace_uuid,NEW.metadata_storage_uuid,NEW.core_schema_oid,NEW.metadata_schema_oid, + NEW.family_identity,NEW.implementation_fingerprint) + OR NEW.catalog_fingerprint IS DISTINCT FROM mst2_route_family_catalog(NEW.core_schema_oid,NEW.metadata_schema_oid) THEN + RAISE EXCEPTION 'qualified namespace registration fingerprint disagrees with its physical family'; + END IF; + IF NOT EXISTS(SELECT 1 FROM mst2_qualified_family_policy p WHERE p.singleton=1 + AND p.implementation_fingerprint=NEW.implementation_fingerprint + AND p.authority_catalog=mst2_route_family_catalog(NEW.core_schema_oid,0::oid,NEW.metadata_schema_oid) + AND p.expected_shape=mst2_route_family_shape(NEW.metadata_schema_oid,NEW.namespace_uuid,NEW.metadata_storage_uuid)) THEN + RAISE EXCEPTION 'qualified namespace does not have the trusted complete physical family shape'; + END IF; + RETURN NEW; +END $$; +DROP TRIGGER mst2_route_namespace_registration_closed ON mst2_metadata_namespace; +CREATE TRIGGER mst2_route_namespace_registration_guard BEFORE INSERT ON mst2_metadata_namespace + FOR EACH ROW EXECUTE FUNCTION mst2_route_family_registration_guard(); + +CREATE FUNCTION mst2_metadata_has_generic_overlap(p bytea) RETURNS boolean LANGUAGE sql VOLATILE +SET search_path=$CORE_SCHEMA$,pg_catalog,pg_temp AS $$ + SELECT EXISTS(SELECT 1 FROM mst2_retention_node WHERE node_id='page:sha256:'||encode(p,'hex')) + OR EXISTS(SELECT 1 FROM mst2_retention_gc_op WHERE node_id='page:sha256:'||encode(p,'hex')) +$$; + +-- Existing generic route derivation must never select the newly registered Q row. +$GENERIC_INSERT_GUARD$ +$SOURCE_REVISION_SQL$ +$ROOTED_ROUTES_SQL$ diff --git a/src/jupiter/migration/m20261008_000200_rooted_routes.sql b/src/jupiter/migration/m20261008_000200_rooted_routes.sql new file mode 100644 index 00000000..e4cef146 --- /dev/null +++ b/src/jupiter/migration/m20261008_000200_rooted_routes.sql @@ -0,0 +1,171 @@ +CREATE FUNCTION mst2_route_qualified_namespace_valid(id uuid) RETURNS boolean LANGUAGE sql VOLATILE +SET search_path=$CORE_SCHEMA$,pg_catalog,pg_temp AS $$ + SELECT mst2_route_scope_valid($CORE_LITERAL$) AND EXISTS(SELECT 1 FROM mst2_metadata_namespace q + JOIN mst2_metadata_namespace g ON g.singleton=1 + JOIN mst2_qualified_family_policy policy ON policy.singleton=1 + JOIN pg_catalog.pg_namespace physical ON physical.oid=q.metadata_schema_oid AND physical.nspname=q.metadata_schema + WHERE q.namespace_uuid=id AND q.singleton IS NULL AND q.graph_domain='qualified-v1' + AND q.family_identity='v3-rooted-qualified-1' AND q.admission_state='ROOTED_Q_ADMITTED' AND q.collector_state='ENABLED' + AND q.metadata_schema='mst2q_'||replace(q.namespace_uuid::text,'-','') + AND q.metadata_schema_oid<>q.core_schema_oid + AND ROW(q.core_schema,q.core_schema_oid,q.database_name,q.database_oid,q.storage_uuid, + q.server_address,q.server_port,q.mono_lock_key2) IS NOT DISTINCT FROM + ROW(g.core_schema,g.core_schema_oid,g.database_name,g.database_oid,g.storage_uuid, + g.server_address,g.server_port,g.mono_lock_key2) + AND q.implementation_fingerprint=policy.implementation_fingerprint + AND policy.authority_catalog=mst2_route_family_catalog(q.core_schema_oid,0::oid,q.metadata_schema_oid) + AND q.catalog_fingerprint=mst2_route_family_catalog(q.core_schema_oid,q.metadata_schema_oid) + AND policy.expected_shape=mst2_route_family_shape(q.metadata_schema_oid,q.namespace_uuid,q.metadata_storage_uuid)) +$$; + +CREATE OR REPLACE FUNCTION mst2_route_statement_barrier() RETURNS trigger LANGUAGE plpgsql VOLATILE AS $$ +DECLARE q_id uuid; caller text:=pg_catalog.current_schema(); +BEGIN + IF TG_TABLE_SCHEMA<>$CORE_LITERAL$ OR NOT EXISTS(SELECT 1 FROM pg_catalog.pg_class + WHERE oid=TG_RELID AND relnamespace=$CORE_OID$) THEN + RAISE EXCEPTION 'storage route mutation is outside its captured core schema'; + END IF; + IF caller IS DISTINCT FROM $CORE_LITERAL$ THEN + SELECT n.namespace_uuid INTO q_id FROM $CORE_SCHEMA$.mst2_metadata_namespace n + WHERE n.metadata_schema=caller AND n.graph_domain='qualified-v1'; + IF NOT FOUND OR NOT $CORE_SCHEMA$.mst2_route_qualified_namespace_valid(q_id) THEN + RAISE EXCEPTION 'storage route mutation caller has no exact captured physical family'; END IF; + END IF; + PERFORM $CORE_SCHEMA$.mst2_route_enter($CORE_LITERAL$); + RETURN NULL; +END $$; + +CREATE FUNCTION mst2_route_qualified_snapshot_proof(sid text) RETURNS boolean LANGUAGE plpgsql VOLATILE +SET search_path=$CORE_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE r mst2_snapshot_storage_route%ROWTYPE; n mst2_metadata_namespace%ROWTYPE; valid boolean; +BEGIN + SELECT * INTO r FROM mst2_snapshot_storage_route WHERE snapshot_id=sid; + IF NOT FOUND THEN RETURN false; END IF; + SELECT * INTO n FROM mst2_metadata_namespace WHERE namespace_uuid=r.namespace_uuid; + IF NOT FOUND OR NOT mst2_route_qualified_namespace_valid(n.namespace_uuid) THEN RETURN false; END IF; + EXECUTE pg_catalog.format('SELECT %I.mst2_metadata_snapshot_route_proof($1,$2,$3,$4,$5,$6,$7)',n.metadata_schema) + INTO valid USING r.snapshot_id,r.canonical_descriptor,r.instance_id,r.commit_oid,r.root_tree_oid,r.metadata_root,r.source_profile; + RETURN coalesce(valid,false); +END $$; + +CREATE FUNCTION mst2_route_qualified_lease_proof(lid text) RETURNS boolean LANGUAGE plpgsql VOLATILE +SET search_path=$CORE_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE r mst2_lease_storage_route%ROWTYPE; n mst2_metadata_namespace%ROWTYPE; valid boolean; +BEGIN + SELECT * INTO r FROM mst2_lease_storage_route WHERE lease_id=lid; + IF NOT FOUND THEN RETURN false; END IF; + SELECT * INTO n FROM mst2_metadata_namespace WHERE namespace_uuid=r.namespace_uuid; + IF NOT FOUND OR NOT mst2_route_qualified_namespace_valid(n.namespace_uuid) THEN RETURN false; END IF; + EXECUTE pg_catalog.format('SELECT %I.mst2_metadata_lease_route_proof($1,$2,$3,$4,$5,$6,$7,$8,$9,$10)',n.metadata_schema) + INTO valid USING r.lease_id,r.snapshot_id,r.namespace_uuid,r.session_incarnation,r.prepare_id,r.metadata_root, + r.authorization_epoch,r.publication_sequence,r.writer_epoch,r.certificate_receipt_id; + RETURN coalesce(valid,false); +END $$; + +-- Existing G proof functions remain authoritative for their original family. +CREATE FUNCTION mst2_route_family_for_snapshot(sid text,caller_schema text) +RETURNS TABLE(namespace_uuid uuid,graph_domain text) LANGUAGE plpgsql VOLATILE +SET search_path=$CORE_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE route mst2_snapshot_storage_route%ROWTYPE; namespace mst2_metadata_namespace%ROWTYPE; +BEGIN + IF NOT mst2_route_scope_valid(caller_schema) THEN RAISE EXCEPTION 'storage route captured primary scope is unavailable'; END IF; + SELECT * INTO route FROM mst2_snapshot_storage_route WHERE snapshot_id=sid; + IF NOT FOUND THEN RETURN; END IF; + SELECT * INTO namespace FROM mst2_metadata_namespace n WHERE n.namespace_uuid=route.namespace_uuid; + IF NOT FOUND OR namespace.graph_domain='generic-v1' AND NOT mst2_route_snapshot_proof(sid) + OR namespace.graph_domain='qualified-v1' AND NOT mst2_route_qualified_snapshot_proof(sid) + OR namespace.graph_domain NOT IN ('generic-v1','qualified-v1') THEN + RAISE EXCEPTION 'permanent snapshot route has a corrupt exact family or fixed source binding'; END IF; + RETURN QUERY SELECT namespace.namespace_uuid,namespace.graph_domain; +END $$; + +CREATE FUNCTION mst2_route_family_for_lease(lid text,caller_schema text) +RETURNS TABLE(namespace_uuid uuid,graph_domain text) LANGUAGE plpgsql VOLATILE +SET search_path=$CORE_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE route mst2_lease_storage_route%ROWTYPE; namespace mst2_metadata_namespace%ROWTYPE; +BEGIN + IF NOT mst2_route_scope_valid(caller_schema) THEN RAISE EXCEPTION 'storage route captured primary scope is unavailable'; END IF; + SELECT * INTO route FROM mst2_lease_storage_route WHERE lease_id=lid; + IF NOT FOUND THEN RETURN; END IF; + SELECT * INTO namespace FROM mst2_metadata_namespace n WHERE n.namespace_uuid=route.namespace_uuid; + IF NOT FOUND OR namespace.graph_domain='generic-v1' AND NOT mst2_route_lease_proof(lid) + OR namespace.graph_domain='qualified-v1' AND NOT mst2_route_qualified_lease_proof(lid) + OR namespace.graph_domain NOT IN ('generic-v1','qualified-v1') THEN + RAISE EXCEPTION 'permanent lease route has a corrupt exact family or fixed source binding'; END IF; + RETURN QUERY SELECT namespace.namespace_uuid,namespace.graph_domain; +END $$; + +CREATE OR REPLACE FUNCTION mst2_route_insert_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$CORE_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE namespace mst2_metadata_namespace%ROWTYPE; valid boolean; +BEGIN + SELECT * INTO namespace FROM mst2_metadata_namespace WHERE namespace_uuid=NEW.namespace_uuid; + IF NOT FOUND THEN RAISE EXCEPTION 'storage route target namespace is not registered'; END IF; + IF namespace.graph_domain='qualified-v1' THEN + IF NOT mst2_route_qualified_namespace_valid(namespace.namespace_uuid) THEN + RAISE EXCEPTION 'qualified route has no exact complete physical family catalog'; END IF; + IF TG_TABLE_NAME='mst2_snapshot_storage_route' THEN + EXECUTE pg_catalog.format('SELECT %I.mst2_metadata_snapshot_candidate($1,$2,$3,$4,$5,$6,$7)',namespace.metadata_schema) + INTO valid USING NEW.snapshot_id,NEW.canonical_descriptor,NEW.instance_id,NEW.commit_oid, + NEW.root_tree_oid,NEW.metadata_root,NEW.source_profile; + ELSIF TG_TABLE_NAME='mst2_lease_storage_route' THEN + EXECUTE pg_catalog.format('SELECT %I.mst2_metadata_lease_route_proof($1,$2,$3,$4,$5,$6,$7,$8,$9,$10)',namespace.metadata_schema) + INTO valid USING NEW.lease_id,NEW.snapshot_id,NEW.namespace_uuid,NEW.session_incarnation,NEW.prepare_id, + NEW.metadata_root,NEW.authorization_epoch,NEW.publication_sequence,NEW.writer_epoch,NEW.certificate_receipt_id; + ELSE RAISE EXCEPTION 'qualified route cannot use a generic binding relation'; END IF; + IF NOT coalesce(valid,false) THEN RAISE EXCEPTION 'qualified route is not independently derived from its exact source'; END IF; + RETURN NEW; + END IF; + IF namespace.singleton IS DISTINCT FROM 1 OR namespace.graph_domain<>'generic-v1' THEN + RAISE EXCEPTION 'storage route target family is unsupported'; END IF; + IF TG_TABLE_NAME='mst2_snapshot_storage_route' THEN + IF NOT EXISTS(SELECT 1 FROM mst2_snapshot_context s JOIN mst2_metadata_prepare p ON p.prepare_id=s.prepare_id + JOIN mst2_metadata_namespace n ON n.namespace_uuid=NEW.namespace_uuid + WHERE n.singleton=1 AND n.graph_domain='generic-v1' AND s.snapshot_id=NEW.snapshot_id AND NEW.canonical_descriptor=s.canonical_descriptor + AND NEW.instance_id=s.instance_id AND NEW.commit_oid=s.commit_oid AND NEW.root_tree_oid=s.root_tree_oid + AND NEW.metadata_root=s.metadata_root AND NEW.source_profile=mst2_route_profile(p.source_domain,p.tagged_root_tree_oid, + p.scope,p.schema_version,p.metadata_codec,p.materialization_policy,p.fs_semantics,p.access_projection, + p.verification_revision,p.projection_revision) + AND p.state='COMMITTED' AND p.metadata_root=s.metadata_root AND p.source_domain='native-git' + AND (p.graph_domain IS NULL OR p.graph_domain='generic-v1') + AND p.tagged_root_tree_oid IN ('sha1:'||s.root_tree_oid,'sha256:'||s.root_tree_oid)) THEN + RAISE EXCEPTION 'storage route is not derived from its actual generic context'; + END IF; + ELSIF TG_TABLE_NAME='mst2_generic_session_storage_binding' THEN + IF NOT EXISTS(SELECT 1 FROM mst2_snapshot_context s JOIN mst2_snapshot_storage_route r USING(snapshot_id) + WHERE s.snapshot_id=NEW.snapshot_id AND r.namespace_uuid=NEW.namespace_uuid + AND s.prepare_id=NEW.prepare_id AND s.metadata_root=NEW.metadata_root) THEN + RAISE EXCEPTION 'storage route generic incarnation is not its actual context'; + END IF; + ELSIF TG_TABLE_NAME='mst2_lease_storage_route' THEN + IF NOT EXISTS(SELECT 1 FROM mst2_snapshot_lease l JOIN mst2_generic_session_storage_binding b USING(snapshot_id) + WHERE l.lease_id=NEW.lease_id AND l.snapshot_id=NEW.snapshot_id AND b.namespace_uuid=NEW.namespace_uuid + AND b.session_incarnation=NEW.session_incarnation AND b.prepare_id=NEW.prepare_id AND b.metadata_root=NEW.metadata_root + AND l.authorization_epoch=NEW.authorization_epoch AND l.publication_sequence=NEW.publication_sequence + AND l.writer_epoch=NEW.writer_epoch AND l.certificate_receipt_id=NEW.certificate_receipt_id + AND mst2_route_snapshot_proof(l.snapshot_id)) THEN + RAISE EXCEPTION 'storage route lease is not its exact generic incarnation and source'; + END IF; + ELSE RAISE EXCEPTION 'storage route insertion target is not registered'; END IF; + RETURN NEW; +END $$; + +CREATE FUNCTION mst2_route_permanent_complete() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$CORE_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE domain text; valid boolean; +BEGIN + SELECT graph_domain INTO domain FROM mst2_metadata_namespace WHERE namespace_uuid=NEW.namespace_uuid; + IF TG_TABLE_NAME='mst2_snapshot_storage_route' THEN + valid:=CASE domain WHEN 'generic-v1' THEN mst2_route_snapshot_proof(NEW.snapshot_id) + WHEN 'qualified-v1' THEN mst2_route_qualified_snapshot_proof(NEW.snapshot_id) ELSE false END; + ELSE + valid:=CASE domain WHEN 'generic-v1' THEN mst2_route_lease_proof(NEW.lease_id) + WHEN 'qualified-v1' THEN mst2_route_qualified_lease_proof(NEW.lease_id) ELSE false END; + END IF; + IF NOT coalesce(valid,false) THEN RAISE EXCEPTION 'permanent storage route cannot commit without its exact actual incarnation'; END IF; + RETURN NULL; +END $$; +CREATE CONSTRAINT TRIGGER mst2_route_permanent_complete AFTER INSERT ON mst2_snapshot_storage_route + DEFERRABLE INITIALLY DEFERRED FOR EACH ROW EXECUTE FUNCTION mst2_route_permanent_complete(); +CREATE CONSTRAINT TRIGGER mst2_route_permanent_complete AFTER INSERT ON mst2_lease_storage_route + DEFERRABLE INITIALLY DEFERRED FOR EACH ROW EXECUTE FUNCTION mst2_route_permanent_complete(); diff --git a/src/jupiter/migration/mod.rs b/src/jupiter/migration/mod.rs index f5981f02..2fbc20ce 100644 --- a/src/jupiter/migration/mod.rs +++ b/src/jupiter/migration/mod.rs @@ -152,6 +152,7 @@ mod m20261007_000400_add_mst2_qualified_metadata_gc; mod m20261007_000500_add_mst2_install_capability; mod m20261007_000600_add_mst2_storage_routes; mod m20261008_000100_add_mst2_chunk_maps; +mod m20261008_000200_add_mst2_rooted_qualified_family; mod runner; pub use m20260905_000100_add_push_queue::ensure_queue_control_seed; pub use runner::apply_migrations; @@ -292,6 +293,7 @@ impl MigratorTrait for Migrator { Box::new(m20261007_000500_add_mst2_install_capability::Migration), Box::new(m20261007_000600_add_mst2_storage_routes::Migration), Box::new(m20261008_000100_add_mst2_chunk_maps::Migration), + Box::new(m20261008_000200_add_mst2_rooted_qualified_family::Migration), ] } } @@ -1202,7 +1204,7 @@ mod tests { async fn import_repo_alias_rows_canonicalized() { let names = migration_names(); assert_eq!( - &names[names.len() - 14..names.len() - 5], + &names[names.len() - 16..names.len() - 7], &[ "m20260923_000200_canonicalize_import_repo_paths".to_string(), "m20260925_000100_media_paging".to_string(), @@ -1217,25 +1219,33 @@ mod tests { "native retention, metadata installation and view tables follow media paging" ); assert_eq!( - &names[names.len() - 5], + &names[names.len() - 7], "m20261007_000200_add_mst2_metadata_generations" ); assert_eq!( - &names[names.len() - 4], + &names[names.len() - 6], "m20261007_000300_add_mst2_metadata_lifetime_history" ); assert_eq!( - &names[names.len() - 3], + &names[names.len() - 5], "m20261007_000400_add_mst2_qualified_metadata_gc" ); assert_eq!( - &names[names.len() - 2], + &names[names.len() - 4], "m20261007_000500_add_mst2_install_capability" ); assert_eq!( - names.last().unwrap(), + &names[names.len() - 3], "m20261007_000600_add_mst2_storage_routes" ); + assert_eq!( + &names[names.len() - 2], + "m20261008_000100_add_mst2_chunk_maps" + ); + assert_eq!( + names.last().unwrap(), + "m20261008_000200_add_mst2_rooted_qualified_family" + ); let db = alias_db().await; insert_repo(&db, 1, "/third-party//a").await; diff --git a/src/jupiter/storage/init.rs b/src/jupiter/storage/init.rs index d8480141..47044027 100644 --- a/src/jupiter/storage/init.rs +++ b/src/jupiter/storage/init.rs @@ -10,6 +10,21 @@ use crate::{ jupiter::migration::{apply_migrations, ensure_queue_control_seed}, }; +#[cfg(test)] +tokio::task_local! { + static GENERIC_HISTORY_BOOTSTRAP: (); +} + +#[cfg(test)] +pub(crate) async fn with_generic_history_bootstrap(future: F) -> F::Output { + GENERIC_HISTORY_BOOTSTRAP.scope((), future).await +} + +#[cfg(test)] +pub(super) fn generic_history_bootstrap_active() -> bool { + GENERIC_HISTORY_BOOTSTRAP.try_with(|_| ()).is_ok() +} + /// Create a PostgreSQL database connection. /// /// After a successful connection, applies any pending database migrations and @@ -25,6 +40,11 @@ pub async fn database_connection(db_config: &DbConfig) -> Result>, pub(crate) native_snapshot_sessions: Arc>, + #[cfg(test)] + pub(crate) shadow_qualified_metadata: + Arc>, + pub(crate) rooted_qualified_metadata: Arc< + tokio::sync::OnceCell>, + >, pub(crate) projection_observation_sink: Option>, pub cl_service: CLService, @@ -194,6 +201,46 @@ pub struct Storage { } impl Storage { + pub(crate) async fn rooted_qualified_metadata_writer( + &self, + ) -> Result<&qualified_metadata_family::RootedQualifiedMetadataRepository, MegaError> { + self.rooted_qualified_metadata + .get_or_try_init(|| async { + let repository = Arc::new( + qualified_metadata_family::RootedQualifiedMetadataRepository::open( + self.mono_storage().get_connection(), + &self.config().database, + ) + .await?, + ); + repository + .maintenance_tick(64) + .await + .map_err(|error| MegaError::Other(error.to_string()))?; + qualified_metadata_family::RootedQualifiedMetadataRepository::start_maintenance( + &repository, + ); + Ok::<_, MegaError>(repository) + }) + .await + .map(Arc::as_ref) + } + + #[cfg(test)] + pub(crate) async fn shadow_qualified_metadata_writer( + &self, + ) -> Result<&qualified_metadata_family::ShadowQualifiedMetadataWriter, MegaError> { + self.shadow_qualified_metadata + .get_or_try_init(|| async { + qualified_metadata_family::ShadowQualifiedMetadataWriter::open( + self.mono_storage().get_connection(), + &self.config().database, + ) + .await + }) + .await + } + pub async fn new( config: Arc, object_store: MegaObjectStorageWrapper, @@ -325,11 +372,14 @@ impl Storage { let storage_event_emitter = crate::jupiter::service::storage_event_emitter::StorageEventEmitter::from_config_disabled(&config); - Ok(Storage { + let storage = Storage { app_service: app_service.into(), native_projection_cache: Arc::default(), native_snapshot_sessions: Arc::default(), native_chunk_maps: Arc::default(), + #[cfg(test)] + shadow_qualified_metadata: Arc::default(), + rooted_qualified_metadata: Arc::default(), projection_observation_sink: None, config_handle, config, @@ -349,7 +399,13 @@ impl Storage { view_runtime, entity_store: Arc::new(SharedEntityStore::default()), vault: None, - }) + }; + #[cfg(test)] + if init::generic_history_bootstrap_active() { + return Ok(storage); + } + storage.rooted_qualified_metadata_writer().await?; + Ok(storage) } pub fn config_handle(&self) -> ConfigHandle { @@ -709,6 +765,9 @@ impl Storage { native_projection_cache: Arc::default(), native_snapshot_sessions: Arc::default(), native_chunk_maps: Arc::default(), + #[cfg(test)] + shadow_qualified_metadata: Arc::default(), + rooted_qualified_metadata: Arc::default(), projection_observation_sink: None, // app_service: AppService::mock(), cl_service: CLService::mock(), diff --git a/src/jupiter/storage/native_metadata_generations.rs b/src/jupiter/storage/native_metadata_generations.rs index a2cad697..dde51fc2 100644 --- a/src/jupiter/storage/native_metadata_generations.rs +++ b/src/jupiter/storage/native_metadata_generations.rs @@ -398,7 +398,11 @@ impl PostgresMetadataGenerationRepository { let fixed = self.require_fixed_plan(txn, intent).await?; check_generation_payloads(txn, &fixed).await?; let legacy = if intent.graph_domain == GraphDomain::Qualified { - qualified::finalize_graph(txn, &fixed, &observation.dag).await? + if self.inner.qualified_family.is_some() { + qualified::finalize_canonical_graph(txn, &fixed, &observation.dag).await? + } else { + qualified::finalize_graph(txn, &fixed, &observation.dag).await? + } } else { self.inner .finalize_stored_plan_in_txn(txn, &intent.legacy, &observation.dag, fixed.stored) diff --git a/src/jupiter/storage/native_metadata_install.rs b/src/jupiter/storage/native_metadata_install.rs index 5b97f663..6ecf72c3 100644 --- a/src/jupiter/storage/native_metadata_install.rs +++ b/src/jupiter/storage/native_metadata_install.rs @@ -139,6 +139,7 @@ pub struct PostgresMetadataInstallRepository { connection: DatabaseConnection, barrier_timeout: Duration, storage_scope: PrimaryStorageScope, + qualified_family: Option, } #[derive(Debug, Clone, PartialEq, Eq)] @@ -160,6 +161,7 @@ impl PostgresMetadataInstallRepository { connection, barrier_timeout: Duration::from_secs(5), storage_scope, + qualified_family: None, }) } @@ -654,6 +656,9 @@ impl PostgresMetadataInstallRepository { if txn.get_database_backend() != DbBackend::Postgres { return Err(internal("metadata installation requires PostgreSQL")); } + if let Some(family) = &self.qualified_family { + family.enter(txn).await?; + } if read_storage_scope(txn).await? != self.storage_scope { return Err(internal( "metadata recovery connection is outside the captured primary storage scope", diff --git a/src/jupiter/storage/native_metadata_qualified.rs b/src/jupiter/storage/native_metadata_qualified.rs index 5a29c9c6..221f9c98 100644 --- a/src/jupiter/storage/native_metadata_qualified.rs +++ b/src/jupiter/storage/native_metadata_qualified.rs @@ -115,6 +115,22 @@ impl GcRecord { } impl PostgresQualifiedMetadataRepository { + #[cfg(test)] + pub(crate) async fn registered_shadow( + connection: DatabaseConnection, + family: crate::jupiter::storage::qualified_metadata_family::VerifiedQualifiedNamespace, + ) -> Result { + let mut inner = PostgresMetadataInstallRepository::new(connection).await?; + inner.qualified_family = Some(family); + Ok(Self { + inner: PostgresMetadataGenerationRepository { + inner, + graph_domain: "qualified-v1", + }, + }) + } + + #[cfg(test)] pub async fn new(connection: DatabaseConnection) -> Result { Ok(Self { inner: PostgresMetadataGenerationRepository { @@ -635,8 +651,7 @@ pub(super) async fn allocate_lifetimes( n.state AS graph_state,n.metadata_codec AS graph_codec,n.bytes AS graph_bytes, b.page_id IS NOT NULL AS payload_present,b.generation AS payload_generation,b.metadata_codec AS payload_codec,b.byte_size AS payload_size, EXISTS(SELECT 1 FROM mst2_metadata_gc_op o WHERE o.page_id=l.page_id AND o.generation=l.generation) AS tombstone, - EXISTS(SELECT 1 FROM mst2_retention_node x WHERE x.node_id=l.node_id) - OR EXISTS(SELECT 1 FROM mst2_retention_gc_op x WHERE x.node_id=l.node_id) AS generic_graph + mst2_metadata_has_generic_overlap(l.page_id) AS generic_graph FROM jsonb_to_recordset($1::jsonb) p(id text,size integer) JOIN mst2_metadata_current c ON c.page_id=decode(p.id,'hex') JOIN mst2_metadata_lifetime l USING(page_id,generation) @@ -740,18 +755,16 @@ pub(super) async fn check_lifetimes( OR ($3='COMMITTED' AND l.state<>'LIVE') OR (l.state='LIVE' AND n.page_id IS NULL) OR n.state<>'LIVE' OR n.metadata_codec<>$2 OR n.bytes<>m.expected_size OR EXISTS(SELECT 1 FROM mst2_metadata_gc_op o WHERE o.page_id=m.page_id AND o.generation=m.generation) - OR EXISTS(SELECT 1 FROM mst2_retention_node x WHERE x.node_id=l.node_id) - OR EXISTS(SELECT 1 FROM mst2_retention_gc_op x WHERE x.node_id=l.node_id)) LIMIT 1", + OR mst2_metadata_has_generic_overlap(l.page_id)) LIMIT 1", [stored.record.prepare_id.clone().into(),stored.record.metadata_codec.into(),stored.record.state.clone().into()], )).await.map_err(internal)?.is_some() { return Err(unavailable("fixed qualified incarnation is no longer installable")); } Ok(()) } -pub(super) async fn finalize_graph( - txn: &DatabaseTransaction, +fn validate_fixed_observation( fixed: &FixedPlan, dag: &ValidatedMetadataDag, -) -> Result { +) -> Result<(), SnapshotError> { let actual_pages: BTreeSet<_> = dag.payloads().iter().map(|p| (p.id, p.size)).collect(); let actual_edges: BTreeSet<_> = dag .edges() @@ -780,6 +793,92 @@ pub(super) async fn finalize_graph( "qualified DAG observation differs from immutable plan", )); } + Ok(()) +} + +pub(super) async fn finalize_canonical_graph( + txn: &DatabaseTransaction, + fixed: &FixedPlan, + dag: &ValidatedMetadataDag, +) -> Result { + validate_fixed_observation(fixed, dag)?; + if fixed.stored.record.state == "COMMITTED" { + verify_graph(txn, fixed).await?; + return fixed.stored.receipt(); + } + let mut pending: BTreeMap<_, usize> = fixed.stored.plan.pages.keys().map(|p| (*p, 0)).collect(); + let mut parents: BTreeMap<_, Vec<_>> = BTreeMap::new(); + for &(parent, child) in &fixed.stored.plan.edges { + *pending + .get_mut(&parent) + .ok_or_else(|| integrity("canonical parent is outside its fixed plan"))? += 1; + parents.entry(child).or_default().push(parent); + } + let mut ready: BTreeSet<_> = pending + .iter() + .filter_map(|(p, count)| (*count == 0).then_some(*p)) + .collect(); + let mut ordered = Vec::with_capacity(pending.len()); + while let Some(page) = ready.pop_first() { + ordered.push(json!({"page":hex::encode(page),"generation":fixed.bindings.0[&page].0})); + for parent in parents.get(&page).into_iter().flatten() { + let count = pending + .get_mut(parent) + .ok_or_else(|| integrity("canonical ancestor is outside its fixed plan"))?; + *count = count + .checked_sub(1) + .ok_or_else(|| integrity("canonical edge accounting underflow"))?; + if *count == 0 { + ready.insert(*parent); + } + } + } + if ordered.len() != pending.len() { + return Err(integrity("canonical certification order contains a cycle")); + } + let certified: i32 = txn + .query_one_raw(statement( + "SELECT mst2_metadata_certify_batch($1,$2::jsonb) AS certified", + [ + fixed.intent.prepare_id().into(), + serde_json::to_string(&ordered).map_err(internal)?.into(), + ], + )) + .await + .map_err(internal)? + .ok_or_else(|| integrity("canonical certification batch is missing"))? + .try_get("", "certified") + .map_err(internal)?; + if certified as usize != ordered.len() { + return Err(integrity( + "canonical certification did not cover its exact cold DAG", + )); + } + retain_existing_roots(txn, &fixed.intent).await?; + verify_graph(txn, fixed).await?; + let changed = txn + .execute_raw(statement( + "UPDATE mst2_metadata_prepare SET state='COMMITTED',committed_at=clock_timestamp() + WHERE prepare_id=$1 AND state='PREPARING' AND storage_seal=$2", + [ + fixed.intent.prepare_id().into(), + fixed.intent.storage_seal.to_vec().into(), + ], + )) + .await + .map_err(internal)?; + if changed.rows_affected() != 1 { + return Err(integrity("canonical finalize lost its prepare CAS")); + } + fixed.stored.receipt() +} + +pub(super) async fn finalize_graph( + txn: &DatabaseTransaction, + fixed: &FixedPlan, + dag: &ValidatedMetadataDag, +) -> Result { + validate_fixed_observation(fixed, dag)?; if fixed.stored.record.state == "COMMITTED" { verify_graph(txn, fixed).await?; return fixed.stored.receipt(); diff --git a/src/jupiter/storage/native_snapshot_session.rs b/src/jupiter/storage/native_snapshot_session.rs index fb8d9134..da15a716 100644 --- a/src/jupiter/storage/native_snapshot_session.rs +++ b/src/jupiter/storage/native_snapshot_session.rs @@ -366,9 +366,11 @@ mod routes; #[path = "native_snapshot_metadata_routes.rs"] mod metadata_routes; -pub(crate) use metadata_routes::MetadataRouteRequest; #[cfg(test)] pub(crate) use metadata_routes::with_metadata_read_barriers; +pub(crate) use metadata_routes::{ + MetadataRouteRequest, PersistedMetadataReadWork, PersistedMetadataRouteBatch, +}; const SESSION_SQL: &str = "SELECT s.snapshot_id,s.canonical_descriptor,s.commit_oid,s.root_tree_oid, (SELECT storage_uuid FROM mst2_metadata_storage_scope WHERE singleton=1) AS authority_storage_uuid, @@ -665,7 +667,7 @@ fn forbidden() -> SnapshotError { "snapshot serving state or authorization epoch changed", ) } -fn install_error(error: MetadataInstallError) -> SnapshotError { +pub(crate) fn install_error(error: MetadataInstallError) -> SnapshotError { match error { MetadataInstallError::Rejected(error) if error.code == SnapshotErrorCode::Internal => { tracing::error!(%error,"native metadata installation unavailable"); @@ -690,6 +692,44 @@ fn install_error(error: MetadataInstallError) -> SnapshotError { } impl super::Storage { + pub(crate) async fn snapshot_metadata_family( + &self, + identity: &str, + lease: bool, + ) -> Result, SnapshotError> + { + use super::base_storage::StorageConnector; + super::qualified_metadata_family::select_snapshot_family( + self.mono_storage().get_connection(), + identity, + lease, + ) + .await + } + + pub(crate) async fn snapshot_metadata_routes( + &self, + context: &SnapshotContext, + requests: &[MetadataRouteRequest<'_>], + ) -> Result { + if self + .snapshot_metadata_family(&context.lease_id, true) + .await? + == Some(super::qualified_metadata_family::SnapshotMetadataFamily::Rooted) + { + return self + .rooted_qualified_metadata_writer() + .await + .map_err(internal)? + .metadata_routes(context, requests) + .await; + } + self.snapshot_sessions() + .await + .metadata_routes(context, requests) + .await + } + pub(crate) async fn snapshot_sessions(&self) -> &PostgresNativeSessionRepository { use super::base_storage::StorageConnector; self.native_snapshot_sessions @@ -714,6 +754,16 @@ impl super::Storage { let instance = uuid::Uuid::parse_str(instance) .map_err(|_| not_ready("invalid native instance"))? .to_string(); + if self.snapshot_metadata_family(lease, true).await? + == Some(super::qualified_metadata_family::SnapshotMetadataFamily::Rooted) + { + return self + .rooted_qualified_metadata_writer() + .await + .map_err(internal)? + .context(sid, lease, &instance) + .await; + } self.snapshot_sessions() .await .context(sid, lease, &instance) @@ -740,6 +790,16 @@ impl super::Storage { let instance = uuid::Uuid::parse_str(instance) .map_err(|_| not_ready("invalid native instance"))? .to_string(); + if self.snapshot_metadata_family(lease, true).await? + == Some(super::qualified_metadata_family::SnapshotMetadataFamily::Rooted) + { + return self + .rooted_qualified_metadata_writer() + .await + .map_err(internal)? + .renew(lease, seconds, &instance) + .await; + } self.snapshot_sessions() .await .renew(lease, seconds, &instance) @@ -751,6 +811,16 @@ impl super::Storage { pub(crate) async fn snapshot_release(&self, lease: &str) -> Result { if self.config().mst2.publication_enabled { + if self.snapshot_metadata_family(lease, true).await? + == Some(super::qualified_metadata_family::SnapshotMetadataFamily::Rooted) + { + return self + .rooted_qualified_metadata_writer() + .await + .map_err(internal)? + .release(lease) + .await; + } self.snapshot_sessions().await.release(lease).await } else { Ok(crate::ceres::snapshot::runtime::runtime().release_lease(lease)) diff --git a/src/jupiter/storage/qualified_family_catalog.sql b/src/jupiter/storage/qualified_family_catalog.sql new file mode 100644 index 00000000..acb8cce9 --- /dev/null +++ b/src/jupiter/storage/qualified_family_catalog.sql @@ -0,0 +1,57 @@ +WITH selected_namespaces AS ( + SELECT oid,nspname,nspowner,nspacl FROM pg_catalog.pg_namespace + WHERE oid IN ($CORE_OID$,$Q_OID$) +), selected_relations AS ( + SELECT c.oid,c.relnamespace,c.relname,c.relkind,c.relpersistence,c.relowner,c.relacl, + c.relrowsecurity,c.relforcerowsecurity,c.reloptions,c.relam,c.relhasrules,c.relispartition + FROM pg_catalog.pg_class c + WHERE c.relnamespace=$Q_OID$ + OR (c.relnamespace=$CORE_OID$ AND (c.relname IN ('mst2_metadata_namespace','mst2_qualified_family_policy', + 'mega_tree','mst2_rooted_source_tree_revision','mst2_verified_object','mst2_retention_node','mst2_retention_gc_op', + 'mega_commit','mega_refs','mst2_native_head','mst2_native_publication','mst2_publication','mst2_publication_outbox', + 'mst2_snapshot_storage_route','mst2_lease_storage_route','mst2_generic_session_storage_binding', + 'mst2_snapshot_context','mst2_snapshot_lease') + OR c.oid IN (SELECT i.indexrelid FROM pg_catalog.pg_index i JOIN pg_catalog.pg_class parent ON parent.oid=i.indrelid + WHERE parent.relnamespace=$CORE_OID$ AND parent.relname IN ('mst2_metadata_namespace','mst2_qualified_family_policy', + 'mega_tree','mst2_rooted_source_tree_revision','mst2_verified_object','mst2_retention_node','mst2_retention_gc_op', + 'mega_commit','mega_refs','mst2_native_head','mst2_native_publication','mst2_publication','mst2_publication_outbox', + 'mst2_snapshot_storage_route','mst2_lease_storage_route','mst2_generic_session_storage_binding', + 'mst2_snapshot_context','mst2_snapshot_lease')))) +), selected_triggers AS ( + SELECT t.* FROM pg_catalog.pg_trigger t + WHERE (t.tgrelid IN (SELECT oid FROM selected_relations) + OR EXISTS(SELECT 1 FROM pg_catalog.pg_constraint x + WHERE x.oid=t.tgconstraint AND x.conrelid IN (SELECT oid FROM pg_catalog.pg_class WHERE relnamespace=$Q_OID$))) + AND NOT ($Q_OID$=0 AND $EXEMPT_Q_OID$<>0 AND t.tgisinternal + AND EXISTS(SELECT 1 FROM pg_catalog.pg_constraint fk + JOIN pg_catalog.pg_class source ON source.oid=fk.conrelid + JOIN pg_catalog.pg_class target ON target.oid=fk.confrelid + WHERE fk.oid=t.tgconstraint AND fk.contype='f' AND fk.connamespace=$EXEMPT_Q_OID$ + AND source.relnamespace=$EXEMPT_Q_OID$ AND target.relnamespace=$CORE_OID$ + AND t.tgrelid=fk.confrelid AND t.tgconstrrelid=fk.conrelid)) +), objects AS ( + SELECT 'namespace' AS kind,oid::text AS key,pg_catalog.to_jsonb(n) AS value FROM selected_namespaces n + UNION ALL SELECT 'relation',oid::text,pg_catalog.to_jsonb(c) FROM selected_relations c + UNION ALL SELECT 'column',a.attrelid||':'||a.attnum,pg_catalog.to_jsonb(a) + FROM pg_catalog.pg_attribute a JOIN selected_relations c ON c.oid=a.attrelid WHERE a.attnum>0 + UNION ALL SELECT 'default',d.oid::text,pg_catalog.to_jsonb(d) + FROM pg_catalog.pg_attrdef d JOIN selected_relations c ON c.oid=d.adrelid + UNION ALL SELECT 'constraint',x.oid::text,pg_catalog.to_jsonb(x) + FROM pg_catalog.pg_constraint x JOIN selected_relations c ON c.oid=x.conrelid + UNION ALL SELECT 'index',i.indexrelid::text,pg_catalog.to_jsonb(i) + FROM pg_catalog.pg_index i JOIN selected_relations c ON c.oid=i.indrelid + UNION ALL SELECT 'trigger',t.oid::text,pg_catalog.to_jsonb(t) + FROM selected_triggers t + UNION ALL SELECT 'function',p.oid::text,pg_catalog.to_jsonb(p) + FROM pg_catalog.pg_proc p WHERE p.pronamespace=$Q_OID$ + OR (p.pronamespace=$CORE_OID$ AND (pg_catalog.left(p.proname,11)='mst2_route_' + OR p.proname='mst2_metadata_has_generic_overlap')) + OR EXISTS(SELECT 1 FROM selected_triggers t WHERE t.tgfoid=p.oid) + UNION ALL SELECT 'policy',p.oid::text,pg_catalog.to_jsonb(p) + FROM pg_catalog.pg_policy p JOIN selected_relations c ON c.oid=p.polrelid + UNION ALL SELECT 'rule',r.oid::text,pg_catalog.to_jsonb(r) + FROM pg_catalog.pg_rewrite r JOIN selected_relations c ON c.oid=r.ev_class +) +SELECT pg_catalog.sha256(pg_catalog.convert_to( + pg_catalog.jsonb_agg(pg_catalog.jsonb_build_array(kind,key,value) ORDER BY kind,key)::text,'UTF8')) AS fingerprint +FROM objects diff --git a/src/jupiter/storage/qualified_family_shape.sql b/src/jupiter/storage/qualified_family_shape.sql new file mode 100644 index 00000000..4269fe74 --- /dev/null +++ b/src/jupiter/storage/qualified_family_shape.sql @@ -0,0 +1,65 @@ +WITH q AS ( + SELECT n.oid,n.nspname FROM pg_catalog.pg_namespace n WHERE n.oid=q_oid +), relations AS ( + SELECT c.* FROM pg_catalog.pg_class c WHERE c.relnamespace=q_oid +), objects AS ( + SELECT 'relation' AS kind,c.relname AS key,jsonb_build_array(c.relname,c.relkind,c.relpersistence,c.relowner,c.relacl, + c.relrowsecurity,c.relforcerowsecurity,c.reloptions,c.relispartition,a.amname) AS value + FROM relations c LEFT JOIN pg_catalog.pg_am a ON a.oid=c.relam + UNION ALL SELECT 'column',c.relname||':'||a.attnum,jsonb_build_array(c.relname,a.attnum,a.attname, + CASE WHEN tn.oid=q_oid THEN '$Q_SCHEMA$' ELSE tn.nspname END,t.typname,a.attnotnull,a.attisdropped,a.attidentity,a.attgenerated,a.attislocal,a.attinhcount, + a.atttypmod,a.attndims,a.attacl,a.attlen,a.attbyval,a.attalign,a.attstorage,a.attcompression, + CASE WHEN cn.oid=q_oid THEN '$Q_SCHEMA$' ELSE cn.nspname END,co.collname, + pg_catalog.replace(pg_catalog.pg_get_expr(d.adbin,d.adrelid),(SELECT nspname FROM q),'$Q_SCHEMA$')) + FROM pg_catalog.pg_attribute a JOIN relations c ON c.oid=a.attrelid + JOIN pg_catalog.pg_type t ON t.oid=a.atttypid JOIN pg_catalog.pg_namespace tn ON tn.oid=t.typnamespace + LEFT JOIN pg_catalog.pg_collation co ON co.oid=a.attcollation LEFT JOIN pg_catalog.pg_namespace cn ON cn.oid=co.collnamespace + LEFT JOIN pg_catalog.pg_attrdef d ON d.adrelid=a.attrelid AND d.adnum=a.attnum WHERE a.attnum>0 + UNION ALL SELECT 'constraint',c.relname||':'||x.conname,jsonb_build_array(c.relname,x.conname,x.contype, + pg_catalog.replace(pg_catalog.pg_get_constraintdef(x.oid,true),(SELECT nspname FROM q),'$Q_SCHEMA$'), + x.condeferrable,x.condeferred,x.convalidated,x.conislocal,x.coninhcount,x.connoinherit) + FROM pg_catalog.pg_constraint x JOIN relations c ON c.oid=x.conrelid + UNION ALL SELECT 'index',c.relname||':'||ic.relname,jsonb_build_array(c.relname,ic.relname, + pg_catalog.replace(pg_catalog.pg_get_indexdef(i.indexrelid),(SELECT nspname FROM q),'$Q_SCHEMA$'), + i.indisunique,i.indisprimary,i.indisexclusion,i.indimmediate,i.indisvalid,i.indisready,i.indislive, + i.indnullsnotdistinct,i.indisclustered,i.indisreplident) + FROM pg_catalog.pg_index i JOIN relations c ON c.oid=i.indrelid JOIN relations ic ON ic.oid=i.indexrelid + UNION ALL SELECT 'function',p.proname||':'||pg_catalog.replace(pg_catalog.pg_get_function_identity_arguments(p.oid),(SELECT nspname FROM q),'$Q_SCHEMA$'),jsonb_build_array( + p.proname,pg_catalog.replace(pg_catalog.pg_get_function_identity_arguments(p.oid),(SELECT nspname FROM q),'$Q_SCHEMA$'), + pg_catalog.replace(pg_catalog.pg_get_function_result(p.oid),(SELECT nspname FROM q),'$Q_SCHEMA$'),l.lanname, + p.proowner,p.proacl,p.prokind,p.provolatile,p.proisstrict,p.prosecdef,p.proleakproof,p.proparallel, + p.procost,p.prorows,p.probin,p.pronargdefaults, + pg_catalog.replace(pg_catalog.pg_get_expr(p.proargdefaults,0),(SELECT nspname FROM q),'$Q_SCHEMA$'), + pg_catalog.replace(pg_catalog.replace(pg_catalog.replace(pg_catalog.replace(p.prosrc, + pg_catalog.quote_literal((SELECT nspname FROM q)),pg_catalog.quote_literal('$Q_SCHEMA$')), + pg_catalog.quote_literal(q_oid::text),pg_catalog.quote_literal('$Q_OID$')), + pg_catalog.quote_literal(n_uuid::text),pg_catalog.quote_literal('$NAMESPACE_UUID$')), + pg_catalog.quote_literal(s_uuid),pg_catalog.quote_literal('$STORAGE_UUID$')), + pg_catalog.replace(pg_catalog.array_to_string(p.proconfig,E'\n'),(SELECT nspname FROM q),'$Q_SCHEMA$')) + FROM pg_catalog.pg_proc p JOIN pg_catalog.pg_language l ON l.oid=p.prolang WHERE p.pronamespace=q_oid + UNION ALL SELECT 'trigger',c.relname||':'||p.proname||':'||t.tgtype||':'||coalesce(x.conname,t.tgname),jsonb_build_array( + CASE WHEN c.relnamespace=q_oid THEN '$Q_SCHEMA$' ELSE cns.nspname END,c.relname, + CASE WHEN t.tgisinternal THEN NULL ELSE t.tgname END,t.tgtype,t.tgenabled,t.tgisinternal, + CASE WHEN pn.oid=q_oid THEN '$Q_SCHEMA$' ELSE pn.nspname END,p.proname, + pg_catalog.replace(pg_catalog.pg_get_function_identity_arguments(p.oid),(SELECT nspname FROM q),'$Q_SCHEMA$'), + t.tgdeferrable,t.tginitdeferred,t.tgnargs, + pg_catalog.replace(encode(t.tgargs,'hex'),encode(pg_catalog.convert_to((SELECT nspname FROM q),'UTF8'),'hex'), + encode(pg_catalog.convert_to('$Q_SCHEMA$','UTF8'),'hex')),t.tgattr::text, + pg_catalog.replace(pg_catalog.pg_get_expr(t.tgqual,t.tgrelid),(SELECT nspname FROM q),'$Q_SCHEMA$'),t.tgoldtable,t.tgnewtable,x.conname, + CASE WHEN rn.oid=q_oid THEN '$Q_SCHEMA$' ELSE rn.nspname END,rc.relname, + CASE WHEN t.tgparentid=0 THEN NULL ELSE 'unexpected parent trigger' END) + FROM pg_catalog.pg_trigger t JOIN pg_catalog.pg_class c ON c.oid=t.tgrelid + JOIN pg_catalog.pg_namespace cns ON cns.oid=c.relnamespace + JOIN pg_catalog.pg_proc p ON p.oid=t.tgfoid JOIN pg_catalog.pg_namespace pn ON pn.oid=p.pronamespace + LEFT JOIN pg_catalog.pg_constraint x ON x.oid=t.tgconstraint + LEFT JOIN pg_catalog.pg_class rc ON rc.oid=t.tgconstrrelid LEFT JOIN pg_catalog.pg_namespace rn ON rn.oid=rc.relnamespace + WHERE c.relnamespace=q_oid OR EXISTS(SELECT 1 FROM pg_catalog.pg_constraint fk + WHERE fk.oid=t.tgconstraint AND fk.conrelid IN (SELECT oid FROM relations)) + UNION ALL SELECT 'rule',c.relname||':'||r.rulename,pg_catalog.to_jsonb(r) + FROM pg_catalog.pg_rewrite r JOIN relations c ON c.oid=r.ev_class + UNION ALL SELECT 'policy',c.relname||':'||p.polname,pg_catalog.to_jsonb(p) + FROM pg_catalog.pg_policy p JOIN relations c ON c.oid=p.polrelid +) +SELECT pg_catalog.sha256(pg_catalog.convert_to( + pg_catalog.jsonb_agg(pg_catalog.jsonb_build_array(kind,key,value) ORDER BY kind,key,value::text)::text,'UTF8')) AS fingerprint +FROM objects diff --git a/src/jupiter/storage/qualified_metadata_anchors.sql b/src/jupiter/storage/qualified_metadata_anchors.sql new file mode 100644 index 00000000..08b3d7ea --- /dev/null +++ b/src/jupiter/storage/qualified_metadata_anchors.sql @@ -0,0 +1,426 @@ +CREATE TABLE mst2_metadata_source_root_attestation ( + attestation_id uuid PRIMARY KEY,namespace_uuid uuid NOT NULL REFERENCES mst2_metadata_family_identity(namespace_uuid), + origin_prepare_id text NOT NULL REFERENCES mst2_metadata_prepare(prepare_id), + tagged_tree_oid text NOT NULL CHECK(tagged_tree_oid ~ '^(sha1:[0-9a-f]{40}|sha256:[0-9a-f]{64}|blake3:[0-9a-f]{64})$'), + source_profile jsonb NOT NULL,profile_digest bytea NOT NULL CHECK(octet_length(profile_digest)=32), + source_body_digest bytea NOT NULL CHECK(octet_length(source_body_digest)=32), + source_revision uuid NOT NULL, + root_page bytea NOT NULL,root_generation bigint NOT NULL, + root_certificate_digest bytea NOT NULL CHECK(octet_length(root_certificate_digest)=32), + source_proof jsonb NOT NULL,attestation_digest bytea NOT NULL CHECK(octet_length(attestation_digest)=32), + FOREIGN KEY(root_page,root_generation,root_certificate_digest) + REFERENCES mst2_metadata_page_certificate(page_id,generation,certificate_digest), + UNIQUE(attestation_id,root_page,root_generation,attestation_digest) +); +CREATE INDEX mst2_metadata_source_root_lookup ON mst2_metadata_source_root_attestation(profile_digest,tagged_tree_oid,root_page,root_generation); +CREATE TABLE mst2_metadata_prepare_reuse_root ( + prepare_id text NOT NULL REFERENCES mst2_metadata_prepare(prepare_id),root_page bytea NOT NULL, + root_generation bigint NOT NULL,attestation_id uuid NOT NULL,attestation_digest bytea NOT NULL, + PRIMARY KEY(prepare_id,root_page), + FOREIGN KEY(attestation_id,root_page,root_generation,attestation_digest) + REFERENCES mst2_metadata_source_root_attestation(attestation_id,root_page,root_generation,attestation_digest) +); +CREATE INDEX mst2_metadata_prepare_reuse_root_page ON mst2_metadata_prepare_reuse_root(root_page,root_generation,prepare_id); +CREATE TABLE mst2_metadata_reuse_index ( + profile_digest bytea NOT NULL,tagged_tree_oid text NOT NULL,attestation_id uuid NOT NULL, + root_page bytea NOT NULL,root_generation bigint NOT NULL,attestation_digest bytea NOT NULL, + PRIMARY KEY(profile_digest,tagged_tree_oid), + FOREIGN KEY(attestation_id,root_page,root_generation,attestation_digest) + REFERENCES mst2_metadata_source_root_attestation(attestation_id,root_page,root_generation,attestation_digest) +); +CREATE INDEX mst2_metadata_reuse_index_page ON mst2_metadata_reuse_index(root_page,root_generation); + +ALTER TABLE mst2_qualified_session_incarnation ADD COLUMN canonical_descriptor bytea NOT NULL, + ADD COLUMN attestation_id uuid NOT NULL,ADD COLUMN attestation_digest bytea NOT NULL, + ADD COLUMN state text NOT NULL CHECK(state IN ('READY','RETIRED')), + ADD FOREIGN KEY(attestation_id,metadata_root,root_generation,attestation_digest) + REFERENCES mst2_metadata_source_root_attestation(attestation_id,root_page,root_generation,attestation_digest); +ALTER TABLE mst2_qualified_lease_binding ADD COLUMN namespace_uuid uuid NOT NULL, + ADD COLUMN authorization_epoch bigint NOT NULL,ADD COLUMN publication_sequence bigint NOT NULL, + ADD COLUMN writer_epoch bigint NOT NULL,ADD COLUMN certificate_receipt_id bigint NOT NULL, + ADD COLUMN expires_at_unix bigint NOT NULL,ADD COLUMN state text NOT NULL CHECK(state IN ('ACTIVE','RELEASED','EXPIRED')), + ADD COLUMN lease_epoch bigint NOT NULL CHECK(lease_epoch>0); +CREATE INDEX mst2_qualified_lease_active_incarnation ON mst2_qualified_lease_binding(snapshot_id,session_incarnation,state,lease_id); +CREATE TABLE mst2_metadata_reader_operation ( + operation_id uuid PRIMARY KEY,lease_id text NOT NULL REFERENCES mst2_qualified_lease_binding(lease_id), + snapshot_id text NOT NULL,session_incarnation uuid NOT NULL,root_page bytea NOT NULL,root_generation bigint NOT NULL, + lease_epoch bigint NOT NULL CHECK(lease_epoch>0),hard_deadline_unix bigint NOT NULL, + state text NOT NULL CHECK(state IN ('ACTIVE','FINISHED','EXPIRED')), + FOREIGN KEY(snapshot_id,session_incarnation) REFERENCES mst2_qualified_session_incarnation(snapshot_id,session_incarnation) +); +CREATE INDEX mst2_metadata_reader_active_lease ON mst2_metadata_reader_operation(lease_id,state,operation_id); +CREATE INDEX mst2_metadata_reader_active_deadline ON mst2_metadata_reader_operation(hard_deadline_unix,operation_id) WHERE state='ACTIVE'; +CREATE TABLE mst2_metadata_root_anchor ( + anchor_id uuid PRIMARY KEY,anchor_kind text NOT NULL CHECK(anchor_kind IN ('PREPARE','REUSE','SESSION','LEASE','REQUEST','READER')), + owner_key text NOT NULL CHECK(octet_length(owner_key) BETWEEN 1 AND 512), + root_page bytea NOT NULL,root_generation bigint NOT NULL,root_certificate_digest bytea NOT NULL, + prepare_id text REFERENCES mst2_metadata_prepare(prepare_id), + snapshot_id text,session_incarnation uuid,lease_id text REFERENCES mst2_qualified_lease_binding(lease_id), + reader_operation_id uuid REFERENCES mst2_metadata_reader_operation(operation_id), + UNIQUE(anchor_kind,owner_key,root_page,root_generation), + FOREIGN KEY(root_page,root_generation) REFERENCES mst2_metadata_graph_node(page_id,generation), + FOREIGN KEY(root_page,root_generation,root_certificate_digest) + REFERENCES mst2_metadata_page_certificate(page_id,generation,certificate_digest), + FOREIGN KEY(snapshot_id,session_incarnation) REFERENCES mst2_qualified_session_incarnation(snapshot_id,session_incarnation) +); +CREATE INDEX mst2_metadata_root_anchor_page ON mst2_metadata_root_anchor(root_page,root_generation,anchor_kind,owner_key); +CREATE INDEX mst2_metadata_root_anchor_prepare ON mst2_metadata_root_anchor(prepare_id,anchor_kind,anchor_id); +CREATE INDEX mst2_metadata_root_anchor_lease ON mst2_metadata_root_anchor(lease_id,anchor_kind,anchor_id); + +CREATE FUNCTION mst2_metadata_session_covers_prepare(pid text) RETURNS boolean LANGUAGE sql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ + SELECT EXISTS(SELECT 1 FROM mst2_metadata_prepare q + JOIN mst2_metadata_current cur ON cur.page_id=q.metadata_root + JOIN mst2_metadata_page_certificate certificate USING(page_id,generation) + JOIN mst2_qualified_session_incarnation session ON session.metadata_root=q.metadata_root + AND session.root_generation=cur.generation + JOIN mst2_metadata_source_root_attestation source ON source.attestation_id=session.attestation_id + AND source.root_page=session.metadata_root AND source.root_generation=session.root_generation + AND source.attestation_digest=session.attestation_digest + AND source.root_certificate_digest=certificate.certificate_digest + JOIN mst2_metadata_root_anchor anchor ON anchor.snapshot_id=session.snapshot_id + AND anchor.session_incarnation=session.session_incarnation AND anchor.anchor_kind='SESSION' + AND anchor.owner_key=session.snapshot_id||':'||session.session_incarnation::text + AND anchor.root_page=session.metadata_root AND anchor.root_generation=session.root_generation + AND anchor.root_certificate_digest=certificate.certificate_digest + WHERE q.prepare_id=pid AND q.plan_kind='ROOTED' AND q.state='COMMITTED' AND session.state='READY' + AND session.namespace_uuid=(SELECT namespace_uuid FROM mst2_metadata_family_identity WHERE singleton=1) + AND convert_from(session.source_profile,'UTF8')::jsonb=$CORE_SCHEMA$.mst2_route_profile( + q.source_domain,q.tagged_root_tree_oid,q.scope,q.schema_version,q.metadata_codec, + q.materialization_policy,q.fs_semantics,q.access_projection,q.verification_revision,q.projection_revision)) +$$; + +CREATE FUNCTION mst2_metadata_native_profile(pid text) RETURNS jsonb LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE q record; +BEGIN + SELECT source_domain,tagged_root_tree_oid,schema_version,metadata_codec,materialization_policy, + fs_semantics,access_projection,verification_revision,projection_revision + INTO q FROM mst2_metadata_prepare WHERE prepare_id=pid AND state IN ('PREPARING','COMMITTED') + AND graph_domain='qualified-v1' AND mst2_metadata_scope_matches(primary_scope); + IF NOT FOUND OR q.source_domain<>'native-git' OR q.schema_version<>2 OR q.metadata_codec<>1 + OR q.materialization_policy<>1 OR q.fs_semantics<>1 OR q.access_projection<>0 + OR q.verification_revision<>2 OR q.projection_revision<>1 + OR q.tagged_root_tree_oid !~ '^(sha1:[0-9a-f]{40}|sha256:[0-9a-f]{64}|blake3:[0-9a-f]{64})$' THEN + RAISE EXCEPTION 'source attestation requires the exact current native metadata profile'; + END IF; + RETURN jsonb_build_object('source_domain',q.source_domain,'hash_kind',split_part(q.tagged_root_tree_oid,':',1), + 'schema_version',q.schema_version,'metadata_codec',q.metadata_codec,'materialization_policy',q.materialization_policy, + 'fs_semantics',q.fs_semantics,'access_projection',q.access_projection, + 'verification_revision',q.verification_revision,'projection_revision',q.projection_revision); +END $$; + +CREATE FUNCTION mst2_metadata_decode_git_tree(b bytea,hash_kind text) RETURNS jsonb +LANGUAGE plpgsql IMMUTABLE STRICT SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE width integer; p integer:=0; start integer; mode text; name bytea; oid bytea; kind integer; + result jsonb[]:=ARRAY[]::jsonb[]; sorted jsonb; count integer:=0; +BEGIN + width:=CASE hash_kind WHEN 'sha1' THEN 20 WHEN 'sha256' THEN 32 WHEN 'blake3' THEN 32 ELSE 0 END; + IF width=0 OR octet_length(b)>67108864 THEN RAISE EXCEPTION 'source Git tree kind or byte budget is invalid'; END IF; + WHILE p32 AND p-start<6 LOOP p:=p+1; END LOOP; + IF p>=octet_length(b) OR get_byte(b,p)<>32 THEN RAISE EXCEPTION 'source Git tree mode is malformed'; END IF; + mode:=convert_from(substring(b FROM start+1 FOR p-start),'UTF8'); p:=p+1; start:=p; + WHILE p0 AND p-start<=255 LOOP p:=p+1; END LOOP; + IF p>=octet_length(b) OR get_byte(b,p)<>0 THEN RAISE EXCEPTION 'source Git tree name is malformed'; END IF; + name:=substring(b FROM start+1 FOR p-start); PERFORM mst2_metadata_valid_name(name); p:=p+1; + IF p>octet_length(b)-width THEN RAISE EXCEPTION 'source Git tree object identity is truncated'; END IF; + oid:=substring(b FROM p+1 FOR width); p:=p+width; + kind:=CASE mode WHEN '40000' THEN 4 WHEN '100644' THEN 1 WHEN '100664' THEN 1 WHEN '100640' THEN 1 + WHEN '100755' THEN 2 WHEN '120000' THEN 3 ELSE 0 END; + IF kind=0 THEN RAISE EXCEPTION 'source Git tree contains an unsupported entry'; END IF; + result:=array_append(result,jsonb_build_object('kind',kind,'name',encode(name,'hex'), + 'oid',hash_kind||':'||encode(oid,'hex'))); count:=count+1; + IF count>131072 THEN RAISE EXCEPTION 'source Git tree entry budget exceeded'; END IF; + END LOOP; + IF EXISTS(SELECT 1 FROM unnest(result) AS rows(value) GROUP BY value->>'name' HAVING count(*)>1) THEN + RAISE EXCEPTION 'source Git tree has duplicate names'; + END IF; + SELECT coalesce(jsonb_agg(value ORDER BY decode(value->>'name','hex')),'[]'::jsonb) INTO sorted + FROM unnest(result) AS rows(value); + RETURN sorted; +END $$; + +CREATE FUNCTION mst2_metadata_compute_source_proof(pid text,tree_oid text,p bytea,g bigint) RETURNS jsonb +LANGUAGE plpgsql VOLATILE SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE profile jsonb:=mst2_metadata_native_profile(pid); profile_hash bytea; body bytea; decoded jsonb; + item jsonb; mapped jsonb; entry_rows jsonb[]:=ARRAY[]::jsonb[]; entries jsonb; child_root bytea; + child_binding record; source_entries jsonb; reference_rows jsonb[]:=ARRAY[]::jsonb[]; + reference jsonb; fact record; built jsonb; proof jsonb; source_revision uuid; body_digest bytea; +BEGIN + IF split_part(tree_oid,':',1) IS DISTINCT FROM profile->>'hash_kind' + OR tree_oid !~ '^(sha1:[0-9a-f]{40}|sha256:[0-9a-f]{64}|blake3:[0-9a-f]{64})$' THEN + RAISE EXCEPTION 'source tree attestation crossed its tagged hash kind'; + END IF; + profile_hash:=sha256(convert_to('mega.mst2.native-profile.v1','UTF8')||decode('00','hex')||convert_to(profile::text,'UTF8')); + SELECT sub_trees INTO body FROM $CORE_SCHEMA$.mega_tree WHERE tree_id=split_part(tree_oid,':',2); + IF NOT FOUND THEN RAISE EXCEPTION 'source Git tree is missing from its captured core'; END IF; + body_digest:=sha256(body); + source_revision:=$CORE_SCHEMA$.mst2_route_capture_source_tree(split_part(tree_oid,':',2),body_digest); + decoded:=mst2_metadata_decode_git_tree(body,profile->>'hash_kind'); + FOR item IN SELECT value FROM jsonb_array_elements(decoded) LOOP + mapped:=jsonb_build_object('kind',(item->>'kind')::integer,'name',item->>'name'); + reference:=jsonb_build_object('kind',(item->>'kind')::integer,'git_oid',item->>'oid'); + IF (item->>'kind')::integer=4 THEN + SELECT a.root_page,a.root_generation,a.root_certificate_digest INTO child_binding FROM mst2_metadata_source_root_attestation a + JOIN mst2_metadata_current cur ON cur.page_id=a.root_page AND cur.generation=a.root_generation + JOIN mst2_metadata_lifetime life USING(page_id,generation) + JOIN mst2_metadata_graph_node node USING(page_id,generation) + JOIN mst2_metadata_prepare origin ON origin.prepare_id=a.origin_prepare_id + JOIN $CORE_SCHEMA$.mega_tree child_tree ON child_tree.tree_id=split_part(a.tagged_tree_oid,':',2) + WHERE a.tagged_tree_oid=item->>'oid' AND a.profile_digest=profile_hash AND a.source_profile=profile + AND a.namespace_uuid=(SELECT namespace_uuid FROM mst2_metadata_family_identity WHERE singleton=1) + AND node.state='LIVE' AND node.certificate_digest=a.root_certificate_digest + AND $CORE_SCHEMA$.mst2_route_source_tree_matches(split_part(a.tagged_tree_oid,':',2),a.source_revision,a.source_body_digest) + AND (life.state='LIVE' AND origin.state='COMMITTED' OR life.state='RESERVED' AND origin.prepare_id=pid AND origin.state='PREPARING') + AND NOT EXISTS(SELECT 1 FROM mst2_metadata_gc_op gc WHERE gc.page_id=a.root_page AND gc.generation=a.root_generation) + ORDER BY a.attestation_id LIMIT 1; + IF NOT FOUND THEN RAISE EXCEPTION 'source directory lacks its exact-profile certified child root'; END IF; + child_root:=child_binding.root_page; + mapped:=mapped||jsonb_build_object('child',encode(child_root,'hex')); + reference:=reference||jsonb_build_object('child_root',encode(child_root,'hex'), + 'child_generation',child_binding.root_generation,'child_certificate',encode(child_binding.root_certificate_digest,'hex')); + ELSE + SELECT size,raw_sha256 INTO fact FROM $CORE_SCHEMA$.mst2_verified_object + WHERE storage_domain='git' AND git_oid=split_part(item->>'oid',':',2) AND object_kind='blob' + AND state='VERIFIED' AND verification_version=2 FOR SHARE NOWAIT; + IF NOT FOUND OR fact.size<0 OR fact.size>8796093022208 OR octet_length(fact.raw_sha256)<>32 + OR (item->>'kind')::integer=3 AND fact.size NOT BETWEEN 1 AND 4095 THEN + RAISE EXCEPTION 'source file lacks its valid current verified-object fact'; + END IF; + mapped:=mapped||jsonb_build_object('size',fact.size,'content_id',encode(fact.raw_sha256,'hex')); + reference:=reference||jsonb_build_object('size',fact.size,'content_digest',encode(fact.raw_sha256,'hex')); + END IF; + reference_rows:=array_append(reference_rows,reference||jsonb_build_object('name',item->>'name')); + entry_rows:=array_append(entry_rows,mapped); + END LOOP; + entries:=to_jsonb(entry_rows); + SELECT coalesce(jsonb_object_agg(value->>'name',value-'name'),'{}'::jsonb) INTO source_entries + FROM unnest(reference_rows) input(value); + built:=mst2_metadata_build_map(entries); + IF decode(built->>'page_id','hex')<>p OR NOT EXISTS(SELECT 1 FROM mst2_metadata_page_certificate c + JOIN mst2_metadata_current cur USING(page_id,generation) JOIN mst2_metadata_graph_node n USING(page_id,generation) + WHERE c.page_id=p AND c.generation=g AND n.state='LIVE' AND n.certificate_digest=c.certificate_digest) THEN + RAISE EXCEPTION 'source projection differs from its independently canonical certified root'; + END IF; + proof:=jsonb_build_object('namespace',(SELECT namespace_uuid::text FROM mst2_metadata_family_identity WHERE singleton=1), + 'source_profile',profile,'profile_digest',encode(profile_hash,'hex'),'tagged_tree_oid',tree_oid, + 'source_body_digest',encode(body_digest,'hex'),'source_revision',source_revision::text,'root_page',encode(p,'hex'),'root_generation',g, + 'root_certificate',(SELECT encode(certificate_digest,'hex') FROM mst2_metadata_page_certificate WHERE page_id=p AND generation=g), + 'source_entry_count',jsonb_array_length(entries),'source_entries',source_entries,'source_work_units',built->'source_work_units', + 'encoded_map_digest',encode(sha256(convert_to(entries::text,'UTF8')),'hex')); + RETURN proof||jsonb_build_object('attestation',encode(sha256(convert_to('mega.mst2.source-root.v1','UTF8') + ||decode('00','hex')||convert_to(proof::text,'UTF8')),'hex')); +END $$; + +CREATE FUNCTION mst2_metadata_source_attestation_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE proof jsonb; +BEGIN + IF TG_OP<>'INSERT' THEN RAISE EXCEPTION 'source root attestation history is immutable'; END IF; + proof:=mst2_metadata_compute_source_proof(NEW.origin_prepare_id,NEW.tagged_tree_oid,NEW.root_page,NEW.root_generation); + IF NEW.source_revision IS NULL THEN NEW.source_revision:=(proof->>'source_revision')::uuid; END IF; + IF NEW.namespace_uuid::text IS DISTINCT FROM proof->>'namespace' OR NEW.source_profile IS DISTINCT FROM proof->'source_profile' + OR NEW.profile_digest IS DISTINCT FROM decode(proof->>'profile_digest','hex') + OR NEW.source_body_digest IS DISTINCT FROM decode(proof->>'source_body_digest','hex') + OR NEW.source_revision IS DISTINCT FROM (proof->>'source_revision')::uuid + OR NEW.root_certificate_digest IS DISTINCT FROM decode(proof->>'root_certificate','hex') + OR NEW.attestation_digest IS DISTINCT FROM decode(proof->>'attestation','hex') OR NEW.source_proof IS DISTINCT FROM proof THEN + RAISE EXCEPTION 'source attestation was not independently derived from the captured core and canonical root'; + END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_metadata_source_attestation_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_metadata_source_root_attestation + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_source_attestation_guard(); + +CREATE FUNCTION mst2_metadata_source_attestation_committed() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + IF NOT EXISTS(SELECT 1 FROM mst2_metadata_prepare WHERE prepare_id=NEW.origin_prepare_id AND state='COMMITTED') THEN + RAISE EXCEPTION 'source attestation cannot commit without its definitive finalized preparation'; + END IF; + IF NOT $CORE_SCHEMA$.mst2_route_source_tree_matches(split_part(NEW.tagged_tree_oid,':',2),NEW.source_revision,NEW.source_body_digest) + OR NEW.source_profile IS DISTINCT FROM mst2_metadata_native_profile(NEW.origin_prepare_id) THEN + RAISE EXCEPTION 'source attestation cannot commit after its exact captured source body or profile changed'; + END IF; + RETURN NULL; +END $$; +CREATE CONSTRAINT TRIGGER mst2_metadata_source_attestation_committed AFTER INSERT ON mst2_metadata_source_root_attestation + DEFERRABLE INITIALLY DEFERRED FOR EACH ROW EXECUTE FUNCTION mst2_metadata_source_attestation_committed(); + +CREATE FUNCTION mst2_metadata_reuse_root_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE profile jsonb; profile_hash bytea; +BEGIN + IF TG_OP<>'INSERT' THEN RAISE EXCEPTION 'rooted reuse membership history is immutable'; END IF; + profile:=mst2_metadata_native_profile(NEW.prepare_id); + profile_hash:=sha256(convert_to('mega.mst2.native-profile.v1','UTF8')||decode('00','hex')||convert_to(profile::text,'UTF8')); + IF NOT EXISTS(SELECT 1 FROM mst2_metadata_prepare q,mst2_metadata_source_root_attestation a + JOIN mst2_metadata_current cur ON cur.page_id=a.root_page AND cur.generation=a.root_generation + JOIN mst2_metadata_lifetime life USING(page_id,generation) + JOIN mst2_metadata_graph_node node USING(page_id,generation) + JOIN mst2_metadata_prepare origin ON origin.prepare_id=a.origin_prepare_id + JOIN $CORE_SCHEMA$.mega_tree tree ON tree.tree_id=split_part(a.tagged_tree_oid,':',2) + WHERE q.prepare_id=NEW.prepare_id AND q.state='PREPARING' AND a.attestation_id=NEW.attestation_id + AND a.root_page=NEW.root_page AND a.root_generation=NEW.root_generation AND a.attestation_digest=NEW.attestation_digest + AND a.profile_digest=profile_hash AND a.source_profile=profile + AND $CORE_SCHEMA$.mst2_route_source_tree_matches(split_part(a.tagged_tree_oid,':',2),a.source_revision,a.source_body_digest) + AND origin.state='COMMITTED' AND life.state='LIVE' AND life.graph_domain='qualified-v1' + AND node.state='LIVE' AND node.certificate_digest=a.root_certificate_digest + AND NOT EXISTS(SELECT 1 FROM mst2_metadata_gc_op gc WHERE gc.page_id=a.root_page AND gc.generation=a.root_generation)) THEN + RAISE EXCEPTION 'rooted reuse membership lacks its exact finalized source and current lifetime'; + END IF; + IF (SELECT count(*) FROM mst2_metadata_prepare_reuse_root WHERE prepare_id=NEW.prepare_id)>=4096 THEN + RAISE EXCEPTION 'rooted reuse boundary budget exceeded'; + END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_metadata_reuse_root_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_metadata_prepare_reuse_root + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_reuse_root_guard(); + +CREATE FUNCTION mst2_metadata_reuse_index_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + IF TG_OP='UPDATE' THEN RAISE EXCEPTION 'rooted reuse index cannot retarget an exact lifetime'; END IF; + IF TG_OP='DELETE' THEN + IF NOT EXISTS(SELECT 1 FROM mst2_metadata_gc_op WHERE page_id=OLD.root_page AND generation=OLD.root_generation AND state='PENDING') THEN + RAISE EXCEPTION 'rooted reuse index retirement requires its exact GC claim'; + END IF; + RETURN OLD; + END IF; + IF NOT EXISTS(SELECT 1 FROM mst2_metadata_source_root_attestation a + JOIN mst2_metadata_prepare q ON q.prepare_id=a.origin_prepare_id + JOIN mst2_metadata_current cur ON cur.page_id=a.root_page AND cur.generation=a.root_generation + JOIN mst2_metadata_lifetime life USING(page_id,generation) + JOIN mst2_metadata_graph_node node USING(page_id,generation) + WHERE a.attestation_id=NEW.attestation_id AND a.profile_digest=NEW.profile_digest AND a.tagged_tree_oid=NEW.tagged_tree_oid + AND a.root_page=NEW.root_page AND a.root_generation=NEW.root_generation AND a.attestation_digest=NEW.attestation_digest + AND q.state='COMMITTED' AND life.state='LIVE' AND node.state='LIVE' AND node.certificate_digest=a.root_certificate_digest + AND NOT EXISTS(SELECT 1 FROM mst2_metadata_gc_op gc WHERE gc.page_id=a.root_page AND gc.generation=a.root_generation)) THEN + RAISE EXCEPTION 'rooted reuse index is not bound to its definitive exact current source proof'; + END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_metadata_reuse_index_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_metadata_reuse_index + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_reuse_index_guard(); + +CREATE FUNCTION mst2_metadata_root_anchor_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + IF TG_OP='UPDATE' THEN RAISE EXCEPTION 'rooted anchors cannot retarget immutable owner or root identities'; END IF; + IF TG_OP='DELETE' THEN + IF OLD.anchor_kind IN ('PREPARE','REUSE') THEN + IF NOT EXISTS(SELECT 1 FROM mst2_metadata_prepare WHERE prepare_id=OLD.prepare_id AND state IN ('PREPARING','COMMITTED','ABORTED')) THEN + RAISE EXCEPTION 'temporary anchor owner history is missing'; + END IF; + ELSIF OLD.anchor_kind='SESSION' THEN + IF NOT EXISTS(SELECT 1 FROM mst2_qualified_session_incarnation s + WHERE s.snapshot_id=OLD.snapshot_id AND s.session_incarnation=OLD.session_incarnation AND s.state='RETIRED') + OR EXISTS(SELECT 1 FROM mst2_qualified_lease_binding l WHERE l.snapshot_id=OLD.snapshot_id + AND l.session_incarnation=OLD.session_incarnation AND l.state='ACTIVE') THEN + RAISE EXCEPTION 'session root still has its active incarnation or leases'; + END IF; + ELSIF OLD.anchor_kind='LEASE' THEN + IF NOT EXISTS(SELECT 1 FROM mst2_qualified_lease_binding WHERE lease_id=OLD.lease_id AND state IN ('RELEASED','EXPIRED')) + OR EXISTS(SELECT 1 FROM mst2_metadata_reader_operation WHERE lease_id=OLD.lease_id AND state='ACTIVE') THEN + RAISE EXCEPTION 'lease root still has its active lease or readers'; + END IF; + ELSE + IF NOT EXISTS(SELECT 1 FROM mst2_metadata_reader_operation r WHERE r.operation_id=OLD.reader_operation_id + AND (r.state='FINISHED' OR r.state='EXPIRED' AND r.hard_deadline_unix<=floor(extract(epoch FROM clock_timestamp()))::bigint)) THEN + RAISE EXCEPTION 'reader root still has an active operation'; + END IF; + END IF; + RETURN OLD; + END IF; + IF NOT EXISTS(SELECT 1 FROM mst2_metadata_graph_node node + JOIN mst2_metadata_current cur USING(page_id,generation) + JOIN mst2_metadata_lifetime life USING(page_id,generation) + JOIN mst2_metadata_page_certificate proof USING(page_id,generation) + WHERE node.page_id=NEW.root_page AND node.generation=NEW.root_generation AND node.state='LIVE' + AND node.certificate_digest=NEW.root_certificate_digest AND proof.certificate_digest=NEW.root_certificate_digest + AND life.state IN ('RESERVED','LIVE') AND life.graph_domain='qualified-v1' + AND NOT EXISTS(SELECT 1 FROM mst2_metadata_gc_op gc WHERE gc.page_id=node.page_id AND gc.generation=node.generation)) THEN + RAISE EXCEPTION 'rooted anchor does not protect its exact canonical current graph'; + END IF; + IF NEW.anchor_kind IN ('PREPARE','REUSE') THEN + IF NEW.owner_key IS DISTINCT FROM NEW.prepare_id OR NEW.snapshot_id IS NOT NULL OR NEW.session_incarnation IS NOT NULL + OR NEW.lease_id IS NOT NULL OR NEW.reader_operation_id IS NOT NULL + OR NOT EXISTS(SELECT 1 FROM mst2_metadata_prepare WHERE prepare_id=NEW.prepare_id + AND state IN ('PREPARING','COMMITTED') AND coverage_retired_at IS NULL) + OR NEW.anchor_kind='PREPARE' AND NOT (EXISTS(SELECT 1 FROM mst2_metadata_prepare_page + WHERE prepare_id=NEW.prepare_id AND page_id=NEW.root_page AND generation=NEW.root_generation) + OR EXISTS(SELECT 1 FROM mst2_metadata_prepare q JOIN mst2_metadata_prepare_reuse_root r USING(prepare_id) + WHERE q.prepare_id=NEW.prepare_id AND q.plan_kind='ROOTED' AND q.metadata_root=NEW.root_page + AND r.root_page=NEW.root_page AND r.root_generation=NEW.root_generation)) + OR NEW.anchor_kind='REUSE' AND NOT EXISTS(SELECT 1 FROM mst2_metadata_prepare_reuse_root + WHERE prepare_id=NEW.prepare_id AND root_page=NEW.root_page AND root_generation=NEW.root_generation) THEN + RAISE EXCEPTION 'temporary anchor differs from its immutable delta or reused-root owner'; + END IF; + ELSIF NEW.anchor_kind='SESSION' THEN + IF NEW.owner_key IS DISTINCT FROM NEW.snapshot_id||':'||NEW.session_incarnation::text OR NEW.prepare_id IS NOT NULL + OR NEW.lease_id IS NOT NULL OR NEW.reader_operation_id IS NOT NULL + OR NOT EXISTS(SELECT 1 FROM mst2_qualified_session_incarnation s WHERE s.snapshot_id=NEW.snapshot_id + AND s.session_incarnation=NEW.session_incarnation AND s.metadata_root=NEW.root_page AND s.root_generation=NEW.root_generation AND s.state='READY') THEN + RAISE EXCEPTION 'session anchor differs from its exact ready incarnation'; + END IF; + ELSIF NEW.anchor_kind='LEASE' THEN + IF NEW.owner_key IS DISTINCT FROM NEW.lease_id OR NEW.prepare_id IS NOT NULL OR NEW.reader_operation_id IS NOT NULL + OR NOT EXISTS(SELECT 1 FROM mst2_qualified_lease_binding l WHERE l.lease_id=NEW.lease_id + AND l.snapshot_id=NEW.snapshot_id AND l.session_incarnation=NEW.session_incarnation + AND l.metadata_root=NEW.root_page AND l.root_generation=NEW.root_generation AND l.state='ACTIVE' + AND l.expires_at_unix>floor(extract(epoch FROM clock_timestamp()))::bigint) THEN + RAISE EXCEPTION 'lease anchor differs from its exact active lease'; + END IF; + ELSE + IF NEW.owner_key IS DISTINCT FROM NEW.reader_operation_id::text OR NEW.prepare_id IS NOT NULL + OR NOT EXISTS(SELECT 1 FROM mst2_metadata_reader_operation r JOIN mst2_qualified_lease_binding l USING(lease_id) + WHERE r.operation_id=NEW.reader_operation_id AND r.lease_id=NEW.lease_id AND r.snapshot_id=NEW.snapshot_id + AND r.session_incarnation=NEW.session_incarnation AND r.root_page=NEW.root_page AND r.root_generation=NEW.root_generation + AND r.state='ACTIVE' AND l.state='ACTIVE' AND l.lease_epoch=r.lease_epoch + AND r.hard_deadline_unix<=l.expires_at_unix AND r.hard_deadline_unix>floor(extract(epoch FROM clock_timestamp()))::bigint) THEN + RAISE EXCEPTION 'reader anchor differs from its exact active lease operation'; + END IF; + END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_metadata_root_anchor_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_metadata_root_anchor + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_root_anchor_guard(); + +CREATE FUNCTION mst2_metadata_reuse_root_protected() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + IF EXISTS(SELECT 1 FROM mst2_metadata_prepare WHERE prepare_id=NEW.prepare_id + AND state IN ('PREPARING','COMMITTED') AND coverage_retired_at IS NULL) + AND NOT EXISTS(SELECT 1 FROM mst2_metadata_root_anchor WHERE anchor_kind='REUSE' AND prepare_id=NEW.prepare_id + AND root_page=NEW.root_page AND root_generation=NEW.root_generation) THEN + RAISE EXCEPTION 'reused boundary must commit with continuously owned root protection'; + END IF; + RETURN NULL; +END $$; +CREATE CONSTRAINT TRIGGER mst2_metadata_reuse_root_protected AFTER INSERT ON mst2_metadata_prepare_reuse_root + DEFERRABLE INITIALLY DEFERRED FOR EACH ROW EXECUTE FUNCTION mst2_metadata_reuse_root_protected(); + +CREATE FUNCTION mst2_metadata_temporary_anchor_continuity() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE q mst2_metadata_prepare%ROWTYPE; +BEGIN + IF OLD.anchor_kind NOT IN ('PREPARE','REUSE') THEN RETURN NULL; END IF; + SELECT * INTO STRICT q FROM mst2_metadata_prepare WHERE prepare_id=OLD.prepare_id; + IF q.state='ABORTED' THEN RETURN NULL; END IF; + IF q.coverage_retired_at IS NULL THEN + IF NOT EXISTS(SELECT 1 FROM mst2_metadata_root_anchor a WHERE a.anchor_kind=OLD.anchor_kind + AND a.owner_key=OLD.owner_key AND a.prepare_id=OLD.prepare_id + AND a.root_page=OLD.root_page AND a.root_generation=OLD.root_generation + AND a.root_certificate_digest=OLD.root_certificate_digest) THEN + RAISE EXCEPTION 'active preparation cannot lose its continuously owned canonical root'; + END IF; + ELSIF q.state<>'COMMITTED' OR NOT (mst2_metadata_session_covers_prepare(q.prepare_id) + OR mst2_metadata_orphan_prepare_retired(q.prepare_id)) THEN + RAISE EXCEPTION 'retired preparation coverage requires a definitive independent session root'; + END IF; + RETURN NULL; +END $$; +CREATE CONSTRAINT TRIGGER mst2_metadata_temporary_anchor_continuity AFTER DELETE ON mst2_metadata_root_anchor + DEFERRABLE INITIALLY DEFERRED FOR EACH ROW EXECUTE FUNCTION mst2_metadata_temporary_anchor_continuity(); diff --git a/src/jupiter/storage/qualified_metadata_canonical.sql b/src/jupiter/storage/qualified_metadata_canonical.sql new file mode 100644 index 00000000..2ce0c283 --- /dev/null +++ b/src/jupiter/storage/qualified_metadata_canonical.sql @@ -0,0 +1,227 @@ +CREATE FUNCTION mst2_metadata_read_le(b bytea,p integer,w integer) RETURNS numeric +LANGUAGE plpgsql IMMUTABLE STRICT SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE v numeric:=0; power numeric:=1; i integer; +BEGIN + IF w NOT IN (2,4,8) OR p<0 OR p>octet_length(b)-w THEN + RAISE EXCEPTION 'MTP2 integer is out of bounds'; + END IF; + FOR i IN 0..w-1 LOOP + v:=v+get_byte(b,p+i)*power; power:=power*256; + END LOOP; + RETURN v; +END $$; + +CREATE FUNCTION mst2_metadata_write_le(v numeric,w integer) RETURNS bytea +LANGUAGE plpgsql IMMUTABLE STRICT SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE result bytea; i integer; +BEGIN + IF w NOT IN (2,4,8) OR v<0 OR v<>trunc(v) OR v>=power(256::numeric,w) THEN + RAISE EXCEPTION 'MTP2 output integer is out of range'; + END IF; + result:=decode(repeat('00',w),'hex'); + FOR i IN 0..w-1 LOOP result:=set_byte(result,i,mod(v,256)::integer); v:=trunc(v/256); END LOOP; + RETURN result; +END $$; + +CREATE FUNCTION mst2_metadata_encode_entry(e jsonb) RETURNS bytea +LANGUAGE plpgsql IMMUTABLE STRICT SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE kind integer:=(e->>'kind')::integer; name bytea:=decode(e->>'name','hex'); target bytea; result bytea; +BEGIN + IF kind IS NULL OR kind NOT BETWEEN 1 AND 4 OR name IS NULL THEN RAISE EXCEPTION 'MTP2 output entry is incomplete'; END IF; + PERFORM mst2_metadata_valid_name(name); + result:=set_byte(decode('00','hex'),0,kind)||mst2_metadata_write_le(octet_length(name),2)||name; + IF kind=4 THEN + target:=decode(e->>'child','hex'); + IF target IS NULL OR octet_length(target)<>32 OR target=decode(repeat('00',32),'hex') THEN + RAISE EXCEPTION 'MTP2 output directory reference is invalid'; + END IF; + ELSE + target:=decode(e->>'content_id','hex'); + IF target IS NULL OR octet_length(target)<>32 OR e->>'size' IS NULL THEN + RAISE EXCEPTION 'MTP2 output file reference is invalid'; + END IF; + result:=result||mst2_metadata_write_le((e->>'size')::numeric,8); + END IF; + RETURN result||target; +END $$; + +CREATE FUNCTION mst2_metadata_build_map(items jsonb,budget bigint DEFAULT 67108864) RETURNS jsonb +LANGUAGE plpgsql IMMUTABLE STRICT SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE n integer; item jsonb; previous bytea; name bytea; minimum bytea; maximum bytea; + raw bytea:=''::bytea; result bytea; prefix bytea; terminal jsonb; + partition record; grouped jsonb:='[]'::jsonb; child jsonb; + visits bigint:=0; encoded_bytes bigint:=0; pages bigint:=1; total_bytes bigint; payload bytea; children integer:=0; +BEGIN + IF jsonb_typeof(items)<>'array' OR budget NOT BETWEEN 1 AND 67108864 THEN + RAISE EXCEPTION 'MTP2 canonical map input or work budget is invalid'; + END IF; + n:=jsonb_array_length(items); + IF n>131072 THEN RAISE EXCEPTION 'MTP2 canonical map exceeds its entry budget'; END IF; + FOR item IN SELECT value FROM jsonb_array_elements(items) LOOP + name:=decode(item->>'name','hex'); + IF previous IS NOT NULL AND previous>=name THEN RAISE EXCEPTION 'MTP2 build names are not strictly byte ordered'; END IF; + previous:=name; maximum:=name; IF minimum IS NULL THEN minimum:=name; END IF; + payload:=mst2_metadata_encode_entry(item); visits:=visits+1+octet_length(payload); + IF visits>budget THEN RAISE EXCEPTION 'MTP2 canonical source work budget exceeded'; END IF; + encoded_bytes:=encoded_bytes+octet_length(payload); + IF n<=128 AND 20+encoded_bytes<=16384 THEN raw:=raw||payload; END IF; + END LOOP; + IF n<=128 AND 20+encoded_bytes<=16384 THEN + result:=decode('4d5450320000','hex')||mst2_metadata_write_le(n,2)||mst2_metadata_write_le(n,8) + ||mst2_metadata_write_le(octet_length(raw),4)||raw; + RETURN jsonb_build_object('bytes',encode(result,'hex'),'page_id',encode(sha256( + convert_to('mega.mst2.metapage','UTF8')||decode('00','hex')||result),'hex'), + 'source_work_units',visits,'pages',pages,'metadata_bytes',octet_length(result)); + END IF; + prefix:=mst2_metadata_lcp(minimum,maximum); + SELECT value INTO terminal FROM jsonb_array_elements(items) WHERE decode(value->>'name','hex')=prefix; + visits:=visits+2*n; + IF visits>budget THEN RAISE EXCEPTION 'MTP2 canonical source grouping work budget exceeded'; END IF; + payload:=mst2_metadata_write_le(octet_length(prefix),2)||prefix; + IF terminal IS NULL THEN payload:=payload||decode('00','hex'); + ELSE payload:=payload||decode('01','hex')||mst2_metadata_encode_entry(terminal); END IF; + total_bytes:=0; + FOR partition IN SELECT get_byte(decode(value->>'name','hex'),octet_length(prefix)) AS label, + jsonb_agg(value ORDER BY decode(value->>'name','hex')) AS members + FROM jsonb_array_elements(items) WHERE octet_length(decode(value->>'name','hex'))>octet_length(prefix) + GROUP BY get_byte(decode(value->>'name','hex'),octet_length(prefix)) ORDER BY label LOOP + grouped:=partition.members; + IF visits>=budget THEN RAISE EXCEPTION 'MTP2 canonical source work budget exceeded'; END IF; + child:=mst2_metadata_build_map(grouped,budget-visits); visits:=visits+(child->>'source_work_units')::bigint; + pages:=pages+(child->>'pages')::bigint; total_bytes:=total_bytes+(child->>'metadata_bytes')::bigint; + IF pages>4096 OR total_bytes>67108864 THEN RAISE EXCEPTION 'MTP2 canonical source encoding exceeds metadata budget'; END IF; + payload:=payload||set_byte(decode('00','hex'),0,partition.label)||mst2_metadata_write_le(jsonb_array_length(grouped),8) + ||decode(child->>'page_id','hex'); children:=children+1; + END LOOP; + IF children+(terminal IS NOT NULL)::integer<2 OR 20+octet_length(payload)>16384 THEN + RAISE EXCEPTION 'MTP2 canonical branch shape or size is invalid'; + END IF; + result:=decode('4d5450320100','hex')||mst2_metadata_write_le(children,2)||mst2_metadata_write_le(n,8) + ||mst2_metadata_write_le(octet_length(payload),4)||payload; + total_bytes:=total_bytes+octet_length(result); + IF total_bytes>67108864 THEN RAISE EXCEPTION 'MTP2 canonical source encoding exceeds metadata byte budget'; END IF; + RETURN jsonb_build_object('bytes',encode(result,'hex'),'page_id',encode(sha256( + convert_to('mega.mst2.metapage','UTF8')||decode('00','hex')||result),'hex'), + 'source_work_units',visits,'pages',pages,'metadata_bytes',total_bytes); +END $$; + +CREATE FUNCTION mst2_metadata_valid_name(n bytea) RETURNS boolean +LANGUAGE plpgsql IMMUTABLE STRICT SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE i integer; +BEGIN + IF octet_length(n) NOT BETWEEN 1 AND 255 OR n IN (decode('2e','hex'),decode('2e2e','hex')) THEN + RAISE EXCEPTION 'MTP2 name has an invalid length or dot component'; + END IF; + FOR i IN 0..octet_length(n)-1 LOOP + IF get_byte(n,i) IN (0,47) THEN RAISE EXCEPTION 'MTP2 name contains NUL or slash'; END IF; + END LOOP; + PERFORM convert_from(n,'UTF8'); + RETURN true; +END $$; + +CREATE FUNCTION mst2_metadata_decode_entry(b bytea,p integer) RETURNS jsonb +LANGUAGE plpgsql IMMUTABLE STRICT SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE start integer:=p; kind integer; n integer; name bytea; target bytea; size numeric; +BEGIN + IF p<0 OR p>=octet_length(b) THEN RAISE EXCEPTION 'MTP2 entry kind is truncated'; END IF; + kind:=get_byte(b,p); p:=p+1; + IF kind NOT BETWEEN 1 AND 4 THEN RAISE EXCEPTION 'MTP2 entry kind is invalid'; END IF; + n:=mst2_metadata_read_le(b,p,2)::integer; p:=p+2; + IF n>octet_length(b)-p THEN RAISE EXCEPTION 'MTP2 entry name is truncated'; END IF; + name:=substring(b FROM p+1 FOR n); p:=p+n; + PERFORM mst2_metadata_valid_name(name); + IF kind=4 THEN + IF p>octet_length(b)-32 THEN RAISE EXCEPTION 'MTP2 directory reference is truncated'; END IF; + target:=substring(b FROM p+1 FOR 32); p:=p+32; + IF target=decode(repeat('00',32),'hex') THEN RAISE EXCEPTION 'MTP2 empty directory has a zero reference'; END IF; + RETURN jsonb_build_object('end',p,'kind',kind,'name',encode(name,'hex'), + 'encoded_bytes',p-start,'child',encode(target,'hex')); + END IF; + size:=mst2_metadata_read_le(b,p,8); p:=p+8; + IF p>octet_length(b)-32 THEN RAISE EXCEPTION 'MTP2 file content reference is truncated'; END IF; + target:=substring(b FROM p+1 FOR 32); p:=p+32; + RETURN jsonb_build_object('end',p,'kind',kind,'name',encode(name,'hex'), + 'encoded_bytes',p-start,'size',size,'content_id',encode(target,'hex')); +END $$; + +CREATE FUNCTION mst2_metadata_decode_local(b bytea) RETURNS jsonb +LANGUAGE plpgsql IMMUTABLE STRICT SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE kind integer; n integer; count numeric; declared numeric; p integer:=20; plen integer; + i integer; label integer; previous_label integer:=-1; previous_name bytea; name bytea; prefix bytea; + terminal integer:=0; item jsonb; entries jsonb:='[]'::jsonb; children jsonb:='[]'::jsonb; + refs jsonb:='[]'::jsonb; child bytea; child_count numeric; entry_bytes bigint:=0; +BEGIN + IF octet_length(b) NOT BETWEEN 20 AND 16384 THEN RAISE EXCEPTION 'MTP2 page length is invalid'; END IF; + IF substring(b FROM 1 FOR 4)<>decode('4d545032','hex') OR get_byte(b,5)<>0 THEN + RAISE EXCEPTION 'MTP2 magic or flags are invalid'; + END IF; + kind:=get_byte(b,4); n:=mst2_metadata_read_le(b,6,2)::integer; + count:=mst2_metadata_read_le(b,8,8); + IF count>9223372036854775807 OR mst2_metadata_read_le(b,16,4)<>octet_length(b)-20 THEN + RAISE EXCEPTION 'MTP2 header count or payload length is invalid'; + END IF; + IF kind=0 THEN + IF n>128 OR count<>n THEN RAISE EXCEPTION 'MTP2 leaf count is invalid'; END IF; + IF n>0 THEN + FOR i IN 1..n LOOP + item:=mst2_metadata_decode_entry(b,p); p:=(item->>'end')::integer; + name:=decode(item->>'name','hex'); + IF previous_name IS NOT NULL AND previous_name>=name THEN + RAISE EXCEPTION 'MTP2 leaf names are not strictly byte ordered'; + END IF; + previous_name:=name; entries:=entries||jsonb_build_array(item-'end'); + entry_bytes:=entry_bytes+(item->>'encoded_bytes')::bigint; + IF (item->>'kind')::integer=4 THEN + refs:=refs||jsonb_build_array(jsonb_build_object('kind','DIRECTORY','name',item->>'name','child',item->>'child')); + END IF; + END LOOP; + END IF; + ELSIF kind=1 THEN + plen:=mst2_metadata_read_le(b,p,2)::integer; p:=p+2; + IF plen>octet_length(b)-p THEN RAISE EXCEPTION 'MTP2 branch prefix is truncated'; END IF; + prefix:=substring(b FROM p+1 FOR plen); p:=p+plen; + IF p>=octet_length(b) THEN RAISE EXCEPTION 'MTP2 terminal flag is truncated'; END IF; + terminal:=get_byte(b,p); p:=p+1; + IF terminal NOT IN (0,1) THEN RAISE EXCEPTION 'MTP2 terminal flag is invalid'; END IF; + IF terminal=1 THEN + item:=mst2_metadata_decode_entry(b,p); p:=(item->>'end')::integer; + IF decode(item->>'name','hex')<>prefix THEN RAISE EXCEPTION 'MTP2 terminal name differs from prefix'; END IF; + entries:=entries||jsonb_build_array(item-'end'); entry_bytes:=(item->>'encoded_bytes')::bigint; + IF (item->>'kind')::integer=4 THEN + refs:=refs||jsonb_build_array(jsonb_build_object('kind','DIRECTORY','name',item->>'name','child',item->>'child')); + END IF; + END IF; + IF n>256 OR n+terminal<2 THEN RAISE EXCEPTION 'MTP2 branch group count is invalid'; END IF; + declared:=terminal; + IF n>0 THEN + FOR i IN 1..n LOOP + IF p>=octet_length(b) THEN RAISE EXCEPTION 'MTP2 branch child label is truncated'; END IF; + label:=get_byte(b,p); p:=p+1; + IF label<=previous_label THEN RAISE EXCEPTION 'MTP2 branch labels are not strictly ordered'; END IF; + previous_label:=label; child_count:=mst2_metadata_read_le(b,p,8); p:=p+8; + IF child_count NOT BETWEEN 1 AND 9223372036854775807 THEN RAISE EXCEPTION 'MTP2 child count is invalid'; END IF; + IF p>octet_length(b)-32 THEN RAISE EXCEPTION 'MTP2 branch child digest is truncated'; END IF; + child:=substring(b FROM p+1 FOR 32); p:=p+32; + declared:=declared+child_count; + IF declared>9223372036854775807 THEN RAISE EXCEPTION 'MTP2 branch count overflows'; END IF; + item:=jsonb_build_object('kind','RADIX','label',label,'count',child_count,'child',encode(child,'hex')); + children:=children||jsonb_build_array(item); refs:=refs||jsonb_build_array(item); + END LOOP; + END IF; + IF declared<>count THEN RAISE EXCEPTION 'MTP2 branch header count differs from children'; END IF; + ELSE + RAISE EXCEPTION 'MTP2 page kind is invalid'; + END IF; + IF p<>octet_length(b) THEN RAISE EXCEPTION 'MTP2 page has trailing bytes'; END IF; + RETURN jsonb_build_object('kind',kind,'count',count,'prefix',encode(prefix,'hex'), + 'entries',entries,'children',children,'refs',refs,'direct_entry_bytes',entry_bytes, + 'page_id',encode(sha256(convert_to('mega.mst2.metapage','UTF8')||decode('00','hex')||b),'hex')); +END $$; + +CREATE FUNCTION mst2_metadata_lcp(a bytea,b bytea) RETURNS bytea +LANGUAGE plpgsql IMMUTABLE STRICT SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE n integer:=least(octet_length(a),octet_length(b)); i integer:=0; +BEGIN + WHILE i Value { + Value::Array( + entries + .iter() + .map(|entry| { + json!({"kind":1,"name":hex::encode(&entry.name),"size":u64::MAX, + "content_id":hex::encode([17;32])}) + }) + .collect(), + ) +} + +fn files(names: impl IntoIterator) -> Vec { + let mut entries: Vec<_> = names + .into_iter() + .map(|name| Entry::file(EntryKind::Regular, name.as_bytes(), u64::MAX, [17; 32])) + .collect(); + entries.sort_by(|left, right| left.name.cmp(&right.name)); + entries +} + +#[tokio::test] +async fn database_canonical_builder_matches_pinned_codec_at_split_and_name_boundaries() { + let (_config, _core, _namespace, q, _guard) = fixture().await; + let terminal_names = + std::iter::once("a".to_owned()).chain((0..129).map(|index| format!("a{index:03}"))); + let cases = [ + Vec::new(), + files((0..128).map(|index| format!("f{index:03}"))), + files((0..129).map(|index| format!("f{index:03}"))), + files((0..128).map(|index| format!("{}{index:03}", "x".repeat(197)))), + files(terminal_names), + files((0..160).map(|index| format!("目录{index:03}"))), + ]; + for entries in cases { + let bytes = Page::build(&entries).unwrap(); + let proof: Value = q + .query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT mst2_metadata_build_map($1::jsonb) AS proof", + [encoded_entries(&entries).to_string().into()], + )) + .await + .unwrap() + .unwrap() + .try_get("", "proof") + .unwrap(); + assert_eq!(proof["bytes"].as_str().unwrap(), hex::encode(&bytes)); + assert_eq!( + proof["page_id"].as_str().unwrap(), + hex::encode(page_id(&bytes)) + ); + let decoded: Value = q + .query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT mst2_metadata_decode_local($1) AS decoded", + [bytes.into()], + )) + .await + .unwrap() + .unwrap() + .try_get("", "decoded") + .unwrap(); + assert_eq!(decoded["count"].as_u64().unwrap(), entries.len() as u64); + } +} + +#[tokio::test] +async fn database_parser_rejects_nonexact_bytes_and_invalid_names() { + let (_config, _core, _namespace, q, _guard) = fixture().await; + let bytes = Page::build(&files(["file".to_owned()])).unwrap(); + let mut trailing = bytes.clone(); + trailing.push(0); + let mut wrong_length = bytes.clone(); + wrong_length[16] = wrong_length[16].wrapping_add(1); + let mut flags = bytes.clone(); + flags[5] = 1; + let mut wrong_count = bytes.clone(); + wrong_count[8] = 2; + let mut invalid_utf8 = bytes.clone(); + invalid_utf8[23] = 255; + let mut slash = bytes.clone(); + slash[23] = b'/'; + for malformed in [ + trailing, + wrong_length, + flags, + wrong_count, + invalid_utf8, + slash, + ] { + assert!( + q.query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT mst2_metadata_decode_local($1)", + [malformed.into()], + )) + .await + .is_err() + ); + } + let mut entries = encoded_entries(&files(["one".to_owned(), "two".to_owned()])); + entries.as_array_mut().unwrap().reverse(); + assert!( + q.query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT mst2_metadata_build_map($1::jsonb)", + [entries.to_string().into()], + )) + .await + .is_err() + ); +} + +#[tokio::test] +async fn exact_typed_certificates_keep_shared_directory_occurrences_and_dedup_physical_edges() { + let (config, core, _namespace, q, _guard) = fixture().await; + let writer = ShadowQualifiedMetadataWriter::open(&core, &config) + .await + .unwrap(); + let pages = prepared(); + let receipt = write(&writer, "certified-shared-directory", &pages).await; + assert_eq!( + count(&q, "SELECT count(*) FROM mst2_metadata_page_certificate").await, + 2 + ); + assert_eq!( + count( + &q, + "SELECT count(*) FROM mst2_metadata_verified_ref WHERE reference_kind='DIRECTORY'" + ) + .await, + 2 + ); + assert_eq!( + count(&q, "SELECT count(*) FROM mst2_metadata_graph_edge").await, + 1 + ); + assert_eq!( + count( + &q, + "SELECT max(rank)::bigint FROM mst2_metadata_page_certificate" + ) + .await, + 1 + ); + assert_eq!(writer.finalize(receipt.intent()).await.unwrap(), receipt); + for table in [ + "mst2_metadata_page_certificate", + "mst2_metadata_verified_ref", + "mst2_metadata_source_root_attestation", + "mst2_metadata_prepare_reuse_root", + "mst2_metadata_reuse_index", + "mst2_metadata_root_anchor", + "mst2_metadata_reader_operation", + ] { + assert!( + q.execute_unprepared(&format!("TRUNCATE {table} CASCADE")) + .await + .is_err() + ); + } + for sql in [ + "UPDATE mst2_metadata_page_certificate SET rank=rank+1", + "DELETE FROM mst2_metadata_verified_ref", + "UPDATE mst2_metadata_verified_ref SET reference_ordinal=reference_ordinal+1", + ] { + assert!(q.execute_unprepared(sql).await.is_err()); + } +} + +#[tokio::test] +async fn raw_sql_cannot_certify_a_leafable_branch_from_valid_certified_children() { + let (config, core, namespace, q, _guard) = fixture().await; + let writer = ShadowQualifiedMetadataWriter::open(&core, &config) + .await + .unwrap(); + let receipt = write(&writer, "raw-canonical-oracle", &prepared()).await; + let child = Page::build(&[Entry::file(EntryKind::Regular, b"file", 3, [42; 32])]).unwrap(); + let mut payload = vec![1, 0, b'f', 1, 1, 1, 0, b'f']; + payload.extend_from_slice(&3_u64.to_le_bytes()); + payload.extend_from_slice(&[17; 32]); + payload.push(b'i'); + payload.extend_from_slice(&1_u64.to_le_bytes()); + payload.extend_from_slice(&page_id(&child)); + let mut branch = b"MTP2\x01\x00".to_vec(); + branch.extend_from_slice(&1_u16.to_le_bytes()); + branch.extend_from_slice(&2_u64.to_le_bytes()); + branch.extend_from_slice(&(payload.len() as u32).to_le_bytes()); + branch.extend_from_slice(&payload); + let root = page_id(&branch); + let pid = uuid::Uuid::new_v4().to_string(); + let txn = q + .begin_with_config(Some(IsolationLevel::ReadCommitted), None) + .await + .unwrap(); + namespace.enter(&txn).await.unwrap(); + txn.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "INSERT INTO mst2_metadata_prepare SELECT (jsonb_populate_record(NULL::mst2_metadata_prepare, + to_jsonb(q)||jsonb_build_object('prepare_id',$1::text,'operation_id','raw-leafable-branch','metadata_root',$2::bytea, + 'state','PREPARING','committed_at',NULL,'node_count',2,'edge_count',1,'total_bytes',$3::bigint))).* + FROM mst2_metadata_prepare q WHERE q.prepare_id=$4", + [pid.clone().into(),root.to_vec().into(),((branch.len()+child.len()) as i64).into(),receipt.intent().prepare_id().into()], + )).await.unwrap(); + txn.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "INSERT INTO mst2_metadata_lifetime(page_id,node_id,generation,state,metadata_codec,expected_size,graph_domain) + VALUES($1,'page:sha256:'||encode($1,'hex'),1,'RESERVED',1,$2,'qualified-v1')", + [root.to_vec().into(),(branch.len() as i32).into()], + )).await.unwrap(); + txn.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "INSERT INTO mst2_metadata_current VALUES($1,1)", + [root.to_vec().into()], + )) + .await + .unwrap(); + txn.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "INSERT INTO mst2_metadata_prepare_page SELECT $1,$2,1,$3 + UNION ALL SELECT $1,page_id,generation,expected_size + FROM mst2_metadata_prepare_page WHERE prepare_id=$4 AND page_id=$5", + [ + pid.clone().into(), + root.to_vec().into(), + (branch.len() as i32).into(), + receipt.intent().prepare_id().into(), + page_id(&child).to_vec().into(), + ], + )) + .await + .unwrap(); + txn.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "INSERT INTO mst2_metadata_payload(page_id,generation,metadata_codec,byte_size,payload) VALUES($1,1,1,$2,$3)", + [root.to_vec().into(),(branch.len() as i32).into(),branch.into()], + )).await.unwrap(); + let error = txn + .query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT mst2_metadata_certify_page($1,$2,1)", + [pid.into(), root.to_vec().into()], + )) + .await + .unwrap_err(); + assert!(error.to_string().contains("leafable"), "{error}"); + txn.rollback().await.unwrap(); +} + +#[tokio::test] +async fn temporary_root_cannot_disappear_in_a_later_transaction() { + let (config, core, namespace, q, _guard) = fixture().await; + let writer = ShadowQualifiedMetadataWriter::open(&core, &config) + .await + .unwrap(); + let receipt = write(&writer, "continuous-owned-root", &prepared()).await; + let txn = q + .begin_with_config(Some(IsolationLevel::ReadCommitted), None) + .await + .unwrap(); + namespace.enter(&txn).await.unwrap(); + txn.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "INSERT INTO mst2_metadata_root_anchor(anchor_id,anchor_kind,owner_key,root_page,root_generation,root_certificate_digest,prepare_id) + SELECT $1::uuid,'PREPARE',$2,c.page_id,c.generation,c.certificate_digest,$2 + FROM mst2_metadata_page_certificate c JOIN mst2_metadata_prepare q ON q.metadata_root=c.page_id + WHERE q.prepare_id=$2", + [uuid::Uuid::new_v4().to_string().into(),receipt.intent().prepare_id().into()], + )).await.unwrap(); + txn.commit().await.unwrap(); + let error = q + .execute_unprepared("DELETE FROM mst2_metadata_root_anchor WHERE anchor_kind='PREPARE'") + .await + .unwrap_err(); + assert!( + error + .to_string() + .contains("continuously owned canonical root"), + "{error}" + ); + assert_eq!( + count(&q, "SELECT count(*) FROM mst2_metadata_root_anchor").await, + 1 + ); +} + +#[tokio::test] +async fn source_root_is_derived_from_current_core_body_and_verified_blob_facts() { + let (config, core, namespace, q, _guard) = fixture().await; + let entries = [Entry::file(EntryKind::Regular, b"file", 3, [17; 32])]; + let page = Page::build(&entries).unwrap(); + let mut builder = MetadataDagBuilder::new(MetadataDagLimits::default()); + builder.add_directory(&page, &entries).unwrap(); + let prepared = PreparedNativeMetadataRetention::test_installation( + Arc::new(builder.finish(page_id(&page)).unwrap()), + "/", + ); + let mut body = b"100644 file\0".to_vec(); + body.extend_from_slice(&[187; 20]); + core.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "INSERT INTO mega_tree(id,tree_id,sub_trees,size,created_at,pack_id,pack_offset,commit_id) + VALUES(1,$1,$2,0,now(),'fixture',0,'fixture')", + ["a".repeat(40).into(), body.clone().into()], + )) + .await + .unwrap(); + core.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "INSERT INTO mst2_verified_object(storage_domain,git_oid,object_kind,raw_sha256,size,verification_version,state,created_at) + VALUES('git',$1,'blob',$2,3,2,'VERIFIED',now())", + ["b".repeat(40).into(),vec![17_u8;32].into()], + )).await.unwrap(); + let writer = ShadowQualifiedMetadataWriter::open(&core, &config) + .await + .unwrap(); + let receipt = write(&writer, "bound-real-core-source", &prepared).await; + let aid = uuid::Uuid::new_v4().to_string(); + let txn = q + .begin_with_config(Some(IsolationLevel::ReadCommitted), None) + .await + .unwrap(); + namespace.enter(&txn).await.unwrap(); + txn.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "INSERT INTO mst2_metadata_source_root_attestation(attestation_id,namespace_uuid,origin_prepare_id, + tagged_tree_oid,source_profile,profile_digest,source_body_digest,root_page,root_generation, + root_certificate_digest,source_proof,attestation_digest) + SELECT $1::uuid,(proof->>'namespace')::uuid,$2,$3,proof->'source_profile',decode(proof->>'profile_digest','hex'), + decode(proof->>'source_body_digest','hex'),$4,1,decode(proof->>'root_certificate','hex'),proof, + decode(proof->>'attestation','hex') FROM (SELECT mst2_metadata_compute_source_proof($2,$3,$4,1) AS proof) input", + [aid.into(),receipt.intent().prepare_id().into(),prepared.fixed_root_tree_oid().into(),page_id(&page).to_vec().into()], + )).await.unwrap(); + txn.commit().await.unwrap(); + assert_eq!( + count( + &q, + "SELECT count(*) FROM mst2_metadata_source_root_attestation" + ) + .await, + 1 + ); + body[7] = b'F'; + core.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "UPDATE mega_tree SET sub_trees=$1 WHERE tree_id=$2", + [body.into(), "a".repeat(40).into()], + )) + .await + .unwrap(); + let error = q + .query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT mst2_metadata_compute_source_proof($1,$2,$3,1)", + [ + receipt.intent().prepare_id().into(), + prepared.fixed_root_tree_oid().into(), + page_id(&page).to_vec().into(), + ], + )) + .await + .unwrap_err(); + assert!( + error + .to_string() + .contains("independently canonical certified root"), + "{error}" + ); +} + +pub(super) async fn seeded_rooted_plan( + core: &DatabaseConnection, + tree_char: char, + name: &str, + id: i64, +) -> (RootedMetadataInstallPlan, MetadataPagePayload) { + let entries = [Entry::file( + EntryKind::Regular, + name.as_bytes(), + 3, + [17; 32], + )]; + let bytes = Page::build(&entries).unwrap(); + let page = page_id(&bytes); + let mut builder = MetadataDagBuilder::new(MetadataDagLimits::default()); + builder.add_directory(&bytes, &entries).unwrap(); + let prepared = PreparedNativeMetadataRetention::test_installation( + Arc::new(builder.finish(page).unwrap()), + "/", + ); + let mut identity = prepared.install_plan().unwrap().identity; + identity.tagged_root_tree_oid = format!("sha1:{}", tree_char.to_string().repeat(40)); + let mut body = format!("100644 {name}\0").into_bytes(); + body.extend_from_slice(&[187; 20]); + core.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "INSERT INTO mega_tree(id,tree_id,sub_trees,size,created_at,pack_id,pack_offset,commit_id) + VALUES($1,$2,$3,0,now(),'fixture',0,'fixture')", + [ + id.into(), + tree_char.to_string().repeat(40).into(), + body.into(), + ], + )) + .await + .unwrap(); + core.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "INSERT INTO mst2_verified_object(storage_domain,git_oid,object_kind,raw_sha256,size,verification_version,state,created_at) + VALUES('git',$1,'blob',$2,3,2,'VERIFIED',now()) ON CONFLICT(storage_domain,git_oid,object_kind) DO NOTHING", + ["b".repeat(40).into(),vec![17_u8;32].into()], + )).await.unwrap(); + let source_roots = BTreeMap::from([(identity.tagged_root_tree_oid.clone(), page)]); + let plan = RootedMetadataInstallPlan::new( + identity, + page, + BTreeMap::from([(page, bytes.len() as u64)]), + BTreeSet::new(), + BTreeMap::new(), + source_roots, + ) + .unwrap(); + ( + plan, + MetadataPagePayload { + id: page, + size: bytes.len() as u64, + bytes, + }, + ) +} + +#[tokio::test] +async fn actual_rooted_cold_and_zero_delta_reuse_keep_one_canonical_graph() { + let (config, core, _namespace, q, _guard) = fixture().await; + let (cold, payload) = seeded_rooted_plan(&core, 'a', "file", 1).await; + let writer = RootedQualifiedMetadataRepository::open(&core, &config) + .await + .unwrap(); + let intent = writer + .begin_intent("actual-rooted-cold", &cold) + .await + .unwrap(); + writer.install_pages(&intent, &[payload]).await.unwrap(); + let receipt = writer.finalize(&intent).await.unwrap(); + assert_eq!(receipt.metadata_root(), cold.root); + assert_eq!( + writer.recover("actual-rooted-cold", &cold).await.unwrap(), + Some(receipt.clone()) + ); + let proof=q.query_one_raw(Statement::from_string(DbBackend::Postgres, + "SELECT a.attestation_id::text,a.attestation_digest,a.root_certificate_digest FROM mst2_metadata_source_root_attestation a" + )).await.unwrap().unwrap(); + let reused = RootedReuseRoot { + generation: intent.root_generation(), + attestation_id: uuid::Uuid::parse_str( + &proof.try_get::("", "attestation_id").unwrap(), + ) + .unwrap(), + attestation_digest: proof + .try_get::>("", "attestation_digest") + .unwrap() + .try_into() + .unwrap(), + certificate_digest: proof + .try_get::>("", "root_certificate_digest") + .unwrap() + .try_into() + .unwrap(), + }; + let warm = RootedMetadataInstallPlan::new( + cold.identity.clone(), + cold.root, + BTreeMap::new(), + BTreeSet::new(), + BTreeMap::from([(cold.root, reused)]), + cold.source_roots.clone(), + ) + .unwrap(); + let warm_intent = writer + .begin_intent("actual-zero-delta-root", &warm) + .await + .unwrap(); + assert_eq!( + writer.finalize(&warm_intent).await.unwrap().metadata_root(), + cold.root + ); + assert_eq!( + count(&q, "SELECT count(*) FROM mst2_metadata_payload").await, + 1 + ); + assert_eq!( + count(&q, "SELECT count(*) FROM mst2_metadata_page_certificate").await, + 1 + ); + assert_eq!( + count( + &q, + "SELECT count(*) FROM mst2_metadata_source_root_attestation" + ) + .await, + 1 + ); + assert_eq!(count(&q,"SELECT count(*) FROM mst2_metadata_prepare WHERE plan_kind='ROOTED' AND state='COMMITTED' AND node_count=0").await,1); + let error = q + .execute_unprepared("DELETE FROM mst2_metadata_root_anchor WHERE anchor_kind='REUSE'") + .await + .unwrap_err(); + assert!( + error + .to_string() + .contains("continuously owned canonical root"), + "{error}" + ); +} + +#[tokio::test] +async fn changed_ancestor_installs_only_delta_and_accepts_independently_attested_source_aliases() { + let (config, core, _namespace, q, _guard) = fixture().await; + let writer = RootedQualifiedMetadataRepository::open(&core, &config) + .await + .unwrap(); + let (first, payload) = seeded_rooted_plan(&core, 'a', "file", 1).await; + let first_intent = writer.begin_intent("alias-first", &first).await.unwrap(); + writer + .install_pages(&first_intent, &[payload]) + .await + .unwrap(); + writer.finalize(&first_intent).await.unwrap(); + let (alias, payload) = seeded_rooted_plan(&core, 'c', "file", 2).await; + assert_eq!(alias.root, first.root); + let alias_intent = writer.begin_intent("alias-second", &alias).await.unwrap(); + writer + .install_pages(&alias_intent, &[payload]) + .await + .unwrap(); + writer.finalize(&alias_intent).await.unwrap(); + let first_hint = writer + .lookup_reuse(&first.identity.tagged_root_tree_oid, &first.identity) + .await + .unwrap() + .unwrap(); + let alias_hint = writer + .lookup_reuse(&alias.identity.tagged_root_tree_oid, &alias.identity) + .await + .unwrap() + .unwrap(); + assert_eq!(first_hint.page_id, alias_hint.page_id); + assert_eq!( + first_hint.proof.certificate_digest, + alias_hint.proof.certificate_digest + ); + assert_ne!( + first_hint.proof.attestation_id, + alias_hint.proof.attestation_id + ); + + let entries = [ + Entry::dir(b"one", first.root), + Entry::dir(b"two", first.root), + ]; + let bytes = Page::build(&entries).unwrap(); + let root = page_id(&bytes); + let mut body = b"40000 one\0".to_vec(); + body.extend_from_slice(&[0xaa; 20]); + body.extend_from_slice(b"40000 two\0"); + body.extend_from_slice(&[0xcc; 20]); + core.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "INSERT INTO mega_tree(id,tree_id,sub_trees,size,created_at,pack_id,pack_offset,commit_id) + VALUES(3,$1,$2,0,now(),'fixture',0,'fixture')", + ["d".repeat(40).into(), body.into()], + )) + .await + .unwrap(); + let mut identity = first.identity.clone(); + identity.tagged_root_tree_oid = format!("sha1:{}", "d".repeat(40)); + let plan = RootedMetadataInstallPlan::new( + identity.clone(), + root, + BTreeMap::from([(root, bytes.len() as u64)]), + BTreeSet::from([(root, first.root)]), + BTreeMap::from([(first.root, first_hint.proof)]), + BTreeMap::from([ + (identity.tagged_root_tree_oid, root), + (first.identity.tagged_root_tree_oid.clone(), first.root), + (alias.identity.tagged_root_tree_oid.clone(), first.root), + ]), + ) + .unwrap(); + let intent = writer + .begin_intent("changed-ancestor-aliases", &plan) + .await + .unwrap(); + writer + .install_pages( + &intent, + &[MetadataPagePayload { + id: root, + size: bytes.len() as u64, + bytes, + }], + ) + .await + .unwrap(); + let receipt = writer.finalize(&intent).await.unwrap(); + assert_eq!(receipt.metadata_root(), root); + assert_eq!( + count(&q, "SELECT count(*) FROM mst2_metadata_payload").await, + 2 + ); + assert_eq!( + count(&q, "SELECT count(*) FROM mst2_metadata_graph_edge").await, + 1 + ); + assert_eq!( + count( + &q, + "SELECT count(*) FROM mst2_metadata_verified_ref WHERE reference_kind='DIRECTORY'" + ) + .await, + 2 + ); + assert_eq!( + count( + &q, + "SELECT count(*) FROM mst2_metadata_source_root_attestation" + ) + .await, + 3 + ); + let row = q.query_one_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "SELECT node_count,edge_count,(SELECT count(*) FROM mst2_metadata_prepare_reuse_root r + WHERE r.prepare_id=q.prepare_id)::bigint AS reuse_count FROM mst2_metadata_prepare q WHERE prepare_id=$1", + [intent.prepare_id().into()])).await.unwrap().unwrap(); + assert_eq!(row.try_get::("", "node_count").unwrap(), 1); + assert_eq!(row.try_get::("", "edge_count").unwrap(), 1); + assert_eq!(row.try_get::("", "reuse_count").unwrap(), 1); + assert_eq!( + writer + .recover("changed-ancestor-aliases", &plan) + .await + .unwrap(), + Some(receipt) + ); +} + +#[tokio::test] +async fn late_raw_delta_membership_is_rejected_after_intent_commit() { + let (config, core, _namespace, q, _guard) = fixture().await; + let (plan, payload) = seeded_rooted_plan(&core, 'a', "file", 1).await; + let (donor, donor_payload) = seeded_rooted_plan(&core, 'c', "other", 2).await; + let writer = RootedQualifiedMetadataRepository::open(&core, &config) + .await + .unwrap(); + let donor_intent = writer + .begin_intent("rooted-extra-donor", &donor) + .await + .unwrap(); + writer + .install_pages(&donor_intent, &[donor_payload.clone()]) + .await + .unwrap(); + writer.finalize(&donor_intent).await.unwrap(); + let intent = writer + .begin_intent("rooted-bounded-intent", &plan) + .await + .unwrap(); + writer.install_pages(&intent, &[payload]).await.unwrap(); + let error=q.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "INSERT INTO mst2_metadata_prepare_page(prepare_id,page_id,generation,expected_size) VALUES($1,$2,1,$3)", + [intent.prepare_id().into(),donor.root.to_vec().into(),(donor_payload.size as i32).into()], + )).await.unwrap_err(); + assert!( + error + .to_string() + .contains("missing or extra exact delta/reuse"), + "{error}" + ); + assert_eq!( + writer.finalize(&intent).await.unwrap().metadata_root(), + plan.root + ); +} diff --git a/src/jupiter/storage/qualified_metadata_certificates.sql b/src/jupiter/storage/qualified_metadata_certificates.sql new file mode 100644 index 00000000..6ddd7e66 --- /dev/null +++ b/src/jupiter/storage/qualified_metadata_certificates.sql @@ -0,0 +1,293 @@ +CREATE TABLE mst2_metadata_page_certificate ( + page_id bytea NOT NULL,generation bigint NOT NULL,certificate_digest bytea NOT NULL CHECK(octet_length(certificate_digest)=32), + origin_prepare_id text NOT NULL REFERENCES mst2_metadata_prepare(prepare_id), + namespace_uuid uuid NOT NULL REFERENCES mst2_metadata_family_identity(namespace_uuid), + proof_revision integer NOT NULL CHECK(proof_revision=1),metadata_codec smallint NOT NULL CHECK(metadata_codec=1), + byte_size integer NOT NULL CHECK(byte_size BETWEEN 20 AND 16384), + rank integer NOT NULL CHECK(rank BETWEEN 0 AND 4095), + map_entry_count bigint NOT NULL CHECK(map_entry_count BETWEEN 0 AND 131072), + map_encoded_entry_bytes bigint NOT NULL CHECK(map_encoded_entry_bytes BETWEEN 0 AND 67108864), + min_name bytea,max_name bytea, + relative_path_bytes integer NOT NULL CHECK(relative_path_bytes BETWEEN 0 AND 4096), + relative_components integer NOT NULL CHECK(relative_components BETWEEN 0 AND 256), + closure_nodes_upper integer NOT NULL CHECK(closure_nodes_upper BETWEEN 1 AND 4096), + closure_edges_upper integer NOT NULL CHECK(closure_edges_upper BETWEEN 0 AND 16384), + closure_bytes_upper bigint NOT NULL CHECK(closure_bytes_upper BETWEEN 20 AND 67108864), + closure_entries_upper bigint NOT NULL CHECK(closure_entries_upper BETWEEN 0 AND 131072), + canonical_proof jsonb NOT NULL, + PRIMARY KEY(page_id,generation),UNIQUE(page_id,generation,certificate_digest), + FOREIGN KEY(page_id,generation) REFERENCES mst2_metadata_lifetime(page_id,generation), + CHECK((map_entry_count=0)=(min_name IS NULL AND max_name IS NULL)), + CHECK(map_entry_count=0 OR (min_name IS NOT NULL AND max_name IS NOT NULL AND min_name<=max_name)) +); +CREATE TABLE mst2_metadata_verified_ref ( + parent_page bytea NOT NULL,parent_generation bigint NOT NULL, + reference_ordinal integer NOT NULL CHECK(reference_ordinal BETWEEN 0 AND 256), + reference_kind text NOT NULL CHECK(reference_kind IN ('DIRECTORY','RADIX')), + name bytea,label integer,advertised_count bigint, + child_page bytea NOT NULL,child_generation bigint NOT NULL, + child_certificate_digest bytea NOT NULL CHECK(octet_length(child_certificate_digest)=32), + PRIMARY KEY(parent_page,parent_generation,reference_ordinal), + FOREIGN KEY(parent_page,parent_generation) REFERENCES mst2_metadata_page_certificate(page_id,generation), + FOREIGN KEY(child_page,child_generation,child_certificate_digest) + REFERENCES mst2_metadata_page_certificate(page_id,generation,certificate_digest), + CHECK((reference_kind='DIRECTORY' AND name IS NOT NULL AND label IS NULL AND advertised_count IS NULL) + OR (reference_kind='RADIX' AND name IS NULL AND label IS NOT NULL AND advertised_count IS NOT NULL + AND label BETWEEN 0 AND 255 AND advertised_count>0)) +); +CREATE INDEX mst2_metadata_verified_ref_child ON mst2_metadata_verified_ref(child_page,child_generation,parent_page,parent_generation); +ALTER TABLE mst2_metadata_graph_node ADD COLUMN certificate_digest bytea NOT NULL, + ADD FOREIGN KEY(page_id,generation,certificate_digest) + REFERENCES mst2_metadata_page_certificate(page_id,generation,certificate_digest); + +CREATE FUNCTION mst2_metadata_child_certificate(p bytea,g bigint,pid text) RETURNS jsonb +LANGUAGE plpgsql VOLATILE SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE c mst2_metadata_page_certificate%ROWTYPE; +BEGIN + SELECT proof.* INTO c FROM mst2_metadata_page_certificate proof + JOIN mst2_metadata_current cur USING(page_id,generation) + JOIN mst2_metadata_lifetime life USING(page_id,generation) + JOIN mst2_metadata_graph_node node USING(page_id,generation) + JOIN mst2_metadata_payload body USING(page_id,generation) + WHERE proof.page_id=p AND proof.generation=g AND life.graph_domain='qualified-v1' + AND node.state='LIVE' AND node.certificate_digest=proof.certificate_digest + AND node.metadata_codec=proof.metadata_codec AND body.metadata_codec=proof.metadata_codec + AND body.byte_size=proof.byte_size AND node.bytes=proof.byte_size AND life.expected_size=proof.byte_size + AND (life.state='LIVE' OR (life.state='RESERVED' AND proof.origin_prepare_id=pid + AND EXISTS(SELECT 1 FROM mst2_metadata_prepare q WHERE q.prepare_id=pid AND q.state='PREPARING'))) + AND NOT EXISTS(SELECT 1 FROM mst2_metadata_gc_op op WHERE op.page_id=p AND op.generation=g); + IF NOT FOUND THEN RAISE EXCEPTION 'MTP2 child lacks its exact certified current graph'; END IF; + RETURN jsonb_build_object('page',encode(c.page_id,'hex'),'generation',c.generation, + 'certificate',encode(c.certificate_digest,'hex'),'rank',c.rank,'map_entry_count',c.map_entry_count, + 'map_encoded_entry_bytes',c.map_encoded_entry_bytes,'min_name',encode(c.min_name,'hex'), + 'max_name',encode(c.max_name,'hex'),'relative_path_bytes',c.relative_path_bytes, + 'relative_components',c.relative_components,'closure_nodes_upper',c.closure_nodes_upper, + 'closure_edges_upper',c.closure_edges_upper,'closure_bytes_upper',c.closure_bytes_upper, + 'closure_entries_upper',c.closure_entries_upper); +END $$; + +CREATE FUNCTION mst2_metadata_compute_certificate(p bytea,g bigint,pid text) RETURNS jsonb +LANGUAGE plpgsql VOLATILE SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE body mst2_metadata_payload%ROWTYPE; q mst2_metadata_prepare%ROWTYPE; decoded jsonb; + ref jsonb; entry jsonb; child jsonb; bound jsonb; bound_refs jsonb:='[]'::jsonb; + seen jsonb:='[]'::jsonb; children jsonb:='{}'::jsonb; child_page bytea; child_generation bigint; + key text; name bytea; child_min bytea; child_max bytea; prefix bytea; minimum bytea; maximum bytea; + map_count bigint:=0; map_bytes bigint:=0; path_bytes integer:=0; components integer:=0; rank integer:=0; + nodes bigint:=1; edges bigint:=0; bytes bigint; entries bigint; proof jsonb; +BEGIN + SELECT * INTO q FROM mst2_metadata_prepare WHERE prepare_id=pid AND state='PREPARING' + AND graph_domain='qualified-v1' AND mst2_metadata_scope_matches(primary_scope); + IF NOT FOUND THEN RAISE EXCEPTION 'MTP2 certification needs an active exact preparation'; END IF; + SELECT b.* INTO body FROM mst2_metadata_payload b JOIN mst2_metadata_current cur USING(page_id,generation) + JOIN mst2_metadata_lifetime life USING(page_id,generation) + JOIN mst2_metadata_prepare_page member USING(page_id,generation) + WHERE b.page_id=p AND b.generation=g AND member.prepare_id=pid AND life.state='RESERVED' + AND life.graph_domain='qualified-v1' AND life.metadata_codec=q.metadata_codec + AND b.metadata_codec=q.metadata_codec AND b.byte_size=member.expected_size AND b.byte_size=life.expected_size + AND NOT EXISTS(SELECT 1 FROM mst2_metadata_gc_op op WHERE op.page_id=p AND op.generation=g); + IF NOT FOUND THEN RAISE EXCEPTION 'MTP2 certification crossed its durable payload lifetime'; END IF; + decoded:=mst2_metadata_decode_local(body.payload); + IF decode(decoded->>'page_id','hex')<>p THEN RAISE EXCEPTION 'MTP2 durable payload digest differs from its page'; END IF; + map_count:=jsonb_array_length(decoded->'entries'); map_bytes:=(decoded->>'direct_entry_bytes')::bigint; + bytes:=body.byte_size; entries:=map_count; + FOR ref IN SELECT value FROM jsonb_array_elements(decoded->'refs') LOOP + child_page:=decode(ref->>'child','hex'); + SELECT member.generation INTO child_generation FROM mst2_metadata_prepare_page member + WHERE member.prepare_id=pid AND member.page_id=child_page; + IF NOT FOUND THEN + SELECT reused.root_generation INTO child_generation FROM mst2_metadata_prepare_reuse_root reused + JOIN mst2_metadata_root_anchor anchor ON anchor.prepare_id=reused.prepare_id + AND anchor.anchor_kind='REUSE' AND anchor.owner_key=pid + AND anchor.root_page=reused.root_page AND anchor.root_generation=reused.root_generation + WHERE reused.prepare_id=pid AND reused.root_page=child_page; + IF NOT FOUND THEN RAISE EXCEPTION 'MTP2 reference is outside exact delta or reused-root membership'; END IF; + END IF; + child:=mst2_metadata_child_certificate(child_page,child_generation,pid); + bound:=ref||jsonb_build_object('generation',child_generation,'certificate',child->>'certificate'); + bound_refs:=bound_refs||jsonb_build_array(bound); + key:=encode(child_page,'hex')||':'||child_generation; + children:=children||jsonb_build_object(encode(child_page,'hex'),child); + IF NOT seen ? key THEN + seen:=seen||jsonb_build_array(key); rank:=greatest(rank,(child->>'rank')::integer+1); + nodes:=nodes+(child->>'closure_nodes_upper')::bigint; + edges:=edges+1+(child->>'closure_edges_upper')::bigint; + bytes:=bytes+(child->>'closure_bytes_upper')::bigint; + entries:=entries+(child->>'closure_entries_upper')::bigint; + END IF; + IF ref->>'kind'='RADIX' THEN + child_min:=decode(child->>'min_name','hex'); child_max:=decode(child->>'max_name','hex'); + prefix:=decode(decoded->>'prefix','hex'); + IF child_min IS NULL OR child_max IS NULL OR (child->>'map_entry_count')::bigint<>(ref->>'count')::bigint + OR octet_length(child_min)<=octet_length(prefix) OR octet_length(child_max)<=octet_length(prefix) + OR substring(child_min FROM 1 FOR octet_length(prefix))<>prefix + OR substring(child_max FROM 1 FOR octet_length(prefix))<>prefix + OR get_byte(child_min,octet_length(prefix))<>(ref->>'label')::integer + OR get_byte(child_max,octet_length(prefix))<>(ref->>'label')::integer THEN + RAISE EXCEPTION 'MTP2 radix child count or name partition differs from its canonical certificate'; + END IF; + IF minimum IS NULL OR child_minmaximum THEN maximum:=child_max; END IF; + map_count:=map_count+(child->>'map_entry_count')::bigint; + map_bytes:=map_bytes+(child->>'map_encoded_entry_bytes')::bigint; + path_bytes:=greatest(path_bytes,(child->>'relative_path_bytes')::integer); + components:=greatest(components,(child->>'relative_components')::integer); + END IF; + IF rank>4095 OR nodes>4096 OR edges>16384 OR bytes>67108864 OR entries>131072 THEN + RAISE EXCEPTION 'MTP2 certified closure upper bound exceeds its fixed budget'; + END IF; + END LOOP; + FOR entry IN SELECT value FROM jsonb_array_elements(decoded->'entries') LOOP + name:=decode(entry->>'name','hex'); + IF minimum IS NULL OR namemaximum THEN maximum:=name; END IF; + IF (entry->>'kind')::integer=4 THEN + child:=children->(entry->>'child'); + path_bytes:=greatest(path_bytes,1+octet_length(name)+(child->>'relative_path_bytes')::integer); + components:=greatest(components,1+(child->>'relative_components')::integer); + ELSE + path_bytes:=greatest(path_bytes,1+octet_length(name)); components:=greatest(components,1); + END IF; + END LOOP; + IF map_count<>(decoded->>'count')::bigint OR map_count>131072 OR map_bytes>67108864 + OR path_bytes>4096 OR components>256 THEN RAISE EXCEPTION 'MTP2 canonical map or path budget is invalid'; END IF; + IF (decoded->>'kind')::integer=1 AND ( + (map_count<=128 AND 20+map_bytes<=16384) + OR decode(decoded->>'prefix','hex') IS DISTINCT FROM mst2_metadata_lcp(minimum,maximum)) THEN + RAISE EXCEPTION 'MTP2 branch is leafable or its prefix is not the true canonical LCP'; + END IF; + proof:=jsonb_build_object('namespace',(SELECT namespace_uuid::text FROM mst2_metadata_family_identity WHERE singleton=1), + 'page',encode(p,'hex'),'generation',g,'codec',body.metadata_codec,'byte_size',body.byte_size, + 'proof_revision',1,'page_kind',(decoded->>'kind')::integer,'references',bound_refs,'rank',rank, + 'map_entry_count',map_count,'map_encoded_entry_bytes',map_bytes,'min_name',encode(minimum,'hex'), + 'max_name',encode(maximum,'hex'),'relative_path_bytes',path_bytes,'relative_components',components, + 'closure_nodes_upper',nodes,'closure_edges_upper',edges,'closure_bytes_upper',bytes,'closure_entries_upper',entries); + RETURN proof||jsonb_build_object('certificate',encode(sha256(convert_to('mega.mst2.canonical-proof.v1','UTF8') + ||decode('00','hex')||convert_to(proof::text,'UTF8')),'hex')); +END $$; + +CREATE FUNCTION mst2_metadata_certificate_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE proof jsonb; +BEGIN + IF TG_OP<>'INSERT' THEN RAISE EXCEPTION 'MTP2 canonical certificate history is immutable'; END IF; + proof:=mst2_metadata_compute_certificate(NEW.page_id,NEW.generation,NEW.origin_prepare_id); + IF NEW.certificate_digest<>decode(proof->>'certificate','hex') OR NEW.canonical_proof<>proof + OR NEW.namespace_uuid::text<>proof->>'namespace' OR NEW.proof_revision<>1 OR NEW.metadata_codec<>(proof->>'codec')::smallint + OR NEW.byte_size<>(proof->>'byte_size')::integer OR NEW.rank<>(proof->>'rank')::integer + OR NEW.map_entry_count<>(proof->>'map_entry_count')::bigint + OR NEW.map_encoded_entry_bytes<>(proof->>'map_encoded_entry_bytes')::bigint + OR NEW.min_name IS DISTINCT FROM decode(proof->>'min_name','hex') + OR NEW.max_name IS DISTINCT FROM decode(proof->>'max_name','hex') + OR NEW.relative_path_bytes<>(proof->>'relative_path_bytes')::integer + OR NEW.relative_components<>(proof->>'relative_components')::integer + OR NEW.closure_nodes_upper<>(proof->>'closure_nodes_upper')::integer + OR NEW.closure_edges_upper<>(proof->>'closure_edges_upper')::integer + OR NEW.closure_bytes_upper<>(proof->>'closure_bytes_upper')::bigint + OR NEW.closure_entries_upper<>(proof->>'closure_entries_upper')::bigint THEN + RAISE EXCEPTION 'MTP2 canonical certificate was not derived from its durable bytes and exact child proofs'; + END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_metadata_certificate_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_metadata_page_certificate + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_certificate_guard(); + +CREATE FUNCTION mst2_metadata_verified_ref_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE ref jsonb; +BEGIN + IF TG_OP<>'INSERT' THEN RAISE EXCEPTION 'MTP2 canonical reference history is immutable'; END IF; + SELECT canonical_proof->'references'->NEW.reference_ordinal INTO ref FROM mst2_metadata_page_certificate + WHERE page_id=NEW.parent_page AND generation=NEW.parent_generation; + IF ref IS NULL OR NEW.reference_kind<>ref->>'kind' OR NEW.child_page<>decode(ref->>'child','hex') + OR NEW.child_generation<>(ref->>'generation')::bigint + OR NEW.child_certificate_digest<>decode(ref->>'certificate','hex') + OR NEW.name IS DISTINCT FROM decode(ref->>'name','hex') + OR NEW.label IS DISTINCT FROM (ref->>'label')::integer + OR NEW.advertised_count IS DISTINCT FROM (ref->>'count')::bigint THEN + RAISE EXCEPTION 'MTP2 canonical reference differs from its independently derived occurrence'; + END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_metadata_verified_ref_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_metadata_verified_ref + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_verified_ref_guard(); + +CREATE FUNCTION mst2_metadata_certificate_complete() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE p bytea; g bigint; proof jsonb; +BEGIN + IF TG_TABLE_NAME IN ('mst2_metadata_verified_ref','mst2_metadata_graph_edge') THEN p:=NEW.parent_page; g:=NEW.parent_generation; + ELSE p:=NEW.page_id; g:=NEW.generation; END IF; + SELECT canonical_proof INTO proof FROM mst2_metadata_page_certificate WHERE page_id=p AND generation=g; + IF proof IS NULL OR (SELECT count(*) FROM mst2_metadata_verified_ref WHERE parent_page=p AND parent_generation=g) + <>jsonb_array_length(proof->'references') + OR NOT EXISTS(SELECT 1 FROM mst2_metadata_graph_node n WHERE n.page_id=p AND n.generation=g + AND n.state='LIVE' AND n.certificate_digest=decode(proof->>'certificate','hex')) + OR EXISTS((SELECT child_page,child_generation FROM mst2_metadata_verified_ref WHERE parent_page=p AND parent_generation=g) + EXCEPT (SELECT child_page,child_generation FROM mst2_metadata_graph_edge WHERE parent_page=p AND parent_generation=g)) + OR EXISTS((SELECT child_page,child_generation FROM mst2_metadata_graph_edge WHERE parent_page=p AND parent_generation=g) + EXCEPT (SELECT child_page,child_generation FROM mst2_metadata_verified_ref WHERE parent_page=p AND parent_generation=g)) THEN + RAISE EXCEPTION 'MTP2 canonical certificate must commit with its complete exact graph and typed references'; + END IF; + RETURN NULL; +END $$; +CREATE CONSTRAINT TRIGGER mst2_metadata_certificate_complete AFTER INSERT ON mst2_metadata_page_certificate + DEFERRABLE INITIALLY DEFERRED FOR EACH ROW EXECUTE FUNCTION mst2_metadata_certificate_complete(); +CREATE CONSTRAINT TRIGGER mst2_metadata_verified_refs_complete AFTER INSERT ON mst2_metadata_verified_ref + DEFERRABLE INITIALLY DEFERRED FOR EACH ROW EXECUTE FUNCTION mst2_metadata_certificate_complete(); +CREATE CONSTRAINT TRIGGER mst2_metadata_graph_refs_complete AFTER INSERT ON mst2_metadata_graph_edge + DEFERRABLE INITIALLY DEFERRED FOR EACH ROW EXECUTE FUNCTION mst2_metadata_certificate_complete(); + +CREATE FUNCTION mst2_metadata_certify_page(pid text,p bytea,g bigint) RETURNS bytea LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE proof jsonb; existing bytea; +BEGIN + SELECT c.certificate_digest INTO existing FROM mst2_metadata_page_certificate c + JOIN mst2_metadata_current cur USING(page_id,generation) JOIN mst2_metadata_lifetime life USING(page_id,generation) + JOIN mst2_metadata_graph_node n USING(page_id,generation) JOIN mst2_metadata_prepare_page member USING(page_id,generation) + JOIN mst2_metadata_prepare q USING(prepare_id) + WHERE c.page_id=p AND c.generation=g AND member.prepare_id=pid AND q.state='PREPARING' + AND mst2_metadata_scope_matches(q.primary_scope) AND life.state='LIVE' AND life.graph_domain='qualified-v1' + AND n.state='LIVE' AND n.certificate_digest=c.certificate_digest AND n.bytes=member.expected_size + AND n.metadata_codec=q.metadata_codec + AND NOT EXISTS(SELECT 1 FROM mst2_metadata_gc_op op WHERE op.page_id=p AND op.generation=g); + IF FOUND THEN RETURN existing; END IF; + proof:=mst2_metadata_compute_certificate(p,g,pid); + INSERT INTO mst2_metadata_page_certificate(page_id,generation,certificate_digest,origin_prepare_id,namespace_uuid, + proof_revision,metadata_codec,byte_size,rank,map_entry_count,map_encoded_entry_bytes,min_name,max_name, + relative_path_bytes,relative_components,closure_nodes_upper,closure_edges_upper,closure_bytes_upper,closure_entries_upper,canonical_proof) + VALUES(p,g,decode(proof->>'certificate','hex'),pid,(proof->>'namespace')::uuid,1,(proof->>'codec')::smallint, + (proof->>'byte_size')::integer,(proof->>'rank')::integer,(proof->>'map_entry_count')::bigint, + (proof->>'map_encoded_entry_bytes')::bigint,decode(proof->>'min_name','hex'),decode(proof->>'max_name','hex'), + (proof->>'relative_path_bytes')::integer,(proof->>'relative_components')::integer, + (proof->>'closure_nodes_upper')::integer,(proof->>'closure_edges_upper')::integer, + (proof->>'closure_bytes_upper')::bigint,(proof->>'closure_entries_upper')::bigint,proof); + INSERT INTO mst2_metadata_verified_ref(parent_page,parent_generation,reference_ordinal,reference_kind, + name,label,advertised_count,child_page,child_generation,child_certificate_digest) + SELECT p,g,ordinality::integer-1,value->>'kind',decode(value->>'name','hex'),(value->>'label')::integer, + (value->>'count')::bigint,decode(value->>'child','hex'),(value->>'generation')::bigint,decode(value->>'certificate','hex') + FROM jsonb_array_elements(proof->'references') WITH ORDINALITY refs(value,ordinality); + INSERT INTO mst2_metadata_graph_node(page_id,generation,state,metadata_codec,bytes,incoming_refs,certificate_digest) + VALUES(p,g,'LIVE',(proof->>'codec')::smallint,(proof->>'byte_size')::integer,0,decode(proof->>'certificate','hex')); + INSERT INTO mst2_metadata_graph_edge(parent_page,parent_generation,child_page,child_generation) + SELECT DISTINCT p,g,child_page,child_generation FROM mst2_metadata_verified_ref WHERE parent_page=p AND parent_generation=g; + RETURN decode(proof->>'certificate','hex'); +END $$; + +CREATE FUNCTION mst2_metadata_certify_batch(pid text,members jsonb) RETURNS integer LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE member jsonb; certified integer:=0; +BEGIN + IF members IS NULL OR jsonb_typeof(members)<>'array' OR jsonb_array_length(members) NOT BETWEEN 1 AND 4096 THEN + RAISE EXCEPTION 'canonical certification batch exceeds its fixed member budget'; + END IF; + -- Array order is supplied by the independently validated cold DAG. Each + -- invocation still proves its durable bytes and already certified children. + FOR member IN SELECT value FROM jsonb_array_elements(members) WITH ORDINALITY AS input(value,ordinal) + ORDER BY ordinal LOOP + IF jsonb_typeof(member)<>'object' OR member->>'page' !~ '^[0-9a-f]{64}$' + OR member->>'generation' IS NULL THEN RAISE EXCEPTION 'canonical certification batch member is malformed'; END IF; + PERFORM mst2_metadata_certify_page(pid,decode(member->>'page','hex'),(member->>'generation')::bigint); + certified:=certified+1; + END LOOP; + RETURN certified; +END $$; diff --git a/src/jupiter/storage/qualified_metadata_family.rs b/src/jupiter/storage/qualified_metadata_family.rs new file mode 100644 index 00000000..52594dd3 --- /dev/null +++ b/src/jupiter/storage/qualified_metadata_family.rs @@ -0,0 +1,624 @@ +//! Captured physical Q provisioning and the sealed rooted production repository. + +use mst2_codec::metapage::{HEADER_LEN, PAGE_MAX_BYTES}; +use sea_orm::{ + ConnectionTrait, DatabaseConnection, DatabaseTransaction, DbBackend, IsolationLevel, Statement, + TransactionTrait, +}; +use sha2::{Digest, Sha256}; +use url::Url; + +#[cfg(test)] +use super::native_metadata_install::generations::{ + GenerationMetadataReceipt, GenerationPrepareIntent, + qualified::PostgresQualifiedMetadataRepository, +}; +use super::{init::postgres_connection, native_metadata_install::MetadataInstallError}; +#[cfg(test)] +use crate::ceres::snapshot::pages::PreparedNativeMetadataRetention; +use crate::{ + ceres::snapshot::{error::SnapshotError, retention_dag::MetadataPagePayload}, + common::errors::MegaError, + config::DbConfig, +}; + +const FAMILY: &str = "v3-rooted-qualified-1"; +const FAMILY_SQL: &str = include_str!("qualified_metadata_family.sql"); +const CANONICAL_SQL: &str = include_str!("qualified_metadata_canonical.sql"); +const CERTIFICATES_SQL: &str = include_str!("qualified_metadata_certificates.sql"); +const ANCHORS_SQL: &str = include_str!("qualified_metadata_anchors.sql"); +const SOURCE_READ_SQL: &str = include_str!("qualified_metadata_source_read.sql"); +const SOURCE_REVISION_SQL: &str = include_str!("qualified_source_revision.sql"); +const ROOTED_SQL: &str = include_str!("qualified_metadata_rooted.sql"); +const SERVING_SQL: &str = include_str!("qualified_metadata_serving.sql"); +const GC_SQL: &str = include_str!("qualified_metadata_gc.sql"); + +#[path = "qualified_metadata_rooted.rs"] +mod rooted; +pub(crate) use rooted::{ + RootedDirectoryWindow, RootedLookupBatch, RootedLookupStatus, RootedQualifiedMetadataRepository, +}; +#[cfg(test)] +pub(crate) use rooted::{ + RootedPrepareIntent, with_rooted_reader_barriers, with_rooted_source_fact_barriers, + with_rooted_source_temporary_shadow, +}; + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(crate) enum SnapshotMetadataFamily { + Generic, + Rooted, +} + +pub(crate) async fn select_snapshot_family( + connection: &DatabaseConnection, + identity: &str, + lease: bool, +) -> Result, SnapshotError> { + let result = async { + let txn = connection + .begin_with_config(Some(IsolationLevel::ReadCommitted), None) + .await?; + let (schema, _) = captured_core(&txn).await?; + let function = if lease { + "mst2_route_family_for_lease" + } else { + "mst2_route_family_for_snapshot" + }; + let rows = txn + .query_all_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + format!( + "SELECT namespace_uuid::text,graph_domain FROM {}.{function}($1,$2)", + identifier(&schema) + ), + [identity.into(), schema.into()], + )) + .await?; + let family = match rows.as_slice() { + [] => None, + [row] => Some(match row.try_get::("", "graph_domain")?.as_str() { + "generic-v1" => SnapshotMetadataFamily::Generic, + "qualified-v1" => SnapshotMetadataFamily::Rooted, + _ => { + return Err(rejected( + "snapshot route selected an unsupported physical family", + )); + } + }), + _ => { + return Err(rejected( + "snapshot route selected multiple physical families", + )); + } + }; + txn.commit().await?; + Ok::<_, MegaError>(family) + } + .await; + result.map_err(|error| { + if let MegaError::Db(db_error) = &error + && rooted::is_lock_unavailable(db_error) + { + return SnapshotError::new( + crate::ceres::snapshot::error::SnapshotErrorCode::TemporaryUnavailable, + "fixed source is being updated; retry the operation", + ); + } + SnapshotError::new( + crate::ceres::snapshot::error::SnapshotErrorCode::IntegrityError, + error.to_string(), + ) + }) +} + +#[derive(Debug, Clone, PartialEq, Eq)] +pub(crate) struct VerifiedQualifiedNamespace { + core_schema: String, + core_oid: i64, + schema: String, + schema_oid: i64, + namespace_uuid: String, + storage_uuid: String, + catalog_fingerprint: Vec, +} + +fn identifier(value: &str) -> String { + format!("\"{}\"", value.replace('"', "\"\"")) +} +fn literal(value: &str) -> String { + format!("'{}'", value.replace('\'', "''")) +} +pub(crate) fn implementation_fingerprint() -> Vec { + let mut hash = Sha256::new(); + hash.update(b"mega.mst2.rooted-qualified-implementation.v1\0"); + // Length-prefix every source component so source selection and proof + // revisions cannot change while retaining an accepted physical stamp. + for (name, bytes) in [ + ("family", FAMILY.as_bytes()), + ("family-ddl", FAMILY_SQL.as_bytes()), + ("canonical-proof-revision-1", CANONICAL_SQL.as_bytes()), + ("indexed-source-read-revision-1", SOURCE_READ_SQL.as_bytes()), + ("captured-source-revision-1", SOURCE_REVISION_SQL.as_bytes()), + ("rooted-collector-revision-1", GC_SQL.as_bytes()), + ("typed-certificate-revision-1", CERTIFICATES_SQL.as_bytes()), + ( + "source-attestation-and-anchor-revision-1", + ANCHORS_SQL.as_bytes(), + ), + ("rooted-plan-and-bindings-revision-1", ROOTED_SQL.as_bytes()), + ( + "rooted-session-and-reader-revision-1", + SERVING_SQL.as_bytes(), + ), + ( + "rooted-physical-route-revision-1", + include_bytes!("../migration/m20261008_000200_rooted_routes.sql").as_slice(), + ), + ( + "authority-catalog-selector", + include_bytes!("qualified_family_catalog.sql").as_slice(), + ), + ( + "normalized-family-shape", + include_bytes!("qualified_family_shape.sql").as_slice(), + ), + ( + "initial-core-registration", + include_bytes!("../migration/m20261008_000200_rooted_qualified_family.sql").as_slice(), + ), + ] { + hash.update((name.len() as u64).to_le_bytes()); + hash.update(name.as_bytes()); + hash.update((bytes.len() as u64).to_le_bytes()); + hash.update(bytes); + } + hash.finalize().to_vec() +} +pub(crate) fn render_family( + core_schema: &str, + core_oid: i64, + q_schema: &str, + q_oid: i64, + n_uuid: &str, + s_uuid: &str, +) -> String { + FAMILY_SQL + .replace("$CANONICAL_SQL$", CANONICAL_SQL) + .replace("$CERTIFICATES_SQL$", CERTIFICATES_SQL) + .replace("$ANCHORS_SQL$", ANCHORS_SQL) + .replace("$SOURCE_READ_SQL$", SOURCE_READ_SQL) + .replace("$ROOTED_SQL$", ROOTED_SQL) + .replace("$SERVING_SQL$", SERVING_SQL) + .replace("$GC_SQL$", GC_SQL) + .replace("$CORE_SCHEMA$", &identifier(core_schema)) + .replace("$CORE_LITERAL$", &literal(core_schema)) + .replace("$Q_SCHEMA$", &identifier(q_schema)) + .replace("$Q_LITERAL$", &literal(q_schema)) + .replace("$CORE_OID$", &core_oid.to_string()) + .replace("$Q_OID$", &q_oid.to_string()) + .replace("$NAMESPACE_UUID$", n_uuid) + .replace("$STORAGE_UUID$", s_uuid) + .replace( + "$IMPLEMENTATION_SHA$", + &hex::encode(implementation_fingerprint()), + ) + .replace("$HEADER_LEN$", &HEADER_LEN.to_string()) + .replace("$PAGE_MAX_BYTES$", &PAGE_MAX_BYTES.to_string()) +} +fn rejected(message: &str) -> MegaError { + MegaError::Other(message.into()) +} + +async fn captured_core(connection: &C) -> Result<(String, i64), MegaError> { + let row = connection.query_one_raw(Statement::from_string(DbBackend::Postgres, + "SELECT current_schema() AS schema,n.oid::bigint AS oid FROM pg_catalog.pg_namespace n WHERE n.nspname=current_schema()")) + .await?.ok_or_else(|| rejected("qualified family core schema is missing"))?; + let schema: String = row.try_get("", "schema")?; + let oid: i64 = row.try_get("", "oid")?; + let valid = connection + .query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + format!( + "SELECT {}.mst2_route_scope_valid($1) AS valid", + identifier(&schema) + ), + [schema.clone().into()], + )) + .await? + .ok_or_else(|| rejected("qualified family core identity is missing"))? + .try_get::("", "valid")?; + if !valid { + return Err(rejected( + "qualified family requires its registered primary core schema", + )); + } + Ok((schema, oid)) +} + +async fn catalog( + connection: &C, + core_oid: i64, + q_oid: i64, +) -> Result, MegaError> { + catalog_with_exemption(connection, core_oid, q_oid, 0).await +} + +async fn catalog_with_exemption( + connection: &C, + core_oid: i64, + q_oid: i64, + exempt_q_oid: i64, +) -> Result, MegaError> { + // Inspect the actual catalogs directly. A replaced helper function cannot + // turn a bad physical structure into a fresh accepted fingerprint. + let sql = include_str!("qualified_family_catalog.sql") + .replace("$CORE_OID$", "$1::bigint::oid") + .replace("$Q_OID$", "$2::bigint::oid") + .replace("$EXEMPT_Q_OID$", "$3::bigint::oid"); + let row = connection + .query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + sql, + [core_oid.into(), q_oid.into(), exempt_q_oid.into()], + )) + .await? + .ok_or_else(|| rejected("qualified family catalog is missing"))?; + Ok(row.try_get("", "fingerprint")?) +} + +async fn registered( + connection: &C, + core: &(String, i64), +) -> Result, MegaError> { + let policy=connection.query_one_raw(Statement::from_string(DbBackend::Postgres,format!( + "SELECT implementation_fingerprint,expected_shape,authority_catalog FROM {}.mst2_qualified_family_policy WHERE singleton=1",identifier(&core.0)))) + .await?.ok_or_else(||rejected("qualified trusted family policy is missing"))?; + if policy.try_get::>("", "implementation_fingerprint")? != implementation_fingerprint() + { + return Err(rejected("qualified trusted core authority catalog changed")); + } + let rows=connection.query_all_raw(Statement::from_string(DbBackend::Postgres,format!( + "SELECT n.namespace_uuid::text,n.metadata_schema,n.metadata_schema_oid::bigint,n.metadata_storage_uuid, + n.implementation_fingerprint,n.catalog_fingerprint, + n.family_identity,n.admission_state,n.collector_state,n.storage_uuid AS core_storage_uuid, + n.core_schema,n.core_schema_oid::bigint, + (SELECT count(*) FROM {c}.mst2_metadata_namespace)::bigint AS namespace_count, + EXISTS(SELECT 1 FROM pg_catalog.pg_namespace p WHERE p.oid=n.metadata_schema_oid AND p.nspname=n.metadata_schema) AS schema_present, + {c}.mst2_route_scope_valid({core_literal}) AS core_valid + FROM {c}.mst2_metadata_namespace n WHERE n.graph_domain='qualified-v1'",c=identifier(&core.0),core_literal=literal(&core.0)))) + .await?; + if rows.is_empty() { + if policy.try_get::>("", "authority_catalog")? + != catalog(connection, core.1, 0).await? + { + return Err(rejected("qualified trusted core authority catalog changed")); + } + return Ok(None); + } + if rows.len() != 1 { + return Err(rejected( + "qualified family registry exceeds its one-Q hard limit", + )); + } + let row = &rows[0]; + let schema: String = row.try_get("", "metadata_schema")?; + let namespace_uuid: String = row.try_get("", "namespace_uuid")?; + let storage_uuid: String = row.try_get("", "metadata_storage_uuid")?; + let schema_oid: i64 = row.try_get("", "metadata_schema_oid")?; + let expected_schema = format!("mst2q_{}", namespace_uuid.replace('-', "")); + let namespace_identity = uuid::Uuid::parse_str(&namespace_uuid) + .map_err(|_| rejected("qualified namespace UUID is invalid"))?; + let storage_identity = uuid::Uuid::parse_str(&storage_uuid) + .map_err(|_| rejected("qualified storage UUID is invalid"))?; + if schema != expected_schema + || namespace_identity.get_version_num() != 4 + || namespace_identity.get_variant() != uuid::Variant::RFC4122 + || namespace_identity.to_string() != namespace_uuid + || storage_identity.get_version_num() != 4 + || storage_identity.get_variant() != uuid::Variant::RFC4122 + || storage_identity.to_string() != storage_uuid + || storage_uuid == row.try_get::("", "core_storage_uuid")? + || row.try_get::("", "core_schema")? != core.0 + || row.try_get::("", "core_schema_oid")? != core.1 + || row.try_get::("", "family_identity")? != FAMILY + || row.try_get::("", "admission_state")? != "ROOTED_Q_ADMITTED" + || row.try_get::("", "collector_state")? != "ENABLED" + || row.try_get::("", "namespace_count")? != 2 + || !row.try_get::("", "schema_present")? + || !row.try_get::("", "core_valid")? + || row.try_get::>("", "implementation_fingerprint")? != implementation_fingerprint() + { + return Err(rejected( + "qualified family registry and physical identity disagree", + )); + } + // A single registered, physically present Q identity is the only permitted + // source of core-side RI triggers omitted from the core authority stamp. + // The complete catalog and trusted shape below still bind all of them. + if policy.try_get::>("", "authority_catalog")? + != catalog_with_exemption(connection, core.1, 0, schema_oid).await? + { + return Err(rejected("qualified trusted core authority catalog changed")); + } + let fingerprint: Vec = row.try_get("", "catalog_fingerprint")?; + if fingerprint.len() != 32 || catalog(connection, core.1, schema_oid).await? != fingerprint { + return Err(rejected( + "qualified family actual catalog fingerprint changed", + )); + } + let shape: Vec = connection + .query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + format!( + "SELECT {}.mst2_route_family_shape($1::bigint::oid,$2::uuid,$3) AS fingerprint", + identifier(&core.0) + ), + [ + schema_oid.into(), + namespace_uuid.clone().into(), + storage_uuid.clone().into(), + ], + )) + .await? + .ok_or_else(|| rejected("qualified family actual shape is missing"))? + .try_get("", "fingerprint")?; + if shape != policy.try_get::>("", "expected_shape")? { + return Err(rejected( + "qualified namespace does not have the trusted complete physical family shape", + )); + } + let stamp=connection.query_one_raw(Statement::from_sql_and_values(DbBackend::Postgres,format!( + "SELECT EXISTS(SELECT 1 FROM {q}.mst2_metadata_family_identity i JOIN {q}.mst2_metadata_storage_scope s USING(singleton) + WHERE i.singleton=1 AND i.namespace_uuid=$1::uuid AND i.storage_uuid=$2 AND s.storage_uuid=$2 + AND i.core_schema_oid=$3::bigint::oid AND i.metadata_schema_oid=$4::bigint::oid + AND i.family_identity=$5 AND i.implementation_fingerprint=$6) AS valid",q=identifier(&schema)), + [namespace_uuid.clone().into(),storage_uuid.clone().into(),core.1.into(),schema_oid.into(),FAMILY.into(),implementation_fingerprint().into()])) + .await?.ok_or_else(||rejected("qualified family physical stamp is missing"))?; + if !stamp.try_get::("", "valid")? { + return Err(rejected("qualified family physical stamp changed")); + } + Ok(Some(VerifiedQualifiedNamespace { + core_schema: core.0.clone(), + core_oid: core.1, + schema, + schema_oid, + namespace_uuid, + storage_uuid, + catalog_fingerprint: fingerprint, + })) +} + +/// Production bootstrap calls this after core migrations, before returning a +/// writable app connection. Existing registrations are verified, never reset. +pub(crate) async fn provision_or_verify_rooted_qualified_family( + connection: &DatabaseConnection, +) -> Result { + let core = captured_core(connection).await?; + for attempt in 0..2 { + let txn = connection + .begin_with_config(Some(IsolationLevel::ReadCommitted), None) + .await?; + if let Some(namespace) = registered(&txn, &core).await? { + txn.execute_unprepared(&format!( + "SELECT {}.mst2_route_enter({})", + identifier(&core.0), + literal(&core.0) + )) + .await?; + let locked = registered(&txn, &core) + .await? + .ok_or_else(|| rejected("qualified registration disappeared"))?; + if locked != namespace { + return Err(rejected("qualified registration changed during bootstrap")); + } + txn.commit().await?; + return Ok(locked); + } + let candidate = uuid::Uuid::new_v4().to_string(); + let schema = format!("mst2q_{}", candidate.replace('-', "")); + let enter = txn + .execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + format!( + "SELECT {}.mst2_route_family_candidate_enter($1,$2::uuid,$3)", + identifier(&core.0) + ), + [ + core.0.clone().into(), + candidate.clone().into(), + schema.clone().into(), + ], + )) + .await; + if let Err(error) = enter { + txn.rollback().await?; + if attempt == 0 && error.to_string().contains("lost bootstrap serialization") { + continue; + } + return Err(error.into()); + } + txn.execute_unprepared(&format!("CREATE SCHEMA {}", identifier(&schema))) + .await?; + let row = txn + .query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT oid::bigint AS oid FROM pg_catalog.pg_namespace WHERE nspname=$1", + [schema.clone().into()], + )) + .await? + .ok_or_else(|| rejected("qualified provisioning schema was not created"))?; + let schema_oid: i64 = row.try_get("", "oid")?; + let storage_uuid = uuid::Uuid::new_v4().to_string(); + let sql = render_family( + &core.0, + core.1, + &schema, + schema_oid, + &candidate, + &storage_uuid, + ); + txn.execute_unprepared(&sql).await?; + let fingerprint = catalog(&txn, core.1, schema_oid).await?; + txn.execute_unprepared(&format!( + "SET LOCAL search_path={},pg_catalog,pg_temp", + identifier(&core.0) + )) + .await?; + txn.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres,format!( + "INSERT INTO {c}.mst2_metadata_namespace(singleton,namespace_uuid,core_schema,core_schema_oid, + database_name,database_oid,storage_uuid,server_address,server_port,mono_lock_key2,metadata_schema, + metadata_schema_oid,family_identity,graph_domain,admission_state,collector_state, + metadata_storage_uuid,implementation_fingerprint,catalog_fingerprint) + SELECT NULL,$1::uuid,g.core_schema,g.core_schema_oid,g.database_name,g.database_oid,g.storage_uuid, + g.server_address,g.server_port,g.mono_lock_key2,$2,$3::bigint::oid,$4,'qualified-v1','ROOTED_Q_ADMITTED','ENABLED',$5,$6,$7 + FROM {c}.mst2_metadata_namespace g WHERE singleton=1",c=identifier(&core.0)), + [candidate.into(),schema.into(),schema_oid.into(),FAMILY.into(),storage_uuid.into(),implementation_fingerprint().into(),fingerprint.into()])).await?; + let namespace = registered(&txn, &core) + .await? + .ok_or_else(|| rejected("qualified provisioning did not register its family"))?; + txn.commit().await?; + return Ok(namespace); + } + Err(rejected( + "qualified provisioning could not serialize bootstrap", + )) +} + +impl VerifiedQualifiedNamespace { + pub(crate) async fn enter(&self, txn: &DatabaseTransaction) -> Result<(), SnapshotError> { + let result: Result<(), MegaError> = async { + let actual = txn + .query_one_raw(Statement::from_string( + DbBackend::Postgres, + "SELECT current_schema() AS schema", + )) + .await? + .ok_or_else(|| rejected("qualified writer schema is missing"))? + .try_get::("", "schema")?; + if actual != self.schema { + return Err(rejected( + "qualified writer escaped its physical pool schema", + )); + } + txn.execute_unprepared(&format!( + "SELECT {}.mst2_route_enter({})", + identifier(&self.core_schema), + literal(&self.core_schema) + )) + .await?; + if registered(txn, &(self.core_schema.clone(), self.core_oid)) + .await? + .as_ref() + != Some(self) + { + return Err(rejected("qualified writer physical namespace changed")); + } + Ok(()) + } + .await; + result.map_err(|e| { + if let MegaError::Db(error) = &e + && rooted::is_lock_unavailable(error) + { + return SnapshotError::new( + crate::ceres::snapshot::error::SnapshotErrorCode::TemporaryUnavailable, + "qualified source is being updated; retry the operation", + ); + } + SnapshotError::new( + crate::ceres::snapshot::error::SnapshotErrorCode::SnapshotNotReady, + e.to_string(), + ) + }) + } +} + +fn pool_url(db_url: &str, namespace: &VerifiedQualifiedNamespace) -> Result { + let mut url = Url::parse(db_url) + .map_err(|_| rejected("qualified writer requires a valid PostgreSQL URL"))?; + let mut options = Vec::new(); + let mut others = Vec::new(); + for (key, value) in url.query_pairs() { + if key == "options" { + options.push(value.into_owned()); + } else { + others.push((key.into_owned(), value.into_owned())); + } + } + // Only server-generated lowercase schema names can reach this option. + options.push(format!( + "-csearch_path={},pg_catalog,pg_temp", + namespace.schema + )); + url.set_query(None); + { + let mut pairs = url.query_pairs_mut(); + for (key, value) in others { + pairs.append_pair(&key, &value); + } + pairs.append_pair("options", &options.join(" ")); + } + Ok(url.to_string()) +} + +/// The adapter has no Deref or unsealed connection accessor. It permits only +/// test-only cold preparation writes, without a production factory. +#[cfg(test)] +pub(crate) struct ShadowQualifiedMetadataWriter { + repository: PostgresQualifiedMetadataRepository, +} +#[cfg(test)] +impl ShadowQualifiedMetadataWriter { + pub(crate) async fn open( + core: &DatabaseConnection, + config: &DbConfig, + ) -> Result { + let captured = captured_core(core).await?; + let namespace = registered(core, &captured) + .await? + .ok_or_else(|| rejected("qualified shadow family is not provisioned at bootstrap"))?; + let mut q_config = config.clone(); + q_config.db_url = pool_url(&config.db_url, &namespace)?; + q_config.max_connection = q_config.max_connection.clamp(1, 2); + q_config.min_connection = 1; + let connection = postgres_connection(&q_config).await?; + let txn = connection + .begin_with_config(Some(IsolationLevel::ReadCommitted), None) + .await?; + namespace + .enter(&txn) + .await + .map_err(|e| rejected(&e.to_string()))?; + txn.commit().await?; + let repository = + PostgresQualifiedMetadataRepository::registered_shadow(connection, namespace) + .await + .map_err(|e| rejected(&e.to_string()))?; + Ok(Self { repository }) + } + pub(crate) async fn begin_intent( + &self, + operation_id: &str, + prepared: &PreparedNativeMetadataRetention, + ) -> Result { + self.repository.begin_intent(operation_id, prepared).await + } + pub(crate) async fn install_pages( + &self, + intent: &GenerationPrepareIntent, + payloads: &[MetadataPagePayload], + ) -> Result<(), MetadataInstallError> { + self.repository.install_pages(intent, payloads).await + } + pub(crate) async fn finalize( + &self, + intent: &GenerationPrepareIntent, + ) -> Result { + self.repository.finalize(intent).await + } +} + +#[cfg(test)] +#[path = "qualified_metadata_family_tests.rs"] +mod tests; diff --git a/src/jupiter/storage/qualified_metadata_family.sql b/src/jupiter/storage/qualified_metadata_family.sql new file mode 100644 index 00000000..853834e5 --- /dev/null +++ b/src/jupiter/storage/qualified_metadata_family.sql @@ -0,0 +1,491 @@ +-- Dedicated physical v3 Q family with exact generations and owned roots. +SET LOCAL search_path=$Q_SCHEMA$,pg_catalog,pg_temp; +CREATE TABLE mst2_metadata_storage_scope ( + singleton smallint PRIMARY KEY CHECK(singleton=1),storage_uuid text NOT NULL UNIQUE +); +INSERT INTO mst2_metadata_storage_scope VALUES(1,'$STORAGE_UUID$'); +CREATE TABLE mst2_metadata_family_identity ( + singleton smallint PRIMARY KEY CHECK(singleton=1),namespace_uuid uuid NOT NULL UNIQUE, + storage_uuid text NOT NULL UNIQUE,core_schema_oid oid NOT NULL,metadata_schema_oid oid NOT NULL, + family_identity text NOT NULL CHECK(family_identity='v3-rooted-qualified-1'), + implementation_fingerprint bytea NOT NULL CHECK(octet_length(implementation_fingerprint)=32) +); +INSERT INTO mst2_metadata_family_identity VALUES(1,'$NAMESPACE_UUID$'::uuid,'$STORAGE_UUID$', + $CORE_OID$,$Q_OID$,'v3-rooted-qualified-1',decode('$IMPLEMENTATION_SHA$','hex')); +CREATE TABLE mst2_metadata_lifetime ( + page_id bytea NOT NULL CHECK(octet_length(page_id)=32), + node_id text NOT NULL CHECK(node_id='page:sha256:'||encode(page_id,'hex')), + generation bigint NOT NULL CHECK(generation>0), + state text NOT NULL CHECK(state IN ('RESERVED','LIVE','DELETING','REMOVED')), + metadata_codec smallint NOT NULL CHECK(metadata_codec=1), + expected_size integer NOT NULL CHECK(expected_size BETWEEN $HEADER_LEN$ AND $PAGE_MAX_BYTES$), + graph_domain text NOT NULL CHECK(graph_domain='qualified-v1'),PRIMARY KEY(page_id,generation) +); +CREATE INDEX idx_mst2_metadata_lifetime_state_page ON mst2_metadata_lifetime(state,page_id); +CREATE INDEX idx_mst2_metadata_lifetime_node_generation ON mst2_metadata_lifetime(node_id,generation); +CREATE TABLE mst2_metadata_current ( + page_id bytea PRIMARY KEY CHECK(octet_length(page_id)=32),generation bigint NOT NULL CHECK(generation>0), + FOREIGN KEY(page_id,generation) REFERENCES mst2_metadata_lifetime(page_id,generation) +); +CREATE TABLE mst2_metadata_payload ( + page_id bytea PRIMARY KEY CHECK(octet_length(page_id)=32),generation bigint NOT NULL CHECK(generation>0), + metadata_codec smallint NOT NULL CHECK(metadata_codec=1), + byte_size integer NOT NULL CHECK(byte_size BETWEEN $HEADER_LEN$ AND $PAGE_MAX_BYTES$), + payload bytea NOT NULL CHECK(octet_length(payload)=byte_size),created_at timestamptz NOT NULL DEFAULT now(), + FOREIGN KEY(page_id,generation) REFERENCES mst2_metadata_lifetime(page_id,generation) +); +CREATE TABLE mst2_metadata_prepare ( + prepare_id text PRIMARY KEY CHECK(prepare_id ~ '^[0-9a-f]{8}-[0-9a-f]{4}-4[0-9a-f]{3}-[89ab][0-9a-f]{3}-[0-9a-f]{12}$'), + operation_id text NOT NULL UNIQUE CHECK(octet_length(operation_id) BETWEEN 1 AND 255), + manifest_digest bytea NOT NULL CHECK(octet_length(manifest_digest)=32), + canonical_plan bytea NOT NULL CHECK(octet_length(canonical_plan)<=2097152), + source_domain text NOT NULL CHECK(source_domain='native-git'),tagged_root_tree_oid text NOT NULL, + plan_kind text NOT NULL DEFAULT 'COLD' CHECK(plan_kind IN ('COLD','ROOTED')), + bindings_revision bigint NOT NULL DEFAULT 0 CHECK(bindings_revision>=0), + scope text NOT NULL CHECK(octet_length(scope)<=4096),schema_version smallint NOT NULL, + metadata_codec smallint NOT NULL CHECK(metadata_codec=1),materialization_policy smallint NOT NULL, + fs_semantics smallint NOT NULL,access_projection smallint NOT NULL,verification_revision integer NOT NULL, + projection_revision smallint NOT NULL,metadata_root bytea NOT NULL CHECK(octet_length(metadata_root)=32), + node_count integer NOT NULL CHECK(node_count BETWEEN 0 AND 4096), + edge_count integer NOT NULL CHECK(edge_count BETWEEN 0 AND 16384), + total_bytes bigint NOT NULL CHECK(total_bytes BETWEEN 0 AND 67108864), + state text NOT NULL CHECK(state IN ('PREPARING','COMMITTED','ABORTED')), + created_at timestamptz NOT NULL DEFAULT now(),committed_at timestamptz,aborted_at timestamptz, + coverage_retired_at timestamptz, + canonical_bindings bytea NOT NULL CHECK(octet_length(canonical_bindings) BETWEEN 12 AND 196620), + bindings_digest bytea NOT NULL CHECK(octet_length(bindings_digest)=32), + primary_scope bytea NOT NULL CHECK(octet_length(primary_scope) BETWEEN 1 AND 16384), + storage_seal bytea NOT NULL CHECK(octet_length(storage_seal)=32), + graph_domain text NOT NULL CHECK(graph_domain='qualified-v1'),UNIQUE(prepare_id,storage_seal), + CHECK((state='COMMITTED')=(committed_at IS NOT NULL) AND (state='ABORTED')=(aborted_at IS NOT NULL) + AND (coverage_retired_at IS NULL OR state='COMMITTED')) +); +CREATE TABLE mst2_metadata_prepare_page ( + prepare_id text NOT NULL REFERENCES mst2_metadata_prepare(prepare_id), + page_id bytea NOT NULL CHECK(octet_length(page_id)=32),generation bigint NOT NULL CHECK(generation>0), + expected_size integer NOT NULL CHECK(expected_size BETWEEN $HEADER_LEN$ AND $PAGE_MAX_BYTES$), + PRIMARY KEY(prepare_id,page_id),UNIQUE(prepare_id,page_id,generation), + FOREIGN KEY(page_id,generation) REFERENCES mst2_metadata_lifetime(page_id,generation) +); +CREATE INDEX idx_mst2_metadata_prepare_page_lifetime ON mst2_metadata_prepare_page(page_id,generation,prepare_id); +CREATE TABLE mst2_metadata_graph_node ( + page_id bytea NOT NULL CHECK (octet_length(page_id)=32), + generation bigint NOT NULL CHECK (generation>0), + state text NOT NULL CHECK (state IN ('LIVE','DELETING')), + metadata_codec smallint NOT NULL CHECK (metadata_codec=1), + bytes bigint NOT NULL CHECK (bytes BETWEEN $HEADER_LEN$ AND $PAGE_MAX_BYTES$), + incoming_refs bigint NOT NULL DEFAULT 0 CHECK (incoming_refs>=0), + PRIMARY KEY(page_id,generation), + FOREIGN KEY(page_id,generation) REFERENCES mst2_metadata_lifetime(page_id,generation) +); +CREATE INDEX idx_mst2_metadata_graph_node_gc ON mst2_metadata_graph_node(state,incoming_refs,page_id,generation); +CREATE TABLE mst2_metadata_graph_edge ( + parent_page bytea NOT NULL, + parent_generation bigint NOT NULL, + child_page bytea NOT NULL, + child_generation bigint NOT NULL, + PRIMARY KEY(parent_page,parent_generation,child_page,child_generation), + FOREIGN KEY(parent_page,parent_generation) REFERENCES mst2_metadata_graph_node(page_id,generation), + FOREIGN KEY(child_page,child_generation) REFERENCES mst2_metadata_graph_node(page_id,generation), + CHECK (parent_page<>child_page OR parent_generation<>child_generation) +); +CREATE INDEX idx_mst2_metadata_graph_edge_child ON mst2_metadata_graph_edge(child_page,child_generation,parent_page,parent_generation); +CREATE TABLE mst2_metadata_graph_root ( + prepare_id text NOT NULL, + storage_seal bytea NOT NULL CHECK (octet_length(storage_seal)=32), + page_id bytea NOT NULL, + generation bigint NOT NULL, + PRIMARY KEY(prepare_id,page_id,generation), + FOREIGN KEY(prepare_id,storage_seal) REFERENCES mst2_metadata_prepare(prepare_id,storage_seal), + FOREIGN KEY(prepare_id,page_id,generation) REFERENCES mst2_metadata_prepare_page(prepare_id,page_id,generation), + FOREIGN KEY(page_id,generation) REFERENCES mst2_metadata_graph_node(page_id,generation) +); +CREATE INDEX idx_mst2_metadata_graph_root_page ON mst2_metadata_graph_root(page_id,generation,prepare_id); +CREATE TABLE mst2_metadata_gc_op ( + operation_id uuid PRIMARY KEY, + page_id bytea NOT NULL CHECK (octet_length(page_id)=32), + generation bigint NOT NULL CHECK (generation>0), + primary_scope bytea NOT NULL CHECK (octet_length(primary_scope) BETWEEN 1 AND 16384), + graph_domain text NOT NULL CHECK (graph_domain='qualified-v1'), + metadata_codec smallint NOT NULL CHECK (metadata_codec=1), + expected_size integer NOT NULL CHECK (expected_size BETWEEN $HEADER_LEN$ AND $PAGE_MAX_BYTES$), + graph_present boolean NOT NULL, + had_payload boolean NOT NULL, + payload_delete_xid bigint, + state text NOT NULL CHECK (state IN ('PENDING','APPLIED')), + created_at timestamptz NOT NULL DEFAULT clock_timestamp(), + completed_at timestamptz, + UNIQUE(page_id,generation), + FOREIGN KEY(page_id,generation) REFERENCES mst2_metadata_lifetime(page_id,generation), + CHECK ((state='APPLIED')=(completed_at IS NOT NULL)) +); +CREATE INDEX idx_mst2_metadata_gc_op_pending ON mst2_metadata_gc_op(created_at,operation_id) WHERE state='PENDING'; + +-- Incarnations are historical; SID alone is never a unique physical binding. +CREATE TABLE mst2_qualified_session_incarnation ( + snapshot_id text NOT NULL,session_incarnation uuid NOT NULL,namespace_uuid uuid NOT NULL, + prepare_id text NOT NULL,storage_seal bytea NOT NULL CHECK(octet_length(storage_seal)=32), + metadata_root bytea NOT NULL CHECK(octet_length(metadata_root)=32),root_generation bigint NOT NULL CHECK(root_generation>0), + source_profile bytea NOT NULL,instance_id text NOT NULL,commit_oid text NOT NULL,root_tree_oid text NOT NULL, + authorization_epoch bigint NOT NULL,publication_sequence bigint NOT NULL,writer_epoch bigint NOT NULL, + certificate_receipt_id bigint NOT NULL,PRIMARY KEY(snapshot_id,session_incarnation), + UNIQUE(snapshot_id,session_incarnation,prepare_id,storage_seal,metadata_root,root_generation), + FOREIGN KEY(snapshot_id,namespace_uuid) REFERENCES $CORE_SCHEMA$.mst2_snapshot_storage_route(snapshot_id,namespace_uuid), + FOREIGN KEY(prepare_id,storage_seal) REFERENCES mst2_metadata_prepare(prepare_id,storage_seal) +); +CREATE TABLE mst2_qualified_lease_binding ( + lease_id text PRIMARY KEY,snapshot_id text NOT NULL,session_incarnation uuid NOT NULL, + prepare_id text NOT NULL,storage_seal bytea NOT NULL,metadata_root bytea NOT NULL,root_generation bigint NOT NULL, + FOREIGN KEY(snapshot_id,session_incarnation,prepare_id,storage_seal,metadata_root,root_generation) + REFERENCES mst2_qualified_session_incarnation(snapshot_id,session_incarnation,prepare_id,storage_seal,metadata_root,root_generation) +); + +$CANONICAL_SQL$ +$CERTIFICATES_SQL$ +$ANCHORS_SQL$ +$SOURCE_READ_SQL$ +$ROOTED_SQL$ + +CREATE FUNCTION mst2_metadata_closed() RETURNS trigger LANGUAGE plpgsql +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN RAISE EXCEPTION 'qualified session serving and collector are closed'; END $$; +CREATE FUNCTION mst2_metadata_gc_apply(id uuid) RETURNS void LANGUAGE plpgsql +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN RAISE EXCEPTION 'qualified collector is closed'; END $$; +CREATE FUNCTION mst2_metadata_gc_finish(id uuid) RETURNS void LANGUAGE plpgsql +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN RAISE EXCEPTION 'qualified collector is closed'; END $$; +CREATE FUNCTION mst2_metadata_immutable() RETURNS trigger LANGUAGE plpgsql +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN RAISE EXCEPTION 'qualified family identity and history are immutable'; END $$; + +-- Observe the caller before entering fixed trusted functions. Core fallback +-- never enters the pool search path; all cross-family authority is explicit. +CREATE FUNCTION mst2_metadata_dml_barrier() RETURNS trigger LANGUAGE plpgsql VOLATILE AS $$ +BEGIN + IF pg_catalog.current_schema() IS DISTINCT FROM $Q_LITERAL$ OR TG_TABLE_SCHEMA IS DISTINCT FROM $Q_LITERAL$ + OR NOT EXISTS(SELECT 1 FROM pg_catalog.pg_class WHERE oid=TG_RELID AND relnamespace='$Q_OID$'::oid) + OR NOT EXISTS(SELECT 1 FROM pg_catalog.pg_namespace WHERE oid='$Q_OID$'::oid AND nspname=$Q_LITERAL$) + OR pg_catalog.pg_is_in_recovery() OR pg_catalog.current_setting('transaction_isolation')<>'read committed' THEN + RAISE EXCEPTION 'qualified mutation requires its captured primary family and READ COMMITTED'; + END IF; + PERFORM $CORE_SCHEMA$.mst2_route_enter($CORE_LITERAL$); + IF NOT EXISTS(SELECT 1 FROM $CORE_SCHEMA$.mst2_metadata_namespace n + WHERE n.namespace_uuid='$NAMESPACE_UUID$'::uuid AND n.metadata_schema=$Q_LITERAL$ + AND n.metadata_schema_oid='$Q_OID$'::oid AND n.core_schema_oid=$CORE_OID$ + AND n.metadata_storage_uuid='$STORAGE_UUID$' AND n.admission_state='ROOTED_Q_ADMITTED' + AND n.collector_state='ENABLED' AND n.implementation_fingerprint=pg_catalog.decode('$IMPLEMENTATION_SHA$','hex') + AND n.catalog_fingerprint=$CORE_SCHEMA$.mst2_route_family_catalog($CORE_OID$,'$Q_OID$'::oid)) THEN + RAISE EXCEPTION 'qualified physical family catalog fingerprint is unavailable'; + END IF; + RETURN NULL; +END $$; + + +CREATE FUNCTION mst2_metadata_scope_matches(s bytea) RETURNS boolean LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE actual jsonb; +BEGIN + IF pg_is_in_recovery() OR current_setting('transaction_isolation')<>'read committed' THEN RETURN false; END IF; + SELECT jsonb_build_array(x.storage_uuid,current_database(),d.oid::bigint,$Q_LITERAL$,'$Q_OID$'::bigint, + inet_server_addr()::text,inet_server_port()) INTO actual FROM mst2_metadata_storage_scope x + JOIN pg_database d ON d.datname=current_database() WHERE x.singleton=1; + RETURN actual IS NOT NULL AND convert_from(s,'UTF8')::jsonb=actual; +EXCEPTION WHEN OTHERS THEN RETURN false; +END $$; +CREATE FUNCTION mst2_metadata_has_generic_overlap(p bytea) RETURNS boolean LANGUAGE sql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ SELECT false $$; + +CREATE FUNCTION mst2_metadata_lifetime_guard() RETURNS trigger LANGUAGE plpgsql +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + IF TG_OP='DELETE' THEN RAISE EXCEPTION 'metadata lifetime watermark cannot be deleted'; END IF; + IF TG_OP='INSERT' THEN + IF NEW.generation<>1 OR NEW.state<>'RESERVED' OR EXISTS(SELECT 1 FROM mst2_metadata_current WHERE page_id=NEW.page_id) THEN + RAISE EXCEPTION 'qualified initial lifetime cannot adopt old history; collector is closed'; + END IF; + RETURN NEW; + END IF; + IF (to_jsonb(NEW)-'state') IS DISTINCT FROM (to_jsonb(OLD)-'state') THEN + RAISE EXCEPTION 'metadata lifetime identity is immutable'; + END IF; + IF NEW.state=OLD.state THEN RETURN NEW; END IF; + IF OLD.state='RESERVED' AND NEW.state='LIVE' THEN + IF NOT EXISTS(SELECT 1 FROM mst2_metadata_graph_node n JOIN mst2_metadata_payload b USING(page_id,generation) + WHERE n.page_id=OLD.page_id AND n.generation=OLD.generation AND n.state='LIVE' + AND n.metadata_codec=OLD.metadata_codec AND n.bytes=OLD.expected_size + AND b.metadata_codec=OLD.metadata_codec AND b.byte_size=OLD.expected_size) + OR NOT (EXISTS(SELECT 1 FROM mst2_metadata_graph_root r WHERE r.page_id=OLD.page_id AND r.generation=OLD.generation) + OR EXISTS(SELECT 1 FROM mst2_metadata_prepare_page member JOIN mst2_metadata_prepare q USING(prepare_id) + JOIN mst2_metadata_root_anchor anchor ON anchor.prepare_id=q.prepare_id AND anchor.anchor_kind='PREPARE' + AND anchor.owner_key=q.prepare_id AND anchor.root_page=q.metadata_root + WHERE member.page_id=OLD.page_id AND member.generation=OLD.generation AND q.plan_kind='ROOTED' + AND q.state='COMMITTED' AND q.coverage_retired_at IS NULL)) THEN + RAISE EXCEPTION 'qualified LIVE transition needs its graph payload and prepare root'; + END IF; + RETURN NEW; + END IF; + RAISE EXCEPTION 'qualified lifetime collection is closed'; +END $$; +CREATE TRIGGER mst2_metadata_lifetime_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_metadata_lifetime + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_lifetime_guard(); +CREATE FUNCTION mst2_metadata_current_guard() RETURNS trigger LANGUAGE plpgsql +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + IF TG_OP<>'INSERT' THEN RAISE EXCEPTION 'qualified current collection is closed'; END IF; + IF NEW.generation<>1 OR EXISTS(SELECT 1 FROM mst2_metadata_lifetime WHERE page_id=NEW.page_id AND generation<>1) + OR EXISTS(SELECT 1 FROM mst2_metadata_payload WHERE page_id=NEW.page_id) + OR EXISTS(SELECT 1 FROM mst2_metadata_graph_node WHERE page_id=NEW.page_id) THEN + RAISE EXCEPTION 'initial qualified current cannot reset or adopt old history'; + END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_metadata_current_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_metadata_current + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_current_guard(); +CREATE FUNCTION mst2_metadata_current_protected() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + IF EXISTS(SELECT 1 FROM mst2_metadata_lifetime WHERE page_id=NEW.page_id AND generation=NEW.generation AND graph_domain='qualified-v1') + AND NOT EXISTS(SELECT 1 FROM mst2_metadata_prepare_page m JOIN mst2_metadata_prepare q USING(prepare_id) + JOIN mst2_metadata_current c ON c.page_id=m.page_id AND c.generation=m.generation + WHERE m.page_id=NEW.page_id AND m.generation=NEW.generation AND q.graph_domain='qualified-v1' + AND q.storage_seal IS NOT NULL AND (q.state='PREPARING' OR (q.state='COMMITTED' AND + (q.coverage_retired_at IS NULL OR mst2_metadata_session_covers_prepare(q.prepare_id))))) THEN + RAISE EXCEPTION 'qualified current change must commit with fresh prepare protection'; + END IF; + RETURN NULL; +END $$; +CREATE CONSTRAINT TRIGGER mst2_metadata_current_protected AFTER INSERT OR UPDATE ON mst2_metadata_current + DEFERRABLE INITIALLY DEFERRED FOR EACH ROW EXECUTE FUNCTION mst2_metadata_current_protected(); + +CREATE FUNCTION mst2_metadata_prepare_guard() RETURNS trigger LANGUAGE plpgsql +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + IF TG_OP='DELETE' THEN RAISE EXCEPTION 'qualified preparation history is immutable'; END IF; + IF TG_OP='INSERT' THEN + IF NEW.state<>'PREPARING' OR NEW.committed_at IS NOT NULL OR NEW.aborted_at IS NOT NULL + OR NEW.coverage_retired_at IS NOT NULL OR NEW.bindings_revision<>0 OR NOT mst2_metadata_scope_matches(NEW.primary_scope) + OR NEW.manifest_digest IS DISTINCT FROM sha256(NEW.canonical_plan) + OR NEW.bindings_digest IS DISTINCT FROM sha256(NEW.canonical_bindings) THEN + RAISE EXCEPTION 'qualified prepare needs its actual fixed primary plan and bindings'; + END IF; + IF NEW.plan_kind='ROOTED' THEN PERFORM mst2_metadata_rooted_manifest(NEW); + ELSIF NEW.node_count=0 OR octet_length(NEW.canonical_bindings)<60 THEN + RAISE EXCEPTION 'cold preparation requires its nonempty full closure'; + END IF; + RETURN NEW; + END IF; + IF (to_jsonb(NEW)-ARRAY['state','committed_at','aborted_at','coverage_retired_at','bindings_revision']) IS DISTINCT FROM + (to_jsonb(OLD)-ARRAY['state','committed_at','aborted_at','coverage_retired_at','bindings_revision']) THEN + RAISE EXCEPTION 'qualified preparation complete identity is immutable'; + END IF; + IF NEW.bindings_revision<>OLD.bindings_revision AND (OLD.plan_kind<>'ROOTED' OR OLD.state<>'PREPARING' + OR NEW.state<>'PREPARING' OR NEW.bindings_revision<>OLD.bindings_revision+1) THEN + RAISE EXCEPTION 'rooted membership revision must advance its exact active preparation'; + END IF; + IF (OLD.state IN ('COMMITTED','ABORTED') AND NEW.state IS DISTINCT FROM OLD.state) + OR (OLD.committed_at IS NOT NULL AND NEW.committed_at IS DISTINCT FROM OLD.committed_at) + OR (OLD.aborted_at IS NOT NULL AND NEW.aborted_at IS DISTINCT FROM OLD.aborted_at) + OR (OLD.coverage_retired_at IS NOT NULL AND NEW.coverage_retired_at IS DISTINCT FROM OLD.coverage_retired_at) THEN + RAISE EXCEPTION 'qualified terminal receipt cannot be revived or rewritten'; + END IF; + IF OLD.state='PREPARING' AND NEW.state='COMMITTED' THEN + IF NEW.plan_kind='ROOTED' THEN + PERFORM mst2_metadata_rooted_finalize_proof(NEW.prepare_id); + ELSE + IF (SELECT count(*) FROM mst2_metadata_prepare_page WHERE prepare_id=NEW.prepare_id)<>NEW.node_count + OR (SELECT coalesce(sum(expected_size),0) FROM mst2_metadata_prepare_page WHERE prepare_id=NEW.prepare_id)<>NEW.total_bytes + OR NOT EXISTS(SELECT 1 FROM mst2_metadata_prepare_page WHERE prepare_id=NEW.prepare_id AND page_id=NEW.metadata_root) + OR EXISTS(SELECT 1 FROM mst2_metadata_prepare_page m + LEFT JOIN mst2_metadata_current c USING(page_id,generation) + LEFT JOIN mst2_metadata_lifetime l USING(page_id,generation) + LEFT JOIN mst2_metadata_payload b USING(page_id,generation) + LEFT JOIN mst2_metadata_graph_node n USING(page_id,generation) + LEFT JOIN mst2_metadata_graph_root r ON r.prepare_id=m.prepare_id AND r.page_id=m.page_id AND r.generation=m.generation + WHERE m.prepare_id=NEW.prepare_id AND (c.page_id IS NULL OR l.page_id IS NULL OR b.page_id IS NULL OR n.page_id IS NULL + OR l.state NOT IN ('RESERVED','LIVE') OR n.state<>'LIVE' OR l.metadata_codec<>NEW.metadata_codec + OR b.metadata_codec<>NEW.metadata_codec OR n.metadata_codec<>NEW.metadata_codec + OR l.expected_size<>m.expected_size OR b.byte_size<>m.expected_size OR n.bytes<>m.expected_size + OR r.storage_seal IS DISTINCT FROM NEW.storage_seal + OR n.incoming_refs<>(SELECT count(*) FROM mst2_metadata_graph_edge WHERE child_page=m.page_id AND child_generation=m.generation))) + OR (SELECT count(*) FROM mst2_metadata_prepare_page m JOIN mst2_metadata_graph_edge e + ON e.parent_page=m.page_id AND e.parent_generation=m.generation WHERE m.prepare_id=NEW.prepare_id)<>NEW.edge_count THEN + RAISE EXCEPTION 'qualified COMMITTED transition lacks its complete exact graph payload and root coverage'; + END IF; + END IF; + END IF; + IF NEW.state='ABORTED' THEN + IF EXISTS(SELECT 1 FROM mst2_metadata_graph_root WHERE prepare_id=NEW.prepare_id) + OR EXISTS(SELECT 1 FROM mst2_qualified_session_incarnation WHERE prepare_id=NEW.prepare_id) THEN + RAISE EXCEPTION 'qualified terminal transition still has exact coverage'; + END IF; + ELSIF NEW.coverage_retired_at IS NOT NULL AND OLD.coverage_retired_at IS NULL THEN + IF NEW.plan_kind<>'ROOTED' THEN + IF EXISTS(SELECT 1 FROM mst2_metadata_graph_root WHERE prepare_id=NEW.prepare_id) + OR EXISTS(SELECT 1 FROM mst2_qualified_session_incarnation WHERE prepare_id=NEW.prepare_id) THEN + RAISE EXCEPTION 'cold terminal transition still has exact coverage'; + END IF; + ELSIF NEW.state<>'COMMITTED' OR EXISTS(SELECT 1 FROM mst2_metadata_graph_root WHERE prepare_id=NEW.prepare_id) + OR NOT (mst2_metadata_session_covers_prepare(NEW.prepare_id) + OR mst2_metadata_orphan_prepare_eligible(NEW.prepare_id)) THEN + RAISE EXCEPTION 'rooted coverage retirement needs its definitive independent ready session root'; + END IF; + END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_metadata_prepare_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_metadata_prepare + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_prepare_guard(); +CREATE FUNCTION mst2_metadata_generation_mapping_guard() RETURNS trigger LANGUAGE plpgsql +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + IF TG_OP<>'INSERT' THEN RAISE EXCEPTION 'metadata generation mappings are immutable'; END IF; + IF NOT EXISTS(SELECT 1 FROM mst2_metadata_prepare q JOIN mst2_metadata_current c ON c.page_id=NEW.page_id AND c.generation=NEW.generation + JOIN mst2_metadata_lifetime l USING(page_id,generation) WHERE q.prepare_id=NEW.prepare_id AND q.state='PREPARING' + AND q.graph_domain='qualified-v1' AND q.storage_seal IS NOT NULL AND mst2_metadata_scope_matches(q.primary_scope) + AND l.state IN ('RESERVED','LIVE') AND l.expected_size=NEW.expected_size AND l.metadata_codec=q.metadata_codec) THEN + RAISE EXCEPTION 'qualified mapping cannot cross domain/current/state fence'; + END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_metadata_generation_mapping_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_metadata_prepare_page + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_generation_mapping_guard(); +CREATE FUNCTION mst2_metadata_payload_fenced() RETURNS trigger LANGUAGE plpgsql +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + IF TG_OP<>'INSERT' THEN RAISE EXCEPTION 'qualified payload mutation and collector are closed'; END IF; + IF NOT EXISTS(SELECT 1 FROM mst2_metadata_current c JOIN mst2_metadata_lifetime l USING(page_id,generation) + JOIN mst2_metadata_prepare_page m USING(page_id,generation) JOIN mst2_metadata_prepare q USING(prepare_id) + WHERE c.page_id=NEW.page_id AND c.generation=NEW.generation AND l.state IN ('RESERVED','LIVE') + AND l.metadata_codec=NEW.metadata_codec AND l.expected_size=NEW.byte_size AND q.state='PREPARING' + AND q.graph_domain='qualified-v1' AND mst2_metadata_scope_matches(q.primary_scope)) THEN + RAISE EXCEPTION 'qualified payload INSERT crossed its exact active generation'; + END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_metadata_payload_fenced BEFORE INSERT OR UPDATE OR DELETE ON mst2_metadata_payload + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_payload_fenced(); +CREATE FUNCTION mst2_metadata_graph_node_guard() RETURNS trigger LANGUAGE plpgsql +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE l mst2_metadata_lifetime%ROWTYPE; +BEGIN + IF TG_OP='DELETE' THEN RAISE EXCEPTION 'qualified graph collection is closed'; END IF; + SELECT l0.* INTO l FROM mst2_metadata_current c JOIN mst2_metadata_lifetime l0 USING(page_id,generation) + WHERE c.page_id=NEW.page_id AND c.generation=NEW.generation; + IF NOT FOUND OR l.graph_domain<>'qualified-v1' OR l.metadata_codec<>NEW.metadata_codec OR l.expected_size<>NEW.bytes + OR NEW.state<>'LIVE' OR l.state NOT IN ('RESERVED','LIVE') + OR NEW.incoming_refs<>(SELECT count(*) FROM mst2_metadata_graph_edge WHERE child_page=NEW.page_id AND child_generation=NEW.generation) THEN + RAISE EXCEPTION 'qualified node does not match exact lifetime and actual counters'; + END IF; + IF TG_OP='INSERT' AND NOT EXISTS(SELECT 1 FROM mst2_metadata_prepare_page m JOIN mst2_metadata_prepare q USING(prepare_id) + WHERE m.page_id=NEW.page_id AND m.generation=NEW.generation AND q.state='PREPARING') THEN + RAISE EXCEPTION 'qualified node creation needs active fixed preparation'; + END IF; + IF NOT EXISTS(SELECT 1 FROM mst2_metadata_page_certificate c WHERE c.page_id=NEW.page_id AND c.generation=NEW.generation + AND c.certificate_digest=NEW.certificate_digest AND c.metadata_codec=NEW.metadata_codec AND c.byte_size=NEW.bytes) THEN + RAISE EXCEPTION 'qualified graph node lacks its exact independently canonical certificate'; + END IF; + IF TG_OP='UPDATE' AND (to_jsonb(NEW)-'incoming_refs') IS DISTINCT FROM (to_jsonb(OLD)-'incoming_refs') THEN + RAISE EXCEPTION 'qualified node identity is immutable'; + END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_metadata_graph_node_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_metadata_graph_node + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_graph_node_guard(); +CREATE FUNCTION mst2_metadata_graph_root_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + IF TG_OP='UPDATE' THEN RAISE EXCEPTION 'qualified roots cannot be retargeted'; END IF; + IF TG_OP='DELETE' THEN + IF EXISTS(SELECT 1 FROM mst2_qualified_session_incarnation WHERE prepare_id=OLD.prepare_id) THEN + RAISE EXCEPTION 'qualified root still has historical incarnation coverage'; + END IF; + RETURN OLD; + END IF; + IF NOT EXISTS(SELECT 1 FROM mst2_metadata_prepare q JOIN mst2_metadata_current c ON c.page_id=NEW.page_id AND c.generation=NEW.generation + JOIN mst2_metadata_lifetime l USING(page_id,generation) JOIN mst2_metadata_graph_node n USING(page_id,generation) + WHERE q.prepare_id=NEW.prepare_id AND q.storage_seal=NEW.storage_seal AND q.state='PREPARING' + AND q.graph_domain='qualified-v1' AND l.graph_domain='qualified-v1' AND l.state IN ('RESERVED','LIVE') + AND n.state='LIVE' AND mst2_metadata_scope_matches(q.primary_scope)) THEN + RAISE EXCEPTION 'qualified root requires its active exact sealed preparation'; + END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_metadata_graph_root_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_metadata_graph_root + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_graph_root_guard(); + +CREATE FUNCTION mst2_metadata_graph_edge_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE op mst2_metadata_gc_op%ROWTYPE; +BEGIN + IF TG_OP='UPDATE' THEN RAISE EXCEPTION 'qualified edge identity is immutable'; END IF; + IF TG_OP='DELETE' THEN RAISE EXCEPTION 'qualified edge collection is closed'; END IF; + IF NOT EXISTS(SELECT 1 FROM mst2_metadata_verified_ref ref + JOIN mst2_metadata_page_certificate parent ON parent.page_id=ref.parent_page AND parent.generation=ref.parent_generation + JOIN mst2_metadata_page_certificate child ON child.page_id=ref.child_page AND child.generation=ref.child_generation + WHERE ref.parent_page=NEW.parent_page AND ref.parent_generation=NEW.parent_generation + AND ref.child_page=NEW.child_page AND ref.child_generation=NEW.child_generation + AND ref.child_certificate_digest=child.certificate_digest AND parent.rank>child.rank) THEN + RAISE EXCEPTION 'qualified edge differs from its exact canonical reference or strict certified rank'; + END IF; + IF (SELECT count(*) FROM mst2_metadata_graph_node n JOIN mst2_metadata_current c USING(page_id,generation) + JOIN mst2_metadata_lifetime l USING(page_id,generation) + WHERE ((n.page_id=NEW.parent_page AND n.generation=NEW.parent_generation) + OR (n.page_id=NEW.child_page AND n.generation=NEW.child_generation)) + AND n.state='LIVE' AND l.state IN ('RESERVED','LIVE') AND l.graph_domain='qualified-v1')<>2 + OR NOT EXISTS(SELECT 1 FROM mst2_metadata_prepare_page a JOIN mst2_metadata_prepare q USING(prepare_id) + WHERE a.page_id=NEW.parent_page AND a.generation=NEW.parent_generation AND q.state='PREPARING' + AND q.graph_domain='qualified-v1' AND mst2_metadata_scope_matches(q.primary_scope) + AND (EXISTS(SELECT 1 FROM mst2_metadata_prepare_page b WHERE b.prepare_id=a.prepare_id + AND b.page_id=NEW.child_page AND b.generation=NEW.child_generation) + OR EXISTS(SELECT 1 FROM mst2_metadata_prepare_reuse_root r JOIN mst2_metadata_root_anchor anchor + ON anchor.prepare_id=r.prepare_id AND anchor.anchor_kind='REUSE' AND anchor.owner_key=r.prepare_id + AND anchor.root_page=r.root_page AND anchor.root_generation=r.root_generation + WHERE r.prepare_id=a.prepare_id AND r.root_page=NEW.child_page AND r.root_generation=NEW.child_generation))) THEN + RAISE EXCEPTION 'edge creation requires active exact qualified endpoints'; + END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_metadata_graph_edge_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_metadata_graph_edge + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_graph_edge_guard(); + +CREATE FUNCTION mst2_metadata_edges_added() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + IF (SELECT count(*) FROM added_edges)>16384 THEN RAISE EXCEPTION 'qualified edge batch exceeds limit'; END IF; + IF NOT EXISTS(SELECT 1 FROM added_edges) THEN RETURN NULL; END IF; + IF EXISTS(SELECT 1 FROM mst2_metadata_graph_node n JOIN ( + SELECT child_page,child_generation,count(*) AS delta FROM added_edges GROUP BY child_page,child_generation + ) d ON d.child_page=n.page_id AND d.child_generation=n.generation + WHERE n.incoming_refs+d.delta<>(SELECT count(*) FROM mst2_metadata_graph_edge x WHERE x.child_page=n.page_id AND x.child_generation=n.generation)) THEN + RAISE EXCEPTION 'qualified insertion found preexisting counter drift'; + END IF; + UPDATE mst2_metadata_graph_node n SET incoming_refs=n.incoming_refs+d.delta FROM ( + SELECT child_page,child_generation,count(*) AS delta FROM added_edges GROUP BY child_page,child_generation + ) d WHERE n.page_id=d.child_page AND n.generation=d.child_generation; + RETURN NULL; +END $$; +CREATE TRIGGER mst2_metadata_edges_added AFTER INSERT ON mst2_metadata_graph_edge + REFERENCING NEW TABLE AS added_edges FOR EACH STATEMENT EXECUTE FUNCTION mst2_metadata_edges_added(); + +DO $$ DECLARE t text; BEGIN + FOREACH t IN ARRAY ARRAY['mst2_metadata_storage_scope','mst2_metadata_family_identity', + 'mst2_metadata_lifetime','mst2_metadata_current','mst2_metadata_payload','mst2_metadata_prepare', + 'mst2_metadata_prepare_page','mst2_metadata_graph_node','mst2_metadata_graph_edge','mst2_metadata_graph_root', + 'mst2_metadata_gc_op','mst2_qualified_session_incarnation','mst2_qualified_lease_binding', + 'mst2_metadata_page_certificate','mst2_metadata_verified_ref','mst2_metadata_source_root_attestation', + 'mst2_metadata_prepare_reuse_root','mst2_metadata_reuse_index','mst2_metadata_reader_operation','mst2_metadata_root_anchor', + 'mst2_metadata_source_entry_reference','mst2_metadata_scope_source_reference'] LOOP + EXECUTE format('CREATE TRIGGER mst2_00_family_barrier BEFORE INSERT OR UPDATE OR DELETE ON %I FOR EACH STATEMENT EXECUTE FUNCTION mst2_metadata_dml_barrier()',t); + EXECUTE format('CREATE TRIGGER mst2_metadata_truncate_guard BEFORE TRUNCATE ON %I FOR EACH STATEMENT EXECUTE FUNCTION mst2_metadata_immutable()',t); + END LOOP; + FOREACH t IN ARRAY ARRAY['mst2_metadata_storage_scope','mst2_metadata_family_identity'] LOOP + EXECUTE format('CREATE TRIGGER mst2_metadata_identity_immutable BEFORE INSERT OR UPDATE OR DELETE ON %I FOR EACH STATEMENT EXECUTE FUNCTION mst2_metadata_immutable()',t); + END LOOP; + FOREACH t IN ARRAY ARRAY['mst2_metadata_gc_op','mst2_qualified_session_incarnation','mst2_qualified_lease_binding', + 'mst2_metadata_reader_operation'] LOOP + EXECUTE format('CREATE TRIGGER mst2_01_family_closed BEFORE INSERT OR UPDATE OR DELETE ON %I FOR EACH STATEMENT EXECUTE FUNCTION mst2_metadata_closed()',t); + END LOOP; +END $$; + +$SERVING_SQL$ +$GC_SQL$ diff --git a/src/jupiter/storage/qualified_metadata_family_tests.rs b/src/jupiter/storage/qualified_metadata_family_tests.rs new file mode 100644 index 00000000..54106a11 --- /dev/null +++ b/src/jupiter/storage/qualified_metadata_family_tests.rs @@ -0,0 +1,855 @@ +use std::{sync::Arc, time::Duration}; + +use mst2_codec::metapage::{Entry, EntryKind, Page, page_id}; +use sea_orm::Database; +use sea_orm_migration::MigratorTrait; + +use super::*; +use crate::{ + ceres::snapshot::retention_dag::{MetadataDagBuilder, MetadataDagLimits}, + jupiter::{ + migration::Migrator, + tests::{TestSchemaGuard, test_db_config}, + }, +}; + +#[path = "qualified_metadata_canonical_tests.rs"] +mod canonical_tests; +#[path = "qualified_metadata_gc_tests.rs"] +mod gc_tests; +#[path = "qualified_metadata_source_revision_tests.rs"] +mod source_revision_tests; + +fn prepared() -> PreparedNativeMetadataRetention { + let entries = [Entry::file(EntryKind::Regular, b"file", 3, [42; 32])]; + let child = Page::build(&entries).unwrap(); + let root_entries = [ + Entry::dir(b"one", page_id(&child)), + Entry::dir(b"two", page_id(&child)), + ]; + let root = Page::build(&root_entries).unwrap(); + let mut builder = MetadataDagBuilder::new(MetadataDagLimits::default()); + builder.add_directory(&child, &entries).unwrap(); + builder.add_directory(&root, &root_entries).unwrap(); + PreparedNativeMetadataRetention::test_installation( + Arc::new(builder.finish(page_id(&root)).unwrap()), + "/", + ) +} + +async fn fixture() -> ( + DbConfig, + DatabaseConnection, + VerifiedQualifiedNamespace, + DatabaseConnection, + TestSchemaGuard, +) { + let temp = tempfile::tempdir().unwrap(); + let (mut config, guard) = test_db_config(temp.path()).await; + config.max_connection = 2; + config.min_connection = 1; + // Exercise the actual production bootstrap, not a test-only DDL shortcut. + let core = super::super::init::database_connection(&config) + .await + .unwrap(); + let namespace = provision_or_verify_rooted_qualified_family(&core) + .await + .unwrap(); + let mut q_config = config.clone(); + q_config.db_url = pool_url(&config.db_url, &namespace).unwrap(); + q_config.max_connection = 1; + let q = postgres_connection(&q_config).await.unwrap(); + (config, core, namespace, q, guard) +} + +async fn count(db: &C, sql: &str) -> i64 { + db.query_one_raw(Statement::from_string(DbBackend::Postgres, sql)) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap() +} +async fn write( + writer: &ShadowQualifiedMetadataWriter, + operation: &str, + pages: &PreparedNativeMetadataRetention, +) -> GenerationMetadataReceipt { + let intent = writer.begin_intent(operation, pages).await.unwrap(); + for batch in pages.dag().payloads().chunks(64) { + writer.install_pages(&intent, batch).await.unwrap(); + } + writer.finalize(&intent).await.unwrap() +} + +#[tokio::test] +async fn physical_q_shadow_writer_is_distinct_and_restart_reuses_its_exact_catalog() { + let (config, core, namespace, q, _guard) = fixture().await; + let pages = prepared(); + let generic=super::super::native_metadata_install::generations::PostgresMetadataGenerationRepository::new(core.clone()).await.unwrap(); + let g = generic.begin_intent("same-pages-g", &pages).await.unwrap(); + for batch in pages.dag().payloads().chunks(64) { + generic.install_pages(&g, batch).await.unwrap(); + } + let g_receipt = generic.finalize(&g).await.unwrap(); + let assembly_dir = tempfile::tempdir().unwrap(); + let mut app_config = crate::config::testing::isolated_config(assembly_dir.path()); + app_config.database = config.clone(); + let storage = super::super::Storage::new_with_connection( + Arc::new(app_config), + Arc::new(core.clone()), + super::super::object_storage::mock_object_storage(), + ) + .await + .unwrap(); + let writer = storage.shadow_qualified_metadata_writer().await.unwrap(); + let clone = storage.clone(); + assert!(std::ptr::eq( + writer, + clone.shadow_qualified_metadata_writer().await.unwrap() + )); + let q_receipt = write(writer, "same-pages-q", &pages).await; + assert_eq!(g_receipt.metadata_root(), q_receipt.metadata_root()); + assert_ne!(g.prepare_id(), q_receipt.intent().prepare_id()); + assert_eq!( + count(&core, "SELECT count(*) FROM mst2_metadata_payload").await, + 2 + ); + assert_eq!( + count( + &q, + "SELECT count(*) FROM mst2_metadata_payload WHERE generation=1" + ) + .await, + 2 + ); + assert_eq!(count(&q,"SELECT count(*) FROM mst2_metadata_prepare WHERE graph_domain='qualified-v1' AND state='COMMITTED'").await,1); + assert_eq!( + count( + &q, + "SELECT count(*) FROM mst2_qualified_session_incarnation" + ) + .await, + 0 + ); + assert_eq!( + count(&q, "SELECT count(*) FROM mst2_qualified_lease_binding").await, + 0 + ); + assert_eq!(count(&q,"SELECT count(*) FROM pg_catalog.pg_class WHERE relnamespace=(SELECT oid FROM pg_catalog.pg_namespace WHERE nspname=current_schema()) AND relkind='r'").await,22); + assert_eq!(count(&q,"SELECT count(*) FROM pg_catalog.pg_class WHERE relnamespace=(SELECT oid FROM pg_catalog.pg_namespace WHERE nspname=current_schema()) AND relname IN ('mst2_retention_node','mst2_retention_edge','mst2_snapshot_context','mst2_snapshot_lease','mega_refs','git_repo','seaql_migrations')").await,0); + let q_scope = q + .query_one_raw(Statement::from_string( + DbBackend::Postgres, + "SELECT primary_scope,storage_seal FROM mst2_metadata_prepare", + )) + .await + .unwrap() + .unwrap(); + let g_scope = core + .query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT primary_scope,storage_seal FROM mst2_metadata_prepare WHERE prepare_id=$1", + [g.prepare_id().into()], + )) + .await + .unwrap() + .unwrap(); + assert_ne!( + q_scope.try_get::>("", "primary_scope").unwrap(), + g_scope.try_get::>("", "primary_scope").unwrap() + ); + assert_ne!( + q_scope.try_get::>("", "storage_seal").unwrap(), + g_scope.try_get::>("", "storage_seal").unwrap() + ); + let restarted = super::super::init::database_connection(&config) + .await + .unwrap(); + assert_eq!( + provision_or_verify_rooted_qualified_family(&restarted) + .await + .unwrap(), + namespace + ); + let fresh = ShadowQualifiedMetadataWriter::open(&restarted, &config) + .await + .unwrap(); + assert_eq!(fresh.finalize(q_receipt.intent()).await.unwrap(), q_receipt); + assert_eq!( + catalog(&q, namespace.core_oid, namespace.schema_oid) + .await + .unwrap(), + namespace.catalog_fingerprint + ); +} + +#[tokio::test] +async fn q_catalog_trigger_function_and_index_tamper_are_never_refreshed() { + for tamper in [ + "ALTER TABLE {q}.mst2_metadata_payload DISABLE TRIGGER mst2_metadata_payload_fenced", + "CREATE OR REPLACE FUNCTION {q}.mst2_metadata_has_generic_overlap(p bytea) RETURNS boolean LANGUAGE sql AS 'SELECT true'", + "DROP INDEX {q}.idx_mst2_metadata_graph_edge_child", + "CREATE RULE forged_skip AS ON INSERT TO {q}.mst2_metadata_prepare DO INSTEAD NOTHING", + ] { + let (config, core, namespace, q, _guard) = fixture().await; + let writer = ShadowQualifiedMetadataWriter::open(&core, &config) + .await + .unwrap(); + core.execute_unprepared(&tamper.replace("{q}", &identifier(&namespace.schema))) + .await + .unwrap(); + assert!(writer.begin_intent("tampered", &prepared()).await.is_err()); + let error = provision_or_verify_rooted_qualified_family(&core) + .await + .unwrap_err(); + assert!( + error.to_string().contains("catalog fingerprint changed"), + "{error}" + ); + assert_eq!( + count(&q, "SELECT count(*) FROM mst2_metadata_prepare").await, + 0 + ); + assert_eq!( + count(&core, "SELECT count(*) FROM mst2_metadata_namespace").await, + 2 + ); + let stored:Vec=core.query_one_raw(Statement::from_string(DbBackend::Postgres, + "SELECT catalog_fingerprint FROM mst2_metadata_namespace WHERE graph_domain='qualified-v1'")) + .await.unwrap().unwrap().try_get_by_index(0).unwrap(); + assert_eq!(stored, namespace.catalog_fingerprint); + } +} + +#[tokio::test] +async fn q_schema_rename_fails_closed_without_reprovisioning_or_g_fallback() { + let (config, core, namespace, _q, _guard) = fixture().await; + let writer = ShadowQualifiedMetadataWriter::open(&core, &config) + .await + .unwrap(); + let renamed = format!("{}_renamed", namespace.schema); + core.execute_unprepared(&format!( + "ALTER SCHEMA {} RENAME TO {}", + identifier(&namespace.schema), + identifier(&renamed) + )) + .await + .unwrap(); + assert!(writer.begin_intent("renamed", &prepared()).await.is_err()); + assert!( + provision_or_verify_rooted_qualified_family(&core) + .await + .is_err() + ); + assert_eq!( + count(&core, "SELECT count(*) FROM mst2_metadata_namespace").await, + 2 + ); + assert_eq!( + count(&core, "SELECT count(*) FROM mst2_metadata_prepare").await, + 0 + ); + assert_eq!( + count( + &core, + &format!( + "SELECT count(*) FROM {}.mst2_metadata_prepare", + identifier(&renamed) + ) + ) + .await, + 0 + ); +} + +#[tokio::test] +async fn q_actual_schema_rr_temp_shadow_and_admitted_ledgers_keep_their_guards() { + let (_config, core, namespace, q, _guard) = fixture().await; + let bad = core + .execute_unprepared(&format!( + "INSERT INTO {}.mst2_metadata_gc_op SELECT * FROM {}.mst2_metadata_gc_op WHERE false", + identifier(&namespace.schema), + identifier(&namespace.schema) + )) + .await + .unwrap_err(); + assert!(bad.to_string().contains("captured primary family"), "{bad}"); + let rr = q + .begin_with_config(Some(IsolationLevel::RepeatableRead), None) + .await + .unwrap(); + let error = rr + .execute_unprepared( + "INSERT INTO mst2_metadata_gc_op SELECT * FROM mst2_metadata_gc_op WHERE false", + ) + .await + .unwrap_err(); + assert!(error.to_string().contains("READ COMMITTED"), "{error}"); + rr.rollback().await.unwrap(); + for statement in [ + "SELECT mst2_metadata_gc_apply(gen_random_uuid())", + "SELECT mst2_metadata_gc_finish(gen_random_uuid())", + ] { + let error = q.execute_unprepared(statement).await.unwrap_err(); + assert!( + error.to_string().contains("query returned no rows"), + "{error}" + ); + } + assert_eq!( + count(&q, "SELECT count(*) FROM mst2_metadata_gc_op").await, + 0 + ); + for table in [ + "mst2_qualified_session_incarnation", + "mst2_qualified_lease_binding", + ] { + assert!( + q.execute_unprepared(&format!("INSERT INTO {table} DEFAULT VALUES")) + .await + .is_err() + ); + assert_eq!(count(&q, &format!("SELECT count(*) FROM {table}")).await, 0); + } + q.execute_unprepared("CREATE TEMP TABLE mst2_metadata_prepare (prepare_id text)") + .await + .unwrap(); + let repository = + PostgresQualifiedMetadataRepository::registered_shadow(q.clone(), namespace.clone()) + .await + .unwrap(); + repository + .begin_intent("temp-shadow", &prepared()) + .await + .unwrap(); + assert_eq!( + count( + &q, + &format!( + "SELECT count(*) FROM {}.mst2_metadata_prepare", + identifier(&namespace.schema) + ) + ) + .await, + 1 + ); + assert_eq!( + count(&q, "SELECT count(*) FROM pg_temp.mst2_metadata_prepare").await, + 0 + ); + let duplicate=core.execute_unprepared("INSERT INTO mst2_metadata_namespace SELECT * FROM mst2_metadata_namespace WHERE graph_domain='qualified-v1'").await.unwrap_err(); + assert!( + duplicate + .to_string() + .contains("exact admitted rooted family scope"), + "{duplicate}" + ); + assert_eq!( + count(&core, "SELECT count(*) FROM mst2_metadata_namespace").await, + 2 + ); + let pk:String=q.query_one_raw(Statement::from_string(DbBackend::Postgres, + "SELECT conkey::text FROM pg_catalog.pg_constraint WHERE conrelid='mst2_qualified_session_incarnation'::regclass AND contype='p'")) + .await.unwrap().unwrap().try_get_by_index(0).unwrap(); + assert_eq!( + pk, "{1,2}", + "same SID can have future distinct incarnations" + ); +} + +#[tokio::test] +async fn provisioning_locks_candidate_and_registered_union_in_uuid_order() { + let temp = tempfile::tempdir().unwrap(); + let (config, _guard) = test_db_config(temp.path()).await; + let core = Database::connect(config.db_url.clone()).await.unwrap(); + Migrator::up(&core, None).await.unwrap(); + let other = Database::connect(config.db_url).await.unwrap(); + let captured = captured_core(&core).await.unwrap(); + let candidate = "00000000-0000-4000-8000-000000000001"; + let schema = format!("mst2q_{}", candidate.replace('-', "")); + let held = core.begin().await.unwrap(); + held.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT pg_catalog.pg_advisory_xact_lock(1296717362,pg_catalog.hashtext($1))", + [schema.clone().into()], + )) + .await + .unwrap(); + let (core_name, q_name) = (captured.0.clone(), schema.clone()); + let waiter = tokio::spawn(async move { + let txn = other.begin().await.unwrap(); + txn.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + format!( + "SELECT {}.mst2_route_family_candidate_enter($1,$2::uuid,$3)", + identifier(&core_name) + ), + [core_name.into(), candidate.into(), q_name.into()], + )) + .await + .unwrap(); + txn.rollback().await.unwrap(); + }); + let deadline = tokio::time::Instant::now() + Duration::from_secs(4); + loop { + let row=held.query_one_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "SELECT EXISTS(SELECT 1 FROM pg_catalog.pg_locks w WHERE w.locktype='advisory' AND NOT w.granted + AND w.classid=1296717362::oid AND w.objid=pg_catalog.hashtext($1)::oid AND w.objsubid=2 + AND w.database=(SELECT oid FROM pg_catalog.pg_database WHERE datname=current_database()) + AND NOT EXISTS(SELECT 1 FROM pg_catalog.pg_locks g WHERE g.pid=w.pid AND g.locktype='advisory' + AND g.classid=1296717362::oid AND g.objid=pg_catalog.hashtext($2)::oid AND g.objsubid=2 AND g.granted)) AS sorted_wait", + [schema.clone().into(),captured.0.clone().into()])).await.unwrap().unwrap(); + if row.try_get::("", "sorted_wait").unwrap() { + break; + } + assert!( + tokio::time::Instant::now() < deadline, + "candidate must be locked before G when its UUID sorts first" + ); + tokio::task::yield_now().await; + } + held.rollback().await.unwrap(); + waiter.await.unwrap(); + assert_eq!( + count(&core, "SELECT count(*) FROM mst2_metadata_namespace").await, + 1 + ); +} + +#[tokio::test] +async fn q_incomplete_commit_null_binding_and_unsealed_plan_are_rejected() { + let (config, core, namespace, q, _guard) = fixture().await; + let writer = ShadowQualifiedMetadataWriter::open(&core, &config) + .await + .unwrap(); + let intent = writer + .begin_intent("fixed-incomplete", &prepared()) + .await + .unwrap(); + let error=q.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "UPDATE mst2_metadata_prepare SET state='COMMITTED',committed_at=clock_timestamp() WHERE prepare_id=$1",[intent.prepare_id().into()])).await.unwrap_err(); + assert!( + error.to_string().contains("complete exact graph payload"), + "{error}" + ); + for mutation in [ + "graph_domain='generic-v1'", + "canonical_plan=canonical_plan||decode('ff','hex')", + "storage_seal=NULL", + "primary_scope=NULL", + ] { + let error = q + .execute_unprepared(&format!("UPDATE mst2_metadata_prepare SET {mutation}")) + .await + .unwrap_err(); + assert!( + error.to_string().contains("complete identity is immutable"), + "{error}" + ); + } + for batch in prepared().dag().payloads().chunks(64) { + writer.install_pages(&intent, batch).await.unwrap(); + } + writer.finalize(&intent).await.unwrap(); + assert_eq!( + count( + &q, + "SELECT count(*) FROM mst2_metadata_prepare WHERE state='COMMITTED'" + ) + .await, + 1 + ); + assert_eq!( + catalog(&q, namespace.core_oid, namespace.schema_oid) + .await + .unwrap(), + namespace.catalog_fingerprint + ); +} + +#[tokio::test] +async fn first_registration_cannot_self_sign_a_minimal_or_half_installed_family() { + for complete in [false, true] { + let temp = tempfile::tempdir().unwrap(); + let (config, _guard) = test_db_config(temp.path()).await; + let core = Database::connect(config.db_url).await.unwrap(); + Migrator::up(&core, None).await.unwrap(); + let captured = captured_core(&core).await.unwrap(); + let candidate = uuid::Uuid::new_v4().to_string(); + let schema = format!("mst2q_{}", candidate.replace('-', "")); + let storage = uuid::Uuid::new_v4().to_string(); + let txn = core.begin().await.unwrap(); + txn.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + format!( + "SELECT {}.mst2_route_family_candidate_enter($1,$2::uuid,$3)", + identifier(&captured.0) + ), + [ + captured.0.clone().into(), + candidate.clone().into(), + schema.clone().into(), + ], + )) + .await + .unwrap(); + txn.execute_unprepared(&format!("CREATE SCHEMA {}", identifier(&schema))) + .await + .unwrap(); + let oid: i64 = txn + .query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT oid::bigint FROM pg_catalog.pg_namespace WHERE nspname=$1", + [schema.clone().into()], + )) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap(); + if complete { + txn.execute_unprepared(&render_family( + &captured.0, + captured.1, + &schema, + oid, + &candidate, + &storage, + )) + .await + .unwrap(); + } else { + txn.execute_unprepared(&format!( + "CREATE TABLE {q}.mst2_metadata_storage_scope(singleton integer,storage_uuid text); + INSERT INTO {q}.mst2_metadata_storage_scope VALUES(1,{storage}); + CREATE TABLE {q}.mst2_metadata_family_identity(singleton integer,namespace_uuid uuid,storage_uuid text, + core_schema_oid oid,metadata_schema_oid oid,family_identity text,implementation_fingerprint bytea); + INSERT INTO {q}.mst2_metadata_family_identity VALUES(1,{candidate}::uuid,{storage},{core_oid},{oid},'{family}',decode('{implementation}','hex'))", + q=identifier(&schema),storage=literal(&storage),candidate=literal(&candidate),core_oid=captured.1, + family=FAMILY,implementation=hex::encode(implementation_fingerprint()))).await.unwrap(); + } + let fingerprint = if complete { + vec![0xff; 32] + } else { + catalog(&txn, captured.1, oid).await.unwrap() + }; + txn.execute_unprepared(&format!( + "SET LOCAL search_path={},pg_catalog,pg_temp", + identifier(&captured.0) + )) + .await + .unwrap(); + let error=txn.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres,format!( + "INSERT INTO {c}.mst2_metadata_namespace(singleton,namespace_uuid,core_schema,core_schema_oid,database_name, + database_oid,storage_uuid,server_address,server_port,mono_lock_key2,metadata_schema,metadata_schema_oid, + family_identity,graph_domain,admission_state,collector_state,metadata_storage_uuid,implementation_fingerprint,catalog_fingerprint) + SELECT NULL,$1::uuid,g.core_schema,g.core_schema_oid,g.database_name,g.database_oid,g.storage_uuid,g.server_address, + g.server_port,g.mono_lock_key2,$2,$3::bigint::oid,$4,'qualified-v1','ROOTED_Q_ADMITTED','ENABLED',$5,$6,$7 + FROM {c}.mst2_metadata_namespace g WHERE singleton=1",c=identifier(&captured.0)), + [candidate.into(),schema.clone().into(),oid.into(),FAMILY.into(),storage.into(),implementation_fingerprint().into(),fingerprint.into()])).await.unwrap_err(); + let expected = if complete { + "fingerprint disagrees" + } else { + "trusted complete physical family shape" + }; + assert!(error.to_string().contains(expected), "{error}"); + txn.rollback().await.unwrap(); + assert_eq!( + count(&core, "SELECT count(*) FROM mst2_metadata_namespace").await, + 1 + ); + assert_eq!( + count( + &core, + &format!( + "SELECT count(*) FROM pg_catalog.pg_namespace WHERE nspname={}", + literal(&schema) + ) + ) + .await, + 0 + ); + assert_eq!( + count(&core, "SELECT count(*) FROM mst2_qualified_family_policy").await, + 1 + ); + } +} + +#[tokio::test] +async fn simultaneous_production_bootstraps_reuse_one_physical_q_namespace() { + let temp = tempfile::tempdir().unwrap(); + let (config, _guard) = test_db_config(temp.path()).await; + let core = postgres_connection(&config).await.unwrap(); + Migrator::up(&core, None).await.unwrap(); + let (left, right) = tokio::join!( + super::super::init::database_connection(&config), + super::super::init::database_connection(&config) + ); + let left = left.unwrap(); + let right = right.unwrap(); + let a = provision_or_verify_rooted_qualified_family(&left) + .await + .unwrap(); + let b = provision_or_verify_rooted_qualified_family(&right) + .await + .unwrap(); + assert_eq!(a, b); + assert_eq!( + count(&core, "SELECT count(*) FROM mst2_metadata_namespace").await, + 2 + ); + assert_eq!( + count( + &core, + &format!( + "SELECT count(*) FROM pg_catalog.pg_namespace WHERE nspname={}", + literal(&a.schema) + ) + ) + .await, + 1 + ); +} + +#[tokio::test] +async fn trusted_shape_policy_and_core_validation_functions_are_immutable_or_fail_closed() { + let (config, core, namespace, _q, _guard) = fixture().await; + for sql in [ + "UPDATE mst2_qualified_family_policy SET expected_shape=decode(repeat('ff',32),'hex')", + "DELETE FROM mst2_qualified_family_policy", + "TRUNCATE mst2_qualified_family_policy", + ] { + let error = core.execute_unprepared(sql).await.unwrap_err(); + assert!(error.to_string().contains("immutable"), "{error}"); + } + let writer = ShadowQualifiedMetadataWriter::open(&core, &config) + .await + .unwrap(); + core.execute_unprepared(&format!( + "CREATE OR REPLACE FUNCTION {}.mst2_route_family_shape(q_oid oid,n_uuid uuid,s_uuid text) + RETURNS bytea LANGUAGE sql AS 'SELECT decode(repeat(''ff'',32),''hex'')'", + identifier(&namespace.core_schema) + )) + .await + .unwrap(); + let error = provision_or_verify_rooted_qualified_family(&core) + .await + .unwrap_err(); + assert!( + error + .to_string() + .contains("trusted core authority catalog changed"), + "{error}" + ); + assert!( + writer + .begin_intent("bad-core-validator", &prepared()) + .await + .is_err() + ); + assert_eq!( + count(&core, "SELECT count(*) FROM mst2_metadata_namespace").await, + 2 + ); +} + +#[tokio::test] +async fn random_q_templates_have_one_trusted_shape_and_changed_composite_signature_is_rejected() { + let temp = tempfile::tempdir().unwrap(); + let (config, _guard) = test_db_config(temp.path()).await; + let core = postgres_connection(&config).await.unwrap(); + Migrator::up(&core, None).await.unwrap(); + let captured = captured_core(&core).await.unwrap(); + let expected: Vec = core + .query_one_raw(Statement::from_string( + DbBackend::Postgres, + "SELECT expected_shape FROM mst2_qualified_family_policy WHERE singleton=1", + )) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap(); + let authority = catalog(&core, captured.1, 0).await.unwrap(); + let mut schemas = Vec::new(); + for _ in 0..2 { + let namespace = uuid::Uuid::new_v4().to_string(); + let schema = format!("mst2q_{}", namespace.replace('-', "")); + let storage = uuid::Uuid::new_v4().to_string(); + let txn = core.begin().await.unwrap(); + txn.execute_unprepared(&format!("CREATE SCHEMA {}", identifier(&schema))) + .await + .unwrap(); + let oid = count( + &txn, + &format!( + "SELECT oid::bigint FROM pg_catalog.pg_namespace WHERE nspname={}", + literal(&schema), + ), + ) + .await; + txn.execute_unprepared(&render_family( + &captured.0, + captured.1, + &schema, + oid, + &namespace, + &storage, + )) + .await + .unwrap(); + let shape = || { + Statement::from_sql_and_values( + DbBackend::Postgres, + format!( + "SELECT {}.mst2_route_family_shape($1::bigint::oid,$2::uuid,$3)", + identifier(&captured.0), + ), + [oid.into(), namespace.clone().into(), storage.clone().into()], + ) + }; + let actual: Vec = txn + .query_one_raw(shape()) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap(); + assert_eq!( + actual, expected, + "random template must retain its trusted structural signature" + ); + assert_eq!( + catalog_with_exemption(&txn, captured.1, 0, oid) + .await + .unwrap(), + authority + ); + assert_ne!( + catalog(&txn, captured.1, 0).await.unwrap(), + authority, + "an unregistered candidate is never implicitly exempted" + ); + txn.execute_unprepared(&format!( + "DROP FUNCTION {q}.mst2_metadata_rooted_manifest({q}.mst2_metadata_prepare); + CREATE FUNCTION {q}.mst2_metadata_rooted_manifest(q text) RETURNS jsonb + LANGUAGE sql IMMUTABLE STRICT AS 'SELECT to_jsonb(q)'", + q = identifier(&schema), + )) + .await + .unwrap(); + let altered: Vec = txn + .query_one_raw(shape()) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap(); + assert_ne!( + altered, expected, + "normalization must preserve the composite argument type" + ); + txn.rollback().await.unwrap(); + schemas.push(schema); + } + assert_ne!(schemas[0], schemas[1]); + assert_eq!(catalog(&core, captured.1, 0).await.unwrap(), authority); + assert!(registered(&core, &captured).await.unwrap().is_none()); +} + +#[tokio::test] +async fn fresh_q_core_authority_exempts_only_exact_q_ri_and_restart_keeps_full_catalog_binding() { + for tamper in ["fk", "ri", "external"] { + let (config, core, namespace, _q, _guard) = fixture().await; + let captured = (namespace.core_schema.clone(), namespace.core_oid); + let authority: Vec = core + .query_one_raw(Statement::from_string( + DbBackend::Postgres, + "SELECT authority_catalog FROM mst2_qualified_family_policy WHERE singleton=1", + )) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap(); + assert_eq!( + catalog_with_exemption(&core, captured.1, 0, namespace.schema_oid) + .await + .unwrap(), + authority + ); + assert_ne!(catalog(&core, captured.1, 0).await.unwrap(), authority); + for _ in 0..2 { + assert_eq!( + registered(&core, &captured).await.unwrap(), + Some(namespace.clone()) + ); + } + let restarted = postgres_connection(&config).await.unwrap(); + assert_eq!( + provision_or_verify_rooted_qualified_family(&restarted) + .await + .unwrap(), + namespace + ); + let txn = core.begin().await.unwrap(); + if tamper == "external" { + let external = format!("hostile_{}", uuid::Uuid::new_v4().simple()); + txn.execute_unprepared(&format!( + "CREATE SCHEMA {external}; CREATE TABLE {external}.forged_route(snapshot_id text,namespace_uuid uuid, + FOREIGN KEY(snapshot_id,namespace_uuid) REFERENCES {c}.mst2_snapshot_storage_route(snapshot_id,namespace_uuid))", + external=identifier(&external),c=identifier(&captured.0), + )).await.unwrap(); + assert_ne!( + catalog_with_exemption(&txn, captured.1, 0, namespace.schema_oid) + .await + .unwrap(), + authority, + "another namespace's internal RI triggers remain part of core authority" + ); + } else { + let row=txn.query_one_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "SELECT fk.conname,t.tgname FROM pg_catalog.pg_constraint fk + JOIN pg_catalog.pg_trigger t ON t.tgconstraint=fk.oid AND t.tgrelid=fk.confrelid AND t.tgisinternal + WHERE fk.connamespace=$1::bigint::oid AND fk.confrelid=$2::regclass AND fk.contype='f' LIMIT 1", + [namespace.schema_oid.into(),format!("{}.mst2_snapshot_storage_route",identifier(&captured.0)).into()])) + .await.unwrap().unwrap(); + let conname: String = row.try_get("", "conname").unwrap(); + let trigger: String = row.try_get("", "tgname").unwrap(); + let sql = if tamper == "fk" { + format!( + "ALTER TABLE {}.mst2_qualified_session_incarnation DROP CONSTRAINT {}", + identifier(&namespace.schema), + identifier(&conname) + ) + } else { + format!( + "ALTER TABLE {}.mst2_snapshot_storage_route DISABLE TRIGGER {}", + identifier(&captured.0), + identifier(&trigger) + ) + }; + txn.execute_unprepared(&sql).await.unwrap(); + assert_eq!( + catalog_with_exemption(&txn, captured.1, 0, namespace.schema_oid) + .await + .unwrap(), + authority + ); + assert_ne!( + catalog(&txn, captured.1, namespace.schema_oid) + .await + .unwrap(), + namespace.catalog_fingerprint + ); + } + assert!( + registered(&txn, &captured).await.is_err(), + "{tamper} tampering cannot refresh admission" + ); + txn.rollback().await.unwrap(); + assert_eq!(registered(&core, &captured).await.unwrap(), Some(namespace)); + } +} diff --git a/src/jupiter/storage/qualified_metadata_gc.rs b/src/jupiter/storage/qualified_metadata_gc.rs new file mode 100644 index 00000000..394d9335 --- /dev/null +++ b/src/jupiter/storage/qualified_metadata_gc.rs @@ -0,0 +1,540 @@ +//! Bounded rooted maintenance. Local retry state is a hint, never GC authority. + +use super::*; + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +enum GcPhase { + Claim, + Apply, +} + +#[derive(Debug, Clone, PartialEq, Eq)] +struct GcSeal { + operation_id: String, + page_id: [u8; 32], + generation: i64, + primary_scope: Vec, + metadata_codec: i16, + expected_size: i32, + graph_present: bool, + had_payload: bool, + certificate_digest: Option<[u8; 32]>, +} + +#[derive(Debug, Clone)] +struct RetryOperation { + seal: GcSeal, + phase: GcPhase, +} + +#[derive(Debug, Default)] +pub(super) struct RootedMaintenanceState { + retry: Option, + cursor: Option<([u8; 32], i64)>, +} + +#[derive(Debug, Default, Clone, serde::Serialize)] +pub(crate) struct RootedMaintenanceWork { + pub collector_enabled: bool, + pub examined: u16, + pub lifetimes_examined: u16, + pub owners_examined: u16, + pub pending_replayed: u16, + pub uncertain_receipts_observed: u16, + pub absent_claims_observed: u16, + pub gc_claimed: u16, + pub gc_applied: u16, + pub payload_pages_removed: u16, + pub payload_bytes_removed: u64, + pub readers_expired: u16, + pub leases_expired: u16, + pub prepares_aborted: u16, + pub handovers_retired: u16, + pub orphans_retired: u16, +} + +struct GcRecord { + seal: GcSeal, + applied: bool, +} + +fn uncertain(operation: &RetryOperation) -> SnapshotError { + SnapshotError::new( + crate::ceres::snapshot::error::SnapshotErrorCode::TemporaryUnavailable, + format!( + "qualified GC outcome unknown for operation {} generation {} phase {:?}; retry the sealed operation on the same primary", + operation.seal.operation_id, operation.seal.generation, operation.phase + ), + ) +} + +fn gc_record(row: &QueryResult) -> Result { + let operation_id: String = row.try_get("", "operation_id").map_err(internal)?; + let id = uuid::Uuid::parse_str(&operation_id).map_err(internal)?; + let certificate: Option> = row.try_get("", "certificate_digest").map_err(internal)?; + let seal = GcSeal { + operation_id, + page_id: digest_column(row, "page_id")?, + generation: row.try_get("", "generation").map_err(internal)?, + primary_scope: row.try_get("", "primary_scope").map_err(internal)?, + metadata_codec: row.try_get("", "metadata_codec").map_err(internal)?, + expected_size: row.try_get("", "expected_size").map_err(internal)?, + graph_present: row.try_get("", "graph_present").map_err(internal)?, + had_payload: row.try_get("", "had_payload").map_err(internal)?, + certificate_digest: certificate + .map(|bytes| bytes.as_slice().try_into().map_err(internal)) + .transpose()?, + }; + if id.get_version_num() != 4 + || id.to_string() != seal.operation_id + || seal.generation <= 0 + || seal.metadata_codec != 1 + || seal.expected_size <= 0 + || seal.graph_present != seal.certificate_digest.is_some() + || row + .try_get::("", "graph_domain") + .map_err(internal)? + != "qualified-v1" + { + return Err(integrity( + "qualified GC immutable operation profile is invalid", + )); + } + let state: String = row.try_get("", "state").map_err(internal)?; + let applied = match state.as_str() { + "PENDING" => false, + "APPLIED" => true, + _ => return Err(integrity("qualified GC operation has an unknown state")), + }; + Ok(GcRecord { seal, applied }) +} + +async fn load_gc( + db: &C, + operation: &str, +) -> Result, SnapshotError> { + db.query_one_raw(sql( + "SELECT operation_id::text,page_id,generation,primary_scope,graph_domain,metadata_codec, + expected_size,graph_present,had_payload,certificate_digest,state + FROM mst2_metadata_gc_op WHERE operation_id=$1::uuid", + [operation.into()], + )) + .await + .map_err(internal)? + .as_ref() + .map(gc_record) + .transpose() +} + +impl RootedQualifiedMetadataRepository { + pub(crate) fn start_maintenance(repository: &std::sync::Arc) { + let repository = std::sync::Arc::downgrade(repository); + tokio::spawn(async move { + let mut interval = tokio::time::interval(std::time::Duration::from_secs(30)); + interval.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Skip); + interval.tick().await; + loop { + interval.tick().await; + let Some(repository) = repository.upgrade() else { + break; + }; + match tokio::time::timeout( + std::time::Duration::from_secs(15), + repository.maintenance_tick(64), + ) + .await + { + Ok(Ok(_)) => {} + Ok(Err(error)) => { + tracing::warn!(code=?error.code, "rooted metadata maintenance failed; durable roots and sealed operations remain recoverable") + } + Err(_) => tracing::warn!( + "rooted metadata maintenance reached its deadline; retrying durable state on the next tick" + ), + } + } + }); + } + + async fn collector_enabled(&self) -> Result { + let txn = self.transaction().await?; + let result = async { + let row = txn + .query_one_raw(sql("SELECT mst2_metadata_gc_enabled() AS enabled", [])) + .await + .map_err(internal)? + .ok_or_else(|| integrity("qualified collector policy is missing"))?; + row.try_get("", "enabled").map_err(internal) + } + .await; + let _ = txn.rollback().await; + result + } + + fn require_gc_seal(&self, seal: &GcSeal) -> Result<(), SnapshotError> { + if seal.primary_scope != self.primary_scope { + return Err(integrity( + "qualified GC seal belongs to another captured primary", + )); + } + Ok(()) + } + + async fn prove_gc_record( + &self, + txn: &DatabaseTransaction, + record: &GcRecord, + ) -> Result<(), SnapshotError> { + self.require_gc_seal(&record.seal)?; + let seal = &record.seal; + txn.query_one_raw(sql( + "SELECT mst2_metadata_gc_proof($1,$2,$3,$4,$5,$6)", + [ + seal.page_id.to_vec().into(), + seal.generation.into(), + if record.applied { "APPLIED" } else { "PENDING" }.into(), + (!record.applied && seal.graph_present).into(), + (!record.applied && seal.had_payload).into(), + seal.certificate_digest.map(|value| value.to_vec()).into(), + ], + )) + .await + .map_err(internal)?; + Ok(()) + } + + async fn observe_gc_retry( + &self, + operation: &RetryOperation, + ) -> Result, SnapshotError> { + self.require_gc_seal(&operation.seal)?; + // transaction() acquires the same-primary route completion barrier. + // A failed barrier or read never proves that an uncertain claim is absent. + let txn = self.transaction().await.map_err(|_| uncertain(operation))?; + let result = async { + let record = load_gc(&txn, &operation.seal.operation_id).await?; + if let Some(record) = &record { + if record.seal != operation.seal { + return Err(integrity( + "qualified GC retry retargeted its immutable sealed identity", + )); + } + self.prove_gc_record(&txn, record).await?; + } + Ok(record) + } + .await; + txn.rollback().await.map_err(|_| uncertain(operation))?; + result.map_err(|error: SnapshotError| { + if error.code == crate::ceres::snapshot::error::SnapshotErrorCode::Internal { + uncertain(operation) + } else { + error + } + }) + } + + async fn claim_gc_candidate( + &self, + candidate: GcSeal, + state: &mut RootedMaintenanceState, + ) -> Result { + self.require_gc_seal(&candidate)?; + let txn = self.transaction().await?; + let result = async { + txn.query_one_raw(sql( + "SELECT mst2_metadata_gc_claim($1,$2,$3,$4::uuid)", + [ + candidate.page_id.to_vec().into(), + candidate.generation.into(), + candidate.primary_scope.clone().into(), + candidate.operation_id.clone().into(), + ], + )) + .await + .map_err(internal)?; + let record = load_gc(&txn, &candidate.operation_id) + .await? + .ok_or_else(|| integrity("qualified claimed GC operation disappeared"))?; + if record.seal != candidate || record.applied { + return Err(integrity( + "qualified claim differs from its exact captured candidate", + )); + } + self.prove_gc_record(&txn, &record).await?; + Ok(record.seal) + } + .await; + let seal = match result { + Ok(seal) => seal, + Err(error) => { + let _ = txn.rollback().await; + return Err(error); + } + }; + let retry = RetryOperation { + seal: seal.clone(), + phase: GcPhase::Claim, + }; + // Retain the hint before COMMIT so task cancellation cannot lose an + // already transmitted commit whose durable outcome is still unknown. + state.retry = Some(retry.clone()); + txn.commit().await.map_err(|_| uncertain(&retry))?; + state.retry = None; + Ok(seal) + } + + async fn apply_gc_seal( + &self, + seal: &GcSeal, + state: &mut RootedMaintenanceState, + ) -> Result { + self.require_gc_seal(seal)?; + let txn = self.transaction().await?; + let result = async { + let record = load_gc(&txn, &seal.operation_id) + .await? + .ok_or_else(|| integrity("qualified committed GC operation is missing"))?; + if &record.seal != seal { + return Err(integrity( + "qualified apply retargeted its exact immutable GC operation", + )); + } + self.prove_gc_record(&txn, &record).await?; + if record.applied { + return Ok(false); + } + txn.query_one_raw(sql( + "SELECT mst2_metadata_gc_apply($1::uuid)", + [seal.operation_id.clone().into()], + )) + .await + .map_err(internal)?; + let applied = load_gc(&txn, &seal.operation_id) + .await? + .ok_or_else(|| integrity("qualified GC apply lost its durable receipt"))?; + if &applied.seal != seal || !applied.applied { + return Err(integrity( + "qualified GC apply did not preserve its sealed APPLIED identity", + )); + } + self.prove_gc_record(&txn, &applied).await?; + Ok(true) + } + .await; + let changed = match result { + Ok(changed) => changed, + Err(error) => { + let _ = txn.rollback().await; + return Err(error); + } + }; + let retry = RetryOperation { + seal: seal.clone(), + phase: GcPhase::Apply, + }; + state.retry = Some(retry.clone()); + txn.commit().await.map_err(|_| uncertain(&retry))?; + state.retry = None; + Ok(changed) + } + + async fn pending_gc_seals(&self, limit: u16) -> Result, SnapshotError> { + let txn = self.transaction().await?; + let result = async { + let rows = txn.query_all_raw(sql( + "SELECT operation_id::text,page_id,generation,primary_scope,graph_domain,metadata_codec, + expected_size,graph_present,had_payload,certificate_digest,state + FROM mst2_metadata_gc_op WHERE state='PENDING' ORDER BY created_at,operation_id LIMIT $1", + [i64::from(limit).into()], + )).await.map_err(internal)?; + rows.iter().map(|row| { + let record=gc_record(row)?; + self.require_gc_seal(&record.seal)?; + Ok(record.seal) + }).collect() + }.await; + let _ = txn.rollback().await; + result + } + + async fn cleanup_gc_owners( + &self, + limit: u16, + work: &mut RootedMaintenanceWork, + ) -> Result<(), SnapshotError> { + let txn = self.transaction().await?; + let result = async { + txn.query_one_raw(sql( + "SELECT * FROM mst2_metadata_gc_owner_cleanup($1::integer)", + [i32::from(limit).into()], + )) + .await + .map_err(internal)? + .ok_or_else(|| integrity("qualified owner cleanup lost its bounded counters")) + } + .await; + let row = sessions::finish(txn, result).await?; + let count = |name: &str| -> Result { + u16::try_from(row.try_get::("", name).map_err(internal)?).map_err(internal) + }; + let examined = count("examined")?; + let readers = count("readers_expired")?; + let leases = count("leases_expired")?; + let prepares = count("prepares_aborted")?; + let handovers = count("handovers_retired")?; + let orphans = count("orphans_retired")?; + if examined > limit || readers + leases + prepares + handovers + orphans > examined { + return Err(integrity( + "qualified owner cleanup counters exceed their shared budget", + )); + } + work.examined += examined; + work.owners_examined += examined; + work.readers_expired += readers; + work.leases_expired += leases; + work.prepares_aborted += prepares; + work.handovers_retired += handovers; + work.orphans_retired += orphans; + Ok(()) + } + + async fn gc_candidates( + &self, + after: Option<([u8; 32], i64)>, + limit: u16, + ) -> Result, SnapshotError> { + let (page, generation) = after.map(|(p, g)| (p.to_vec(), g)).unwrap_or_default(); + let txn = self.transaction().await?; + let result = async { + let rows = txn.query_all_raw(sql( + "SELECT l.page_id,l.generation,l.metadata_codec,l.expected_size, + n.page_id IS NOT NULL AS graph_present,n.certificate_digest, + EXISTS(SELECT 1 FROM mst2_metadata_payload b WHERE b.page_id=l.page_id AND b.generation=l.generation) AS had_payload, + NOT EXISTS(SELECT 1 FROM mst2_metadata_root_anchor a WHERE a.root_page=l.page_id AND a.root_generation=l.generation) + AND NOT EXISTS(SELECT 1 FROM mst2_metadata_graph_root r WHERE r.page_id=l.page_id AND r.generation=l.generation) + AND NOT EXISTS(SELECT 1 FROM mst2_metadata_graph_edge e WHERE e.child_page=l.page_id AND e.child_generation=l.generation) + AND NOT EXISTS(SELECT 1 FROM mst2_metadata_prepare_page m JOIN mst2_metadata_prepare q USING(prepare_id) + WHERE m.page_id=l.page_id AND m.generation=l.generation + AND (q.state='PREPARING' OR q.state='COMMITTED' AND q.coverage_retired_at IS NULL)) + AND NOT EXISTS(SELECT 1 FROM mst2_metadata_prepare_reuse_root r JOIN mst2_metadata_prepare q USING(prepare_id) + WHERE r.root_page=l.page_id AND r.root_generation=l.generation + AND (q.state='PREPARING' OR q.state='COMMITTED' AND q.coverage_retired_at IS NULL)) + AND (l.state='LIVE' OR EXISTS(SELECT 1 FROM mst2_metadata_prepare_page m JOIN mst2_metadata_prepare q USING(prepare_id) + WHERE m.page_id=l.page_id AND m.generation=l.generation AND q.state='ABORTED' AND q.storage_seal IS NOT NULL) + AND NOT EXISTS(SELECT 1 FROM mst2_metadata_prepare_page m JOIN mst2_metadata_prepare q USING(prepare_id) + WHERE m.page_id=l.page_id AND m.generation=l.generation AND q.state<>'ABORTED')) AS eligible + FROM mst2_metadata_lifetime l JOIN mst2_metadata_current c USING(page_id,generation) + LEFT JOIN mst2_metadata_graph_node n USING(page_id,generation) + WHERE l.state IN ('RESERVED','LIVE') AND (l.page_id,l.generation)>($1,$2) + ORDER BY l.page_id,l.generation LIMIT $3", + [page.into(),generation.into(),i64::from(limit).into()], + )).await.map_err(internal)?; + rows.iter().map(|row| { + let certificate:Option>=row.try_get("","certificate_digest").map_err(internal)?; + Ok((GcSeal { + operation_id:uuid::Uuid::new_v4().to_string(), + page_id:digest_column(row,"page_id")?, + generation:row.try_get("","generation").map_err(internal)?, + primary_scope:self.primary_scope.clone(), + metadata_codec:row.try_get("","metadata_codec").map_err(internal)?, + expected_size:row.try_get("","expected_size").map_err(internal)?, + graph_present:row.try_get("","graph_present").map_err(internal)?, + had_payload:row.try_get("","had_payload").map_err(internal)?, + certificate_digest:certificate.map(|bytes|bytes.as_slice().try_into().map_err(internal)).transpose()?, + },row.try_get("","eligible").map_err(internal)?)) + }).collect() + }.await; + let _ = txn.rollback().await; + result + } + + /// A shared owner/page work budget; SQL proof and capacity scans are separately bounded. + pub(crate) async fn maintenance_tick( + &self, + limit: u16, + ) -> Result { + if !(1..=64).contains(&limit) { + return Err(integrity("qualified maintenance limit must be 1..=64")); + } + let mut state = self.maintenance_state.lock().await; + let mut work = RootedMaintenanceWork { + collector_enabled: self.collector_enabled().await?, + ..Default::default() + }; + if let Some(retry) = state.retry.clone() { + work.examined += 1; + work.lifetimes_examined += 1; + match self.observe_gc_retry(&retry).await? { + Some(record) if record.applied => { + state.retry = None; + work.uncertain_receipts_observed += 1; + } + Some(record) if work.collector_enabled => { + let changed = self.apply_gc_seal(&record.seal, &mut state).await?; + work.pending_replayed += 1; + work.record_applied(&record.seal, changed)?; + } + Some(_) => {} + None if retry.phase == GcPhase::Claim => { + state.retry = None; + work.absent_claims_observed += 1; + } + None => { + return Err(integrity( + "qualified committed GC apply lost its sealed operation history", + )); + } + } + } + if work.collector_enabled && work.examined < limit { + for seal in self.pending_gc_seals(limit - work.examined).await? { + work.examined += 1; + work.lifetimes_examined += 1; + let changed = self.apply_gc_seal(&seal, &mut state).await?; + work.pending_replayed += 1; + work.record_applied(&seal, changed)?; + } + } + if work.examined < limit { + let remaining = limit - work.examined; + let owner_budget = if work.collector_enabled { + remaining.div_ceil(2) + } else { + remaining + }; + self.cleanup_gc_owners(owner_budget, &mut work).await?; + } + if work.collector_enabled && work.examined < limit { + let remaining = limit - work.examined; + let candidates = self.gc_candidates(state.cursor, remaining).await?; + let reached_end = candidates.len() < usize::from(remaining); + for (candidate, eligible) in candidates { + work.examined += 1; + work.lifetimes_examined += 1; + state.cursor = Some((candidate.page_id, candidate.generation)); + if eligible { + let seal = self.claim_gc_candidate(candidate, &mut state).await?; + work.gc_claimed += 1; + let changed = self.apply_gc_seal(&seal, &mut state).await?; + work.record_applied(&seal, changed)?; + } + } + if reached_end { + state.cursor = None; + } + } + Ok(work) + } +} + +impl RootedMaintenanceWork { + fn record_applied(&mut self, seal: &GcSeal, changed: bool) -> Result<(), SnapshotError> { + self.gc_applied += 1; + if changed && seal.had_payload { + self.payload_pages_removed += 1; + self.payload_bytes_removed += u64::try_from(seal.expected_size).map_err(internal)?; + } + Ok(()) + } +} diff --git a/src/jupiter/storage/qualified_metadata_gc.sql b/src/jupiter/storage/qualified_metadata_gc.sql new file mode 100644 index 00000000..13b41998 --- /dev/null +++ b/src/jupiter/storage/qualified_metadata_gc.sql @@ -0,0 +1,574 @@ +-- Historical certificates and source/session records do not own live graph rows. +ALTER TABLE mst2_metadata_gc_op ADD COLUMN certificate_digest bytea, + ADD CONSTRAINT mst2_gc_certificate_binding CHECK( + (graph_present AND octet_length(certificate_digest)=32 OR NOT graph_present AND certificate_digest IS NULL) IS TRUE); +ALTER TABLE mst2_metadata_prepare ADD COLUMN orphan_expires_at timestamptz NOT NULL; +CREATE INDEX mst2_metadata_prepare_orphan_scan ON mst2_metadata_prepare(orphan_expires_at,prepare_id) + WHERE state='PREPARING' OR state='COMMITTED' AND coverage_retired_at IS NULL; +CREATE INDEX mst2_metadata_current_active_scan ON mst2_metadata_lifetime(page_id,generation) + WHERE state IN ('RESERVED','LIVE'); + +CREATE FUNCTION mst2_metadata_prepare_expiry_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + IF TG_OP='INSERT' THEN NEW.created_at:=clock_timestamp(); NEW.orphan_expires_at:=NEW.created_at+interval '3600 seconds'; + ELSIF NEW.created_at IS DISTINCT FROM OLD.created_at OR NEW.orphan_expires_at IS DISTINCT FROM OLD.orphan_expires_at THEN + RAISE EXCEPTION 'qualified preparation expiry is immutable database evidence'; END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_metadata_prepare_00_expiry_guard BEFORE INSERT OR UPDATE ON mst2_metadata_prepare + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_prepare_expiry_guard(); + +CREATE FUNCTION mst2_metadata_orphan_prepare_eligible(pid text) RETURNS boolean LANGUAGE sql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ + SELECT EXISTS(SELECT 1 FROM mst2_metadata_prepare q WHERE q.prepare_id=$1 AND q.state='COMMITTED' + AND q.plan_kind='ROOTED' AND q.graph_domain='qualified-v1' AND mst2_metadata_scope_matches(q.primary_scope) + AND q.orphan_expires_at=q.created_at+interval '3600 seconds' AND q.orphan_expires_at<=clock_timestamp() + AND NOT EXISTS(SELECT 1 FROM mst2_qualified_session_incarnation s WHERE s.state='READY' + AND (s.prepare_id=q.prepare_id OR s.namespace_uuid='$NAMESPACE_UUID$'::uuid AND s.metadata_root=q.metadata_root + AND s.source_profile=convert_to($CORE_SCHEMA$.mst2_route_profile(q.source_domain,q.tagged_root_tree_oid,q.scope, + q.schema_version,q.metadata_codec,q.materialization_policy,q.fs_semantics,q.access_projection, + q.verification_revision,q.projection_revision)::text,'UTF8') + AND s.root_generation IN (SELECT generation FROM mst2_metadata_prepare_page m + WHERE m.prepare_id=q.prepare_id AND m.page_id=q.metadata_root UNION ALL + SELECT root_generation FROM mst2_metadata_prepare_reuse_root r WHERE r.prepare_id=q.prepare_id AND r.root_page=q.metadata_root))) + AND NOT EXISTS(SELECT 1 FROM mst2_qualified_lease_binding l WHERE l.prepare_id=q.prepare_id AND l.state='ACTIVE') + AND NOT EXISTS(SELECT 1 FROM mst2_metadata_reader_operation r JOIN mst2_qualified_lease_binding l USING(lease_id) + WHERE l.prepare_id=q.prepare_id AND r.state='ACTIVE')) +$$; +CREATE FUNCTION mst2_metadata_orphan_prepare_retired(pid text) RETURNS boolean LANGUAGE sql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ + SELECT mst2_metadata_orphan_prepare_eligible($1) AND EXISTS(SELECT 1 FROM mst2_metadata_prepare q + WHERE q.prepare_id=$1 AND q.coverage_retired_at>=q.orphan_expires_at) + AND NOT EXISTS(SELECT 1 FROM mst2_metadata_root_anchor WHERE prepare_id=$1 AND anchor_kind IN ('PREPARE','REUSE')) +$$; + +CREATE FUNCTION mst2_metadata_gc_enabled() RETURNS boolean LANGUAGE sql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ + SELECT NOT pg_is_in_recovery() AND current_setting('transaction_isolation')='read committed' + AND EXISTS(SELECT 1 FROM $CORE_SCHEMA$.mst2_metadata_namespace n WHERE n.namespace_uuid='$NAMESPACE_UUID$'::uuid + AND n.graph_domain='qualified-v1' AND n.metadata_schema=$Q_LITERAL$ AND n.metadata_schema_oid='$Q_OID$'::oid + AND n.core_schema_oid=$CORE_OID$ AND n.metadata_storage_uuid='$STORAGE_UUID$' + AND n.admission_state='ROOTED_Q_ADMITTED' AND n.collector_state='ENABLED' + AND n.implementation_fingerprint=decode('$IMPLEMENTATION_SHA$','hex') + AND n.catalog_fingerprint=$CORE_SCHEMA$.mst2_route_family_catalog($CORE_OID$,'$Q_OID$'::oid)) +$$; +CREATE FUNCTION mst2_metadata_gc_enter() RETURNS void LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + PERFORM $CORE_SCHEMA$.mst2_route_enter($CORE_LITERAL$); + IF NOT mst2_metadata_gc_enabled() THEN RAISE EXCEPTION 'qualified collector is not independently admitted'; END IF; +END $$; + +CREATE FUNCTION mst2_metadata_assert_uncovered(p bytea,g bigint) RETURNS void LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + IF EXISTS(SELECT 1 FROM mst2_metadata_root_anchor WHERE root_page=p AND root_generation=g) + OR EXISTS(SELECT 1 FROM mst2_metadata_graph_root WHERE page_id=p AND generation=g) + OR EXISTS(SELECT 1 FROM mst2_metadata_graph_edge WHERE child_page=p AND child_generation=g) + OR EXISTS(SELECT 1 FROM mst2_metadata_prepare_page m JOIN mst2_metadata_prepare q USING(prepare_id) + WHERE m.page_id=p AND m.generation=g AND (q.state='PREPARING' OR q.state='COMMITTED' AND q.coverage_retired_at IS NULL)) + OR EXISTS(SELECT 1 FROM mst2_metadata_prepare_reuse_root r JOIN mst2_metadata_prepare q USING(prepare_id) + WHERE r.root_page=p AND r.root_generation=g AND (q.state='PREPARING' OR q.state='COMMITTED' AND q.coverage_retired_at IS NULL)) THEN + RAISE EXCEPTION 'qualified current lifetime still has exact owned coverage or incoming edges'; END IF; +END $$; + +CREATE FUNCTION mst2_metadata_gc_proof(p bytea,g bigint,stage text,graph_present boolean,payload_present boolean,c bytea) +RETURNS void LANGUAGE plpgsql VOLATILE SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE life mst2_metadata_lifetime%ROWTYPE; node mst2_metadata_graph_node%ROWTYPE; + body mst2_metadata_payload%ROWTYPE; certificate mst2_metadata_page_certificate%ROWTYPE; actual boolean; +BEGIN + IF p IS NULL OR octet_length(p)<>32 OR g IS NULL OR g<=0 OR stage IS NULL + OR graph_present IS NULL OR payload_present IS NULL OR graph_present AND c IS NULL THEN + RAISE EXCEPTION 'qualified GC proof has an incomplete exact identity'; END IF; + SELECT * INTO life FROM mst2_metadata_lifetime WHERE page_id=p AND generation=g FOR UPDATE; + IF NOT FOUND OR life.graph_domain<>'qualified-v1' OR life.metadata_codec<>1 OR stage NOT IN ('CLAIM','PENDING','APPLIED') + OR stage='CLAIM' AND life.state NOT IN ('RESERVED','LIVE') OR stage='PENDING' AND life.state<>'DELETING' + OR stage='APPLIED' AND life.state<>'REMOVED' THEN RAISE EXCEPTION 'qualified GC stage is not its exact historical lifetime'; END IF; + IF stage<>'APPLIED' AND NOT EXISTS(SELECT 1 FROM mst2_metadata_current WHERE page_id=p AND generation=g) THEN + RAISE EXCEPTION 'qualified GC does not name its exact current generation'; END IF; + PERFORM mst2_metadata_assert_uncovered(p,g); + SELECT * INTO node FROM mst2_metadata_graph_node WHERE page_id=p AND generation=g FOR UPDATE; + actual:=FOUND; + IF actual IS DISTINCT FROM graph_present OR actual AND (node.incoming_refs<>0 OR node.metadata_codec<>life.metadata_codec + OR node.bytes<>life.expected_size OR node.certificate_digest IS DISTINCT FROM c + OR stage='CLAIM' AND node.state<>'LIVE' OR stage='PENDING' AND node.state<>'DELETING') THEN + RAISE EXCEPTION 'qualified GC graph identity or actual counter changed'; END IF; + SELECT * INTO body FROM mst2_metadata_payload WHERE page_id=p AND generation=g FOR UPDATE; + actual:=FOUND; + IF actual IS DISTINCT FROM payload_present OR actual AND (body.metadata_codec<>life.metadata_codec + OR body.byte_size<>life.expected_size OR octet_length(body.payload)<>body.byte_size + OR sha256(convert_to('mega.mst2.metapage','UTF8')||decode('00','hex')||body.payload)<>p) THEN + RAISE EXCEPTION 'qualified GC durable bytes differ from their exact lifetime and page digest'; END IF; + IF payload_present THEN PERFORM mst2_metadata_decode_local(body.payload); END IF; + IF c IS NOT NULL THEN + SELECT * INTO certificate FROM mst2_metadata_page_certificate WHERE page_id=p AND generation=g AND certificate_digest=c; + IF NOT FOUND OR certificate.namespace_uuid<>'$NAMESPACE_UUID$'::uuid OR certificate.metadata_codec<>life.metadata_codec + OR certificate.byte_size<>life.expected_size OR certificate.proof_revision<>1 + OR (SELECT count(*) FROM (SELECT 1 FROM mst2_metadata_verified_ref WHERE parent_page=p AND parent_generation=g LIMIT 258) refs) + <>jsonb_array_length(certificate.canonical_proof->'references') THEN + RAISE EXCEPTION 'qualified GC lost its immutable canonical certificate and typed reference evidence'; END IF; + ELSIF graph_present OR EXISTS(SELECT 1 FROM mst2_metadata_page_certificate WHERE page_id=p AND generation=g) THEN + RAISE EXCEPTION 'qualified GC cannot omit a certified graph identity'; END IF; + IF graph_present THEN + IF (SELECT count(*) FROM (SELECT 1 FROM mst2_metadata_graph_edge WHERE parent_page=p AND parent_generation=g LIMIT 258) edges)>257 THEN + RAISE EXCEPTION 'qualified canonical node has too many physical outgoing edges'; END IF; + IF payload_present AND (EXISTS((SELECT child_page,child_generation FROM mst2_metadata_verified_ref WHERE parent_page=p AND parent_generation=g) + EXCEPT (SELECT child_page,child_generation FROM mst2_metadata_graph_edge WHERE parent_page=p AND parent_generation=g)) + OR EXISTS((SELECT child_page,child_generation FROM mst2_metadata_graph_edge WHERE parent_page=p AND parent_generation=g) + EXCEPT (SELECT child_page,child_generation FROM mst2_metadata_verified_ref WHERE parent_page=p AND parent_generation=g))) THEN + RAISE EXCEPTION 'qualified GC physical outgoing edges differ from exact typed occurrences'; END IF; + END IF; + IF stage='CLAIM' AND life.state='LIVE' AND NOT (graph_present AND payload_present AND c IS NOT NULL) THEN + RAISE EXCEPTION 'LIVE metadata corruption cannot be collected'; END IF; + IF stage='CLAIM' AND graph_present AND NOT payload_present THEN + RAISE EXCEPTION 'certified RESERVED graph corruption cannot be collected without its durable bytes'; END IF; + IF stage='CLAIM' AND life.state='RESERVED' AND (NOT EXISTS(SELECT 1 FROM mst2_metadata_prepare_page m + JOIN mst2_metadata_prepare q USING(prepare_id) WHERE m.page_id=p AND m.generation=g AND m.expected_size=life.expected_size + AND q.graph_domain='qualified-v1' AND q.state='ABORTED' AND q.storage_seal IS NOT NULL) + OR EXISTS(SELECT 1 FROM mst2_metadata_prepare_page m JOIN mst2_metadata_prepare q USING(prepare_id) + WHERE m.page_id=p AND m.generation=g AND q.state<>'ABORTED')) THEN + RAISE EXCEPTION 'RESERVED metadata requires an explicitly aborted exact origin'; END IF; + IF stage='APPLIED' AND (EXISTS(SELECT 1 FROM mst2_metadata_graph_edge WHERE parent_page=p AND parent_generation=g) + OR EXISTS(SELECT 1 FROM mst2_metadata_reuse_index WHERE root_page=p AND root_generation=g)) THEN + RAISE EXCEPTION 'APPLIED qualified receipt still has old physical edges or reuse hints'; END IF; +END $$; + +CREATE FUNCTION mst2_metadata_gc_op_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE life mst2_metadata_lifetime%ROWTYPE; +BEGIN + IF TG_OP='DELETE' THEN RAISE EXCEPTION 'qualified GC operation history is immutable'; END IF; + PERFORM mst2_metadata_gc_enter(); + IF NOT mst2_metadata_scope_matches(NEW.primary_scope) THEN RAISE EXCEPTION 'qualified GC left its captured primary scope'; END IF; + IF TG_OP='INSERT' THEN + IF NEW.state<>'PENDING' OR NEW.completed_at IS NOT NULL OR NEW.payload_delete_xid IS NOT NULL + OR substr(NEW.operation_id::text,15,1)<>'4' OR substr(NEW.operation_id::text,20,1) NOT IN ('8','9','a','b') THEN + RAISE EXCEPTION 'qualified GC must begin with a fresh UUID and unmodified PENDING proof'; END IF; + PERFORM mst2_metadata_gc_proof(NEW.page_id,NEW.generation,'CLAIM',NEW.graph_present,NEW.had_payload,NEW.certificate_digest); + SELECT * INTO STRICT life FROM mst2_metadata_lifetime WHERE page_id=NEW.page_id AND generation=NEW.generation; + IF ROW(NEW.graph_domain,NEW.metadata_codec,NEW.expected_size) IS DISTINCT FROM + ROW(life.graph_domain,life.metadata_codec,life.expected_size) THEN RAISE EXCEPTION 'qualified GC copied profile is incorrect'; END IF; + NEW.created_at:=clock_timestamp(); + ELSE + IF (to_jsonb(NEW)-ARRAY['state','completed_at','payload_delete_xid']) IS DISTINCT FROM + (to_jsonb(OLD)-ARRAY['state','completed_at','payload_delete_xid']) THEN RAISE EXCEPTION 'qualified GC operation cannot retarget immutable evidence'; END IF; + IF OLD.state='APPLIED' AND NEW IS DISTINCT FROM OLD THEN RAISE EXCEPTION 'qualified GC receipt cannot revive or change'; END IF; + IF NEW.payload_delete_xid IS DISTINCT FROM OLD.payload_delete_xid THEN + IF OLD.state<>'PENDING' OR NEW.state<>'PENDING' OR NOT OLD.had_payload OR OLD.payload_delete_xid IS NOT NULL THEN + RAISE EXCEPTION 'qualified payload deletion has no fresh same-transaction phase'; END IF; + PERFORM mst2_metadata_gc_proof(OLD.page_id,OLD.generation,'PENDING',OLD.graph_present,true,OLD.certificate_digest); + NEW.payload_delete_xid:=txid_current(); + END IF; + IF OLD.state='PENDING' AND NEW.state='APPLIED' THEN + IF OLD.had_payload AND OLD.payload_delete_xid IS DISTINCT FROM txid_current() THEN + RAISE EXCEPTION 'qualified GC completion has no same-transaction payload removal'; END IF; + PERFORM mst2_metadata_gc_proof(OLD.page_id,OLD.generation,'APPLIED',false,false,OLD.certificate_digest); + NEW.completed_at:=clock_timestamp(); + ELSIF NEW.state IS DISTINCT FROM OLD.state OR NEW.completed_at IS DISTINCT FROM OLD.completed_at THEN + RAISE EXCEPTION 'qualified GC state transition is invalid'; END IF; + END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_metadata_gc_op_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_metadata_gc_op + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_gc_op_guard(); + +CREATE FUNCTION mst2_metadata_gc_complete() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE op mst2_metadata_gc_op%ROWTYPE; +BEGIN + SELECT * INTO STRICT op FROM mst2_metadata_gc_op WHERE operation_id=NEW.operation_id; + IF op.state='PENDING' THEN + IF op.payload_delete_xid IS NOT NULL THEN RAISE EXCEPTION 'qualified payload deletion phase cannot escape its atomic apply transaction'; END IF; + PERFORM mst2_metadata_gc_proof(op.page_id,op.generation,'PENDING',op.graph_present,op.had_payload,op.certificate_digest); + ELSE PERFORM mst2_metadata_gc_proof(op.page_id,op.generation,'APPLIED',false,false,op.certificate_digest); END IF; + RETURN NULL; +END $$; +CREATE CONSTRAINT TRIGGER mst2_metadata_gc_complete AFTER INSERT OR UPDATE ON mst2_metadata_gc_op + DEFERRABLE INITIALLY DEFERRED FOR EACH ROW EXECUTE FUNCTION mst2_metadata_gc_complete(); + +CREATE OR REPLACE FUNCTION mst2_metadata_lifetime_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE current_root mst2_metadata_current%ROWTYPE; op mst2_metadata_gc_op%ROWTYPE; previous mst2_metadata_lifetime%ROWTYPE; +BEGIN + IF TG_OP='DELETE' THEN RAISE EXCEPTION 'qualified lifetime watermarks are immutable'; END IF; + IF TG_OP='INSERT' THEN + IF NEW.state<>'RESERVED' THEN RAISE EXCEPTION 'qualified lifetime must begin RESERVED'; END IF; + SELECT * INTO current_root FROM mst2_metadata_current WHERE page_id=NEW.page_id; + IF FOUND THEN + SELECT * INTO previous FROM mst2_metadata_lifetime WHERE page_id=NEW.page_id AND generation=current_root.generation; + SELECT * INTO op FROM mst2_metadata_gc_op WHERE page_id=NEW.page_id AND generation=current_root.generation AND state='APPLIED'; + IF NOT FOUND OR previous.state<>'REMOVED' OR current_root.generation=9223372036854775807 + OR NEW.generation<>current_root.generation+1 OR NEW.metadata_codec<>op.metadata_codec OR NEW.expected_size<>op.expected_size + OR NOT mst2_metadata_scope_matches(op.primary_scope) THEN RAISE EXCEPTION 'qualified reservation needs its exact next-generation APPLIED predecessor'; END IF; + PERFORM mst2_metadata_gc_proof(NEW.page_id,current_root.generation,'APPLIED',false,false,op.certificate_digest); + ELSIF NEW.generation<>1 OR EXISTS(SELECT 1 FROM mst2_metadata_lifetime WHERE page_id=NEW.page_id) + OR EXISTS(SELECT 1 FROM mst2_metadata_payload WHERE page_id=NEW.page_id) + OR EXISTS(SELECT 1 FROM mst2_metadata_graph_node WHERE page_id=NEW.page_id) THEN + RAISE EXCEPTION 'qualified initial reservation cannot adopt history'; END IF; + RETURN NEW; + END IF; + IF (to_jsonb(NEW)-'state') IS DISTINCT FROM (to_jsonb(OLD)-'state') THEN RAISE EXCEPTION 'qualified lifetime identity is immutable'; END IF; + IF NEW.state=OLD.state THEN RETURN NEW; END IF; + IF OLD.state='RESERVED' AND NEW.state='LIVE' THEN + IF NOT EXISTS(SELECT 1 FROM mst2_metadata_graph_node node JOIN mst2_metadata_payload body USING(page_id,generation) + JOIN mst2_metadata_page_certificate certificate USING(page_id,generation) + WHERE node.page_id=OLD.page_id AND node.generation=OLD.generation AND node.state='LIVE' + AND node.certificate_digest=certificate.certificate_digest AND node.metadata_codec=OLD.metadata_codec + AND body.metadata_codec=OLD.metadata_codec AND body.byte_size=OLD.expected_size AND node.bytes=OLD.expected_size) + OR NOT (EXISTS(SELECT 1 FROM mst2_metadata_graph_root root JOIN mst2_metadata_prepare q USING(prepare_id) + WHERE root.page_id=OLD.page_id AND root.generation=OLD.generation AND q.plan_kind='COLD' + AND q.state='COMMITTED' AND root.storage_seal=q.storage_seal AND q.coverage_retired_at IS NULL) + OR EXISTS(SELECT 1 FROM mst2_metadata_prepare_page m JOIN mst2_metadata_prepare q USING(prepare_id) + JOIN mst2_metadata_root_anchor anchor ON anchor.prepare_id=q.prepare_id AND anchor.anchor_kind='PREPARE' + AND anchor.root_page=q.metadata_root AND anchor.owner_key=q.prepare_id + WHERE m.page_id=OLD.page_id AND m.generation=OLD.generation AND q.plan_kind='ROOTED' + AND q.state='COMMITTED' AND q.coverage_retired_at IS NULL)) THEN + RAISE EXCEPTION 'qualified LIVE transition lacks its certified graph and exact owned preparation'; END IF; + RETURN NEW; + END IF; + PERFORM mst2_metadata_gc_enter(); + SELECT * INTO op FROM mst2_metadata_gc_op WHERE page_id=OLD.page_id AND generation=OLD.generation AND state='PENDING'; + IF NOT FOUND OR NOT mst2_metadata_scope_matches(op.primary_scope) THEN RAISE EXCEPTION 'qualified lifecycle transition lacks its exact pending operation'; END IF; + IF OLD.state IN ('RESERVED','LIVE') AND NEW.state='DELETING' THEN + PERFORM mst2_metadata_gc_proof(OLD.page_id,OLD.generation,'CLAIM',op.graph_present,op.had_payload,op.certificate_digest); + ELSIF OLD.state='DELETING' AND NEW.state='REMOVED' THEN + IF op.had_payload AND op.payload_delete_xid IS DISTINCT FROM txid_current() THEN RAISE EXCEPTION 'qualified removal has no atomic payload deletion'; END IF; + PERFORM mst2_metadata_gc_proof(OLD.page_id,OLD.generation,'PENDING',false,false,op.certificate_digest); + IF EXISTS(SELECT 1 FROM mst2_metadata_graph_edge WHERE parent_page=OLD.page_id AND parent_generation=OLD.generation) + OR EXISTS(SELECT 1 FROM mst2_metadata_reuse_index WHERE root_page=OLD.page_id AND root_generation=OLD.generation) THEN + RAISE EXCEPTION 'qualified removal still has old physical references'; END IF; + ELSE RAISE EXCEPTION 'qualified lifecycle transition is invalid'; END IF; + RETURN NEW; +END $$; + +CREATE OR REPLACE FUNCTION mst2_metadata_current_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE op mst2_metadata_gc_op%ROWTYPE; life mst2_metadata_lifetime%ROWTYPE; +BEGIN + IF TG_OP='DELETE' THEN RAISE EXCEPTION 'qualified current watermark cannot be deleted'; END IF; + IF TG_OP='INSERT' THEN + IF NEW.generation<>1 OR EXISTS(SELECT 1 FROM mst2_metadata_lifetime WHERE page_id=NEW.page_id AND generation<>1) + OR EXISTS(SELECT 1 FROM mst2_metadata_payload WHERE page_id=NEW.page_id) + OR EXISTS(SELECT 1 FROM mst2_metadata_graph_node WHERE page_id=NEW.page_id) THEN + RAISE EXCEPTION 'qualified initial current cannot adopt history'; END IF; + RETURN NEW; + END IF; + IF NEW.page_id<>OLD.page_id OR OLD.generation=9223372036854775807 OR NEW.generation<>OLD.generation+1 THEN + RAISE EXCEPTION 'qualified current requires exact next-generation CAS'; END IF; + SELECT * INTO op FROM mst2_metadata_gc_op WHERE page_id=OLD.page_id AND generation=OLD.generation AND state='APPLIED'; + IF NOT FOUND OR NOT mst2_metadata_scope_matches(op.primary_scope) + OR EXISTS(SELECT 1 FROM mst2_metadata_payload WHERE page_id=OLD.page_id) + OR EXISTS(SELECT 1 FROM mst2_metadata_graph_node WHERE page_id=OLD.page_id AND generation=OLD.generation) THEN + RAISE EXCEPTION 'qualified current has no definitive old-generation removal'; END IF; + PERFORM mst2_metadata_gc_proof(OLD.page_id,OLD.generation,'APPLIED',false,false,op.certificate_digest); + SELECT * INTO life FROM mst2_metadata_lifetime WHERE page_id=NEW.page_id AND generation=NEW.generation; + IF NOT FOUND OR life.state<>'RESERVED' OR life.graph_domain<>'qualified-v1' + OR life.metadata_codec<>op.metadata_codec OR life.expected_size<>op.expected_size THEN + RAISE EXCEPTION 'qualified fresh current differs from its immutable page profile'; END IF; + RETURN NEW; +END $$; + +CREATE FUNCTION mst2_metadata_capacity() RETURNS TABLE(resident_pages bigint,resident_bytes bigint) LANGUAGE sql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ + SELECT count(*)::bigint,coalesce(sum(byte_size),0)::bigint FROM (SELECT byte_size FROM mst2_metadata_payload LIMIT 16385) actual +$$; +CREATE FUNCTION mst2_metadata_capacity_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE pages bigint; bytes bigint; actual bigint; +BEGIN + IF TG_TABLE_NAME='mst2_metadata_payload' THEN + SELECT resident_pages,resident_bytes INTO pages,bytes FROM mst2_metadata_capacity(); + IF pages>16384 OR bytes>268435456 THEN RAISE EXCEPTION 'qualified resident payload capacity is exceeded'; END IF; + ELSIF TG_TABLE_NAME='mst2_metadata_source_entry_reference' THEN + SELECT count(*) INTO actual FROM (SELECT 1 FROM mst2_metadata_source_entry_reference LIMIT 262145) bounded; + IF actual>262144 THEN RAISE EXCEPTION 'qualified retained source dictionary capacity is exceeded'; END IF; + RETURN NULL; + ELSIF TG_TABLE_NAME='mst2_qualified_session_incarnation' THEN + SELECT count(*) INTO actual FROM (SELECT 1 FROM mst2_qualified_session_incarnation WHERE state='READY' LIMIT 4097) bounded; + ELSIF TG_TABLE_NAME='mst2_qualified_lease_binding' THEN + SELECT count(*) INTO actual FROM (SELECT 1 FROM mst2_qualified_lease_binding WHERE state='ACTIVE' LIMIT 4097) bounded; + ELSE SELECT count(*) INTO actual FROM (SELECT 1 FROM mst2_metadata_reader_operation WHERE state='ACTIVE' LIMIT 4097) bounded; END IF; + IF actual>4096 THEN RAISE EXCEPTION 'qualified active serving capacity is exceeded'; END IF; + RETURN NULL; +END $$; +DO $$ DECLARE name text; BEGIN + FOREACH name IN ARRAY ARRAY['mst2_metadata_payload','mst2_qualified_session_incarnation', + 'mst2_qualified_lease_binding','mst2_metadata_reader_operation','mst2_metadata_source_entry_reference'] LOOP + EXECUTE format('CREATE TRIGGER mst2_metadata_capacity_guard AFTER INSERT OR UPDATE ON %I FOR EACH STATEMENT EXECUTE FUNCTION mst2_metadata_capacity_guard()',name); + END LOOP; +END $$; + +-- Append-only identity and replay evidence remains bounded independently of +-- resident payloads. The admission check reads actual stored rows, never hints. +CREATE FUNCTION mst2_metadata_history_capacity_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE actual bigint; bytes bigint; row_limit integer; size_expression text; +BEGIN + IF TG_TABLE_NAME IN ('mst2_metadata_prepare_page','mst2_metadata_prepare_reuse_root','mst2_metadata_verified_ref') THEN + row_limit:=262144; size_expression:='0'; + ELSIF TG_TABLE_NAME='mst2_metadata_prepare' THEN + row_limit:=16384; + size_expression:='pg_column_size(canonical_plan)::bigint+pg_column_size(canonical_bindings)+pg_column_size(primary_scope)'; + ELSIF TG_TABLE_NAME='mst2_metadata_source_root_attestation' THEN + row_limit:=16384; size_expression:='pg_column_size(source_proof)::bigint+pg_column_size(source_profile)'; + ELSIF TG_TABLE_NAME='mst2_metadata_page_certificate' THEN + row_limit:=16384; size_expression:='pg_column_size(canonical_proof)::bigint'; + ELSIF TG_TABLE_NAME='mst2_metadata_scope_source_reference' THEN + row_limit:=16384; size_expression:='pg_column_size(ancestor_revisions)::bigint'; + ELSIF TG_TABLE_NAME='mst2_qualified_session_incarnation' THEN + row_limit:=65536; size_expression:='pg_column_size(canonical_descriptor)::bigint+pg_column_size(source_profile)'; + ELSIF TG_TABLE_NAME IN ('mst2_metadata_lifetime','mst2_metadata_current','mst2_metadata_gc_op', + 'mst2_qualified_lease_binding','mst2_metadata_reader_operation') THEN + row_limit:=65536; size_expression:='0'; + ELSE RAISE EXCEPTION 'qualified history quota has an unknown physical relation'; END IF; + EXECUTE format('SELECT count(*),coalesce(sum(size),0) FROM (SELECT %s AS size FROM %I.%I LIMIT %s) actual', + size_expression,TG_TABLE_SCHEMA,TG_TABLE_NAME,row_limit+1) INTO actual,bytes; + IF actual>row_limit OR bytes>268435456 THEN + RAISE EXCEPTION 'qualified immutable history capacity is exceeded for %',TG_TABLE_NAME; END IF; + RETURN NULL; +END $$; +DO $$ DECLARE name text; BEGIN + FOREACH name IN ARRAY ARRAY['mst2_metadata_prepare','mst2_metadata_prepare_page','mst2_metadata_prepare_reuse_root', + 'mst2_metadata_verified_ref','mst2_metadata_source_root_attestation','mst2_metadata_page_certificate','mst2_metadata_scope_source_reference', + 'mst2_qualified_session_incarnation','mst2_metadata_lifetime','mst2_metadata_current','mst2_metadata_gc_op', + 'mst2_qualified_lease_binding','mst2_metadata_reader_operation'] LOOP + EXECUTE format('CREATE TRIGGER mst2_metadata_history_capacity_guard AFTER INSERT ON %I FOR EACH STATEMENT EXECUTE FUNCTION mst2_metadata_history_capacity_guard()',name); + END LOOP; +END $$; + +CREATE FUNCTION mst2_metadata_gc_claim(p bytea,g bigint,scope bytea,id uuid) RETURNS uuid LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE life mst2_metadata_lifetime%ROWTYPE; node mst2_metadata_graph_node%ROWTYPE; op mst2_metadata_gc_op%ROWTYPE; + has_graph boolean; has_body boolean; +BEGIN + PERFORM mst2_metadata_gc_enter(); + IF NOT mst2_metadata_scope_matches(scope) THEN RAISE EXCEPTION 'qualified claim scope differs from its captured primary'; END IF; + SELECT * INTO op FROM mst2_metadata_gc_op WHERE operation_id=id; + IF FOUND THEN + IF op.page_id IS DISTINCT FROM p OR op.generation<>g OR op.primary_scope IS DISTINCT FROM scope THEN + RAISE EXCEPTION 'qualified GC operation cannot be reused for a different lifetime'; END IF; + RETURN id; + END IF; + SELECT * INTO STRICT life FROM mst2_metadata_lifetime WHERE page_id=p AND generation=g; + SELECT * INTO node FROM mst2_metadata_graph_node WHERE page_id=p AND generation=g; + has_graph:=FOUND; has_body:=EXISTS(SELECT 1 FROM mst2_metadata_payload WHERE page_id=p AND generation=g); + INSERT INTO mst2_metadata_gc_op(operation_id,page_id,generation,primary_scope,graph_domain,metadata_codec, + expected_size,graph_present,had_payload,certificate_digest,state) + VALUES(id,p,g,scope,'qualified-v1',life.metadata_codec,life.expected_size,has_graph,has_body, + CASE WHEN has_graph THEN node.certificate_digest ELSE NULL END,'PENDING'); + UPDATE mst2_metadata_lifetime SET state='DELETING' WHERE page_id=p AND generation=g; + IF has_graph THEN UPDATE mst2_metadata_graph_node SET state='DELETING' WHERE page_id=p AND generation=g; END IF; + RETURN id; +END $$; + +CREATE OR REPLACE FUNCTION mst2_metadata_gc_apply(id uuid) RETURNS void LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE op mst2_metadata_gc_op%ROWTYPE; +BEGIN + PERFORM mst2_metadata_gc_enter(); + SELECT * INTO STRICT op FROM mst2_metadata_gc_op WHERE operation_id=id FOR UPDATE; + IF NOT mst2_metadata_scope_matches(op.primary_scope) THEN RAISE EXCEPTION 'qualified GC replay left its captured primary'; END IF; + IF op.state='APPLIED' THEN + PERFORM mst2_metadata_gc_proof(op.page_id,op.generation,'APPLIED',false,false,op.certificate_digest); RETURN; + END IF; + PERFORM mst2_metadata_gc_proof(op.page_id,op.generation,'PENDING',op.graph_present,op.had_payload,op.certificate_digest); + IF op.had_payload THEN + UPDATE mst2_metadata_gc_op SET payload_delete_xid=txid_current() WHERE operation_id=id; + DELETE FROM mst2_metadata_payload WHERE page_id=op.page_id AND generation=op.generation; + IF NOT FOUND THEN RAISE EXCEPTION 'qualified atomic payload deletion lost its exact row'; END IF; + END IF; + DELETE FROM mst2_metadata_graph_edge WHERE parent_page=op.page_id AND parent_generation=op.generation; + DELETE FROM mst2_metadata_reuse_index WHERE root_page=op.page_id AND root_generation=op.generation; + DELETE FROM mst2_metadata_graph_node WHERE page_id=op.page_id AND generation=op.generation; + UPDATE mst2_metadata_lifetime SET state='REMOVED' WHERE page_id=op.page_id AND generation=op.generation AND state='DELETING'; + IF NOT FOUND THEN RAISE EXCEPTION 'qualified atomic removal lost its exact lifecycle'; END IF; + UPDATE mst2_metadata_gc_op SET state='APPLIED' WHERE operation_id=id; +END $$; +CREATE OR REPLACE FUNCTION mst2_metadata_gc_finish(id uuid) RETURNS void LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ BEGIN PERFORM mst2_metadata_gc_apply(id); END $$; + +-- Additional graph/payload guards follow; no history relation is collected. +CREATE OR REPLACE FUNCTION mst2_metadata_payload_fenced() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE op mst2_metadata_gc_op%ROWTYPE; +BEGIN + IF TG_OP='UPDATE' THEN RAISE EXCEPTION 'qualified immutable payload cannot be updated'; END IF; + IF TG_OP='DELETE' THEN + PERFORM mst2_metadata_gc_enter(); + SELECT * INTO op FROM mst2_metadata_gc_op WHERE page_id=OLD.page_id AND generation=OLD.generation AND state='PENDING'; + IF NOT FOUND OR NOT op.had_payload OR op.payload_delete_xid IS DISTINCT FROM txid_current() + OR NOT mst2_metadata_scope_matches(op.primary_scope) OR op.expected_size<>OLD.byte_size OR op.metadata_codec<>OLD.metadata_codec THEN + RAISE EXCEPTION 'qualified payload removal needs its exact same-transaction pending evidence'; END IF; + PERFORM mst2_metadata_gc_proof(OLD.page_id,OLD.generation,'PENDING',op.graph_present,true,op.certificate_digest); + RETURN OLD; + END IF; + IF NOT EXISTS(SELECT 1 FROM mst2_metadata_current c JOIN mst2_metadata_lifetime l USING(page_id,generation) + JOIN mst2_metadata_prepare_page m USING(page_id,generation) JOIN mst2_metadata_prepare q USING(prepare_id) + WHERE c.page_id=NEW.page_id AND c.generation=NEW.generation AND l.state IN ('RESERVED','LIVE') + AND l.metadata_codec=NEW.metadata_codec AND l.expected_size=NEW.byte_size AND q.state='PREPARING' + AND q.graph_domain='qualified-v1' AND mst2_metadata_scope_matches(q.primary_scope)) THEN + RAISE EXCEPTION 'qualified payload INSERT crossed its exact active generation'; END IF; + RETURN NEW; +END $$; + +CREATE OR REPLACE FUNCTION mst2_metadata_graph_node_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE life mst2_metadata_lifetime%ROWTYPE; op mst2_metadata_gc_op%ROWTYPE; +BEGIN + IF TG_OP='DELETE' THEN + PERFORM mst2_metadata_gc_enter(); + SELECT * INTO op FROM mst2_metadata_gc_op WHERE page_id=OLD.page_id AND generation=OLD.generation AND state='PENDING'; + IF NOT FOUND OR NOT op.graph_present OR op.certificate_digest IS DISTINCT FROM OLD.certificate_digest + OR OLD.state<>'DELETING' OR NOT mst2_metadata_scope_matches(op.primary_scope) + OR op.had_payload AND op.payload_delete_xid IS DISTINCT FROM txid_current() + OR EXISTS(SELECT 1 FROM mst2_metadata_graph_edge WHERE parent_page=OLD.page_id AND parent_generation=OLD.generation) THEN + RAISE EXCEPTION 'qualified graph removal lacks its atomic byte removal and empty outgoing edges'; END IF; + PERFORM mst2_metadata_gc_proof(OLD.page_id,OLD.generation,'PENDING',true,false,op.certificate_digest); + RETURN OLD; + END IF; + SELECT l.* INTO life FROM mst2_metadata_current c JOIN mst2_metadata_lifetime l USING(page_id,generation) + WHERE c.page_id=NEW.page_id AND c.generation=NEW.generation; + IF NOT FOUND OR life.graph_domain<>'qualified-v1' OR life.metadata_codec<>NEW.metadata_codec OR life.expected_size<>NEW.bytes + OR NEW.incoming_refs<>(SELECT count(*) FROM mst2_metadata_graph_edge WHERE child_page=NEW.page_id AND child_generation=NEW.generation) + OR NOT EXISTS(SELECT 1 FROM mst2_metadata_page_certificate certificate WHERE certificate.page_id=NEW.page_id + AND certificate.generation=NEW.generation AND certificate.certificate_digest=NEW.certificate_digest + AND certificate.metadata_codec=NEW.metadata_codec AND certificate.byte_size=NEW.bytes) THEN + RAISE EXCEPTION 'qualified graph identity, certificate or physical counter changed'; END IF; + IF TG_OP='INSERT' THEN + IF NEW.state<>'LIVE' OR life.state NOT IN ('RESERVED','LIVE') + OR NOT EXISTS(SELECT 1 FROM mst2_metadata_prepare_page m JOIN mst2_metadata_prepare q USING(prepare_id) + WHERE m.page_id=NEW.page_id AND m.generation=NEW.generation AND q.state='PREPARING') THEN + RAISE EXCEPTION 'qualified graph creation requires an active exact preparation'; END IF; + ELSE + IF (to_jsonb(NEW)-ARRAY['state','incoming_refs']) IS DISTINCT FROM (to_jsonb(OLD)-ARRAY['state','incoming_refs']) THEN + RAISE EXCEPTION 'qualified graph identity is immutable'; END IF; + IF NEW.state<>OLD.state THEN + PERFORM mst2_metadata_gc_enter(); + SELECT * INTO op FROM mst2_metadata_gc_op WHERE page_id=OLD.page_id AND generation=OLD.generation AND state='PENDING'; + IF NOT FOUND OR OLD.state<>'LIVE' OR NEW.state<>'DELETING' OR life.state<>'DELETING' OR NEW.incoming_refs<>0 + OR op.certificate_digest IS DISTINCT FROM NEW.certificate_digest OR NOT mst2_metadata_scope_matches(op.primary_scope) THEN + RAISE EXCEPTION 'qualified graph state change has no exact pending claim'; END IF; + PERFORM mst2_metadata_assert_uncovered(OLD.page_id,OLD.generation); + ELSIF NEW.state='LIVE' AND life.state NOT IN ('RESERVED','LIVE') OR NEW.state='DELETING' AND life.state<>'DELETING' THEN + RAISE EXCEPTION 'qualified graph state differs from its current lifecycle'; END IF; + END IF; + RETURN NEW; +END $$; + +CREATE OR REPLACE FUNCTION mst2_metadata_graph_edge_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE op mst2_metadata_gc_op%ROWTYPE; +BEGIN + IF TG_OP='UPDATE' THEN RAISE EXCEPTION 'qualified edge identity is immutable'; END IF; + IF TG_OP='DELETE' THEN + PERFORM mst2_metadata_gc_enter(); + SELECT * INTO op FROM mst2_metadata_gc_op WHERE page_id=OLD.parent_page AND generation=OLD.parent_generation AND state='PENDING'; + IF NOT FOUND OR NOT op.graph_present OR NOT mst2_metadata_scope_matches(op.primary_scope) + OR op.had_payload AND op.payload_delete_xid IS DISTINCT FROM txid_current() + OR EXISTS(SELECT 1 FROM mst2_metadata_payload WHERE page_id=OLD.parent_page AND generation=OLD.parent_generation) + OR NOT EXISTS(SELECT 1 FROM mst2_metadata_graph_node node JOIN mst2_metadata_current cur USING(page_id,generation) + JOIN mst2_metadata_lifetime life USING(page_id,generation) + WHERE node.page_id=OLD.parent_page AND node.generation=OLD.parent_generation AND node.state='DELETING' + AND life.state='DELETING' AND node.incoming_refs=0 AND node.certificate_digest=op.certificate_digest) THEN + RAISE EXCEPTION 'qualified edge removal is outside its exact atomic graph deletion phase'; END IF; + PERFORM mst2_metadata_assert_uncovered(OLD.parent_page,OLD.parent_generation); + RETURN OLD; + END IF; + IF NOT EXISTS(SELECT 1 FROM mst2_metadata_verified_ref ref + JOIN mst2_metadata_page_certificate parent ON parent.page_id=ref.parent_page AND parent.generation=ref.parent_generation + JOIN mst2_metadata_page_certificate child ON child.page_id=ref.child_page AND child.generation=ref.child_generation + WHERE ref.parent_page=NEW.parent_page AND ref.parent_generation=NEW.parent_generation + AND ref.child_page=NEW.child_page AND ref.child_generation=NEW.child_generation + AND ref.child_certificate_digest=child.certificate_digest AND parent.rank>child.rank) THEN + RAISE EXCEPTION 'qualified edge differs from its exact canonical reference or strict certified rank'; END IF; + IF (SELECT count(*) FROM mst2_metadata_graph_node n JOIN mst2_metadata_current c USING(page_id,generation) + JOIN mst2_metadata_lifetime l USING(page_id,generation) + WHERE ((n.page_id=NEW.parent_page AND n.generation=NEW.parent_generation) + OR (n.page_id=NEW.child_page AND n.generation=NEW.child_generation)) + AND n.state='LIVE' AND l.state IN ('RESERVED','LIVE') AND l.graph_domain='qualified-v1')<>2 + OR NOT EXISTS(SELECT 1 FROM mst2_metadata_prepare_page a JOIN mst2_metadata_prepare q USING(prepare_id) + WHERE a.page_id=NEW.parent_page AND a.generation=NEW.parent_generation AND q.state='PREPARING' + AND q.graph_domain='qualified-v1' AND mst2_metadata_scope_matches(q.primary_scope) + AND (EXISTS(SELECT 1 FROM mst2_metadata_prepare_page b WHERE b.prepare_id=a.prepare_id + AND b.page_id=NEW.child_page AND b.generation=NEW.child_generation) + OR EXISTS(SELECT 1 FROM mst2_metadata_prepare_reuse_root r JOIN mst2_metadata_root_anchor anchor + ON anchor.prepare_id=r.prepare_id AND anchor.anchor_kind='REUSE' AND anchor.owner_key=r.prepare_id + AND anchor.root_page=r.root_page AND anchor.root_generation=r.root_generation + WHERE r.prepare_id=a.prepare_id AND r.root_page=NEW.child_page AND r.root_generation=NEW.child_generation))) THEN + RAISE EXCEPTION 'edge creation requires active exact qualified endpoints'; END IF; + RETURN NEW; +END $$; + +CREATE FUNCTION mst2_metadata_edges_removed() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + IF (SELECT count(*) FROM removed_edges)>257 THEN RAISE EXCEPTION 'qualified GC edge deletion exceeds one canonical node'; END IF; + IF EXISTS(SELECT 1 FROM mst2_metadata_graph_node node JOIN (SELECT child_page,child_generation,count(*) AS delta + FROM removed_edges GROUP BY child_page,child_generation) removed + ON removed.child_page=node.page_id AND removed.child_generation=node.generation + WHERE node.incoming_refs-removed.delta<>(SELECT count(*) FROM mst2_metadata_graph_edge edge + WHERE edge.child_page=node.page_id AND edge.child_generation=node.generation)) THEN + RAISE EXCEPTION 'qualified removal discovered physical incoming counter drift'; END IF; + UPDATE mst2_metadata_graph_node node SET incoming_refs=node.incoming_refs-removed.delta + FROM (SELECT child_page,child_generation,count(*) AS delta FROM removed_edges GROUP BY child_page,child_generation) removed + WHERE node.page_id=removed.child_page AND node.generation=removed.child_generation; + RETURN NULL; +END $$; +CREATE TRIGGER mst2_metadata_edges_removed AFTER DELETE ON mst2_metadata_graph_edge + REFERENCING OLD TABLE AS removed_edges FOR EACH STATEMENT EXECUTE FUNCTION mst2_metadata_edges_removed(); + +CREATE FUNCTION mst2_metadata_gc_owner_cleanup(maximum integer) +RETURNS TABLE(examined bigint,readers_expired bigint,leases_expired bigint,prepares_aborted bigint, + handovers_retired bigint,orphans_retired bigint) +LANGUAGE plpgsql VOLATILE SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE bound integer:=maximum; item record; q mst2_metadata_prepare%ROWTYPE; now_unix bigint; +BEGIN + PERFORM $CORE_SCHEMA$.mst2_route_enter($CORE_LITERAL$); + IF maximum IS NULL OR maximum NOT BETWEEN 0 AND 64 THEN RAISE EXCEPTION 'qualified owner cleanup budget must be 0..=64'; END IF; + examined:=0; readers_expired:=0; leases_expired:=0; prepares_aborted:=0; handovers_retired:=0; orphans_retired:=0; + now_unix:=floor(extract(epoch FROM clock_timestamp()))::bigint; + FOR item IN SELECT * FROM ( + (SELECT 'READER'::text AS kind,operation_id::text AS owner,lease_id,hard_deadline_unix AS deadline + FROM mst2_metadata_reader_operation WHERE state='ACTIVE' AND hard_deadline_unix<=now_unix + ORDER BY hard_deadline_unix,operation_id LIMIT bound) + UNION ALL + (SELECT 'LEASE'::text,lease_id,lease_id,expires_at_unix FROM mst2_qualified_lease_binding + WHERE state='ACTIVE' AND expires_at_unix<=now_unix ORDER BY expires_at_unix,lease_id LIMIT bound) + UNION ALL + (SELECT 'PREPARE'::text,prepare_id,NULL::text,floor(extract(epoch FROM orphan_expires_at))::bigint + FROM mst2_metadata_prepare WHERE (state='PREPARING' OR state='COMMITTED' AND coverage_retired_at IS NULL) + AND orphan_expires_at<=clock_timestamp() ORDER BY orphan_expires_at,prepare_id LIMIT bound) + ) expired ORDER BY deadline,kind,owner LIMIT bound LOOP + examined:=examined+1; + IF item.kind='READER' THEN + UPDATE mst2_metadata_reader_operation SET state='EXPIRED' WHERE operation_id=item.owner::uuid AND state='ACTIVE'; + IF NOT FOUND THEN RAISE EXCEPTION 'qualified expired reader changed behind its mutation barrier'; END IF; + readers_expired:=readers_expired+1; + DELETE FROM mst2_metadata_root_anchor WHERE reader_operation_id=item.owner::uuid AND anchor_kind IN ('REQUEST','READER'); + PERFORM mst2_metadata_cleanup_lease(item.lease_id); + ELSIF item.kind='LEASE' THEN + UPDATE mst2_qualified_lease_binding SET state='EXPIRED',lease_epoch=lease_epoch+1 WHERE lease_id=item.owner AND state='ACTIVE'; + IF NOT FOUND THEN RAISE EXCEPTION 'qualified expired lease changed behind its mutation barrier'; END IF; + leases_expired:=leases_expired+1; PERFORM mst2_metadata_cleanup_lease(item.lease_id); + ELSE + SELECT * INTO STRICT q FROM mst2_metadata_prepare WHERE prepare_id=item.owner; + IF q.state='PREPARING' THEN + DELETE FROM mst2_metadata_graph_root WHERE prepare_id=q.prepare_id; + UPDATE mst2_metadata_prepare SET state='ABORTED',aborted_at=clock_timestamp() WHERE prepare_id=q.prepare_id; + DELETE FROM mst2_metadata_root_anchor WHERE prepare_id=q.prepare_id AND anchor_kind IN ('PREPARE','REUSE'); + prepares_aborted:=prepares_aborted+1; + ELSIF mst2_metadata_session_covers_prepare(q.prepare_id) THEN + UPDATE mst2_metadata_prepare SET coverage_retired_at=clock_timestamp() WHERE prepare_id=q.prepare_id; + DELETE FROM mst2_metadata_root_anchor WHERE prepare_id=q.prepare_id AND anchor_kind IN ('PREPARE','REUSE'); + handovers_retired:=handovers_retired+1; + ELSIF mst2_metadata_orphan_prepare_eligible(q.prepare_id) THEN + UPDATE mst2_metadata_prepare SET coverage_retired_at=clock_timestamp() WHERE prepare_id=q.prepare_id; + DELETE FROM mst2_metadata_root_anchor WHERE prepare_id=q.prepare_id AND anchor_kind IN ('PREPARE','REUSE'); + orphans_retired:=orphans_retired+1; + END IF; + END IF; + END LOOP; + RETURN NEXT; +END $$; +DROP TRIGGER mst2_01_family_closed ON mst2_metadata_gc_op; diff --git a/src/jupiter/storage/qualified_metadata_gc_tests.rs b/src/jupiter/storage/qualified_metadata_gc_tests.rs new file mode 100644 index 00000000..346254bf --- /dev/null +++ b/src/jupiter/storage/qualified_metadata_gc_tests.rs @@ -0,0 +1,456 @@ +use std::collections::{BTreeMap, BTreeSet}; + +use super::{canonical_tests::seeded_rooted_plan, *}; +use crate::ceres::snapshot::{ + rooted_metadata_install::RootedMetadataInstallPlan, + rooted_metadata_projection::RootedReuseLookup, +}; + +async fn write_rooted( + writer: &RootedQualifiedMetadataRepository, + operation: &str, + plan: &RootedMetadataInstallPlan, + payload: MetadataPagePayload, +) -> RootedPrepareIntent { + let intent = writer.begin_intent(operation, plan).await.unwrap(); + writer.install_pages(&intent, &[payload]).await.unwrap(); + writer.finalize(&intent).await.unwrap(); + intent +} + +// Fault injection only: expire database evidence in the isolated test schema, +// restore the original guards/catalog, then use normal production maintenance. +async fn age_orphan(q: &DatabaseConnection, namespace: &VerifiedQualifiedNamespace, prepare: &str) { + let txn = q.begin().await.unwrap(); + for trigger in [ + "mst2_00_family_barrier", + "mst2_metadata_prepare_guard", + "mst2_metadata_prepare_00_expiry_guard", + ] { + txn.execute_unprepared(&format!( + "ALTER TABLE mst2_metadata_prepare DISABLE TRIGGER {trigger}" + )) + .await + .unwrap(); + } + txn.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "UPDATE mst2_metadata_prepare SET created_at=clock_timestamp()-interval '3601 seconds', + orphan_expires_at=clock_timestamp()-interval '1 second' WHERE prepare_id=$1", + [prepare.into()], + )) + .await + .unwrap(); + // clock_timestamp() calls differ; preserve the exact database TTL relation. + txn.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "UPDATE mst2_metadata_prepare SET orphan_expires_at=created_at+interval '3600 seconds' WHERE prepare_id=$1", + [prepare.into()])).await.unwrap(); + for trigger in [ + "mst2_00_family_barrier", + "mst2_metadata_prepare_guard", + "mst2_metadata_prepare_00_expiry_guard", + ] { + txn.execute_unprepared(&format!( + "ALTER TABLE mst2_metadata_prepare ENABLE TRIGGER {trigger}" + )) + .await + .unwrap(); + } + txn.commit().await.unwrap(); + assert_eq!( + catalog(q, namespace.core_oid, namespace.schema_oid) + .await + .unwrap(), + namespace.catalog_fingerprint + ); +} + +async fn claim( + q: &DatabaseConnection, + intent: &RootedPrepareIntent, +) -> Result { + let operation = uuid::Uuid::new_v4().to_string(); + q.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "SELECT mst2_metadata_gc_claim($1,$2,(SELECT primary_scope FROM mst2_metadata_prepare WHERE prepare_id=$3),$4::uuid)", + [intent.metadata_root().to_vec().into(),intent.root_generation().into(),intent.prepare_id().into(),operation.clone().into()] + )).await?; + Ok(operation) +} + +#[tokio::test] +async fn admitted_orphan_gc_preserves_replay_identity_and_advances_exact_generation() { + let (config, core, namespace, q, _guard) = fixture().await; + let (plan, payload) = seeded_rooted_plan(&core, 'a', "file", 1).await; + let writer = RootedQualifiedMetadataRepository::open(&core, &config) + .await + .unwrap(); + let first = write_rooted(&writer, "gc-generation-one", &plan, payload.clone()).await; + assert_eq!( + count(&q, "SELECT mst2_metadata_gc_enabled()::bigint").await, + 1 + ); + assert!( + claim(&q, &first) + .await + .unwrap_err() + .to_string() + .contains("owned coverage") + ); + assert_eq!( + count(&q, "SELECT count(*) FROM mst2_metadata_gc_op").await, + 0 + ); + age_orphan(&q, &namespace, first.prepare_id()).await; + let work = writer.maintenance_tick(64).await.unwrap(); + assert!(work.collector_enabled); + assert!(work.examined <= 64); + assert_eq!(work.orphans_retired, 1); + assert_eq!(work.payload_pages_removed, 1); + assert_eq!(work.payload_bytes_removed, payload.size); + assert_eq!( + count(&q, "SELECT count(*) FROM mst2_metadata_payload").await, + 0 + ); + assert_eq!( + count(&q, "SELECT count(*) FROM mst2_metadata_graph_node").await, + 0 + ); + assert_eq!( + count(&q, "SELECT count(*) FROM mst2_metadata_page_certificate").await, + 1 + ); + assert_eq!( + count( + &q, + "SELECT count(*) FROM mst2_metadata_source_root_attestation" + ) + .await, + 1 + ); + assert_eq!( + count( + &q, + "SELECT count(*) FROM mst2_metadata_gc_op WHERE state='APPLIED'" + ) + .await, + 1 + ); + let operation: String = q + .query_one_raw(Statement::from_string( + DbBackend::Postgres, + "SELECT operation_id::text FROM mst2_metadata_gc_op", + )) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap(); + let fresh = write_rooted(&writer, "gc-generation-two", &plan, payload.clone()).await; + assert_eq!(fresh.root_generation(), first.root_generation() + 1); + assert!(writer.recover("gc-generation-one", &plan).await.is_err()); + for _ in 0..2 { + q.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT mst2_metadata_gc_apply($1::uuid)", + [operation.clone().into()], + )) + .await + .unwrap(); + } + assert_eq!( + count( + &q, + "SELECT count(*) FROM mst2_metadata_payload WHERE generation=2" + ) + .await, + 1 + ); + assert_eq!( + count( + &q, + "SELECT count(*) FROM mst2_metadata_lifetime WHERE generation=1 AND state='REMOVED'" + ) + .await, + 1 + ); + assert_eq!( + count( + &q, + "SELECT count(*) FROM mst2_metadata_lifetime WHERE generation=2 AND state='LIVE'" + ) + .await, + 1 + ); + assert_eq!( + count(&q, "SELECT count(*) FROM mst2_metadata_page_certificate").await, + 2 + ); + assert_eq!( + count( + &q, + "SELECT count(*) FROM mst2_metadata_current WHERE generation=2" + ) + .await, + 1 + ); +} + +#[tokio::test] +async fn pending_gc_survives_rebuild_and_payload_stamp_cannot_commit_alone() { + let (config, core, namespace, q, _guard) = fixture().await; + let (plan, payload) = seeded_rooted_plan(&core, 'a', "file", 1).await; + let writer = RootedQualifiedMetadataRepository::open(&core, &config) + .await + .unwrap(); + let intent = write_rooted(&writer, "gc-pending", &plan, payload).await; + age_orphan(&q, &namespace, intent.prepare_id()).await; + assert_eq!(writer.maintenance_tick(1).await.unwrap().orphans_retired, 1); + let operation = claim(&q, &intent).await.unwrap(); + assert_eq!( + count( + &q, + "SELECT count(*) FROM mst2_metadata_gc_op WHERE state='PENDING'" + ) + .await, + 1 + ); + let txn = q.begin().await.unwrap(); + txn.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "UPDATE mst2_metadata_gc_op SET payload_delete_xid=txid_current() WHERE operation_id=$1::uuid", + [operation.into()])).await.unwrap(); + let error = txn.commit().await.unwrap_err(); + assert!( + error.to_string().contains("cannot escape its atomic apply"), + "{error}" + ); + assert_eq!( + count(&q, "SELECT count(*) FROM mst2_metadata_payload").await, + 1 + ); + assert_eq!(count(&q,"SELECT count(*) FROM mst2_metadata_gc_op WHERE state='PENDING' AND payload_delete_xid IS NULL").await,1); + drop(writer); + let rebuilt = RootedQualifiedMetadataRepository::open(&core, &config) + .await + .unwrap(); + let work = rebuilt.maintenance_tick(1).await.unwrap(); + assert_eq!(work.examined, 1); + assert_eq!(work.pending_replayed, 1); + assert_eq!(work.gc_applied, 1); + assert_eq!(work.payload_pages_removed, 1); + assert_eq!( + count(&q, "SELECT count(*) FROM mst2_metadata_payload").await, + 0 + ); + assert_eq!( + count( + &q, + "SELECT count(*) FROM mst2_metadata_gc_op WHERE state='APPLIED'" + ) + .await, + 1 + ); +} + +#[tokio::test] +async fn actual_incoming_edge_blocks_child_gc_until_parent_removal() { + let (config, core, namespace, q, _guard) = fixture().await; + let (child, payload) = seeded_rooted_plan(&core, 'a', "file", 1).await; + let writer = RootedQualifiedMetadataRepository::open(&core, &config) + .await + .unwrap(); + let child_intent = write_rooted(&writer, "gc-edge-child", &child, payload).await; + let hint = writer + .lookup_reuse(&child.identity.tagged_root_tree_oid, &child.identity) + .await + .unwrap() + .unwrap(); + let entries = [ + Entry::dir(b"one", child.root), + Entry::dir(b"two", child.root), + ]; + let bytes = Page::build(&entries).unwrap(); + let root = page_id(&bytes); + let mut body = b"40000 one\0".to_vec(); + body.extend_from_slice(&[0xaa; 20]); + body.extend_from_slice(b"40000 two\0"); + body.extend_from_slice(&[0xaa; 20]); + core.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "INSERT INTO mega_tree(id,tree_id,sub_trees,size,created_at,pack_id,pack_offset,commit_id) + VALUES(2,$1,$2,0,now(),'fixture',0,'fixture')", + ["d".repeat(40).into(), body.into()], + )) + .await + .unwrap(); + let mut identity = child.identity.clone(); + identity.tagged_root_tree_oid = format!("sha1:{}", "d".repeat(40)); + let parent = RootedMetadataInstallPlan::new( + identity.clone(), + root, + BTreeMap::from([(root, bytes.len() as u64)]), + BTreeSet::from([(root, child.root)]), + BTreeMap::from([(child.root, hint.proof)]), + BTreeMap::from([ + (identity.tagged_root_tree_oid, root), + (child.identity.tagged_root_tree_oid.clone(), child.root), + ]), + ) + .unwrap(); + let parent_intent = write_rooted( + &writer, + "gc-edge-parent", + &parent, + MetadataPagePayload { + id: root, + size: bytes.len() as u64, + bytes, + }, + ) + .await; + age_orphan(&q, &namespace, child_intent.prepare_id()).await; + age_orphan(&q, &namespace, parent_intent.prepare_id()).await; + for _ in 0..2 { + assert_eq!(writer.maintenance_tick(1).await.unwrap().orphans_retired, 1); + } + assert_eq!( + count(&q, "SELECT count(*) FROM mst2_metadata_root_anchor").await, + 0 + ); + assert_eq!( + count(&q, "SELECT count(*) FROM mst2_metadata_graph_edge").await, + 1 + ); + assert_eq!( + count( + &q, + "SELECT sum(incoming_refs)::bigint FROM mst2_metadata_graph_node" + ) + .await, + 1 + ); + assert!( + claim(&q, &child_intent) + .await + .unwrap_err() + .to_string() + .contains("incoming edges") + ); + let operation = claim(&q, &parent_intent).await.unwrap(); + q.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT mst2_metadata_gc_apply($1::uuid)", + [operation.into()], + )) + .await + .unwrap(); + assert_eq!( + count(&q, "SELECT count(*) FROM mst2_metadata_graph_edge").await, + 0 + ); + assert_eq!( + count( + &q, + "SELECT sum(incoming_refs)::bigint FROM mst2_metadata_graph_node" + ) + .await, + 0 + ); + let operation = claim(&q, &child_intent).await.unwrap(); + q.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT mst2_metadata_gc_apply($1::uuid)", + [operation.into()], + )) + .await + .unwrap(); + assert_eq!( + count(&q, "SELECT count(*) FROM mst2_metadata_payload").await, + 0 + ); + assert_eq!( + count(&q, "SELECT count(*) FROM mst2_metadata_verified_ref").await, + 2 + ); +} + +#[tokio::test] +async fn restored_guards_refuse_corrupt_payload_size_counter_and_current_bindings() { + for case in 0..4 { + let (config, core, namespace, q, _guard) = fixture().await; + let (plan, payload) = seeded_rooted_plan(&core, 'a', "file", 1).await; + let writer = RootedQualifiedMetadataRepository::open(&core, &config) + .await + .unwrap(); + let intent = write_rooted(&writer, "gc-corruption", &plan, payload).await; + age_orphan(&q, &namespace, intent.prepare_id()).await; + assert_eq!(writer.maintenance_tick(1).await.unwrap().orphans_retired, 1); + let (table, guard, statement) = match case { + 0 => ( + "mst2_metadata_payload", + "mst2_metadata_payload_fenced", + "UPDATE mst2_metadata_payload SET payload=set_byte(payload,20,255)", + ), + 1 => ( + "mst2_metadata_graph_node", + "mst2_metadata_graph_node_guard", + "UPDATE mst2_metadata_graph_node SET bytes=bytes+1", + ), + 2 => ( + "mst2_metadata_graph_node", + "mst2_metadata_graph_node_guard", + "UPDATE mst2_metadata_graph_node SET incoming_refs=1", + ), + _ => ( + "mst2_metadata_current", + "mst2_metadata_current_guard", + "DELETE FROM mst2_metadata_current", + ), + }; + let txn = q.begin().await.unwrap(); + for trigger in ["mst2_00_family_barrier", guard] { + txn.execute_unprepared(&format!("ALTER TABLE {table} DISABLE TRIGGER {trigger}")) + .await + .unwrap(); + } + txn.execute_unprepared(statement).await.unwrap(); + for trigger in ["mst2_00_family_barrier", guard] { + txn.execute_unprepared(&format!("ALTER TABLE {table} ENABLE TRIGGER {trigger}")) + .await + .unwrap(); + } + txn.commit().await.unwrap(); + assert_eq!( + catalog(&q, namespace.core_oid, namespace.schema_oid) + .await + .unwrap(), + namespace.catalog_fingerprint + ); + let error = claim(&q, &intent).await.unwrap_err(); + assert!( + error.to_string().contains(if case == 0 { + "durable bytes" + } else if case < 3 { + "actual counter" + } else { + "current generation" + }), + "{error}" + ); + assert_eq!( + count(&q, "SELECT count(*) FROM mst2_metadata_gc_op").await, + 0 + ); + assert_eq!( + count(&q, "SELECT count(*) FROM mst2_metadata_payload").await, + 1 + ); + assert_eq!( + count( + &q, + "SELECT count(*) FROM mst2_metadata_lifetime WHERE state='LIVE'" + ) + .await, + 1 + ); + } +} diff --git a/src/jupiter/storage/qualified_metadata_reader.rs b/src/jupiter/storage/qualified_metadata_reader.rs new file mode 100644 index 00000000..63ccbdcd --- /dev/null +++ b/src/jupiter/storage/qualified_metadata_reader.rs @@ -0,0 +1,961 @@ +//! Persisted routes traverse exact certified occurrences, including reused roots. + +use std::collections::{BTreeSet, HashMap}; + +use mst2_codec::metapage::{HEADER_LEN, PAGE_MAX_BYTES}; +use serde::Deserialize; + +use super::*; +use crate::{ + ceres::snapshot::{ + retention_dag::MetadataDagLimits, runtime::SnapshotContext, + view::validate_scope_relative_path, + }, + jupiter::storage::native_snapshot_session::{ + MetadataRouteRequest, PersistedMetadataReadWork, PersistedMetadataRouteBatch, + }, +}; + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +struct Binding { + generation: i64, + certificate: [u8; 32], +} + +#[cfg(test)] +tokio::task_local! { + static READER_ADMISSION_BARRIERS: (Arc, Arc); + static SOURCE_FACT_BARRIERS: (Arc, Arc); + pub(super) static READER_TEMP_SOURCE_SHADOW: bool; +} + +#[cfg(test)] +pub(crate) async fn with_rooted_source_fact_barriers( + admitted: Arc, + resume: Arc, + future: F, +) -> F::Output { + SOURCE_FACT_BARRIERS.scope((admitted, resume), future).await +} + +#[cfg(test)] +pub(crate) async fn with_rooted_source_temporary_shadow( + future: F, +) -> F::Output { + READER_TEMP_SOURCE_SHADOW.scope(true, future).await +} + +#[cfg(test)] +pub(crate) async fn with_rooted_reader_barriers( + admitted: Arc, + resume: Arc, + future: F, +) -> F::Output { + READER_ADMISSION_BARRIERS + .scope((admitted, resume), future) + .await +} + +#[derive(Debug, Deserialize)] +struct Reference { + kind: String, + child: String, + generation: i64, + certificate: String, + name: Option, + label: Option, + count: Option, +} + +fn digest_hex(value: &str) -> Result<[u8; 32], SnapshotError> { + hex::decode(value) + .map_err(internal)? + .as_slice() + .try_into() + .map_err(internal) +} +fn limit(message: &str) -> SnapshotError { + SnapshotError::new( + crate::ceres::snapshot::error::SnapshotErrorCode::LimitExceeded, + message, + ) +} +fn absent(message: &str) -> SnapshotError { + SnapshotError::new( + crate::ceres::snapshot::error::SnapshotErrorCode::PathNotFound, + message, + ) +} + +struct StoredPage { + bytes: Vec, + page: Page, + entries: u64, +} + +pub(crate) struct RootedDirectoryWindow { + pub directory_root: [u8; 32], + pub entry_count: u64, + pub entries: Vec, + pub has_more: bool, + pub proof_pages: Vec<([u8; 32], Vec)>, + pub ancestors: Vec<(String, [u8; 32])>, +} + +pub(crate) enum RootedLookupStatus { + Directory([u8; 32]), + File { + entry: mst2_codec::metapage::Entry, + git_oid: String, + }, + Absent, + NotDirectory { + symlink: bool, + }, +} +pub(crate) struct RootedLookupBatch { + pub results: Vec, + pub proof_pages: Vec<([u8; 32], Vec)>, +} +struct Reader<'a> { + txn: &'a DatabaseTransaction, + operation: uuid::Uuid, + bindings: HashMap<[u8; 32], Binding>, + cache: HashMap<[u8; 32], StoredPage>, + work: PersistedMetadataReadWork, + limits: MetadataDagLimits, + directories: BTreeMap, + source_ids: BTreeMap, + file_oids: HashMap<(uuid::Uuid, Vec), String>, +} + +impl RootedQualifiedMetadataRepository { + pub(crate) async fn fixed_path_metadata( + &self, + pinned: &SnapshotContext, + path: &str, + ) -> Result { + validate_scope_relative_path(path)?; + let absolute = if pinned.built.descriptor.scope == "/" { + path.to_owned() + } else if path == "/" { + pinned.built.descriptor.scope.clone() + } else { + format!("{}{}", pinned.built.descriptor.scope, path) + }; + validate_scope_relative_path(&absolute)?; + let (operation, binding, source) = self.admit_reader(pinned).await?; + let result = async { + let txn = self.read_transaction().await?; + let root = pinned.built.descriptor.metadata_root; + let mut reader = Reader::new(&txn, operation, root, binding, source); + let result = reader.lookup(root, path).await; + sessions::finish(txn, result).await + } + .await; + self.finish_reader(operation, result).await + } + + pub(crate) async fn directory_window( + &self, + pinned: &SnapshotContext, + path: &str, + after: Option<&str>, + count: usize, + ) -> Result { + validate_scope_relative_path(path)?; + if !(1..=256).contains(&count) { + return Err(limit("directory limit must be 1..256")); + } + let absolute = if pinned.built.descriptor.scope == "/" { + path.to_owned() + } else if path == "/" { + pinned.built.descriptor.scope.clone() + } else { + format!("{}{}", pinned.built.descriptor.scope, path) + }; + validate_scope_relative_path(&absolute)?; + let (operation, binding, source) = self.admit_reader(pinned).await?; + let result = async { + let txn = self.read_transaction().await?; + let read = async { + let root = pinned.built.descriptor.metadata_root; + let mut reader = Reader::new(&txn, operation, root, binding, source); + let directory = reader.directory(root, path).await?; + reader.load(directory).await?; + let total = reader.cache[&directory].entries; + let mut entries = Vec::with_capacity(count + 1); + reader + .range(directory, after.map(str::as_bytes), count + 1, &mut entries) + .await?; + let has_more = entries.len() > count; + entries.truncate(count); + reader + .source_entries(reader.source_ids[path], directory, &entries) + .await?; + Ok(RootedDirectoryWindow { + directory_root: directory, + entry_count: total, + entries, + has_more, + proof_pages: reader.proofs()?, + ancestors: reader.directories.into_iter().collect(), + }) + } + .await; + sessions::finish(txn, read).await + } + .await; + self.finish_reader(operation, result).await + } + + pub(crate) async fn lookup_metadata( + &self, + pinned: &SnapshotContext, + paths: &[String], + ) -> Result { + if paths.len() > 128 { + return Err(limit("at most 128 paths per lookup")); + } + for path in paths { + validate_scope_relative_path(path)?; + let absolute = if pinned.built.descriptor.scope == "/" { + path.clone() + } else if path == "/" { + pinned.built.descriptor.scope.clone() + } else { + format!("{}{}", pinned.built.descriptor.scope, path) + }; + validate_scope_relative_path(&absolute)?; + } + let (operation, binding, source) = self.admit_reader(pinned).await?; + let result = async { + let txn = self.read_transaction().await?; + let read = async { + let root = pinned.built.descriptor.metadata_root; + let mut reader = Reader::new(&txn, operation, root, binding, source); + let mut results = Vec::with_capacity(paths.len()); + for path in paths { + results.push(reader.lookup(root, path).await?); + } + Ok(RootedLookupBatch { + results, + proof_pages: reader.proofs()?, + }) + } + .await; + sessions::finish(txn, read).await + } + .await; + self.finish_reader(operation, result).await + } + pub(crate) async fn metadata_routes( + &self, + pinned: &SnapshotContext, + requests: &[MetadataRouteRequest<'_>], + ) -> Result { + if requests.is_empty() || requests.len() > 64 { + return Err(limit("items must hold 1..64 entries")); + } + for request in requests { + validate_scope_relative_path(request.directory_path)?; + let scope = &pinned.built.descriptor.scope; + let absolute = if scope == "/" { + request.directory_path.to_owned() + } else if request.directory_path == "/" { + scope.clone() + } else { + format!("{scope}{}", request.directory_path) + }; + validate_scope_relative_path(&absolute)?; + } + // Acquire REQUEST and READER ownership atomically, then release the + // mutation barrier before fetching payloads. Cleanup never precedes + // ownership of the returned byte buffers. + let (operation, binding, source) = self.admit_reader(pinned).await?; + let result = self + .read_routes(pinned, requests, operation, binding, source) + .await; + self.finish_reader(operation, result).await + } + + async fn admit_reader( + &self, + pinned: &SnapshotContext, + ) -> Result<(uuid::Uuid, Binding, uuid::Uuid), SnapshotError> { + let txn = self.transaction().await?; + let admitted = async { + let row = self + .session_row( + &txn, + &pinned.built.snapshot_id, + &pinned.lease_id, + &pinned.built.instance_id, + ) + .await?; + let current = sessions::context( + &row, + &pinned.built.snapshot_id, + &pinned.lease_id, + &pinned.built.instance_id, + )?; + if current.built.descriptor != pinned.built.descriptor + || current.commit_oid != pinned.commit_oid + || current.root_tree_oid != pinned.root_tree_oid + || current.authorization_epoch != pinned.authorization_epoch + { + return Err(integrity( + "qualified fixed session changed during reader admission", + )); + } + let reader = txn + .query_one_raw(sql( + "SELECT operation_id::text,root_generation,certificate_digest + FROM mst2_metadata_begin_reader($1,$2,$3)", + [ + pinned.built.snapshot_id.clone().into(), + pinned.lease_id.clone().into(), + pinned.built.instance_id.clone().into(), + ], + )) + .await + .map_err(database_error)? + .ok_or_else(|| unavailable("qualified reader admission returned no owned root"))?; + Ok(( + uuid::Uuid::parse_str( + &reader + .try_get::("", "operation_id") + .map_err(internal)?, + ) + .map_err(internal)?, + Binding { + generation: reader.try_get("", "root_generation").map_err(internal)?, + certificate: digest_column(&reader, "certificate_digest")?, + }, + row.try_get("", "attestation_id").map_err(internal)?, + )) + } + .await; + let owned = sessions::finish(txn, admitted).await?; + #[cfg(test)] + if let Ok((admitted, resume)) = READER_ADMISSION_BARRIERS.try_with(Clone::clone) { + admitted.wait().await; + resume.wait().await; + } + Ok(owned) + } + + async fn finish_reader( + &self, + operation: uuid::Uuid, + result: Result, + ) -> Result { + let cleanup = self.transaction().await; + let cleanup = match cleanup { + Ok(txn) => { + let finished = txn + .execute_raw(sql( + "SELECT mst2_metadata_finish_reader($1::uuid)", + [operation.to_string().into()], + )) + .await + .map_err(internal) + .map(|_| ()); + sessions::finish(txn, finished).await + } + Err(error) => Err(error), + }; + // A failed cleanup leaves durable roots until the bounded hard deadline. + // Report the failure instead of pretending that a protected operation + // has been definitively finished. + match (result, cleanup) { + (Err(error), _) => Err(error), + (Ok(_), Err(error)) => Err(error), + (Ok(value), Ok(())) => Ok(value), + } + } + + async fn read_routes( + &self, + pinned: &SnapshotContext, + requests: &[MetadataRouteRequest<'_>], + operation: uuid::Uuid, + binding: Binding, + source: uuid::Uuid, + ) -> Result { + let txn = self.read_transaction().await?; + let result = async { + let root = pinned.built.descriptor.metadata_root; + let mut reader = Reader::new(&txn, operation, root, binding, source); + let mut seen = BTreeSet::new(); + let mut pages = Vec::new(); + for request in requests { + let directory = reader.directory(root, request.directory_path).await?; + let route = reader.route(directory, request.route).await?; + let reached = route + .last() + .ok_or_else(|| integrity("qualified metadata route is empty"))?; + if let Some(expected) = request.expected_digest + && expected != format!("sha256:{}", hex::encode(reached)) + { + return Err(SnapshotError::new( + crate::ceres::snapshot::error::SnapshotErrorCode::DigestMismatch, + "qualified metadata route does not reach expected_digest", + )); + } + for page in route { + if seen.insert(page) { + pages.push((page, reader.cache[&page].bytes.clone())); + } + } + } + Ok(PersistedMetadataRouteBatch { + pages, + work: reader.work, + }) + } + .await; + sessions::finish(txn, result).await + } +} + +impl Reader<'_> { + fn new( + txn: &DatabaseTransaction, + operation: uuid::Uuid, + root: [u8; 32], + binding: Binding, + source: uuid::Uuid, + ) -> Reader<'_> { + Reader { + txn, + operation, + bindings: HashMap::from([(root, binding)]), + cache: HashMap::new(), + work: PersistedMetadataReadWork::default(), + limits: MetadataDagLimits::default(), + directories: BTreeMap::from([("/".into(), root)]), + source_ids: BTreeMap::from([("/".into(), source)]), + file_oids: HashMap::new(), + } + } + + async fn source_entries( + &mut self, + source: uuid::Uuid, + directory: [u8; 32], + entries: &[mst2_codec::metapage::Entry], + ) -> Result, uuid::Uuid>, SnapshotError> { + let binding = self + .bindings + .get(&directory) + .ok_or_else(|| integrity("qualified source directory has no certified binding"))?; + let names: Vec<_> = entries + .iter() + .map(|entry| hex::encode(&entry.name)) + .collect(); + let rows = self.txn.query_all_raw(sql("SELECT * FROM mst2_metadata_read_source_entries($1::uuid,$2::uuid,$3,$4,$5,$6::jsonb)", + [self.operation.to_string().into(), source.to_string().into(), directory.to_vec().into(), + binding.generation.into(), binding.certificate.to_vec().into(),json!(names).into()])).await.map_err(database_error)?; + #[cfg(test)] + if entries.iter().any(|entry| !entry.is_dir()) + && let Ok((admitted, resume)) = SOURCE_FACT_BARRIERS.try_with(Clone::clone) + { + admitted.wait().await; + resume.wait().await; + } + if rows.len() != entries.len() { + return Err(integrity( + "qualified source-name index differs from selected actual page entries", + )); + } + let mut children = HashMap::new(); + for row in rows { + let name: Vec = row.try_get("", "name").map_err(internal)?; + let entry = entries + .iter() + .find(|entry| entry.name == name) + .ok_or_else(|| { + integrity("qualified source-name read returned an unrequested occurrence") + })?; + let state: String = row.try_get("", "fact_state").map_err(internal)?; + match state.as_str() { + "READY" => {} + "MISSING" => { + return Err(SnapshotError::new( + crate::ceres::snapshot::error::SnapshotErrorCode::MetadataNotReady, + "selected fixed file has no current verified object metadata", + )); + } + "SOURCE_UNAVAILABLE" => { + return Err(unavailable( + "selected fixed directory has no current exact source attestation", + )); + } + _ => { + return Err(integrity( + "selected fixed file has invalid or different current verified object metadata", + )); + } + } + let kind: i16 = row.try_get("", "kind").map_err(internal)?; + if kind as u8 != entry.kind as u8 { + return Err(integrity( + "qualified selected occurrence changed its fixed filesystem kind", + )); + } + if entry.is_dir() { + let child = self.bindings.get(&entry.child_root).ok_or_else(|| { + integrity("qualified source child has no certified occurrence") + })?; + let root = digest_column(&row, "child_root")?; + if root != entry.child_root + || child.generation + != row + .try_get::("", "child_generation") + .map_err(internal)? + || child.certificate != digest_column(&row, "child_certificate_digest")? + { + return Err(integrity( + "qualified source-name directory differs from its exact certified lifetime", + )); + } + children.insert( + name.clone(), + row.try_get("", "child_attestation_id").map_err(internal)?, + ); + } else if row.try_get::("", "byte_size").map_err(internal)? as u64 != entry.size + || digest_column(&row, "content_digest")? != entry.content_id + { + return Err(integrity( + "qualified source-name file differs from its certified actual page", + )); + } + if !entry.is_dir() { + self.file_oids.insert( + (source, name), + row.try_get("", "git_oid").map_err(internal)?, + ); + } + } + Ok(children) + } + + fn proofs(&self) -> Result)>, SnapshotError> { + let bytes: usize = self.cache.values().map(|page| page.bytes.len()).sum(); + if bytes > 1_048_576 { + return Err(SnapshotError::new( + crate::ceres::snapshot::error::SnapshotErrorCode::ProofBudgetExceeded, + "qualified proof pages exceed the response budget; use metadata/pages", + )); + } + let mut pages: Vec<_> = self + .cache + .iter() + .map(|(id, page)| (*id, page.bytes.clone())) + .collect(); + pages.sort_by_key(|item| item.0); + Ok(pages) + } + + async fn range( + &mut self, + id: [u8; 32], + after: Option<&[u8]>, + count: usize, + out: &mut Vec, + ) -> Result<(), SnapshotError> { + if out.len() >= count { + return Ok(()); + } + self.load(id).await?; + match self.cache[&id].page.clone() { + Page::Leaf { entries } => { + let start = after + .map(|name| entries.partition_point(|entry| entry.name.as_slice() <= name)) + .unwrap_or(0); + out.extend(entries.into_iter().skip(start).take(count - out.len())); + } + Page::Branch { + prefix, + terminal, + children, + } => { + if let Some(entry) = terminal + && after.is_none_or(|name| entry.name.as_slice() > name) + { + out.push(entry); + } + for child in children { + if out.len() >= count { + break; + } + let mut partition = prefix.clone(); + partition.push(child.label); + if after.is_some_and(|name| { + partition.as_slice() < name && !name.starts_with(&partition) + }) { + continue; + } + self.descend(id, child.label).await?; + Box::pin(self.range(child.child_page_id, after, count, out)).await?; + } + } + } + Ok(()) + } + + async fn find_entry( + &mut self, + mut id: [u8; 32], + name: &[u8], + ) -> Result, SnapshotError> { + loop { + self.load(id).await?; + match &self.cache[&id].page { + Page::Leaf { entries } => { + return Ok(entries.iter().find(|entry| entry.name == name).cloned()); + } + Page::Branch { + prefix, + terminal, + children, + } => { + if name == prefix { + return Ok(terminal.clone()); + } + if !name.starts_with(prefix) { + return Ok(None); + } + let Some(label) = name.get(prefix.len()).copied() else { + return Ok(None); + }; + if !children.iter().any(|child| child.label == label) { + return Ok(None); + } + id = self.descend(id, label).await?; + } + } + } + } + + async fn lookup( + &mut self, + mut root: [u8; 32], + path: &str, + ) -> Result { + if path == "/" { + self.load(root).await?; + return Ok(RootedLookupStatus::Directory(root)); + } + let mut prefix = String::new(); + let mut source = self.source_ids["/"]; + let mut parts = path[1..].split('/').peekable(); + while let Some(name) = parts.next() { + let Some(entry) = self.find_entry(root, name.as_bytes()).await? else { + return Ok(RootedLookupStatus::Absent); + }; + if !entry.is_dir() { + if parts.peek().is_some() { + return Ok(RootedLookupStatus::NotDirectory { + symlink: entry.kind == mst2_codec::metapage::EntryKind::Symlink, + }); + } + self.source_entries(source, root, std::slice::from_ref(&entry)) + .await?; + let git_oid = self + .file_oids + .get(&(source, entry.name.clone())) + .cloned() + .ok_or_else(|| { + integrity("selected fixed file has no exact source OID binding") + })?; + return Ok(RootedLookupStatus::File { entry, git_oid }); + } + let children = self + .source_entries(source, root, std::slice::from_ref(&entry)) + .await?; + source = *children.get(&entry.name).ok_or_else(|| { + integrity("qualified named directory has no independently derived source child") + })?; + root = entry.child_root; + prefix.push('/'); + prefix.push_str(name); + self.directories.insert(prefix.clone(), root); + self.source_ids.insert(prefix.clone(), source); + } + self.load(root).await?; + Ok(RootedLookupStatus::Directory(root)) + } + async fn load(&mut self, id: [u8; 32]) -> Result<(), SnapshotError> { + self.work.walk_visits += 1; + if self.work.walk_visits > self.limits.prepare_entry_visits as u64 { + return Err(limit("qualified route work budget exceeded")); + } + if self.cache.contains_key(&id) { + return Ok(()); + } + if self.cache.len() >= self.limits.nodes { + return Err(limit("qualified route page budget exceeded")); + } + let binding = *self.bindings.get(&id).ok_or_else(|| { + integrity("qualified page was not reached through a certified occurrence") + })?; + self.work.page_queries += 1; + let row = self + .txn + .query_one_raw(sql( + PAGE_SQL, + [ + id.to_vec().into(), + binding.generation.into(), + binding.certificate.to_vec().into(), + self.operation.to_string().into(), + ], + )) + .await + .map_err(internal)? + .ok_or_else(|| { + unavailable("qualified certified page or reader ownership is unavailable") + })?; + let bytes: Vec = row.try_get("", "payload").map_err(internal)?; + let size: i32 = row.try_get("", "byte_size").map_err(internal)?; + if !(HEADER_LEN..=PAGE_MAX_BYTES).contains(&bytes.len()) + || bytes.len() != size as usize + || page_id(&bytes) != id + { + return Err(integrity( + "qualified durable metadata bytes differ from their certified page", + )); + } + let (page, entries) = Page::decode(&bytes).map_err(internal)?; + self.work.payload_bytes = self + .work + .payload_bytes + .checked_add(bytes.len() as u64) + .filter(|value| *value <= self.limits.payload_bytes) + .ok_or_else(|| limit("qualified route byte budget exceeded"))?; + let raw_refs: serde_json::Value = row.try_get("", "references").map_err(internal)?; + let refs: Vec = serde_json::from_value(raw_refs).map_err(internal)?; + if refs.len() > 257 { + return Err(integrity("qualified canonical reference count is invalid")); + } + let mut expected = Vec::new(); + let direct = match &page { + Page::Leaf { entries } => entries.as_slice(), + Page::Branch { + terminal, children, .. + } => { + for child in children { + expected.push(( + "RADIX", + child.child_page_id, + None, + Some(child.label), + Some(child.subtree_entries), + )); + } + terminal.as_slice() + } + }; + // Reference ordering is direct DIRECTORY occurrences followed by RADIX, + // as defined by the independent canonical parser. Compare occurrences + // as a set here; SQL validates the certificate's ordinal completeness. + for entry in direct.iter().filter(|entry| entry.is_dir()) { + expected.push(( + "DIRECTORY", + entry.child_root, + Some(hex::encode(&entry.name)), + None, + None, + )); + } + let mut actual = BTreeSet::new(); + for reference in refs { + let child = digest_hex(&reference.child)?; + let binding = Binding { + generation: reference.generation, + certificate: digest_hex(&reference.certificate)?, + }; + if binding.generation <= 0 + || self + .bindings + .get(&child) + .is_some_and(|known| *known != binding) + { + return Err(integrity( + "qualified occurrence retargeted its exact lifetime or certificate", + )); + } + self.bindings.insert(child, binding); + if !actual.insert(( + reference.kind, + child, + reference.name, + reference.label, + reference.count, + )) { + return Err(integrity("qualified typed occurrence has a duplicate")); + } + } + let expected: BTreeSet<_> = expected + .into_iter() + .map(|(kind, child, name, label, count)| (kind.to_owned(), child, name, label, count)) + .collect(); + if actual != expected { + return Err(integrity( + "qualified certificate references differ from its actual durable page", + )); + } + self.work.edge_references_checked += actual.len() as u64; + if self.work.edge_references_checked > self.limits.edges as u64 { + return Err(limit("qualified route reference budget exceeded")); + } + self.work.pages_loaded += 1; + self.cache.insert( + id, + StoredPage { + bytes, + page, + entries, + }, + ); + Ok(()) + } + + async fn descend(&mut self, parent: [u8; 32], label: u8) -> Result<[u8; 32], SnapshotError> { + self.load(parent).await?; + let (prefix, child) = match &self.cache[&parent].page { + Page::Branch { + prefix, children, .. + } => ( + prefix.clone(), + children + .iter() + .find(|child| child.label == label) + .ok_or_else(|| absent("route label is absent in fixed qualified metadata"))? + .clone(), + ), + Page::Leaf { .. } => return Err(absent("route descends past a qualified leaf")), + }; + let id = child.child_page_id; + self.load(id).await?; + let received = &self.cache[&id]; + let mut partition = prefix; + partition.push(label); + let valid = match &received.page { + Page::Leaf { entries } => entries + .iter() + .all(|entry| entry.name.starts_with(&partition)), + Page::Branch { prefix, .. } => prefix.starts_with(&partition), + }; + if !valid || received.entries != child.subtree_entries { + return Err(integrity( + "qualified radix partition differs from its exact child", + )); + } + Ok(id) + } + + async fn directory( + &mut self, + mut root: [u8; 32], + path: &str, + ) -> Result<[u8; 32], SnapshotError> { + if path == "/" { + return Ok(root); + } + let mut directory_path = String::new(); + let mut source = self.source_ids["/"]; + for name in path[1..].split('/') { + let mut id = root; + loop { + self.load(id).await?; + let entry = match &self.cache[&id].page { + Page::Leaf { entries } => entries + .iter() + .find(|entry| entry.name == name.as_bytes()) + .cloned(), + Page::Branch { + prefix, terminal, .. + } => { + if name.as_bytes() == prefix { + terminal.clone() + } else { + if !name.as_bytes().starts_with(prefix) { + return Err(absent("name is absent in qualified directory")); + } + let label = + name.as_bytes().get(prefix.len()).copied().ok_or_else(|| { + absent("name is absent in qualified directory") + })?; + id = self.descend(id, label).await?; + continue; + } + } + } + .ok_or_else(|| absent("name is absent in qualified directory"))?; + if !entry.is_dir() { + return Err(SnapshotError::new( + crate::ceres::snapshot::error::SnapshotErrorCode::NotDirectory, + "fixed qualified path is not a directory", + )); + } + let children = self + .source_entries(source, root, std::slice::from_ref(&entry)) + .await?; + source = *children.get(&entry.name).ok_or_else(|| { + integrity("qualified directory route lost its exact source-name binding") + })?; + root = entry.child_root; + directory_path.push('/'); + directory_path.push_str(name); + self.directories.insert(directory_path.clone(), root); + self.source_ids.insert(directory_path.clone(), source); + break; + } + } + Ok(root) + } + async fn route( + &mut self, + root: [u8; 32], + labels: &[u8], + ) -> Result, SnapshotError> { + self.load(root).await?; + let mut pages = vec![root]; + let mut current = root; + for label in labels { + current = self.descend(current, *label).await?; + pages.push(current); + } + Ok(pages) + } +} + +const PAGE_SQL:&str="SELECT body.payload,body.byte_size,proof.canonical_proof->'references' AS references + FROM mst2_metadata_page_certificate proof JOIN mst2_metadata_current cur USING(page_id,generation) + JOIN mst2_metadata_lifetime life USING(page_id,generation) JOIN mst2_metadata_graph_node node USING(page_id,generation) + JOIN mst2_metadata_payload body USING(page_id,generation) + WHERE proof.page_id=$1 AND proof.generation=$2 AND proof.certificate_digest=$3 + AND proof.namespace_uuid=(SELECT namespace_uuid FROM mst2_metadata_family_identity WHERE singleton=1) + AND life.state='LIVE' AND life.graph_domain='qualified-v1' AND node.state='LIVE' + AND node.certificate_digest=proof.certificate_digest AND node.bytes=proof.byte_size + AND life.expected_size=proof.byte_size AND body.byte_size=proof.byte_size AND body.metadata_codec=1 + AND octet_length(body.payload)=proof.byte_size AND NOT EXISTS(SELECT 1 FROM mst2_metadata_gc_op gc + WHERE gc.page_id=proof.page_id AND gc.generation=proof.generation) + AND EXISTS(SELECT 1 FROM mst2_metadata_reader_operation reader JOIN mst2_metadata_root_anchor anchor + ON anchor.reader_operation_id=reader.operation_id AND anchor.anchor_kind='READER' + AND anchor.root_page=reader.root_page AND anchor.root_generation=reader.root_generation + WHERE reader.operation_id=$4::uuid AND reader.state='ACTIVE' + AND reader.hard_deadline_unix>floor(extract(epoch FROM clock_timestamp()))::bigint) + AND (SELECT count(*) FROM mst2_metadata_verified_ref ref WHERE ref.parent_page=proof.page_id + AND ref.parent_generation=proof.generation)=jsonb_array_length(proof.canonical_proof->'references') + AND NOT EXISTS((SELECT ref.child_page,ref.child_generation FROM mst2_metadata_verified_ref ref + WHERE ref.parent_page=proof.page_id AND ref.parent_generation=proof.generation) EXCEPT + (SELECT edge.child_page,edge.child_generation FROM mst2_metadata_graph_edge edge + WHERE edge.parent_page=proof.page_id AND edge.parent_generation=proof.generation)) + AND NOT EXISTS((SELECT edge.child_page,edge.child_generation FROM mst2_metadata_graph_edge edge + WHERE edge.parent_page=proof.page_id AND edge.parent_generation=proof.generation) EXCEPT + (SELECT ref.child_page,ref.child_generation FROM mst2_metadata_verified_ref ref + WHERE ref.parent_page=proof.page_id AND ref.parent_generation=proof.generation))"; diff --git a/src/jupiter/storage/qualified_metadata_rooted.rs b/src/jupiter/storage/qualified_metadata_rooted.rs new file mode 100644 index 00000000..5dc2f4be --- /dev/null +++ b/src/jupiter/storage/qualified_metadata_rooted.rs @@ -0,0 +1,766 @@ +use std::{collections::BTreeMap, sync::Arc}; + +use mst2_codec::metapage::{Page, page_id}; +use sea_orm::{QueryResult, Value}; +use serde_json::json; + +use super::*; +use crate::ceres::snapshot::{ + metadata_install::MetadataInstallIdentity, + rooted_metadata_install::{RootedMetadataInstallPlan, RootedReuseRoot}, + rooted_metadata_projection::{CertifiedReusableDirectory, RootedReuseLookup}, +}; + +#[path = "qualified_metadata_session.rs"] +mod sessions; + +#[path = "qualified_metadata_gc.rs"] +mod gc; +#[path = "qualified_metadata_reader.rs"] +mod reader; +pub(crate) use reader::{RootedDirectoryWindow, RootedLookupBatch, RootedLookupStatus}; +#[cfg(test)] +pub(crate) use reader::{ + with_rooted_reader_barriers, with_rooted_source_fact_barriers, + with_rooted_source_temporary_shadow, +}; + +#[derive(Debug, Clone, PartialEq, Eq)] +pub(crate) struct RootedPrepareIntent { + prepare_id: String, + operation_id: String, + plan: Arc, + bindings: BTreeMap<[u8; 32], (i64, u64)>, + manifest_digest: [u8; 32], + bindings_digest: [u8; 32], + primary_scope: Vec, + storage_seal: [u8; 32], + root_generation: i64, +} + +#[derive(Debug, Clone, PartialEq, Eq)] +pub(crate) struct RootedMetadataReceipt { + intent: RootedPrepareIntent, + certificate_digest: [u8; 32], + attestation_id: uuid::Uuid, + attestation_digest: [u8; 32], +} + +impl RootedPrepareIntent { + #[cfg(test)] + pub(crate) fn prepare_id(&self) -> &str { + &self.prepare_id + } + #[cfg(test)] + pub(crate) fn metadata_root(&self) -> [u8; 32] { + self.plan.root + } + #[cfg(test)] + pub(crate) fn root_generation(&self) -> i64 { + self.root_generation + } +} +impl RootedMetadataReceipt { + pub(crate) fn metadata_root(&self) -> [u8; 32] { + self.intent.plan.root + } +} + +pub(crate) struct RootedQualifiedMetadataRepository { + connection: DatabaseConnection, + namespace: VerifiedQualifiedNamespace, + primary_scope: Vec, + maintenance_state: tokio::sync::Mutex, +} + +#[async_trait::async_trait] +impl RootedReuseLookup for RootedQualifiedMetadataRepository { + async fn lookup_reuse( + &self, + tree_oid: &str, + identity: &MetadataInstallIdentity, + ) -> Result, SnapshotError> { + // This is a bounded projection hint. It acquires no writer/retention + // lock and grants no installation authority: begin_intent and finalize + // independently bind the exact attestation and current lifetime. + let kind = identity + .tagged_root_tree_oid + .split_once(':') + .ok_or_else(|| integrity("rooted source identity has no tagged hash kind"))? + .0; + let profile = json!({"source_domain": identity.source_domain,"hash_kind": kind, + "schema_version": identity.schema_version,"metadata_codec": identity.metadata_codec, + "materialization_policy": identity.materialization_policy,"fs_semantics": identity.fs_semantics, + "access_projection": identity.access_projection,"verification_revision": identity.verification_revision, + "projection_revision": identity.projection_revision}); + let q = identifier(&self.namespace.schema); + let c = identifier(&self.namespace.core_schema); + let row = self.connection.query_one_raw(sql(format!( + "SELECT a.root_page,a.root_generation,a.attestation_id::text,a.attestation_digest, + p.certificate_digest,p.relative_path_bytes,p.relative_components,p.closure_nodes_upper, + p.closure_edges_upper,p.closure_bytes_upper,p.closure_entries_upper + FROM {q}.mst2_metadata_reuse_index i JOIN {q}.mst2_metadata_source_root_attestation a + ON a.attestation_id=i.attestation_id AND a.root_page=i.root_page AND a.root_generation=i.root_generation + AND a.attestation_digest=i.attestation_digest + JOIN {q}.mst2_metadata_page_certificate p ON p.page_id=a.root_page AND p.generation=a.root_generation + AND p.certificate_digest=a.root_certificate_digest + JOIN {q}.mst2_metadata_current cur ON cur.page_id=p.page_id AND cur.generation=p.generation + JOIN {q}.mst2_metadata_lifetime life ON life.page_id=p.page_id AND life.generation=p.generation + JOIN {q}.mst2_metadata_graph_node node ON node.page_id=p.page_id AND node.generation=p.generation + JOIN {q}.mst2_metadata_payload body ON body.page_id=p.page_id AND body.generation=p.generation + JOIN {q}.mst2_metadata_prepare origin ON origin.prepare_id=a.origin_prepare_id + JOIN {c}.mega_tree source ON source.tree_id=split_part(a.tagged_tree_oid,':',2) + WHERE i.tagged_tree_oid=$1 AND i.profile_digest=sha256(convert_to('mega.mst2.native-profile.v1','UTF8') + ||decode('00','hex')||convert_to($2::jsonb::text,'UTF8')) AND a.source_profile=$2::jsonb + AND a.namespace_uuid=$3::uuid AND {c}.mst2_route_source_tree_matches(split_part(a.tagged_tree_oid,':',2),a.source_revision,a.source_body_digest) + AND life.state='LIVE' AND life.graph_domain='qualified-v1' AND node.state='LIVE' + AND node.certificate_digest=p.certificate_digest AND node.bytes=p.byte_size + AND body.byte_size=p.byte_size AND body.metadata_codec=p.metadata_codec + AND life.expected_size=p.byte_size AND origin.state='COMMITTED' AND NOT pg_is_in_recovery() + AND NOT EXISTS(SELECT 1 FROM {q}.mst2_metadata_gc_op gc WHERE gc.page_id=p.page_id AND gc.generation=p.generation)" + ),[tree_oid.into(),profile.into(),self.namespace.namespace_uuid.clone().into()])) + .await.map_err(database_error)?; + let Some(row) = row else { + return Ok(None); + }; + let count = |name: &str| -> Result { + usize::try_from(row.try_get::("", name).map_err(internal)?).map_err(internal) + }; + let wide = |name: &str| -> Result { + u64::try_from(row.try_get::("", name).map_err(internal)?).map_err(internal) + }; + Ok(Some(CertifiedReusableDirectory { + page_id: digest_column(&row, "root_page")?, + proof: RootedReuseRoot { + generation: row.try_get("", "root_generation").map_err(internal)?, + attestation_id: uuid::Uuid::parse_str( + &row.try_get::("", "attestation_id") + .map_err(internal)?, + ) + .map_err(internal)?, + attestation_digest: digest_column(&row, "attestation_digest")?, + certificate_digest: digest_column(&row, "certificate_digest")?, + }, + relative_path_bytes: count("relative_path_bytes")?, + relative_components: count("relative_components")?, + closure_nodes_upper: count("closure_nodes_upper")?, + closure_edges_upper: count("closure_edges_upper")?, + closure_bytes_upper: wide("closure_bytes_upper")?, + closure_entries_upper: usize::try_from(wide("closure_entries_upper")?) + .map_err(internal)?, + })) + } +} + +fn sql(text: impl Into, values: impl IntoIterator) -> Statement { + Statement::from_sql_and_values(DbBackend::Postgres, text, values) +} +fn internal(error: impl std::fmt::Display) -> SnapshotError { + SnapshotError::new( + crate::ceres::snapshot::error::SnapshotErrorCode::Internal, + error.to_string(), + ) +} +pub(super) fn is_lock_unavailable(error: &sea_orm::DbErr) -> bool { + let runtime = match error { + sea_orm::DbErr::Exec(runtime) | sea_orm::DbErr::Query(runtime) => runtime, + _ => return false, + }; + let sea_orm::RuntimeErr::SqlxError(sqlx_error) = runtime else { + return false; + }; + let sea_orm::sqlx::Error::Database(database_error) = sqlx_error.as_ref() else { + return false; + }; + database_error.code().as_deref() == Some("55P03") +} +fn database_error(error: sea_orm::DbErr) -> SnapshotError { + if is_lock_unavailable(&error) { + return SnapshotError::new( + crate::ceres::snapshot::error::SnapshotErrorCode::TemporaryUnavailable, + "qualified source is being updated; retry the operation", + ); + } + internal(error) +} +fn integrity(message: &str) -> SnapshotError { + SnapshotError::new( + crate::ceres::snapshot::error::SnapshotErrorCode::IntegrityError, + message, + ) +} +fn unavailable(message: &str) -> SnapshotError { + SnapshotError::new( + crate::ceres::snapshot::error::SnapshotErrorCode::ObjectUnavailable, + message, + ) +} +fn digest_column(row: &QueryResult, name: &str) -> Result<[u8; 32], SnapshotError> { + let bytes: Vec = row.try_get("", name).map_err(internal)?; + bytes.as_slice().try_into().map_err(internal) +} +fn encode_bindings(bindings: &BTreeMap<[u8; 32], (i64, u64)>) -> Vec { + let mut bytes = Vec::with_capacity(12 + 48 * bindings.len()); + bytes.extend_from_slice(b"MST2GEN1"); + bytes.extend_from_slice(&(bindings.len() as u32).to_be_bytes()); + for (page, (generation, size)) in bindings { + bytes.extend_from_slice(page); + bytes.extend_from_slice(&generation.to_be_bytes()); + bytes.extend_from_slice(&size.to_be_bytes()); + } + bytes +} +fn decode_bindings( + bytes: &[u8], + plan: &RootedMetadataInstallPlan, +) -> Result, SnapshotError> { + if bytes.len() != 12 + 48 * plan.delta.len() || bytes.get(..8) != Some(b"MST2GEN1") { + return Err(integrity( + "rooted delta binding encoding differs from its plan", + )); + } + let count = u32::from_be_bytes(bytes[8..12].try_into().map_err(internal)?) as usize; + if count != plan.delta.len() { + return Err(integrity( + "rooted delta binding count differs from its plan", + )); + } + let mut bindings = BTreeMap::new(); + for ((expected, size), record) in plan.delta.iter().zip(bytes[12..].as_chunks::<48>().0) { + let page: [u8; 32] = record[..32].try_into().map_err(internal)?; + let generation = i64::from_be_bytes(record[32..40].try_into().map_err(internal)?); + let recorded_size = u64::from_be_bytes(record[40..48].try_into().map_err(internal)?); + if page != *expected || recorded_size != *size || generation <= 0 { + return Err(integrity("rooted exact delta lifetime binding is invalid")); + } + bindings.insert(page, (generation, recorded_size)); + } + Ok(bindings) +} + +async fn committed( + txn: DatabaseTransaction, + result: Result, + operation: &str, + digest: [u8; 32], + phase: super::super::native_metadata_install::MetadataCommitPhase, +) -> Result { + match result { + Err(error) => { + let _ = txn.rollback().await; + Err(error.into()) + } + Ok(value) => match txn.commit().await { + Ok(()) => Ok(value), + Err(_) => Err(MetadataInstallError::CommitUncertain { + operation_id: operation.into(), + manifest_digest: digest, + phase, + }), + }, + } +} + +impl RootedQualifiedMetadataRepository { + pub(crate) async fn open( + core: &DatabaseConnection, + config: &DbConfig, + ) -> Result { + let captured = captured_core(core).await?; + let namespace = registered(core, &captured) + .await? + .ok_or_else(|| rejected("rooted qualified family is not provisioned"))?; + let mut q_config = config.clone(); + q_config.db_url = pool_url(&config.db_url, &namespace)?; + q_config.max_connection = q_config.max_connection.clamp(1, 4); + q_config.min_connection = 1; + let connection = postgres_connection(&q_config).await?; + let txn = connection + .begin_with_config(Some(IsolationLevel::ReadCommitted), None) + .await?; + namespace + .enter(&txn) + .await + .map_err(|e| rejected(&e.to_string()))?; + let scope:serde_json::Value=txn.query_one_raw(sql( + "SELECT jsonb_build_array(s.storage_uuid,current_database(),d.oid::bigint,current_schema(),n.oid::bigint, + inet_server_addr()::text,inet_server_port()) AS scope FROM mst2_metadata_storage_scope s + JOIN pg_catalog.pg_database d ON d.datname=current_database() + JOIN pg_catalog.pg_namespace n ON n.nspname=current_schema() WHERE s.singleton=1 AND NOT pg_is_in_recovery()",[], + )).await?.ok_or_else(||rejected("rooted captured primary scope is missing"))?.try_get("","scope")?; + let primary_scope = serde_json::to_vec(&scope).map_err(|e| rejected(&e.to_string()))?; + txn.commit().await?; + Ok(Self { + connection, + namespace, + primary_scope, + maintenance_state: tokio::sync::Mutex::default(), + }) + } + + async fn transaction(&self) -> Result { + let txn = self + .connection + .begin_with_config(Some(IsolationLevel::ReadCommitted), None) + .await + .map_err(database_error)?; + self.namespace.enter(&txn).await?; + let valid: bool = txn + .query_one_raw(sql( + "SELECT mst2_metadata_scope_matches($1) AS valid", + [self.primary_scope.clone().into()], + )) + .await + .map_err(database_error)? + .ok_or_else(|| integrity("rooted primary scope is missing"))? + .try_get("", "valid") + .map_err(internal)?; + if !valid { + return Err(integrity("rooted writer left its captured primary scope")); + } + Ok(txn) + } + + async fn load_intent( + &self, + db: &C, + operation: &str, + plan: &RootedMetadataInstallPlan, + ) -> Result, SnapshotError> { + let Some(row)=db.query_one_raw(sql("SELECT prepare_id,plan_kind,state,manifest_digest,canonical_plan,canonical_bindings, + bindings_digest,primary_scope,storage_seal FROM mst2_metadata_prepare WHERE operation_id=$1",[operation.into()])) + .await.map_err(internal)? else {return Ok(None);}; + let digest = plan.digest()?; + if row.try_get::("", "plan_kind").map_err(internal)? != "ROOTED" + || digest_column(&row, "manifest_digest")? != digest + { + return Err(SnapshotError::new( + crate::ceres::snapshot::error::SnapshotErrorCode::Conflict, + "operation is bound to a different rooted manifest", + )); + } + let stored: Vec = row.try_get("", "canonical_plan").map_err(internal)?; + if RootedMetadataInstallPlan::decode(&stored, &digest)? != *plan { + return Err(integrity( + "rooted stored plan differs from its exact source identity", + )); + } + let encoded_bindings: Vec = row.try_get("", "canonical_bindings").map_err(internal)?; + let bindings_digest: [u8; 32] = Sha256::digest(&encoded_bindings).into(); + if bindings_digest != digest_column(&row, "bindings_digest")? { + return Err(integrity( + "rooted stored lifetime binding digest is invalid", + )); + } + let primary_scope: Vec = row.try_get("", "primary_scope").map_err(internal)?; + if primary_scope != self.primary_scope { + return Err(integrity( + "rooted operation belongs to another physical primary", + )); + } + let bindings = decode_bindings(&encoded_bindings, plan)?; + let root_generation = bindings + .get(&plan.root) + .map(|b| b.0) + .or_else(|| plan.reused.get(&plan.root).map(|b| b.generation)) + .ok_or_else(|| integrity("rooted root lifetime is missing"))?; + Ok(Some(( + RootedPrepareIntent { + prepare_id: row.try_get("", "prepare_id").map_err(internal)?, + operation_id: operation.into(), + plan: Arc::new(plan.clone()), + bindings, + manifest_digest: digest, + bindings_digest, + primary_scope, + storage_seal: digest_column(&row, "storage_seal")?, + root_generation, + }, + row.try_get("", "state").map_err(internal)?, + ))) + } + + pub(crate) async fn begin_intent( + &self, + operation: &str, + plan: &RootedMetadataInstallPlan, + ) -> Result { + plan.validate()?; + if operation.is_empty() || operation.len() > 255 || operation.contains('\0') { + return Err(integrity("invalid rooted operation ID").into()); + } + let manifest_digest = plan.digest()?; + let txn = self.transaction().await?; + let result=async { + if let Some((intent,state))=self.load_intent(&txn,operation,plan).await? { + if state=="ABORTED" {return Err(unavailable("rooted preparation was definitively aborted"));} + return Ok(intent); + } + let pages:Vec<_>=plan.delta.iter().map(|(page,size)|json!({"page":hex::encode(page),"size":size})).collect(); + let encoded=serde_json::to_string(&pages).map_err(internal)?; + txn.execute_raw(sql("INSERT INTO mst2_metadata_lifetime(page_id,node_id,generation,state,metadata_codec,expected_size,graph_domain) + SELECT decode(p.page,'hex'),'page:sha256:'||p.page,coalesce(cur.generation+1,1),'RESERVED',1,p.size,'qualified-v1' + FROM jsonb_to_recordset($1::jsonb) p(page text,size integer) + LEFT JOIN mst2_metadata_current cur ON cur.page_id=decode(p.page,'hex') + LEFT JOIN mst2_metadata_lifetime previous ON previous.page_id=cur.page_id AND previous.generation=cur.generation + WHERE (cur.page_id IS NULL AND NOT EXISTS(SELECT 1 FROM mst2_metadata_lifetime life WHERE life.page_id=decode(p.page,'hex'))) + OR (previous.state='REMOVED' AND cur.generation<9223372036854775807 + AND EXISTS(SELECT 1 FROM mst2_metadata_gc_op proof WHERE proof.page_id=cur.page_id + AND proof.generation=cur.generation AND proof.state='APPLIED'))", + [encoded.clone().into()])).await.map_err(internal)?; + txn.execute_raw(sql("INSERT INTO mst2_metadata_current(page_id,generation) + SELECT life.page_id,life.generation FROM jsonb_to_recordset($1::jsonb) p(page text,size integer) + JOIN mst2_metadata_lifetime life ON life.page_id=decode(p.page,'hex') AND life.generation=1 AND life.state='RESERVED' + WHERE NOT EXISTS(SELECT 1 FROM mst2_metadata_current cur WHERE cur.page_id=life.page_id)", + [encoded.clone().into()])).await.map_err(internal)?; + txn.execute_raw(sql("UPDATE mst2_metadata_current cur SET generation=fresh.generation + FROM jsonb_to_recordset($1::jsonb) p(page text,size integer),mst2_metadata_lifetime previous, + mst2_metadata_lifetime fresh + WHERE cur.page_id=decode(p.page,'hex') AND previous.page_id=cur.page_id AND previous.generation=cur.generation + AND previous.state='REMOVED' AND cur.generation<9223372036854775807 + AND fresh.page_id=cur.page_id AND fresh.generation=cur.generation+1 AND fresh.state='RESERVED' + AND fresh.expected_size=p.size AND fresh.metadata_codec=1 AND fresh.graph_domain='qualified-v1' + AND EXISTS(SELECT 1 FROM mst2_metadata_gc_op proof WHERE proof.page_id=cur.page_id + AND proof.generation=cur.generation AND proof.state='APPLIED')",[encoded.clone().into()])) + .await.map_err(internal)?; + let rows=txn.query_all_raw(sql("SELECT cur.page_id,cur.generation,life.expected_size,life.metadata_codec,life.graph_domain,life.state, + n.state AS graph_state,n.certificate_digest,body.byte_size FROM jsonb_to_recordset($1::jsonb) p(page text,size integer) + JOIN mst2_metadata_current cur ON cur.page_id=decode(p.page,'hex') JOIN mst2_metadata_lifetime life USING(page_id,generation) + LEFT JOIN mst2_metadata_graph_node n USING(page_id,generation) LEFT JOIN mst2_metadata_payload body USING(page_id,generation) + WHERE NOT EXISTS(SELECT 1 FROM mst2_metadata_gc_op gc WHERE gc.page_id=cur.page_id AND gc.generation=cur.generation) + ORDER BY cur.page_id",[encoded.into()])).await.map_err(internal)?; + let mut bindings=BTreeMap::new(); + for row in rows { + let page=digest_column(&row,"page_id")?; let size=row.try_get::("","expected_size").map_err(internal)? as u64; + let state:String=row.try_get("","state").map_err(internal)?; + if plan.delta.get(&page)!=Some(&size) || row.try_get::("","metadata_codec").map_err(internal)?!=1 + || row.try_get::("","graph_domain").map_err(internal)?!="qualified-v1" || !matches!(state.as_str(),"RESERVED"|"LIVE") + || state=="LIVE" && (row.try_get::>("","graph_state").map_err(internal)?.as_deref()!=Some("LIVE") + || row.try_get::>("","byte_size").map_err(internal)?!=Some(size as i32)) { + return Err(unavailable("rooted requested delta lifetime is not exactly installable")); + } + bindings.insert(page,(row.try_get("","generation").map_err(internal)?,size)); + } + if bindings.len()!=plan.delta.len() {return Err(unavailable("rooted delta lifetime coverage is incomplete"));} + let canonical_bindings=encode_bindings(&bindings); let bindings_digest:[u8;32]=Sha256::digest(&canonical_bindings).into(); + let prepare_id=uuid::Uuid::new_v4().to_string(); + let root_generation=bindings.get(&plan.root).map(|b|b.0).or_else(||plan.reused.get(&plan.root).map(|b|b.generation)) + .ok_or_else(||integrity("rooted root lifetime is missing"))?; + let mut seal=Sha256::new(); seal.update(b"mega.mst2.rooted-storage-seal.v1\0"); + seal.update(prepare_id.as_bytes()); seal.update(manifest_digest); seal.update(bindings_digest); + seal.update((self.primary_scope.len() as u64).to_be_bytes()); seal.update(&self.primary_scope); + seal.update(plan.root); seal.update(root_generation.to_be_bytes()); + let storage_seal:[u8;32]=seal.finalize().into(); let identity=&plan.identity; + txn.execute_raw(sql("INSERT INTO mst2_metadata_prepare(prepare_id,operation_id,manifest_digest,canonical_plan,plan_kind, + source_domain,tagged_root_tree_oid,scope,schema_version,metadata_codec,materialization_policy,fs_semantics,access_projection, + verification_revision,projection_revision,metadata_root,node_count,edge_count,total_bytes,state,canonical_bindings,bindings_digest, + primary_scope,storage_seal,graph_domain) VALUES($1,$2,$3,$4,'ROOTED',$5,$6,$7,$8,$9,$10,$11,$12,$13,$14,$15,$16,$17,$18,'PREPARING',$19,$20,$21,$22,'qualified-v1')", + [prepare_id.clone().into(),operation.into(),manifest_digest.to_vec().into(),plan.encode()?.into(),identity.source_domain.clone().into(), + identity.tagged_root_tree_oid.clone().into(),identity.scope.clone().into(),(identity.schema_version as i16).into(), + (identity.metadata_codec as i16).into(),(identity.materialization_policy as i16).into(),(identity.fs_semantics as i16).into(), + (identity.access_projection as i16).into(),identity.verification_revision.into(),(identity.projection_revision as i16).into(), + plan.root.to_vec().into(),(plan.delta.len() as i32).into(),(plan.edges.len() as i32).into(),(plan.delta_bytes()? as i64).into(), + canonical_bindings.into(),bindings_digest.to_vec().into(),self.primary_scope.clone().into(),storage_seal.to_vec().into()])).await.map_err(internal)?; + let delta:Vec<_>=bindings.iter().map(|(page,(generation,size))|json!({"page":hex::encode(page),"generation":generation,"size":size})).collect(); + txn.execute_raw(sql("INSERT INTO mst2_metadata_prepare_page(prepare_id,page_id,generation,expected_size) + SELECT $1,decode(p.page,'hex'),p.generation,p.size FROM jsonb_to_recordset($2::jsonb) p(page text,generation bigint,size integer)", + [prepare_id.clone().into(),serde_json::to_string(&delta).map_err(internal)?.into()])).await.map_err(internal)?; + let reused:Vec<_>=plan.reused.iter().map(|(page,proof)|json!({"page":hex::encode(page),"generation":proof.generation, + "attestation_id":proof.attestation_id.to_string(),"attestation_digest":hex::encode(proof.attestation_digest)})).collect(); + txn.execute_raw(sql("INSERT INTO mst2_metadata_prepare_reuse_root(prepare_id,root_page,root_generation,attestation_id,attestation_digest) + SELECT $1,decode(r.page,'hex'),r.generation,r.attestation_id::uuid,decode(r.attestation_digest,'hex') + FROM jsonb_to_recordset($2::jsonb) r(page text,generation bigint,attestation_id text,attestation_digest text)", + [prepare_id.clone().into(),serde_json::to_string(&reused).map_err(internal)?.into()])).await.map_err(internal)?; + txn.execute_raw(sql("INSERT INTO mst2_metadata_root_anchor(anchor_id,anchor_kind,owner_key,prepare_id,root_page,root_generation,root_certificate_digest) + SELECT gen_random_uuid(),'REUSE',$1,$1,r.root_page,r.root_generation,a.root_certificate_digest + FROM mst2_metadata_prepare_reuse_root r JOIN mst2_metadata_source_root_attestation a USING(attestation_id) WHERE r.prepare_id=$1", + [prepare_id.clone().into()])).await.map_err(internal)?; + Ok(RootedPrepareIntent {prepare_id,operation_id:operation.into(),plan:Arc::new(plan.clone()),bindings,manifest_digest, + bindings_digest,primary_scope:self.primary_scope.clone(),storage_seal,root_generation}) + }.await; + committed( + txn, + result, + operation, + manifest_digest, + super::super::native_metadata_install::MetadataCommitPhase::Intent, + ) + .await + } + + async fn require_intent( + &self, + db: &C, + expected: &RootedPrepareIntent, + ) -> Result { + let (stored, state) = self + .load_intent(db, &expected.operation_id, &expected.plan) + .await? + .ok_or_else(|| unavailable("rooted preparation is missing"))?; + if stored != *expected { + return Err(integrity( + "rooted preparation crossed its immutable physical binding", + )); + } + if state == "ABORTED" { + return Err(unavailable("rooted preparation was definitively aborted")); + } + Ok(state) + } + + pub(crate) async fn install_pages( + &self, + intent: &RootedPrepareIntent, + payloads: &[MetadataPagePayload], + ) -> Result<(), MetadataInstallError> { + if payloads.is_empty() || payloads.len() > 64 { + return Err(integrity("rooted payload batch requires 1..=64 pages").into()); + } + let mut pages = BTreeMap::new(); + for payload in payloads { + let &(generation, size) = intent + .bindings + .get(&payload.id) + .ok_or_else(|| integrity("rooted payload is outside its fixed delta"))?; + if size != payload.size + || payload.bytes.len() as u64 != size + || page_id(&payload.bytes) != payload.id + { + return Err(integrity( + "rooted delta payload differs from its exact page digest or size", + ) + .into()); + } + Page::decode(&payload.bytes).map_err(internal)?; + if pages + .insert( + payload.id, + json!({"page":hex::encode(payload.id),"generation":generation,"size":size, + "payload":hex::encode(&payload.bytes)}), + ) + .is_some() + { + return Err(integrity("rooted payload batch has duplicate pages").into()); + } + } + let encoded = + serde_json::to_string(&pages.into_values().collect::>()).map_err(internal)?; + let txn = self.transaction().await?; + let result=async { + let state=self.require_intent(&txn,intent).await?; + if state=="PREPARING" { + txn.execute_raw(sql("INSERT INTO mst2_metadata_payload(page_id,generation,metadata_codec,byte_size,payload) + SELECT decode(p.page,'hex'),p.generation,1,p.size,decode(p.payload,'hex') + FROM jsonb_to_recordset($1::jsonb) p(page text,generation bigint,size integer,payload text) + WHERE NOT EXISTS(SELECT 1 FROM mst2_metadata_payload body WHERE body.page_id=decode(p.page,'hex')) + ON CONFLICT(page_id) DO NOTHING",[encoded.clone().into()])).await.map_err(internal)?; + } + if txn.query_one_raw(sql("SELECT p.page FROM jsonb_to_recordset($1::jsonb) p(page text,generation bigint,size integer,payload text) + LEFT JOIN mst2_metadata_payload body ON body.page_id=decode(p.page,'hex') + LEFT JOIN mst2_metadata_current cur ON cur.page_id=body.page_id AND cur.generation=body.generation + WHERE body.page_id IS NULL OR cur.page_id IS NULL OR body.generation<>p.generation OR body.metadata_codec<>1 + OR body.byte_size<>p.size OR body.payload<>decode(p.payload,'hex') LIMIT 1",[encoded.into()])) + .await.map_err(internal)?.is_some() {return Err(integrity("rooted durable delta payload conflicts with its exact incarnation"));} + Ok(()) + }.await; + committed( + txn, + result, + &intent.operation_id, + intent.manifest_digest, + super::super::native_metadata_install::MetadataCommitPhase::Payload, + ) + .await + } + + pub(crate) async fn finalize( + &self, + intent: &RootedPrepareIntent, + ) -> Result { + // Rehash/decode the actual delta outside the core route lock. A cold + // preparation additionally keeps the full Rust DAG validator as oracle. + let payloads = self.read_delta(intent).await?; + if intent.plan.reused.is_empty() { + use crate::ceres::snapshot::retention_dag::{ + MetadataDagCandidate, MetadataDagLimits, ValidatedMetadataDag, + }; + ValidatedMetadataDag::validate( + MetadataDagCandidate { + metadata_codec: intent.plan.identity.metadata_codec, + root: intent.plan.root, + pages: payloads, + edges: intent.plan.edges.iter().copied().collect(), + }, + MetadataDagLimits::default(), + )?; + } + let txn = self.transaction().await?; + let result=async { + txn.query_one_raw(sql("SELECT prepare_id FROM mst2_metadata_prepare WHERE prepare_id=$1 FOR UPDATE", + [intent.prepare_id.clone().into()])).await.map_err(internal)?; + let state=self.require_intent(&txn,intent).await?; + if state=="COMMITTED" {return self.receipt(&txn,intent).await;} + let ordered:Vec<_>=intent.plan.child_first_delta()?.iter().map(|page|json!({"page":hex::encode(page), + "generation":intent.bindings[page].0})).collect(); + if !ordered.is_empty() { + let count:i32=txn.query_one_raw(sql("SELECT mst2_metadata_certify_batch($1,$2::jsonb) AS certified", + [intent.prepare_id.clone().into(),serde_json::to_string(&ordered).map_err(internal)?.into()])) + .await.map_err(internal)?.ok_or_else(||integrity("rooted certification result is missing"))?.try_get("","certified").map_err(internal)?; + if count as usize!=ordered.len() {return Err(integrity("rooted certification did not cover its exact delta"));} + } + txn.execute_raw(sql("INSERT INTO mst2_metadata_root_anchor(anchor_id,anchor_kind,owner_key,prepare_id,root_page,root_generation,root_certificate_digest) + SELECT gen_random_uuid(),'PREPARE',$1,$1,c.page_id,c.generation,c.certificate_digest + FROM mst2_metadata_page_certificate c WHERE c.page_id=$2 AND c.generation=$3 + ON CONFLICT(anchor_kind,owner_key,root_page,root_generation) DO NOTHING", + [intent.prepare_id.clone().into(),intent.plan.root.to_vec().into(),intent.root_generation.into()])).await.map_err(internal)?; + let sources:Vec<_>=intent.plan.source_roots.iter().map(|(tree,page)|json!({"tree":tree,"page":hex::encode(page)})).collect(); + let sources=txn.query_all_raw(sql("SELECT source.tree,cur.page_id FROM jsonb_to_recordset($1::jsonb) source(tree text,page text) + JOIN mst2_metadata_current cur ON cur.page_id=decode(source.page,'hex') + JOIN mst2_metadata_page_certificate proof USING(page_id,generation) ORDER BY proof.rank,source.tree", + [serde_json::to_string(&sources).map_err(internal)?.into()])).await.map_err(internal)?; + if sources.len()!=intent.plan.source_roots.len() {return Err(unavailable("rooted source certificate order is incomplete"));} + for source in sources { + let tree:String=source.try_get("","tree").map_err(internal)?; + let page=digest_column(&source,"page_id")?; + let generation=intent.bindings.get(&page).map(|b|b.0).or_else(||intent.plan.reused.get(&page).map(|b|b.generation)) + .ok_or_else(||integrity("rooted directory source has no exact lifetime"))?; + let reused=txn.query_one_raw(sql("SELECT a.attestation_id FROM mst2_metadata_prepare_reuse_root r + JOIN mst2_metadata_source_root_attestation boundary ON boundary.attestation_id=r.attestation_id + JOIN mst2_metadata_source_root_attestation a ON a.root_page=r.root_page AND a.root_generation=r.root_generation + AND a.root_certificate_digest=boundary.root_certificate_digest + JOIN mst2_metadata_prepare origin ON origin.prepare_id=a.origin_prepare_id + JOIN mst2_metadata_current cur ON cur.page_id=a.root_page AND cur.generation=a.root_generation + JOIN mst2_metadata_lifetime life USING(page_id,generation) + JOIN $CORE$.mega_tree source ON source.tree_id=split_part(a.tagged_tree_oid,':',2) + WHERE r.prepare_id=$1 AND r.root_page=$2 AND r.root_generation=$3 AND a.tagged_tree_oid=$4 + AND origin.state='COMMITTED' AND life.state='LIVE' AND a.source_profile=mst2_metadata_native_profile($1) + AND $CORE$.mst2_route_source_tree_matches(split_part(a.tagged_tree_oid,':',2),a.source_revision,a.source_body_digest) + ORDER BY a.attestation_id LIMIT 1".replace("$CORE$",&identifier(&self.namespace.core_schema)), + [intent.prepare_id.clone().into(),page.to_vec().into(),generation.into(),tree.clone().into()])).await.map_err(internal)?; + if reused.is_some() {continue;} + txn.execute_raw(sql("INSERT INTO mst2_metadata_source_root_attestation(attestation_id,namespace_uuid,origin_prepare_id,tagged_tree_oid, + source_profile,profile_digest,source_body_digest,root_page,root_generation,root_certificate_digest,source_proof,attestation_digest) + SELECT gen_random_uuid(),(proof->>'namespace')::uuid,$1,$2,proof->'source_profile',decode(proof->>'profile_digest','hex'), + decode(proof->>'source_body_digest','hex'),$3,$4,decode(proof->>'root_certificate','hex'),proof,decode(proof->>'attestation','hex') + FROM (SELECT mst2_metadata_compute_source_proof($1,$2,$3,$4) AS proof) input", + [intent.prepare_id.clone().into(),tree.clone().into(),page.to_vec().into(),generation.into()])).await.map_err(internal)?; + } + let changed=txn.execute_raw(sql("UPDATE mst2_metadata_prepare SET state='COMMITTED',committed_at=clock_timestamp() + WHERE prepare_id=$1 AND state='PREPARING' AND storage_seal=$2", + [intent.prepare_id.clone().into(),intent.storage_seal.to_vec().into()])).await.map_err(internal)?; + if changed.rows_affected()!=1 {return Err(integrity("rooted finalize lost its exact prepare transition"));} + txn.execute_raw(sql("UPDATE mst2_metadata_lifetime life SET state='LIVE' FROM mst2_metadata_prepare_page member + WHERE member.prepare_id=$1 AND life.page_id=member.page_id AND life.generation=member.generation AND life.state='RESERVED'", + [intent.prepare_id.clone().into()])).await.map_err(internal)?; + txn.execute_raw(sql("INSERT INTO mst2_metadata_reuse_index(profile_digest,tagged_tree_oid,attestation_id,root_page,root_generation,attestation_digest) + SELECT a.profile_digest,a.tagged_tree_oid,a.attestation_id,a.root_page,a.root_generation,a.attestation_digest + FROM mst2_metadata_source_root_attestation a WHERE a.origin_prepare_id=$1 + ON CONFLICT(profile_digest,tagged_tree_oid) DO NOTHING",[intent.prepare_id.clone().into()])).await.map_err(internal)?; + self.receipt(&txn,intent).await + }.await; + committed( + txn, + result, + &intent.operation_id, + intent.manifest_digest, + super::super::native_metadata_install::MetadataCommitPhase::Finalize, + ) + .await + } + + async fn read_delta( + &self, + intent: &RootedPrepareIntent, + ) -> Result, SnapshotError> { + let rows=self.connection.query_all_raw(sql("SELECT member.page_id,member.generation,member.expected_size,body.metadata_codec, + body.byte_size,body.payload FROM mst2_metadata_prepare_page member JOIN mst2_metadata_payload body USING(page_id,generation) + JOIN mst2_metadata_current cur USING(page_id,generation) WHERE member.prepare_id=$1 ORDER BY member.page_id LIMIT 4097", + [intent.prepare_id.clone().into()])).await.map_err(internal)?; + if rows.len() != intent.bindings.len() { + return Err(unavailable("rooted durable delta is incomplete")); + } + let mut payloads = Vec::with_capacity(rows.len()); + for row in rows { + let page = digest_column(&row, "page_id")?; + let &(generation, size) = intent + .bindings + .get(&page) + .ok_or_else(|| integrity("rooted durable delta has an extra member"))?; + let bytes: Vec = row.try_get("", "payload").map_err(internal)?; + if row.try_get::("", "generation").map_err(internal)? != generation + || row.try_get::("", "expected_size").map_err(internal)? as u64 != size + || row.try_get::("", "byte_size").map_err(internal)? as u64 != size + || row.try_get::("", "metadata_codec").map_err(internal)? != 1 + || bytes.len() as u64 != size + || page_id(&bytes) != page + { + return Err(integrity( + "rooted durable delta crossed its fixed byte and lifetime binding", + )); + } + Page::decode(&bytes).map_err(internal)?; + payloads.push(MetadataPagePayload { + id: page, + size, + bytes, + }); + } + Ok(payloads) + } + + async fn receipt( + &self, + db: &C, + intent: &RootedPrepareIntent, + ) -> Result { + let row=db.query_one_raw(sql("SELECT c.certificate_digest,a.attestation_id::text,a.attestation_digest + FROM mst2_metadata_prepare q JOIN mst2_metadata_current cur ON cur.page_id=q.metadata_root AND cur.generation=$2 + JOIN mst2_metadata_lifetime life USING(page_id,generation) JOIN mst2_metadata_graph_node n USING(page_id,generation) + JOIN mst2_metadata_page_certificate c USING(page_id,generation) + JOIN mst2_metadata_source_root_attestation a ON a.root_page=c.page_id AND a.root_generation=c.generation + JOIN mst2_metadata_prepare origin ON origin.prepare_id=a.origin_prepare_id + JOIN $CORE$.mega_tree source ON source.tree_id=split_part(a.tagged_tree_oid,':',2) + WHERE q.prepare_id=$1 AND q.plan_kind='ROOTED' AND q.state='COMMITTED' AND life.state='LIVE' AND n.state='LIVE' + AND n.certificate_digest=c.certificate_digest AND a.root_certificate_digest=c.certificate_digest + AND origin.state='COMMITTED' AND a.source_profile=mst2_metadata_native_profile($1) + AND a.tagged_tree_oid=mst2_metadata_rooted_scope_tree($1) + AND $CORE$.mst2_route_source_tree_matches(split_part(a.tagged_tree_oid,':',2),a.source_revision,a.source_body_digest) + AND NOT EXISTS(SELECT 1 FROM mst2_metadata_gc_op gc WHERE gc.page_id=c.page_id AND gc.generation=c.generation) + AND (q.coverage_retired_at IS NULL AND EXISTS(SELECT 1 FROM mst2_metadata_root_anchor anchor + WHERE anchor.anchor_kind='PREPARE' AND anchor.prepare_id=q.prepare_id AND anchor.owner_key=q.prepare_id + AND anchor.root_page=c.page_id AND anchor.root_generation=c.generation + AND anchor.root_certificate_digest=c.certificate_digest) OR mst2_metadata_session_covers_prepare(q.prepare_id)) + ORDER BY (a.origin_prepare_id=q.prepare_id) DESC,a.attestation_id LIMIT 1".replace("$CORE$",&identifier(&self.namespace.core_schema)), + [intent.prepare_id.clone().into(),intent.root_generation.into()])).await.map_err(internal)? + .ok_or_else(||unavailable("rooted definitive receipt has no exact active canonical source root"))?; + Ok(RootedMetadataReceipt { + intent: intent.clone(), + certificate_digest: digest_column(&row, "certificate_digest")?, + attestation_id: uuid::Uuid::parse_str( + &row.try_get::("", "attestation_id") + .map_err(internal)?, + ) + .map_err(internal)?, + attestation_digest: digest_column(&row, "attestation_digest")?, + }) + } + + pub(crate) async fn recover( + &self, + operation: &str, + plan: &RootedMetadataInstallPlan, + ) -> Result, SnapshotError> { + let txn = self.transaction().await?; + let Some((intent, state)) = self.load_intent(&txn, operation, plan).await? else { + txn.commit().await.map_err(internal)?; + return Ok(None); + }; + let receipt = if state == "COMMITTED" { + Some(self.receipt(&txn, &intent).await?) + } else { + None + }; + txn.commit().await.map_err(internal)?; + Ok(receipt) + } +} diff --git a/src/jupiter/storage/qualified_metadata_rooted.sql b/src/jupiter/storage/qualified_metadata_rooted.sql new file mode 100644 index 00000000..8116bc61 --- /dev/null +++ b/src/jupiter/storage/qualified_metadata_rooted.sql @@ -0,0 +1,358 @@ +CREATE FUNCTION mst2_metadata_read_be(b bytea,p integer,w integer) RETURNS numeric +LANGUAGE plpgsql IMMUTABLE STRICT SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE v numeric:=0; i integer; +BEGIN + IF w NOT IN (2,4,8) OR p<0 OR p>octet_length(b)-w THEN RAISE EXCEPTION 'rooted integer is out of bounds'; END IF; + FOR i IN 0..w-1 LOOP v:=v*256+get_byte(b,p+i); END LOOP; + RETURN v; +END $$; + +CREATE FUNCTION mst2_metadata_rooted_string(b bytea,p integer,maximum integer) RETURNS jsonb +LANGUAGE plpgsql IMMUTABLE STRICT SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE n numeric; value text; +BEGIN + n:=mst2_metadata_read_be(b,p,4); p:=p+4; + IF n>maximum OR n>octet_length(b)-p THEN RAISE EXCEPTION 'rooted string exceeds its exact byte boundary'; END IF; + value:=convert_from(substring(b FROM p+1 FOR n::integer),'UTF8'); + RETURN jsonb_build_object('end',p+n::integer,'text',value); +END $$; + +CREATE FUNCTION mst2_metadata_decode_rooted_plan(b bytea) RETURNS jsonb +LANGUAGE plpgsql IMMUTABLE STRICT SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE domain bytea:=convert_to('mega.mst2.rooted-install.v1','UTF8')||decode('00','hex'); + cursor_pos integer; part jsonb; source_domain text; tree_oid text; scope text; hash_kind text; identity jsonb; + root bytea; count numeric; i integer; page bytea; parent bytea; child bytea; previous bytea; previous_tree text; + size numeric; generation numeric; attestation bytea; attestation_digest bytea; certificate_digest bytea; + delta_rows jsonb[]:=ARRAY[]::jsonb[]; edge_rows jsonb[]:=ARRAY[]::jsonb[]; + reuse_rows jsonb[]:=ARRAY[]::jsonb[]; source_rows jsonb[]:=ARRAY[]::jsonb[]; total_bytes bigint:=0; + delta_index jsonb; node_index jsonb; adjacency jsonb; +BEGIN + IF octet_length(b)>2097152 OR substring(b FROM 1 FOR octet_length(domain))<>domain THEN + RAISE EXCEPTION 'rooted preparation domain or byte budget is invalid'; + END IF; + cursor_pos:=octet_length(domain); + IF mst2_metadata_read_be(b,cursor_pos,2)<>1 THEN RAISE EXCEPTION 'rooted preparation version is unsupported'; END IF; + cursor_pos:=cursor_pos+2; + part:=mst2_metadata_rooted_string(b,cursor_pos,64); cursor_pos:=(part->>'end')::integer; source_domain:=part->>'text'; + part:=mst2_metadata_rooted_string(b,cursor_pos,128); cursor_pos:=(part->>'end')::integer; tree_oid:=part->>'text'; + part:=mst2_metadata_rooted_string(b,cursor_pos,4096); cursor_pos:=(part->>'end')::integer; scope:=part->>'text'; + IF source_domain<>'native-git' OR tree_oid !~ '^(sha1:[0-9a-f]{40}|sha256:[0-9a-f]{64}|blake3:[0-9a-f]{64})$' + OR scope NOT LIKE '/%' OR scope<>'/' AND (scope LIKE '%/' OR scope LIKE '%//%' OR EXISTS( + SELECT 1 FROM unnest(string_to_array(substring(scope FROM 2),'/')) name WHERE name IN ('','.','..') + OR octet_length(name)>255) OR cardinality(string_to_array(substring(scope FROM 2),'/'))>256) THEN + RAISE EXCEPTION 'rooted source identity or scope is invalid'; END IF; + hash_kind:=split_part(tree_oid,':',1); + identity:=jsonb_build_object('source_domain',source_domain,'tagged_root_tree_oid',tree_oid,'scope',scope, + 'schema_version',mst2_metadata_read_be(b,cursor_pos,2), + 'metadata_codec',mst2_metadata_read_be(b,cursor_pos+2,2), + 'materialization_policy',mst2_metadata_read_be(b,cursor_pos+4,2), + 'fs_semantics',mst2_metadata_read_be(b,cursor_pos+6,2), + 'access_projection',mst2_metadata_read_be(b,cursor_pos+8,2), + 'verification_revision',mst2_metadata_read_be(b,cursor_pos+10,4), + 'projection_revision',mst2_metadata_read_be(b,cursor_pos+14,2)); + cursor_pos:=cursor_pos+16; + IF (identity->>'schema_version')::integer<>2 OR (identity->>'metadata_codec')::integer<>1 + OR (identity->>'materialization_policy')::integer<>1 OR (identity->>'fs_semantics')::integer<>1 + OR (identity->>'access_projection')::integer<>0 OR (identity->>'verification_revision')::integer<>2 + OR (identity->>'projection_revision')::integer<>1 THEN RAISE EXCEPTION 'rooted source profile is not current native'; END IF; + IF cursor_pos>octet_length(b)-32 THEN RAISE EXCEPTION 'rooted metadata root is truncated'; END IF; + root:=substring(b FROM cursor_pos+1 FOR 32); cursor_pos:=cursor_pos+32; + count:=mst2_metadata_read_be(b,cursor_pos,4); cursor_pos:=cursor_pos+4; + IF count>4096 THEN RAISE EXCEPTION 'rooted delta exceeds its node budget'; END IF; + IF count>0 THEN FOR i IN 1..count::integer LOOP + IF cursor_pos>octet_length(b)-40 THEN RAISE EXCEPTION 'rooted delta member is truncated'; END IF; + page:=substring(b FROM cursor_pos+1 FOR 32); size:=mst2_metadata_read_be(b,cursor_pos+32,8); cursor_pos:=cursor_pos+40; + IF previous IS NOT NULL AND previous>=page OR size NOT BETWEEN 20 AND 16384 THEN + RAISE EXCEPTION 'rooted delta is not exactly ordered or has invalid size'; END IF; + previous:=page; total_bytes:=total_bytes+size::bigint; + IF total_bytes>67108864 THEN RAISE EXCEPTION 'rooted delta exceeds its metadata byte budget'; END IF; + delta_rows:=array_append(delta_rows,jsonb_build_object('page',encode(page,'hex'),'size',size)); + END LOOP; END IF; + previous:=NULL; count:=mst2_metadata_read_be(b,cursor_pos,4); cursor_pos:=cursor_pos+4; + IF count>16384 THEN RAISE EXCEPTION 'rooted delta exceeds its edge budget'; END IF; + IF count>0 THEN FOR i IN 1..count::integer LOOP + IF cursor_pos>octet_length(b)-64 THEN RAISE EXCEPTION 'rooted edge is truncated'; END IF; + parent:=substring(b FROM cursor_pos+1 FOR 32); child:=substring(b FROM cursor_pos+33 FOR 32); cursor_pos:=cursor_pos+64; + IF previous IS NOT NULL AND previous>=parent||child OR parent=child THEN RAISE EXCEPTION 'rooted edges are not unique and ordered'; END IF; + previous:=parent||child; + edge_rows:=array_append(edge_rows,jsonb_build_object('parent',encode(parent,'hex'),'child',encode(child,'hex'))); + END LOOP; END IF; + previous:=NULL; count:=mst2_metadata_read_be(b,cursor_pos,4); cursor_pos:=cursor_pos+4; + IF count>4096 OR coalesce(array_length(delta_rows,1),0)+count>4096 THEN RAISE EXCEPTION 'rooted delta and boundaries exceed node budget'; END IF; + IF count>0 THEN FOR i IN 1..count::integer LOOP + IF cursor_pos>octet_length(b)-120 THEN RAISE EXCEPTION 'rooted reuse boundary is truncated'; END IF; + page:=substring(b FROM cursor_pos+1 FOR 32); generation:=mst2_metadata_read_be(b,cursor_pos+32,8); + attestation:=substring(b FROM cursor_pos+41 FOR 16); attestation_digest:=substring(b FROM cursor_pos+57 FOR 32); + certificate_digest:=substring(b FROM cursor_pos+89 FOR 32); cursor_pos:=cursor_pos+120; + IF previous IS NOT NULL AND previous>=page OR generation NOT BETWEEN 1 AND 9223372036854775807 THEN + RAISE EXCEPTION 'rooted reuse boundaries are not exact positive ordered lifetimes'; END IF; + previous:=page; + reuse_rows:=array_append(reuse_rows,jsonb_build_object('page',encode(page,'hex'),'generation',generation, + 'attestation_id',encode(attestation,'hex')::uuid,'attestation_digest',encode(attestation_digest,'hex'), + 'certificate_digest',encode(certificate_digest,'hex'))); + END LOOP; END IF; + count:=mst2_metadata_read_be(b,cursor_pos,4); cursor_pos:=cursor_pos+4; + IF count NOT BETWEEN 1 AND 4096 THEN RAISE EXCEPTION 'rooted source-root budget is invalid'; END IF; + FOR i IN 1..count::integer LOOP + part:=mst2_metadata_rooted_string(b,cursor_pos,128); cursor_pos:=(part->>'end')::integer; tree_oid:=part->>'text'; + IF cursor_pos>octet_length(b)-32 THEN RAISE EXCEPTION 'rooted source root is truncated'; END IF; + page:=substring(b FROM cursor_pos+1 FOR 32); cursor_pos:=cursor_pos+32; + IF tree_oid !~ '^(sha1:[0-9a-f]{40}|sha256:[0-9a-f]{64}|blake3:[0-9a-f]{64})$' + OR split_part(tree_oid,':',1)<>hash_kind OR previous_tree IS NOT NULL AND convert_to(previous_tree,'UTF8')>=convert_to(tree_oid,'UTF8') THEN + RAISE EXCEPTION 'rooted source roots are not unique ordered same-profile identities'; END IF; + previous_tree:=tree_oid; source_rows:=array_append(source_rows,jsonb_build_object('tree_oid',tree_oid,'page',encode(page,'hex'))); + END LOOP; + IF cursor_pos<>octet_length(b) THEN RAISE EXCEPTION 'rooted preparation has trailing bytes'; END IF; + SELECT coalesce(jsonb_object_agg(value->>'page',true),'{}'::jsonb) INTO delta_index FROM unnest(delta_rows) d(value); + SELECT coalesce(jsonb_object_agg(value->>'page',true),'{}'::jsonb) INTO node_index FROM ( + SELECT value FROM unnest(delta_rows) UNION ALL SELECT value FROM unnest(reuse_rows)) nodes; + IF coalesce(array_length(delta_rows,1),0)+coalesce(array_length(reuse_rows,1),0)=0 + OR EXISTS(SELECT 1 FROM unnest(reuse_rows) r(value) WHERE delta_index ? (r.value->>'page')) + OR NOT node_index ? encode(root,'hex') + OR EXISTS(SELECT 1 FROM unnest(edge_rows) e(value) WHERE NOT delta_index ? (e.value->>'parent') + OR NOT node_index ? (e.value->>'child')) + OR EXISTS(SELECT 1 FROM unnest(source_rows) s(value) WHERE NOT node_index ? (s.value->>'page')) + OR NOT EXISTS(SELECT 1 FROM unnest(source_rows) s(value) WHERE value->>'page'=encode(root,'hex')) THEN + RAISE EXCEPTION 'rooted plan has overlap or an unbound graph/source endpoint'; + END IF; + SELECT coalesce(jsonb_object_agg(parent,children),'{}'::jsonb) INTO adjacency FROM ( + SELECT value->>'parent' AS parent,jsonb_agg(value->>'child') AS children FROM unnest(edge_rows) e(value) + GROUP BY value->>'parent') grouped; + IF EXISTS(WITH RECURSIVE reached(page) AS (SELECT encode(root,'hex') UNION + SELECT child FROM reached r CROSS JOIN LATERAL jsonb_array_elements_text(adjacency->r.page) children(child)) + SELECT 1 FROM (SELECT value FROM unnest(delta_rows) UNION ALL SELECT value FROM unnest(reuse_rows)) nodes + WHERE NOT EXISTS(SELECT 1 FROM reached r WHERE r.page=nodes.value->>'page')) THEN + RAISE EXCEPTION 'rooted plan includes members outside its bounded delta and boundary closure'; + END IF; + RETURN identity||jsonb_build_object('root',encode(root,'hex'),'delta',to_jsonb(delta_rows),'edges',to_jsonb(edge_rows), + 'reused',to_jsonb(reuse_rows),'source_roots',to_jsonb(source_rows),'total_delta_bytes',total_bytes); +END $$; + +CREATE FUNCTION mst2_metadata_decode_delta_bindings(b bytea,plan jsonb) RETURNS jsonb +LANGUAGE plpgsql IMMUTABLE STRICT SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE count numeric; cursor_pos integer:=12; i integer; page bytea; generation numeric; size numeric; + previous bytea; binding_rows jsonb[]:=ARRAY[]::jsonb[]; +BEGIN + IF octet_length(b) NOT BETWEEN 12 AND 196620 OR substring(b FROM 1 FOR 8)<>convert_to('MST2GEN1','UTF8') THEN + RAISE EXCEPTION 'rooted generation binding encoding is invalid'; END IF; + count:=mst2_metadata_read_be(b,8,4); + IF count<>jsonb_array_length(plan->'delta') OR octet_length(b)<>12+48*count THEN + RAISE EXCEPTION 'rooted delta binding count differs from its immutable plan'; END IF; + IF count>0 THEN FOR i IN 1..count::integer LOOP + page:=substring(b FROM cursor_pos+1 FOR 32); generation:=mst2_metadata_read_be(b,cursor_pos+32,8); + size:=mst2_metadata_read_be(b,cursor_pos+40,8); cursor_pos:=cursor_pos+48; + IF previous IS NOT NULL AND previous>=page OR generation NOT BETWEEN 1 AND 9223372036854775807 + OR encode(page,'hex') IS DISTINCT FROM plan->'delta'->(i-1)->>'page' + OR size IS DISTINCT FROM (plan->'delta'->(i-1)->>'size')::numeric THEN + RAISE EXCEPTION 'rooted generation binding is not its exact positive ordered delta member'; END IF; + previous:=page; binding_rows:=array_append(binding_rows,jsonb_build_object('page',encode(page,'hex'),'generation',generation,'size',size)); + END LOOP; END IF; + RETURN to_jsonb(binding_rows); +END $$; + +CREATE FUNCTION mst2_metadata_rooted_manifest(q mst2_metadata_prepare) RETURNS jsonb +LANGUAGE plpgsql IMMUTABLE STRICT SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE plan jsonb:=mst2_metadata_decode_rooted_plan(q.canonical_plan); bindings jsonb; +BEGIN + bindings:=mst2_metadata_decode_delta_bindings(q.canonical_bindings,plan); + IF q.plan_kind<>'ROOTED' OR q.source_domain IS DISTINCT FROM plan->>'source_domain' + OR q.tagged_root_tree_oid IS DISTINCT FROM plan->>'tagged_root_tree_oid' OR q.scope IS DISTINCT FROM plan->>'scope' + OR q.schema_version IS DISTINCT FROM (plan->>'schema_version')::smallint + OR q.metadata_codec IS DISTINCT FROM (plan->>'metadata_codec')::smallint + OR q.materialization_policy IS DISTINCT FROM (plan->>'materialization_policy')::smallint + OR q.fs_semantics IS DISTINCT FROM (plan->>'fs_semantics')::smallint + OR q.access_projection IS DISTINCT FROM (plan->>'access_projection')::smallint + OR q.verification_revision IS DISTINCT FROM (plan->>'verification_revision')::integer + OR q.projection_revision IS DISTINCT FROM (plan->>'projection_revision')::smallint + OR q.metadata_root IS DISTINCT FROM decode(plan->>'root','hex') + OR q.node_count<>jsonb_array_length(plan->'delta') OR q.edge_count<>jsonb_array_length(plan->'edges') + OR q.total_bytes<>(plan->>'total_delta_bytes')::bigint THEN + RAISE EXCEPTION 'rooted preparation fields differ from its independently decoded manifest'; END IF; + RETURN plan||jsonb_build_object('bindings',bindings); +END $$; + +CREATE FUNCTION mst2_metadata_check_rooted_members(pid text) RETURNS jsonb +LANGUAGE plpgsql VOLATILE STRICT SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE q mst2_metadata_prepare%ROWTYPE; plan jsonb; +BEGIN + SELECT * INTO STRICT q FROM mst2_metadata_prepare WHERE prepare_id=pid AND plan_kind='ROOTED'; + plan:=mst2_metadata_rooted_manifest(q); + IF EXISTS((SELECT decode(value->>'page','hex'),(value->>'generation')::bigint,(value->>'size')::integer + FROM jsonb_array_elements(plan->'bindings')) EXCEPT + (SELECT page_id,generation,expected_size FROM mst2_metadata_prepare_page WHERE prepare_id=pid)) + OR EXISTS((SELECT page_id,generation,expected_size FROM mst2_metadata_prepare_page WHERE prepare_id=pid) EXCEPT + (SELECT decode(value->>'page','hex'),(value->>'generation')::bigint,(value->>'size')::integer + FROM jsonb_array_elements(plan->'bindings'))) + OR EXISTS((SELECT decode(value->>'page','hex'),(value->>'generation')::bigint,(value->>'attestation_id')::uuid, + decode(value->>'attestation_digest','hex'),decode(value->>'certificate_digest','hex') + FROM jsonb_array_elements(plan->'reused')) EXCEPT + (SELECT r.root_page,r.root_generation,r.attestation_id,r.attestation_digest,a.root_certificate_digest + FROM mst2_metadata_prepare_reuse_root r JOIN mst2_metadata_source_root_attestation a USING(attestation_id) + WHERE r.prepare_id=pid)) + OR EXISTS((SELECT r.root_page,r.root_generation,r.attestation_id,r.attestation_digest,a.root_certificate_digest + FROM mst2_metadata_prepare_reuse_root r JOIN mst2_metadata_source_root_attestation a USING(attestation_id) + WHERE r.prepare_id=pid) EXCEPT + (SELECT decode(value->>'page','hex'),(value->>'generation')::bigint,(value->>'attestation_id')::uuid, + decode(value->>'attestation_digest','hex'),decode(value->>'certificate_digest','hex') + FROM jsonb_array_elements(plan->'reused'))) THEN + RAISE EXCEPTION 'rooted preparation has missing or extra exact delta/reuse lifetime bindings'; END IF; + RETURN plan; +END $$; + +CREATE FUNCTION mst2_metadata_rooted_members_complete() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + IF EXISTS(SELECT 1 FROM mst2_metadata_prepare WHERE prepare_id=NEW.prepare_id AND plan_kind='ROOTED') THEN + PERFORM mst2_metadata_check_rooted_members(NEW.prepare_id); + END IF; + RETURN NULL; +END $$; +CREATE CONSTRAINT TRIGGER mst2_metadata_rooted_prepare_complete AFTER INSERT OR UPDATE OF bindings_revision ON mst2_metadata_prepare + DEFERRABLE INITIALLY DEFERRED FOR EACH ROW EXECUTE FUNCTION mst2_metadata_rooted_members_complete(); + +CREATE FUNCTION mst2_metadata_rooted_members_added() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + -- Every actual membership statement queues one authoritative commit check + -- per affected prepare, including a late raw-SQL addition. No caller state + -- can disable the decoded exact-coverage proof. + UPDATE mst2_metadata_prepare q SET bindings_revision=q.bindings_revision+1 + WHERE q.plan_kind='ROOTED' AND q.prepare_id IN (SELECT DISTINCT prepare_id FROM added_members); + RETURN NULL; +END $$; +CREATE TRIGGER mst2_metadata_rooted_delta_added AFTER INSERT ON mst2_metadata_prepare_page + REFERENCING NEW TABLE AS added_members FOR EACH STATEMENT EXECUTE FUNCTION mst2_metadata_rooted_members_added(); +CREATE TRIGGER mst2_metadata_rooted_reuse_added AFTER INSERT ON mst2_metadata_prepare_reuse_root + REFERENCING NEW TABLE AS added_members FOR EACH STATEMENT EXECUTE FUNCTION mst2_metadata_rooted_members_added(); + +CREATE TABLE mst2_metadata_scope_source_reference ( + prepare_id text PRIMARY KEY REFERENCES mst2_metadata_prepare(prepare_id), + scope_tree_oid text NOT NULL CHECK(scope_tree_oid ~ '^(sha1:[0-9a-f]{40}|sha256:[0-9a-f]{64}|blake3:[0-9a-f]{64})$'), + ancestor_revisions jsonb NOT NULL CHECK((jsonb_typeof(ancestor_revisions)='array' AND jsonb_array_length(ancestor_revisions)<=256) IS TRUE) +); +CREATE FUNCTION mst2_metadata_derive_scope_source(pid text) RETURNS jsonb LANGUAGE plpgsql VOLATILE STRICT +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE q mst2_metadata_prepare%ROWTYPE; tree_oid text; component text; body bytea; item jsonb; scanned_bytes bigint:=0; + digest bytea; revision uuid; ancestors jsonb[]:=ARRAY[]::jsonb[]; +BEGIN + SELECT * INTO STRICT q FROM mst2_metadata_prepare WHERE prepare_id=pid AND plan_kind='ROOTED' AND state='PREPARING'; + tree_oid:=q.tagged_root_tree_oid; + IF q.scope='/' THEN RETURN jsonb_build_object('scope_tree_oid',tree_oid,'ancestor_revisions','[]'::jsonb); END IF; + FOREACH component IN ARRAY string_to_array(substring(q.scope FROM 2),'/') LOOP + SELECT sub_trees INTO body FROM $CORE_SCHEMA$.mega_tree WHERE tree_id=split_part(tree_oid,':',2); + IF NOT FOUND THEN RAISE EXCEPTION 'rooted source scope has a missing fixed ancestor'; END IF; + scanned_bytes:=scanned_bytes+octet_length(body); + IF scanned_bytes>67108864 THEN RAISE EXCEPTION 'rooted source-scope walk exceeds its fixed byte-work budget'; END IF; + digest:=sha256(body); + revision:=$CORE_SCHEMA$.mst2_route_capture_source_tree(split_part(tree_oid,':',2),digest); + ancestors:=array_append(ancestors,jsonb_build_object('tree_oid',tree_oid,'revision',revision::text,'body_digest',encode(digest,'hex'))); + SELECT value INTO item FROM jsonb_array_elements(mst2_metadata_decode_git_tree(body,split_part(tree_oid,':',1))) + WHERE decode(value->>'name','hex')=convert_to(component,'UTF8'); + IF NOT FOUND OR (item->>'kind')::integer<>4 THEN RAISE EXCEPTION 'rooted source scope is not its exact fixed directory'; END IF; + tree_oid:=item->>'oid'; + END LOOP; + RETURN jsonb_build_object('scope_tree_oid',tree_oid,'ancestor_revisions',to_jsonb(ancestors)); +END $$; +CREATE FUNCTION mst2_metadata_scope_source_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE proof jsonb; +BEGIN + IF TG_OP<>'INSERT' THEN RAISE EXCEPTION 'rooted scope source proof is immutable'; END IF; + proof:=mst2_metadata_derive_scope_source(NEW.prepare_id); + IF NEW.scope_tree_oid IS DISTINCT FROM proof->>'scope_tree_oid' + OR NEW.ancestor_revisions IS DISTINCT FROM proof->'ancestor_revisions' THEN + RAISE EXCEPTION 'rooted scope source proof was not independently derived from actual core ancestors'; END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_metadata_scope_source_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_metadata_scope_source_reference + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_scope_source_guard(); + +CREATE FUNCTION mst2_metadata_rooted_scope_tree(pid text) RETURNS text LANGUAGE plpgsql VOLATILE STRICT +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE bound mst2_metadata_scope_source_reference%ROWTYPE; proof jsonb; ancestor jsonb; +BEGIN + SELECT * INTO bound FROM mst2_metadata_scope_source_reference WHERE prepare_id=pid; + IF NOT FOUND THEN + proof:=mst2_metadata_derive_scope_source(pid); + INSERT INTO mst2_metadata_scope_source_reference(prepare_id,scope_tree_oid,ancestor_revisions) + VALUES(pid,proof->>'scope_tree_oid',proof->'ancestor_revisions') RETURNING * INTO bound; + END IF; + FOR ancestor IN SELECT value FROM jsonb_array_elements(bound.ancestor_revisions) LOOP + IF NOT $CORE_SCHEMA$.mst2_route_source_tree_matches(split_part(ancestor->>'tree_oid',':',2), + (ancestor->>'revision')::uuid,decode(ancestor->>'body_digest','hex')) THEN + RAISE EXCEPTION 'rooted fixed scope ancestor lost its exact current source revision'; END IF; + END LOOP; + RETURN bound.scope_tree_oid; +END $$; + +CREATE FUNCTION mst2_metadata_rooted_finalize_proof(pid text) RETURNS jsonb LANGUAGE plpgsql VOLATILE STRICT +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE q mst2_metadata_prepare%ROWTYPE; plan jsonb; selected_root_generation bigint; root_certificate bytea; + profile jsonb; source jsonb; source_generation bigint; scoped_tree text; + generation_index jsonb; +BEGIN + SELECT * INTO STRICT q FROM mst2_metadata_prepare WHERE prepare_id=pid AND plan_kind='ROOTED' AND state='PREPARING'; + plan:=mst2_metadata_check_rooted_members(pid); profile:=mst2_metadata_native_profile(pid); + SELECT jsonb_object_agg(value->>'page',value->'generation') INTO generation_index FROM ( + SELECT value FROM jsonb_array_elements(plan->'bindings') UNION ALL + SELECT value FROM jsonb_array_elements(plan->'reused')) members; + selected_root_generation:=(generation_index->>(plan->>'root'))::bigint; + SELECT c.certificate_digest INTO root_certificate FROM mst2_metadata_page_certificate c + JOIN mst2_metadata_graph_node n USING(page_id,generation) JOIN mst2_metadata_current cur USING(page_id,generation) + JOIN mst2_metadata_lifetime life USING(page_id,generation) + WHERE c.page_id=q.metadata_root AND c.generation=selected_root_generation AND n.state='LIVE' + AND n.certificate_digest=c.certificate_digest AND life.state IN ('RESERVED','LIVE') AND life.graph_domain='qualified-v1' + AND c.relative_path_bytes+CASE WHEN q.scope='/' THEN 0 ELSE octet_length(q.scope) END<=4096 + AND c.relative_components+CASE WHEN q.scope='/' THEN 0 ELSE cardinality(string_to_array(substring(q.scope FROM 2),'/')) END<=256 + AND NOT EXISTS(SELECT 1 FROM mst2_metadata_gc_op gc WHERE gc.page_id=c.page_id AND gc.generation=c.generation); + IF NOT FOUND OR NOT EXISTS(SELECT 1 FROM mst2_metadata_root_anchor a WHERE a.anchor_kind='PREPARE' + AND a.owner_key=pid AND a.prepare_id=pid AND a.root_page=q.metadata_root AND a.root_generation=selected_root_generation + AND a.root_certificate_digest=root_certificate) THEN RAISE EXCEPTION 'rooted finalize lacks its independently certified owned root'; END IF; + IF EXISTS(SELECT 1 FROM mst2_metadata_prepare_page m + LEFT JOIN mst2_metadata_current cur USING(page_id,generation) LEFT JOIN mst2_metadata_lifetime life USING(page_id,generation) + LEFT JOIN mst2_metadata_payload body USING(page_id,generation) LEFT JOIN mst2_metadata_graph_node n USING(page_id,generation) + LEFT JOIN mst2_metadata_page_certificate c USING(page_id,generation) + WHERE m.prepare_id=pid AND (cur.page_id IS NULL OR life.page_id IS NULL OR body.page_id IS NULL OR n.page_id IS NULL OR c.page_id IS NULL + OR life.state NOT IN ('RESERVED','LIVE') OR life.graph_domain<>'qualified-v1' OR life.metadata_codec<>q.metadata_codec + OR body.metadata_codec<>q.metadata_codec OR n.metadata_codec<>q.metadata_codec OR c.metadata_codec<>q.metadata_codec + OR life.expected_size<>m.expected_size OR body.byte_size<>m.expected_size OR n.bytes<>m.expected_size OR c.byte_size<>m.expected_size + OR n.state<>'LIVE' OR n.certificate_digest<>c.certificate_digest + OR life.state='RESERVED' AND c.origin_prepare_id<>pid + OR n.incoming_refs<>(SELECT count(*) FROM mst2_metadata_graph_edge e WHERE e.child_page=m.page_id AND e.child_generation=m.generation) + OR EXISTS(SELECT 1 FROM mst2_metadata_gc_op gc WHERE gc.page_id=m.page_id AND gc.generation=m.generation))) THEN + RAISE EXCEPTION 'rooted finalize lost its exact durable canonical delta graph'; END IF; + IF EXISTS((SELECT decode(value->>'parent','hex'),decode(value->>'child','hex') FROM jsonb_array_elements(plan->'edges')) EXCEPT + (SELECT e.parent_page,e.child_page FROM mst2_metadata_prepare_page m JOIN mst2_metadata_graph_edge e + ON e.parent_page=m.page_id AND e.parent_generation=m.generation WHERE m.prepare_id=pid)) + OR EXISTS((SELECT e.parent_page,e.child_page FROM mst2_metadata_prepare_page m JOIN mst2_metadata_graph_edge e + ON e.parent_page=m.page_id AND e.parent_generation=m.generation WHERE m.prepare_id=pid) EXCEPT + (SELECT decode(value->>'parent','hex'),decode(value->>'child','hex') FROM jsonb_array_elements(plan->'edges'))) + OR EXISTS(SELECT 1 FROM mst2_metadata_prepare_reuse_root r WHERE r.prepare_id=pid AND NOT EXISTS( + SELECT 1 FROM mst2_metadata_root_anchor a WHERE a.prepare_id=pid AND a.owner_key=pid AND a.anchor_kind='REUSE' + AND a.root_page=r.root_page AND a.root_generation=r.root_generation)) THEN + RAISE EXCEPTION 'rooted finalize differs from its exact delta edges or owned reuse boundaries'; END IF; + scoped_tree:=mst2_metadata_rooted_scope_tree(pid); + IF NOT EXISTS(SELECT 1 FROM jsonb_array_elements(plan->'source_roots') binding(value) + WHERE value->>'tree_oid'=scoped_tree AND value->>'page'=plan->>'root') THEN + RAISE EXCEPTION 'rooted metadata root is not bound to its independently selected fixed source scope'; END IF; + FOR source IN SELECT value FROM jsonb_array_elements(plan->'source_roots') LOOP + source_generation:=(generation_index->>(source->>'page'))::bigint; + IF NOT EXISTS(SELECT 1 FROM mst2_metadata_source_root_attestation a + JOIN mst2_metadata_current cur ON cur.page_id=a.root_page AND cur.generation=a.root_generation + JOIN mst2_metadata_lifetime life USING(page_id,generation) JOIN mst2_metadata_graph_node n USING(page_id,generation) + JOIN mst2_metadata_prepare origin ON origin.prepare_id=a.origin_prepare_id + JOIN $CORE_SCHEMA$.mega_tree t ON t.tree_id=split_part(a.tagged_tree_oid,':',2) + WHERE a.tagged_tree_oid=source->>'tree_oid' AND a.root_page=decode(source->>'page','hex') AND a.root_generation=source_generation + AND a.source_profile=profile + AND $CORE_SCHEMA$.mst2_route_source_tree_matches(split_part(a.tagged_tree_oid,':',2),a.source_revision,a.source_body_digest) AND n.state='LIVE' + AND n.certificate_digest=a.root_certificate_digest + AND (a.origin_prepare_id=pid AND origin.state='PREPARING' AND life.state IN ('RESERVED','LIVE') + OR origin.state='COMMITTED' AND life.state='LIVE' AND EXISTS(SELECT 1 FROM mst2_metadata_prepare_reuse_root r + WHERE r.prepare_id=pid AND r.root_page=a.root_page AND r.root_generation=a.root_generation + AND EXISTS(SELECT 1 FROM mst2_metadata_source_root_attestation boundary + WHERE boundary.attestation_id=r.attestation_id AND boundary.root_certificate_digest=a.root_certificate_digest)))) THEN + RAISE EXCEPTION 'rooted source binding lacks its independently attested exact current directory'; END IF; + END LOOP; + RETURN jsonb_build_object('root',plan->>'root','generation',selected_root_generation,'certificate',encode(root_certificate,'hex'), + 'scoped_tree_oid',scoped_tree,'delta_nodes',q.node_count,'delta_edges',q.edge_count,'delta_bytes',q.total_bytes); +END $$; diff --git a/src/jupiter/storage/qualified_metadata_serving.sql b/src/jupiter/storage/qualified_metadata_serving.sql new file mode 100644 index 00000000..fa3d729e --- /dev/null +++ b/src/jupiter/storage/qualified_metadata_serving.sql @@ -0,0 +1,581 @@ +-- Root ownership is derived from durable identities, never an application flag. +CREATE FUNCTION mst2_metadata_publication_valid(instance text,commit_id text,tree_id text, + sequence_id bigint,epoch_id bigint,receipt_id bigint,current_head boolean DEFAULT false) +RETURNS boolean LANGUAGE sql VOLATILE SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ + SELECT EXISTS(SELECT 1 FROM $CORE_SCHEMA$.mst2_native_publication c + JOIN $CORE_SCHEMA$.mst2_publication p ON p.id=c.receipt_id + JOIN $CORE_SCHEMA$.mst2_publication_outbox o ON o.operation_id=p.operation_id + JOIN $CORE_SCHEMA$.mega_commit source ON source.commit_id=c.root_commit AND source.tree=c.root_tree + JOIN $CORE_SCHEMA$.mega_commit previous ON previous.commit_id=c.old_root_commit AND previous.tree=c.old_root_tree + JOIN $CORE_SCHEMA$.mega_commit path_source ON path_source.commit_id=c.path_commit AND path_source.tree=c.path_tree + WHERE c.receipt_id=$6 AND c.namespace='/' AND c.instance_id=$1 + AND c.root_commit=$2 AND c.root_tree=$3 AND c.sequence=$4 AND c.writer_epoch=$5 + AND $4>0 AND $5>0 AND p.writer_epoch=$5 AND p.writer_kind='trunk_push' + AND p.native_certificate_version=1 AND p.request_digest_version=1 + AND p.request_digest ~ '^sha256:[0-9a-f]{64}$' AND c.origin_ref='refs/heads/main' + AND c.origin_path=p.namespace AND c.path_commit=p.new_oid AND c.old_root_commit=p.old_oid + AND c.old_path_commit IS DISTINCT FROM c.path_commit AND o.namespace=p.namespace AND o.sequence=p.sequence + AND (NOT $7 OR (EXISTS(SELECT 1 FROM $CORE_SCHEMA$.mst2_native_head h + WHERE h.namespace='/' AND h.state='READY' AND h.instance_id=$1 AND h.root_commit=$2 + AND h.root_tree=$3 AND h.sequence=$4 AND h.writer_epoch=$5 + AND h.certificate_receipt_id=$6) + AND (SELECT count(*) FROM $CORE_SCHEMA$.mega_refs r + WHERE r.path='/' AND r.ref_name='refs/heads/main' AND NOT r.is_cl)=1 + AND EXISTS(SELECT 1 FROM $CORE_SCHEMA$.mega_refs r WHERE r.path='/' + AND r.ref_name='refs/heads/main' AND NOT r.is_cl AND r.ref_commit_hash=$2 AND r.ref_tree_hash=$3)))) +$$; + +CREATE FUNCTION mst2_metadata_descriptor(pid text,instance text,commit_id text,tree_id text) +RETURNS bytea LANGUAGE plpgsql VOLATILE STRICT SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE p record; scope_bytes bytea; view_digest bytea; instance_bytes bytea; +BEGIN + SELECT tagged_root_tree_oid,scope,metadata_root INTO p FROM mst2_metadata_prepare WHERE prepare_id=pid AND plan_kind='ROOTED' + AND state='COMMITTED' AND graph_domain='qualified-v1' AND mst2_metadata_scope_matches(primary_scope); + IF NOT FOUND OR instance IS DISTINCT FROM (instance::uuid)::text + OR split_part(p.tagged_root_tree_oid,':',2) IS DISTINCT FROM tree_id + OR NOT EXISTS(SELECT 1 FROM $CORE_SCHEMA$.mega_commit c WHERE c.commit_id=$3 AND c.tree=$4) + OR octet_length(commit_id)<>octet_length(tree_id) + OR commit_id !~ '^([0-9a-f]{40}|[0-9a-f]{64})$' THEN + RAISE EXCEPTION 'qualified descriptor has no exact canonical instance and fixed source'; + END IF; + PERFORM mst2_metadata_native_profile(pid); + scope_bytes:=convert_to(p.scope,'UTF8'); + IF p.scope !~ '^/' OR octet_length(scope_bytes)>4096 OR p.scope<>'/' AND + (p.scope ~ '/$|//|/(\.|\.\.)(/|$)' OR cardinality(string_to_array(substring(p.scope FROM 2),'/'))>256) + OR EXISTS(SELECT 1 FROM unnest(string_to_array(substring(p.scope FROM 2),'/')) component + WHERE octet_length(component)>255) THEN + RAISE EXCEPTION 'qualified descriptor scope is not canonical'; + END IF; + instance_bytes:=decode(replace(instance,'-',''),'hex'); + view_digest:=sha256(convert_to('mega.mst2.namespaceview','UTF8')||decode('00','hex')||convert_to(commit_id,'UTF8')); + RETURN convert_to('MSD2','UTF8')||decode('00020001','hex')||instance_bytes||view_digest + ||decode(lpad(to_hex(octet_length(scope_bytes)),4,'0'),'hex')||scope_bytes + ||decode('0001000100000000','hex')||p.metadata_root; +END $$; + +CREATE FUNCTION mst2_metadata_root_live(p bytea,g bigint,c bytea) RETURNS boolean +LANGUAGE sql VOLATILE STRICT SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ + SELECT EXISTS(SELECT 1 FROM mst2_metadata_current cur JOIN mst2_metadata_lifetime life USING(page_id,generation) + JOIN mst2_metadata_graph_node node USING(page_id,generation) + JOIN mst2_metadata_page_certificate proof USING(page_id,generation) + JOIN mst2_metadata_payload body USING(page_id,generation) + WHERE cur.page_id=p AND cur.generation=g AND life.state='LIVE' AND life.graph_domain='qualified-v1' + AND life.metadata_codec=1 AND node.state='LIVE' AND node.metadata_codec=1 AND body.metadata_codec=1 + AND node.certificate_digest=c AND proof.certificate_digest=c AND proof.proof_revision=1 + AND proof.namespace_uuid='$NAMESPACE_UUID$'::uuid AND life.expected_size=proof.byte_size + AND node.bytes=proof.byte_size AND body.byte_size=proof.byte_size AND octet_length(body.payload)=proof.byte_size + AND p=sha256(convert_to('mega.mst2.metapage','UTF8')||decode('00','hex')||body.payload) + AND NOT EXISTS(SELECT 1 FROM mst2_metadata_gc_op gc WHERE gc.page_id=p AND gc.generation=g)) +$$; + +CREATE FUNCTION mst2_metadata_serving_source(pid text,a_id uuid,p bytea,g bigint,c bytea) +RETURNS boolean LANGUAGE plpgsql VOLATILE STRICT SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE a record; scope text; certificate record; +BEGIN + SELECT proof.tagged_tree_oid,proof.source_revision,proof.source_body_digest,proof.profile_digest,proof.source_profile + INTO a FROM mst2_metadata_source_root_attestation proof + JOIN mst2_metadata_prepare origin ON origin.prepare_id=proof.origin_prepare_id + WHERE proof.attestation_id=a_id AND proof.namespace_uuid='$NAMESPACE_UUID$'::uuid + AND proof.root_page=p AND proof.root_generation=g AND proof.root_certificate_digest=c + AND origin.state='COMMITTED' AND proof.source_profile=mst2_metadata_native_profile(pid) + AND proof.tagged_tree_oid=mst2_metadata_rooted_scope_tree(pid); + IF NOT FOUND THEN RETURN false; END IF; + SELECT relative_path_bytes,relative_components INTO certificate FROM mst2_metadata_page_certificate WHERE page_id=p AND generation=g AND certificate_digest=c; + IF NOT FOUND THEN RETURN false; END IF; + SELECT q.scope INTO STRICT scope FROM mst2_metadata_prepare q WHERE q.prepare_id=pid; + IF (CASE WHEN scope='/' THEN 0 ELSE octet_length(scope) END)::bigint+certificate.relative_path_bytes>4096 + OR (CASE WHEN scope='/' THEN 0 ELSE cardinality(string_to_array(substring(scope FROM 2),'/')) END)::bigint + +certificate.relative_components>256 THEN RETURN false; END IF; + RETURN $CORE_SCHEMA$.mst2_route_source_tree_matches(split_part(a.tagged_tree_oid,':',2),a.source_revision,a.source_body_digest) + AND a.profile_digest=sha256(convert_to('mega.mst2.native-profile.v1','UTF8')||decode('00','hex') + ||convert_to(a.source_profile::text,'UTF8')); +END $$; + +CREATE FUNCTION mst2_metadata_snapshot_candidate(sid text,descriptor bytea,instance text,commit_id text, + tree_id text,root bytea,profile jsonb) RETURNS boolean LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE p record; a record; +BEGIN + IF sid IS DISTINCT FROM 'sha256:'||encode(sha256(convert_to('mega.mst2.descriptor','UTF8') + ||decode('00','hex')||descriptor),'hex') THEN RETURN false; END IF; + FOR p IN SELECT q.prepare_id FROM mst2_metadata_prepare q WHERE q.metadata_root=root AND q.state='COMMITTED' + AND q.plan_kind='ROOTED' AND q.graph_domain='qualified-v1' AND q.coverage_retired_at IS NULL + AND q.tagged_root_tree_oid=profile->>'tagged_root_tree_oid' AND q.scope=profile->>'scope' + AND $CORE_SCHEMA$.mst2_route_profile(q.source_domain,q.tagged_root_tree_oid,q.scope,q.schema_version, + q.metadata_codec,q.materialization_policy,q.fs_semantics,q.access_projection, + q.verification_revision,q.projection_revision)=profile ORDER BY q.prepare_id LIMIT 1 LOOP + IF descriptor IS DISTINCT FROM mst2_metadata_descriptor(p.prepare_id,instance,commit_id,tree_id) THEN CONTINUE; END IF; + FOR a IN SELECT proof.attestation_id,proof.root_generation,proof.root_certificate_digest FROM mst2_metadata_source_root_attestation proof + JOIN mst2_metadata_current current_root ON current_root.page_id=proof.root_page AND current_root.generation=proof.root_generation + WHERE proof.root_page=root AND proof.tagged_tree_oid=mst2_metadata_rooted_scope_tree(p.prepare_id) + AND proof.source_profile=mst2_metadata_native_profile(p.prepare_id) + AND $CORE_SCHEMA$.mst2_route_source_tree_matches(split_part(proof.tagged_tree_oid,':',2),proof.source_revision,proof.source_body_digest) + ORDER BY proof.attestation_id LIMIT 1 LOOP + IF mst2_metadata_serving_source(p.prepare_id,a.attestation_id,root,a.root_generation,a.root_certificate_digest) + AND mst2_metadata_root_live(root,a.root_generation,a.root_certificate_digest) THEN RETURN true; END IF; + END LOOP; + END LOOP; + RETURN false; +END $$; + +CREATE FUNCTION mst2_metadata_incarnation_proof(sid text,inc uuid,require_live boolean DEFAULT false) +RETURNS boolean LANGUAGE plpgsql VOLATILE SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE s mst2_qualified_session_incarnation%ROWTYPE; p record; + a record; profile jsonb; +BEGIN + SELECT * INTO s FROM mst2_qualified_session_incarnation WHERE snapshot_id=sid AND session_incarnation=inc; + IF NOT FOUND OR s.namespace_uuid<>'$NAMESPACE_UUID$'::uuid OR s.authorization_epoch<>1 THEN RETURN false; END IF; + SELECT prepare_id,source_domain,tagged_root_tree_oid,scope,schema_version,metadata_codec, + materialization_policy,fs_semantics,access_projection,verification_revision,projection_revision + INTO p FROM mst2_metadata_prepare WHERE prepare_id=s.prepare_id AND state='COMMITTED' + AND plan_kind='ROOTED' AND storage_seal=s.storage_seal AND metadata_root=s.metadata_root; + IF NOT FOUND THEN RETURN false; END IF; + profile:=$CORE_SCHEMA$.mst2_route_profile(p.source_domain,p.tagged_root_tree_oid,p.scope,p.schema_version, + p.metadata_codec,p.materialization_policy,p.fs_semantics,p.access_projection,p.verification_revision,p.projection_revision); + IF s.source_profile IS DISTINCT FROM convert_to(profile::text,'UTF8') + OR s.canonical_descriptor IS DISTINCT FROM mst2_metadata_descriptor(p.prepare_id,s.instance_id,s.commit_oid,s.root_tree_oid) + OR s.snapshot_id IS DISTINCT FROM 'sha256:'||encode(sha256(convert_to('mega.mst2.descriptor','UTF8') + ||decode('00','hex')||s.canonical_descriptor),'hex') + OR NOT EXISTS(SELECT 1 FROM $CORE_SCHEMA$.mst2_snapshot_storage_route r + WHERE r.snapshot_id=s.snapshot_id AND r.namespace_uuid=s.namespace_uuid AND r.canonical_descriptor=s.canonical_descriptor + AND r.instance_id=s.instance_id AND r.commit_oid=s.commit_oid AND r.root_tree_oid=s.root_tree_oid + AND r.metadata_root=s.metadata_root AND r.source_profile=profile) + OR NOT mst2_metadata_publication_valid(s.instance_id,s.commit_oid,s.root_tree_oid, + s.publication_sequence,s.writer_epoch,s.certificate_receipt_id,false) THEN RETURN false; END IF; + SELECT attestation_id,root_certificate_digest INTO a FROM mst2_metadata_source_root_attestation WHERE attestation_id=s.attestation_id + AND attestation_digest=s.attestation_digest AND root_page=s.metadata_root AND root_generation=s.root_generation; + IF NOT FOUND OR NOT mst2_metadata_serving_source(s.prepare_id,a.attestation_id,s.metadata_root, + s.root_generation,a.root_certificate_digest) THEN RETURN false; END IF; + IF require_live AND (s.state<>'READY' OR NOT mst2_metadata_root_live(s.metadata_root,s.root_generation,a.root_certificate_digest) + OR NOT EXISTS(SELECT 1 FROM mst2_metadata_root_anchor anchor WHERE anchor.anchor_kind='SESSION' + AND anchor.owner_key=s.snapshot_id||':'||s.session_incarnation::text AND anchor.snapshot_id=s.snapshot_id + AND anchor.session_incarnation=s.session_incarnation AND anchor.root_page=s.metadata_root + AND anchor.root_generation=s.root_generation AND anchor.root_certificate_digest=a.root_certificate_digest)) THEN RETURN false; END IF; + RETURN true; +END $$; + +CREATE FUNCTION mst2_metadata_snapshot_route_proof(sid text,descriptor bytea,instance text,commit_id text, + tree_id text,root bytea,profile jsonb) RETURNS boolean LANGUAGE sql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ + SELECT EXISTS(SELECT 1 FROM mst2_qualified_session_incarnation s WHERE s.snapshot_id=sid + AND s.canonical_descriptor=descriptor AND s.instance_id=instance AND s.commit_oid=commit_id AND s.root_tree_oid=tree_id + AND s.metadata_root=root AND s.source_profile=convert_to(profile::text,'UTF8') + AND mst2_metadata_incarnation_proof(s.snapshot_id,s.session_incarnation,false)) +$$; + +CREATE FUNCTION mst2_metadata_lease_route_proof(lid text,sid text,namespace uuid,inc uuid,pid text,root bytea, + auth bigint,publication bigint,epoch bigint,receipt bigint) RETURNS boolean LANGUAGE sql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ + SELECT EXISTS(SELECT 1 FROM mst2_qualified_lease_binding l JOIN mst2_qualified_session_incarnation s + ON s.snapshot_id=l.snapshot_id AND s.session_incarnation=l.session_incarnation + WHERE l.lease_id=lid AND l.snapshot_id=sid AND l.namespace_uuid=namespace AND namespace='$NAMESPACE_UUID$'::uuid + AND l.session_incarnation=inc AND l.prepare_id=pid AND l.metadata_root=root + AND l.authorization_epoch=auth AND l.publication_sequence=publication AND l.writer_epoch=epoch + AND l.certificate_receipt_id=receipt AND l.storage_seal=s.storage_seal AND l.root_generation=s.root_generation + AND ROW(l.authorization_epoch,l.publication_sequence,l.writer_epoch,l.certificate_receipt_id) + IS NOT DISTINCT FROM ROW(s.authorization_epoch,s.publication_sequence,s.writer_epoch,s.certificate_receipt_id) + AND mst2_metadata_incarnation_proof(s.snapshot_id,s.session_incarnation,false)) +$$; + +CREATE UNIQUE INDEX mst2_qualified_snapshot_ready ON mst2_qualified_session_incarnation(snapshot_id) WHERE state='READY'; +CREATE INDEX mst2_metadata_committed_source ON mst2_metadata_prepare(metadata_root,tagged_root_tree_oid,scope,prepare_id) + WHERE state='COMMITTED' AND plan_kind='ROOTED'; +CREATE INDEX mst2_qualified_lease_expiry ON mst2_qualified_lease_binding(expires_at_unix,lease_id) WHERE state='ACTIVE'; +CREATE FUNCTION mst2_metadata_session_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE p record; a record; profile jsonb; +BEGIN + IF TG_OP='DELETE' THEN RAISE EXCEPTION 'qualified session incarnation history is immutable'; END IF; + IF TG_OP='UPDATE' THEN + IF (to_jsonb(NEW)-'state') IS DISTINCT FROM (to_jsonb(OLD)-'state') OR OLD.state='RETIRED' AND NEW.state<>'RETIRED' + OR NEW.state NOT IN ('READY','RETIRED') THEN RAISE EXCEPTION 'qualified fixed incarnation cannot change or revive'; END IF; + IF NEW.state='RETIRED' AND EXISTS(SELECT 1 FROM mst2_qualified_lease_binding l + WHERE l.snapshot_id=OLD.snapshot_id AND l.session_incarnation=OLD.session_incarnation AND l.state='ACTIVE') THEN + RAISE EXCEPTION 'qualified ready session still has active lease identities'; END IF; + RETURN NEW; + END IF; + SELECT prepare_id,source_domain,tagged_root_tree_oid,scope,schema_version,metadata_codec, + materialization_policy,fs_semantics,access_projection,verification_revision,projection_revision + INTO p FROM mst2_metadata_prepare WHERE prepare_id=NEW.prepare_id AND state='COMMITTED' + AND plan_kind='ROOTED' AND storage_seal=NEW.storage_seal AND metadata_root=NEW.metadata_root AND coverage_retired_at IS NULL; + IF NOT FOUND OR NEW.namespace_uuid<>'$NAMESPACE_UUID$'::uuid OR NEW.state<>'READY' OR NEW.authorization_epoch<>1 + OR substr(NEW.session_incarnation::text,15,1)<>'4' OR substr(NEW.session_incarnation::text,20,1) NOT IN ('8','9','a','b') + OR EXISTS(SELECT 1 FROM mst2_qualified_session_incarnation WHERE snapshot_id=NEW.snapshot_id AND state='READY') THEN + RAISE EXCEPTION 'qualified incarnation requires a fresh server identity and definitive rooted preparation'; END IF; + profile:=$CORE_SCHEMA$.mst2_route_profile(p.source_domain,p.tagged_root_tree_oid,p.scope,p.schema_version, + p.metadata_codec,p.materialization_policy,p.fs_semantics,p.access_projection,p.verification_revision,p.projection_revision); + SELECT attestation_id,root_certificate_digest INTO a FROM mst2_metadata_source_root_attestation WHERE attestation_id=NEW.attestation_id + AND attestation_digest=NEW.attestation_digest AND root_page=NEW.metadata_root AND root_generation=NEW.root_generation; + IF NOT FOUND OR NEW.source_profile IS DISTINCT FROM convert_to(profile::text,'UTF8') + OR NEW.canonical_descriptor IS DISTINCT FROM mst2_metadata_descriptor(p.prepare_id,NEW.instance_id,NEW.commit_oid,NEW.root_tree_oid) + OR NEW.snapshot_id IS DISTINCT FROM 'sha256:'||encode(sha256(convert_to('mega.mst2.descriptor','UTF8') + ||decode('00','hex')||NEW.canonical_descriptor),'hex') + OR NOT mst2_metadata_serving_source(p.prepare_id,a.attestation_id,NEW.metadata_root,NEW.root_generation,a.root_certificate_digest) + OR NOT mst2_metadata_root_live(NEW.metadata_root,NEW.root_generation,a.root_certificate_digest) + OR NOT EXISTS(SELECT 1 FROM mst2_metadata_root_anchor anchor WHERE anchor.anchor_kind='PREPARE' + AND anchor.prepare_id=NEW.prepare_id AND anchor.owner_key=NEW.prepare_id AND anchor.root_page=NEW.metadata_root + AND anchor.root_generation=NEW.root_generation AND anchor.root_certificate_digest=a.root_certificate_digest) + OR NOT mst2_metadata_publication_valid(NEW.instance_id,NEW.commit_oid,NEW.root_tree_oid, + NEW.publication_sequence,NEW.writer_epoch,NEW.certificate_receipt_id,true) + OR NOT EXISTS(SELECT 1 FROM $CORE_SCHEMA$.mst2_snapshot_storage_route r WHERE r.snapshot_id=NEW.snapshot_id + AND r.namespace_uuid=NEW.namespace_uuid AND r.canonical_descriptor=NEW.canonical_descriptor + AND r.instance_id=NEW.instance_id AND r.commit_oid=NEW.commit_oid AND r.root_tree_oid=NEW.root_tree_oid + AND r.metadata_root=NEW.metadata_root AND r.source_profile=profile) THEN + RAISE EXCEPTION 'qualified incarnation differs from its independently derived source and permanent route'; END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_metadata_session_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_qualified_session_incarnation + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_session_guard(); + +CREATE FUNCTION mst2_metadata_lease_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE s mst2_qualified_session_incarnation%ROWTYPE; now_unix bigint:=floor(extract(epoch FROM clock_timestamp()))::bigint; +BEGIN + IF TG_OP='DELETE' THEN RAISE EXCEPTION 'qualified lease binding history is immutable'; END IF; + IF TG_OP='UPDATE' THEN + IF (to_jsonb(NEW)-ARRAY['state','expires_at_unix','lease_epoch']) IS DISTINCT FROM + (to_jsonb(OLD)-ARRAY['state','expires_at_unix','lease_epoch']) OR OLD.state<>'ACTIVE' AND NEW IS DISTINCT FROM OLD THEN + RAISE EXCEPTION 'qualified fixed lease cannot change or revive'; END IF; + IF NEW.state='ACTIVE' THEN + IF OLD.expires_at_unix<=now_unix OR NEW.expires_at_unix<=now_unix OR NEW.expires_at_unix>now_unix+3600 + OR NEW.lease_epoch<>OLD.lease_epoch THEN RAISE EXCEPTION 'qualified renewal requires its still-active bounded lease'; END IF; + ELSIF NEW.state IN ('RELEASED','EXPIRED') THEN + IF NEW.expires_at_unix<>OLD.expires_at_unix OR NEW.lease_epoch<>OLD.lease_epoch+1 + OR NEW.state='EXPIRED' AND OLD.expires_at_unix>now_unix THEN + RAISE EXCEPTION 'qualified terminal lease differs from its exact deadline and epoch'; END IF; + ELSE RAISE EXCEPTION 'qualified lease state is unsupported'; END IF; + RETURN NEW; + END IF; + SELECT * INTO s FROM mst2_qualified_session_incarnation + WHERE snapshot_id=NEW.snapshot_id AND session_incarnation=NEW.session_incarnation AND state='READY'; + IF NOT FOUND OR NEW.namespace_uuid<>'$NAMESPACE_UUID$'::uuid OR NEW.namespace_uuid<>s.namespace_uuid + OR NEW.state<>'ACTIVE' OR NEW.lease_epoch<>1 OR NEW.expires_at_unix<=now_unix OR NEW.expires_at_unix>now_unix+3600 + OR NEW.lease_id IS DISTINCT FROM (NEW.lease_id::uuid)::text OR substr(NEW.lease_id,15,1)<>'4' + OR substr(NEW.lease_id,20,1) NOT IN ('8','9','a','b') + OR ROW(NEW.prepare_id,NEW.storage_seal,NEW.metadata_root,NEW.root_generation,NEW.authorization_epoch, + NEW.publication_sequence,NEW.writer_epoch,NEW.certificate_receipt_id) IS DISTINCT FROM + ROW(s.prepare_id,s.storage_seal,s.metadata_root,s.root_generation,s.authorization_epoch, + s.publication_sequence,s.writer_epoch,s.certificate_receipt_id) + OR NOT mst2_metadata_incarnation_proof(s.snapshot_id,s.session_incarnation,true) THEN + RAISE EXCEPTION 'qualified lease differs from its exact active incarnation and source'; END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_metadata_lease_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_qualified_lease_binding + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_lease_guard(); + +CREATE FUNCTION mst2_metadata_reader_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE l mst2_qualified_lease_binding%ROWTYPE; now_unix bigint:=floor(extract(epoch FROM clock_timestamp()))::bigint; +BEGIN + IF TG_OP='DELETE' THEN RAISE EXCEPTION 'qualified reader operation history is immutable'; END IF; + IF TG_OP='UPDATE' THEN + IF (to_jsonb(NEW)-'state') IS DISTINCT FROM (to_jsonb(OLD)-'state') + OR OLD.state<>'ACTIVE' AND NEW.state<>OLD.state OR NEW.state NOT IN ('ACTIVE','FINISHED','EXPIRED') + OR NEW.state='EXPIRED' AND OLD.hard_deadline_unix>now_unix THEN + RAISE EXCEPTION 'qualified reader identity cannot change or be prematurely expired'; END IF; + RETURN NEW; + END IF; + SELECT * INTO l FROM mst2_qualified_lease_binding WHERE lease_id=NEW.lease_id AND state='ACTIVE' AND expires_at_unix>now_unix; + IF NOT FOUND OR NEW.state<>'ACTIVE' OR substr(NEW.operation_id::text,15,1)<>'4' + OR substr(NEW.operation_id::text,20,1) NOT IN ('8','9','a','b') + OR ROW(NEW.snapshot_id,NEW.session_incarnation,NEW.root_page,NEW.root_generation,NEW.lease_epoch) IS DISTINCT FROM + ROW(l.snapshot_id,l.session_incarnation,l.metadata_root,l.root_generation,l.lease_epoch) + OR NEW.hard_deadline_unix<=now_unix OR NEW.hard_deadline_unix>least(l.expires_at_unix,now_unix+60) + OR NOT mst2_metadata_incarnation_proof(l.snapshot_id,l.session_incarnation,true) THEN + RAISE EXCEPTION 'qualified reader lacks its exact active lease and bounded deadline'; END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_metadata_reader_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_metadata_reader_operation + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_reader_guard(); + +CREATE FUNCTION mst2_metadata_serving_complete() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE s mst2_qualified_session_incarnation%ROWTYPE; l mst2_qualified_lease_binding%ROWTYPE; + r mst2_metadata_reader_operation%ROWTYPE; +BEGIN + IF TG_TABLE_NAME='mst2_qualified_session_incarnation' THEN + SELECT * INTO STRICT s FROM mst2_qualified_session_incarnation + WHERE snapshot_id=NEW.snapshot_id AND session_incarnation=NEW.session_incarnation; + IF NOT mst2_metadata_incarnation_proof(s.snapshot_id,s.session_incarnation,s.state='READY') + OR s.state='READY' AND NOT EXISTS(SELECT 1 FROM mst2_qualified_lease_binding lease + WHERE lease.snapshot_id=s.snapshot_id AND lease.session_incarnation=s.session_incarnation AND lease.state='ACTIVE') + OR s.state='RETIRED' AND EXISTS(SELECT 1 FROM mst2_metadata_root_anchor a + WHERE a.anchor_kind='SESSION' AND a.snapshot_id=s.snapshot_id AND a.session_incarnation=s.session_incarnation) THEN + RAISE EXCEPTION 'qualified session cannot commit without its exact final serving roots'; END IF; + ELSIF TG_TABLE_NAME='mst2_qualified_lease_binding' THEN + SELECT * INTO STRICT l FROM mst2_qualified_lease_binding WHERE lease_id=NEW.lease_id; + IF NOT EXISTS(SELECT 1 FROM $CORE_SCHEMA$.mst2_lease_storage_route route + WHERE route.lease_id=l.lease_id AND route.namespace_uuid=l.namespace_uuid AND route.snapshot_id=l.snapshot_id + AND route.session_incarnation=l.session_incarnation AND route.prepare_id=l.prepare_id AND route.metadata_root=l.metadata_root + AND route.authorization_epoch=l.authorization_epoch AND route.publication_sequence=l.publication_sequence + AND route.writer_epoch=l.writer_epoch AND route.certificate_receipt_id=l.certificate_receipt_id) + OR l.state='ACTIVE' AND (l.expires_at_unix<=floor(extract(epoch FROM clock_timestamp()))::bigint + OR NOT mst2_metadata_incarnation_proof(l.snapshot_id,l.session_incarnation,true) + OR NOT EXISTS(SELECT 1 FROM mst2_metadata_root_anchor a WHERE a.anchor_kind='LEASE' AND a.owner_key=l.lease_id + AND a.lease_id=l.lease_id AND a.root_page=l.metadata_root AND a.root_generation=l.root_generation)) + OR l.state<>'ACTIVE' AND NOT EXISTS(SELECT 1 FROM mst2_metadata_reader_operation operation + WHERE operation.lease_id=l.lease_id AND operation.state='ACTIVE') + AND EXISTS(SELECT 1 FROM mst2_metadata_root_anchor a WHERE a.anchor_kind='LEASE' AND a.lease_id=l.lease_id) THEN + RAISE EXCEPTION 'qualified lease cannot commit without its exact final route and owned protection'; END IF; + ELSE + SELECT * INTO STRICT r FROM mst2_metadata_reader_operation WHERE operation_id=NEW.operation_id; + IF r.state='ACTIVE' AND (r.hard_deadline_unix<=floor(extract(epoch FROM clock_timestamp()))::bigint + OR (SELECT count(*) FROM mst2_metadata_root_anchor a WHERE a.reader_operation_id=r.operation_id + AND a.anchor_kind IN ('REQUEST','READER') AND a.owner_key=r.operation_id::text + AND a.lease_id=r.lease_id AND a.snapshot_id=r.snapshot_id AND a.session_incarnation=r.session_incarnation + AND a.root_page=r.root_page AND a.root_generation=r.root_generation)<>2) + OR r.state<>'ACTIVE' AND EXISTS(SELECT 1 FROM mst2_metadata_root_anchor a WHERE a.reader_operation_id=r.operation_id) THEN + RAISE EXCEPTION 'qualified reader cannot commit without both exact owned roots or definitive cleanup'; END IF; + END IF; + RETURN NULL; +END $$; +CREATE CONSTRAINT TRIGGER mst2_metadata_session_complete AFTER INSERT OR UPDATE ON mst2_qualified_session_incarnation + DEFERRABLE INITIALLY DEFERRED FOR EACH ROW EXECUTE FUNCTION mst2_metadata_serving_complete(); +CREATE CONSTRAINT TRIGGER mst2_metadata_lease_complete AFTER INSERT OR UPDATE ON mst2_qualified_lease_binding + DEFERRABLE INITIALLY DEFERRED FOR EACH ROW EXECUTE FUNCTION mst2_metadata_serving_complete(); +CREATE CONSTRAINT TRIGGER mst2_metadata_reader_complete AFTER INSERT OR UPDATE ON mst2_metadata_reader_operation + DEFERRABLE INITIALLY DEFERRED FOR EACH ROW EXECUTE FUNCTION mst2_metadata_serving_complete(); + +CREATE FUNCTION mst2_metadata_session_row(sid text,lid text,instance text) +RETURNS TABLE(canonical_descriptor bytea,commit_oid text,root_tree_oid text,expires_at_unix bigint, + authorization_epoch bigint,session_incarnation uuid,root_generation bigint,certificate_digest bytea,attestation_id uuid) +LANGUAGE plpgsql VOLATILE SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE l mst2_qualified_lease_binding%ROWTYPE; s mst2_qualified_session_incarnation%ROWTYPE; + c bytea; +BEGIN + SELECT * INTO l FROM mst2_qualified_lease_binding lease WHERE lease.lease_id=lid AND lease.snapshot_id=sid + AND lease.state='ACTIVE' AND lease.expires_at_unix>floor(extract(epoch FROM clock_timestamp()))::bigint; + IF NOT FOUND THEN RETURN; END IF; + SELECT * INTO s FROM mst2_qualified_session_incarnation session WHERE session.snapshot_id=sid + AND session.session_incarnation=l.session_incarnation AND session.state='READY' AND session.instance_id=instance; + IF NOT FOUND THEN RETURN; END IF; + SELECT proof.root_certificate_digest INTO c FROM mst2_metadata_source_root_attestation proof + WHERE proof.attestation_id=s.attestation_id AND proof.attestation_digest=s.attestation_digest; + IF NOT FOUND OR NOT mst2_metadata_incarnation_proof(sid,s.session_incarnation,true) + OR NOT $CORE_SCHEMA$.mst2_route_qualified_namespace_valid('$NAMESPACE_UUID$'::uuid) + OR NOT EXISTS(SELECT 1 FROM $CORE_SCHEMA$.mst2_lease_storage_route route + WHERE route.lease_id=lid AND route.snapshot_id=sid AND route.namespace_uuid=l.namespace_uuid + AND route.session_incarnation=l.session_incarnation AND route.prepare_id=l.prepare_id AND route.metadata_root=l.metadata_root + AND route.authorization_epoch=l.authorization_epoch AND route.publication_sequence=l.publication_sequence + AND route.writer_epoch=l.writer_epoch AND route.certificate_receipt_id=l.certificate_receipt_id) + OR l.namespace_uuid IS DISTINCT FROM s.namespace_uuid + OR ROW(l.prepare_id,l.storage_seal,l.metadata_root,l.root_generation,l.authorization_epoch, + l.publication_sequence,l.writer_epoch,l.certificate_receipt_id) IS DISTINCT FROM + ROW(s.prepare_id,s.storage_seal,s.metadata_root,s.root_generation,s.authorization_epoch, + s.publication_sequence,s.writer_epoch,s.certificate_receipt_id) + OR NOT EXISTS(SELECT 1 FROM mst2_metadata_root_anchor anchor WHERE anchor.anchor_kind='LEASE' AND anchor.owner_key=lid + AND anchor.lease_id=lid AND anchor.snapshot_id=sid AND anchor.session_incarnation=l.session_incarnation + AND anchor.root_page=l.metadata_root AND anchor.root_generation=l.root_generation AND anchor.root_certificate_digest=c) THEN + RAISE EXCEPTION 'qualified durable session route, source, current lifetime or root ownership changed'; END IF; + RETURN QUERY SELECT s.canonical_descriptor,s.commit_oid,s.root_tree_oid,l.expires_at_unix, + l.authorization_epoch,s.session_incarnation,s.root_generation,c,s.attestation_id; +END $$; + +CREATE FUNCTION mst2_metadata_handoff(request jsonb) RETURNS TABLE(lease_id text,expires_at_unix bigint) +LANGUAGE plpgsql VOLATILE SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE sid text:=request->>'snapshot_id'; descriptor bytea:=decode(request->>'canonical_descriptor','hex'); + instance text:=request->>'instance_id'; commit_id text:=request->>'commit_oid'; tree_id text:=request->>'root_tree_oid'; + lid text:=request->>'lease_id'; inc uuid:=(request->>'session_incarnation')::uuid; pid text:=request->>'prepare_id'; + p record; s mst2_qualified_session_incarnation%ROWTYPE; + a record; profile jsonb; now_unix bigint; deadline bigint; +BEGIN + PERFORM $CORE_SCHEMA$.mst2_route_enter($CORE_LITERAL$); + PERFORM mst2_metadata_cleanup_expired(64); + now_unix:=floor(extract(epoch FROM clock_timestamp()))::bigint; + deadline:=now_unix+least(3600,greatest(1,(request->>'lease_seconds')::bigint)); + IF request->>'authorization_epoch' IS DISTINCT FROM '1' OR lid IS NULL OR inc IS NULL OR descriptor IS NULL + OR NOT mst2_metadata_publication_valid(instance,commit_id,tree_id,(request->>'publication_sequence')::bigint, + (request->>'writer_epoch')::bigint,(request->>'certificate_receipt_id')::bigint,true) THEN + RAISE EXCEPTION 'qualified handoff differs from the independently observed current publication'; END IF; + SELECT * INTO s FROM mst2_qualified_session_incarnation WHERE snapshot_id=sid AND state='READY'; + IF FOUND THEN + IF s.state<>'READY' OR s.canonical_descriptor IS DISTINCT FROM descriptor OR s.instance_id IS DISTINCT FROM instance + OR s.commit_oid IS DISTINCT FROM commit_id OR s.root_tree_oid IS DISTINCT FROM tree_id + OR ROW(s.publication_sequence,s.writer_epoch,s.certificate_receipt_id) IS DISTINCT FROM + ROW((request->>'publication_sequence')::bigint,(request->>'writer_epoch')::bigint,(request->>'certificate_receipt_id')::bigint) + OR NOT mst2_metadata_incarnation_proof(sid,s.session_incarnation,true) THEN + RAISE EXCEPTION 'qualified handoff cannot revive or replace a historical incarnation'; END IF; + ELSE + SELECT metadata_root,storage_seal,source_domain,tagged_root_tree_oid,scope,schema_version,metadata_codec, + materialization_policy,fs_semantics,access_projection,verification_revision,projection_revision + INTO p FROM mst2_metadata_prepare WHERE prepare_id=pid AND state='COMMITTED' AND plan_kind='ROOTED' + AND storage_seal=decode(request->>'storage_seal','hex') AND coverage_retired_at IS NULL; + IF NOT FOUND THEN RAISE EXCEPTION 'qualified handoff has no exact definitive rooted receipt'; END IF; + SELECT attestation_id,attestation_digest,root_generation,root_certificate_digest + INTO a FROM mst2_metadata_source_root_attestation WHERE attestation_id=(request->>'attestation_id')::uuid + AND attestation_digest=decode(request->>'attestation_digest','hex') AND root_page=p.metadata_root + AND root_generation=(request->>'root_generation')::bigint AND root_certificate_digest=decode(request->>'certificate_digest','hex'); + IF NOT FOUND OR descriptor IS DISTINCT FROM mst2_metadata_descriptor(pid,instance,commit_id,tree_id) + OR NOT mst2_metadata_serving_source(pid,a.attestation_id,p.metadata_root,a.root_generation,a.root_certificate_digest) + OR NOT mst2_metadata_root_live(p.metadata_root,a.root_generation,a.root_certificate_digest) + OR NOT EXISTS(SELECT 1 FROM mst2_metadata_root_anchor anchor WHERE anchor.anchor_kind='PREPARE' + AND anchor.prepare_id=pid AND anchor.owner_key=pid AND anchor.root_page=p.metadata_root + AND anchor.root_generation=a.root_generation AND anchor.root_certificate_digest=a.root_certificate_digest) THEN + RAISE EXCEPTION 'qualified handoff lacks its continuously owned exact canonical source root'; END IF; + profile:=$CORE_SCHEMA$.mst2_route_profile(p.source_domain,p.tagged_root_tree_oid,p.scope,p.schema_version, + p.metadata_codec,p.materialization_policy,p.fs_semantics,p.access_projection,p.verification_revision,p.projection_revision); + INSERT INTO $CORE_SCHEMA$.mst2_snapshot_storage_route(snapshot_id,namespace_uuid,canonical_descriptor,instance_id, + commit_oid,root_tree_oid,metadata_root,source_profile) + VALUES(sid,'$NAMESPACE_UUID$'::uuid,descriptor,instance,commit_id,tree_id,p.metadata_root,profile) + ON CONFLICT(snapshot_id) DO NOTHING; + INSERT INTO mst2_qualified_session_incarnation(snapshot_id,session_incarnation,namespace_uuid,prepare_id,storage_seal, + metadata_root,root_generation,source_profile,instance_id,commit_oid,root_tree_oid,authorization_epoch, + publication_sequence,writer_epoch,certificate_receipt_id,canonical_descriptor,attestation_id,attestation_digest,state) + VALUES(sid,inc,'$NAMESPACE_UUID$'::uuid,pid,p.storage_seal,p.metadata_root,a.root_generation,convert_to(profile::text,'UTF8'), + instance,commit_id,tree_id,1,(request->>'publication_sequence')::bigint,(request->>'writer_epoch')::bigint, + (request->>'certificate_receipt_id')::bigint,descriptor,a.attestation_id,a.attestation_digest,'READY') RETURNING * INTO s; + INSERT INTO mst2_metadata_root_anchor(anchor_id,anchor_kind,owner_key,root_page,root_generation,root_certificate_digest, + snapshot_id,session_incarnation) VALUES(gen_random_uuid(),'SESSION',sid||':'||inc::text, + s.metadata_root,s.root_generation,a.root_certificate_digest,sid,inc); + END IF; + SELECT root_certificate_digest INTO STRICT a FROM mst2_metadata_source_root_attestation WHERE attestation_id=s.attestation_id; + INSERT INTO mst2_qualified_lease_binding(lease_id,snapshot_id,session_incarnation,prepare_id,storage_seal,metadata_root, + root_generation,namespace_uuid,authorization_epoch,publication_sequence,writer_epoch,certificate_receipt_id, + expires_at_unix,state,lease_epoch) VALUES(lid,sid,s.session_incarnation,s.prepare_id,s.storage_seal,s.metadata_root, + s.root_generation,s.namespace_uuid,s.authorization_epoch,s.publication_sequence,s.writer_epoch,s.certificate_receipt_id, + deadline,'ACTIVE',1); + INSERT INTO $CORE_SCHEMA$.mst2_lease_storage_route(lease_id,snapshot_id,namespace_uuid,session_incarnation,prepare_id, + metadata_root,authorization_epoch,publication_sequence,writer_epoch,certificate_receipt_id) + VALUES(lid,sid,s.namespace_uuid,s.session_incarnation,s.prepare_id,s.metadata_root,s.authorization_epoch, + s.publication_sequence,s.writer_epoch,s.certificate_receipt_id); + INSERT INTO mst2_metadata_root_anchor(anchor_id,anchor_kind,owner_key,root_page,root_generation,root_certificate_digest, + snapshot_id,session_incarnation,lease_id) VALUES(gen_random_uuid(),'LEASE',lid,s.metadata_root,s.root_generation, + a.root_certificate_digest,sid,s.session_incarnation,lid); + IF pid IS NOT NULL THEN + SELECT coverage_retired_at INTO p FROM mst2_metadata_prepare WHERE prepare_id=pid AND state='COMMITTED' AND plan_kind='ROOTED' + AND storage_seal=decode(request->>'storage_seal','hex') AND metadata_root=s.metadata_root; + IF NOT FOUND OR s.root_generation IS DISTINCT FROM (request->>'root_generation')::bigint + OR a.root_certificate_digest IS DISTINCT FROM decode(request->>'certificate_digest','hex') + OR descriptor IS DISTINCT FROM mst2_metadata_descriptor(pid,instance,commit_id,tree_id) + OR NOT mst2_metadata_session_covers_prepare(pid) + OR NOT EXISTS(SELECT 1 FROM mst2_metadata_source_root_attestation receipt + WHERE receipt.attestation_id=(request->>'attestation_id')::uuid + AND receipt.attestation_digest=decode(request->>'attestation_digest','hex') + AND receipt.root_page=s.metadata_root AND receipt.root_generation=s.root_generation + AND receipt.root_certificate_digest=a.root_certificate_digest + AND mst2_metadata_serving_source(pid,receipt.attestation_id,s.metadata_root,s.root_generation,a.root_certificate_digest)) THEN + RAISE EXCEPTION 'qualified losing preparation cannot retire a different serving identity'; END IF; + IF p.coverage_retired_at IS NULL THEN + UPDATE mst2_metadata_prepare SET coverage_retired_at=clock_timestamp() WHERE prepare_id=pid; + DELETE FROM mst2_metadata_root_anchor WHERE prepare_id=pid AND anchor_kind IN ('PREPARE','REUSE'); + END IF; + END IF; + RETURN QUERY SELECT lid,deadline; +END $$; + +CREATE FUNCTION mst2_metadata_cleanup_lease(lid text) RETURNS void LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE l mst2_qualified_lease_binding%ROWTYPE; +BEGIN + SELECT * INTO l FROM mst2_qualified_lease_binding WHERE lease_id=lid; + IF NOT FOUND THEN RETURN; END IF; + IF l.state<>'ACTIVE' AND NOT EXISTS(SELECT 1 FROM mst2_metadata_reader_operation WHERE lease_id=lid AND state='ACTIVE') THEN + DELETE FROM mst2_metadata_root_anchor WHERE lease_id=lid AND anchor_kind='LEASE'; + END IF; + IF NOT EXISTS(SELECT 1 FROM mst2_qualified_lease_binding lease WHERE lease.snapshot_id=l.snapshot_id + AND lease.session_incarnation=l.session_incarnation AND lease.state='ACTIVE') THEN + UPDATE mst2_qualified_session_incarnation SET state='RETIRED' + WHERE snapshot_id=l.snapshot_id AND session_incarnation=l.session_incarnation AND state='READY'; + DELETE FROM mst2_metadata_root_anchor WHERE snapshot_id=l.snapshot_id + AND session_incarnation=l.session_incarnation AND anchor_kind='SESSION'; + END IF; +END $$; + +CREATE FUNCTION mst2_metadata_cleanup_expired(maximum integer DEFAULT 64) RETURNS void LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE item record; now_unix bigint; bound integer:=least(64,greatest(0,maximum)); +BEGIN + PERFORM $CORE_SCHEMA$.mst2_route_enter($CORE_LITERAL$); + now_unix:=floor(extract(epoch FROM clock_timestamp()))::bigint; + FOR item IN SELECT * FROM ( + (SELECT 'READER'::text AS kind,operation_id::text AS owner,lease_id,hard_deadline_unix AS deadline + FROM mst2_metadata_reader_operation WHERE state='ACTIVE' AND hard_deadline_unix<=now_unix + ORDER BY hard_deadline_unix,operation_id LIMIT bound) + UNION ALL + (SELECT 'LEASE'::text,lease_id,lease_id,expires_at_unix + FROM mst2_qualified_lease_binding WHERE state='ACTIVE' AND expires_at_unix<=now_unix + ORDER BY expires_at_unix,lease_id LIMIT bound) + ) expired ORDER BY deadline,kind,owner LIMIT bound LOOP + IF item.kind='READER' THEN + UPDATE mst2_metadata_reader_operation SET state='EXPIRED' WHERE operation_id=item.owner::uuid AND state='ACTIVE'; + DELETE FROM mst2_metadata_root_anchor WHERE reader_operation_id=item.owner::uuid AND anchor_kind IN ('REQUEST','READER'); + ELSE + UPDATE mst2_qualified_lease_binding SET state='EXPIRED',lease_epoch=lease_epoch+1 WHERE lease_id=item.owner AND state='ACTIVE'; + END IF; + PERFORM mst2_metadata_cleanup_lease(item.lease_id); + END LOOP; +END $$; + +CREATE FUNCTION mst2_metadata_renew_lease(lid text,seconds bigint,instance text) +RETURNS TABLE(snapshot_id text,expires_at_unix bigint) LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE l mst2_qualified_lease_binding%ROWTYPE; now_unix bigint; deadline bigint; +BEGIN + PERFORM $CORE_SCHEMA$.mst2_route_enter($CORE_LITERAL$); + PERFORM mst2_metadata_cleanup_expired(64); + now_unix:=floor(extract(epoch FROM clock_timestamp()))::bigint; + SELECT * INTO l FROM mst2_qualified_lease_binding WHERE lease_id=lid AND state='ACTIVE'; + IF NOT FOUND THEN RETURN; END IF; + IF l.expires_at_unix<=now_unix THEN + UPDATE mst2_qualified_lease_binding SET state='EXPIRED',lease_epoch=lease_epoch+1 WHERE lease_id=lid; + PERFORM mst2_metadata_cleanup_lease(lid); RETURN; + END IF; + IF NOT EXISTS(SELECT 1 FROM mst2_metadata_session_row(l.snapshot_id,lid,instance)) THEN RETURN; END IF; + deadline:=greatest(l.expires_at_unix,now_unix+least(3600,greatest(1,seconds))); + UPDATE mst2_qualified_lease_binding lease SET expires_at_unix=deadline WHERE lease.lease_id=lid; + RETURN QUERY SELECT l.snapshot_id,deadline; +END $$; + +CREATE FUNCTION mst2_metadata_release_lease(lid text) RETURNS boolean LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE l mst2_qualified_lease_binding%ROWTYPE; changed boolean; +BEGIN + PERFORM $CORE_SCHEMA$.mst2_route_enter($CORE_LITERAL$); + PERFORM mst2_metadata_cleanup_expired(64); + SELECT * INTO l FROM mst2_qualified_lease_binding WHERE lease_id=lid; + IF NOT FOUND THEN RETURN false; END IF; + IF NOT mst2_metadata_lease_route_proof(lid,l.snapshot_id,l.namespace_uuid,l.session_incarnation,l.prepare_id,l.metadata_root, + l.authorization_epoch,l.publication_sequence,l.writer_epoch,l.certificate_receipt_id) THEN + RAISE EXCEPTION 'qualified release lost its fixed historical source identity'; END IF; + changed:=l.state='ACTIVE'; + IF changed THEN UPDATE mst2_qualified_lease_binding SET state='RELEASED',lease_epoch=lease_epoch+1 WHERE lease_id=lid; END IF; + PERFORM mst2_metadata_cleanup_lease(lid); + RETURN changed; +END $$; + +CREATE FUNCTION mst2_metadata_begin_reader(sid text,lid text,instance text) +RETURNS TABLE(operation_id uuid,root_generation bigint,certificate_digest bytea) LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE l mst2_qualified_lease_binding%ROWTYPE; s record; op uuid:=gen_random_uuid(); deadline bigint; kind text; +BEGIN + PERFORM $CORE_SCHEMA$.mst2_route_enter($CORE_LITERAL$); + PERFORM mst2_metadata_cleanup_expired(64); + SELECT * INTO s FROM mst2_metadata_session_row(sid,lid,instance); + IF NOT FOUND THEN RETURN; END IF; + SELECT * INTO STRICT l FROM mst2_qualified_lease_binding WHERE lease_id=lid; + deadline:=least(l.expires_at_unix,floor(extract(epoch FROM clock_timestamp()))::bigint+60); + INSERT INTO mst2_metadata_reader_operation(operation_id,lease_id,snapshot_id,session_incarnation,root_page,root_generation, + lease_epoch,hard_deadline_unix,state) VALUES(op,lid,sid,l.session_incarnation,l.metadata_root,l.root_generation,l.lease_epoch,deadline,'ACTIVE'); + FOREACH kind IN ARRAY ARRAY['REQUEST','READER'] LOOP + INSERT INTO mst2_metadata_root_anchor(anchor_id,anchor_kind,owner_key,root_page,root_generation,root_certificate_digest, + snapshot_id,session_incarnation,lease_id,reader_operation_id) VALUES(gen_random_uuid(),kind,op::text, + l.metadata_root,l.root_generation,s.certificate_digest,sid,l.session_incarnation,lid,op); + END LOOP; + RETURN QUERY SELECT op,l.root_generation,s.certificate_digest::bytea; +END $$; + +CREATE FUNCTION mst2_metadata_finish_reader(op uuid) RETURNS void LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE r mst2_metadata_reader_operation%ROWTYPE; +BEGIN + PERFORM $CORE_SCHEMA$.mst2_route_enter($CORE_LITERAL$); + SELECT * INTO r FROM mst2_metadata_reader_operation WHERE operation_id=op; + IF NOT FOUND THEN RETURN; END IF; + IF r.state='ACTIVE' THEN UPDATE mst2_metadata_reader_operation SET state='FINISHED' WHERE operation_id=op; END IF; + DELETE FROM mst2_metadata_root_anchor WHERE reader_operation_id=op AND anchor_kind IN ('REQUEST','READER'); + PERFORM mst2_metadata_cleanup_lease(r.lease_id); +END $$; + +DROP TRIGGER mst2_01_family_closed ON mst2_qualified_session_incarnation; +DROP TRIGGER mst2_01_family_closed ON mst2_qualified_lease_binding; +DROP TRIGGER mst2_01_family_closed ON mst2_metadata_reader_operation; diff --git a/src/jupiter/storage/qualified_metadata_session.rs b/src/jupiter/storage/qualified_metadata_session.rs new file mode 100644 index 00000000..8e29b301 --- /dev/null +++ b/src/jupiter/storage/qualified_metadata_session.rs @@ -0,0 +1,291 @@ +//! Rooted session handoff and lease operations over the captured physical Q. + +use mst2_codec::descriptor::ServingDescriptor; + +use super::*; +use crate::{ + ceres::snapshot::{ + descriptor::BuiltDescriptor, + rooted_metadata_projection::PreparedRootedNativeMetadata, + runtime::{LeaseRenewed, SnapshotContext}, + view::{SnapshotView, hex}, + }, + jupiter::storage::native_publication_storage::NativePublicationHead, +}; + +fn gone() -> SnapshotError { + SnapshotError::new( + crate::ceres::snapshot::error::SnapshotErrorCode::LeaseExpired, + "lease is not active for this snapshot; re-resolve", + ) +} + +pub(super) async fn finish( + txn: DatabaseTransaction, + result: Result, +) -> Result { + match result { + Ok(value) => { + txn.commit().await.map_err(|_| { + SnapshotError::new( + crate::ceres::snapshot::error::SnapshotErrorCode::TemporaryUnavailable, + "qualified transaction outcome unknown; retry the original operation", + ) + })?; + Ok(value) + } + Err(error) => { + let _ = txn.rollback().await; + Err(error) + } + } +} + +pub(super) fn context( + row: &QueryResult, + sid: &str, + lease: &str, + instance: &str, +) -> Result { + let bytes: Vec = row.try_get("", "canonical_descriptor").map_err(internal)?; + let descriptor = ServingDescriptor::decode(&bytes).map_err(internal)?; + let commit: String = row.try_get("", "commit_oid").map_err(internal)?; + let tree: String = row.try_get("", "root_tree_oid").map_err(internal)?; + let view = SnapshotView::from_commit(&commit, &tree); + if format!( + "sha256:{}", + hex(&descriptor.snapshot_id().map_err(internal)?) + ) != sid + || uuid::Uuid::from_bytes(descriptor.instance_uuid).to_string() != instance + || format!("sha256:{}", hex(&descriptor.namespace_view_id)) != view.view_id + { + return Err(integrity( + "qualified durable descriptor differs from its fixed source", + )); + } + let deadline = u64::try_from( + row.try_get::("", "expires_at_unix") + .map_err(internal)?, + ) + .map_err(internal)?; + let epoch = u64::try_from( + row.try_get::("", "authorization_epoch") + .map_err(internal)?, + ) + .map_err(internal)?; + let metadata_root = format!("sha256:{}", hex(&descriptor.metadata_root)); + Ok(SnapshotContext { + built: BuiltDescriptor { + descriptor, + instance_id: instance.into(), + snapshot_id: sid.into(), + metadata_root, + }, + commit_oid: commit, + root_tree_oid: tree, + lease_id: lease.into(), + lease_expires_at_unix: deadline, + authorization_epoch: epoch, + }) +} + +impl RootedQualifiedMetadataRepository { + pub(crate) async fn install( + &self, + built: &BuiltDescriptor, + prepared: &PreparedRootedNativeMetadata, + ) -> Result { + if prepared.plan.root != built.descriptor.metadata_root + || prepared.plan.identity.scope != built.descriptor.scope + { + return Err(integrity("rooted projection differs from the serving descriptor").into()); + } + // Each attempt has its own durable identity. A previous retired + // incarnation may have the same cold plan but a different generation; + // its historical receipt cannot install or resurrect this attempt. + let operation = format!("rooted-http:{}:{}", built.snapshot_id, uuid::Uuid::new_v4()); + let intent = self.begin_intent(&operation, &prepared.plan).await?; + for pages in prepared.payloads.chunks(64) { + self.install_pages(&intent, pages).await?; + } + self.finalize(&intent).await + } + + pub(crate) async fn open_session( + &self, + expected: &NativePublicationHead, + built: &BuiltDescriptor, + receipt: Option<&RootedMetadataReceipt>, + seconds: u64, + ) -> Result, SnapshotError> { + let txn = self.transaction().await?; + let result = async { + if receipt.is_none() { + txn.execute_raw(sql("SELECT mst2_metadata_cleanup_expired(64)",[])).await.map_err(database_error)?; + } + let present = txn.query_one_raw(sql("SELECT session_incarnation FROM mst2_qualified_session_incarnation + WHERE snapshot_id=$1 AND state='READY'", [built.snapshot_id.clone().into()])).await.map_err(database_error)?; + if present.is_none() && receipt.is_none() { return Ok(None); } + if let Some(receipt) = receipt { + self.require_intent(&txn, &receipt.intent).await?; + if self.receipt(&txn, &receipt.intent).await? != *receipt + || receipt.metadata_root() != built.descriptor.metadata_root + { return Err(integrity("qualified handoff receipt differs from its definitive canonical root")); } + } + let request = json!({ + "snapshot_id":built.snapshot_id,"canonical_descriptor":hex::encode(built.descriptor.encode().map_err(internal)?), + "instance_id":expected.instance_id,"commit_oid":expected.root.commit,"root_tree_oid":expected.root.tree, + "publication_sequence":expected.token.sequence,"writer_epoch":expected.token.epoch, + "certificate_receipt_id":expected.token.certificate,"authorization_epoch":1, + "lease_id":uuid::Uuid::new_v4().to_string(),"lease_seconds":seconds.clamp(1,3600), + "session_incarnation":uuid::Uuid::new_v4().to_string(), + "prepare_id":receipt.map(|r|r.intent.prepare_id.as_str()), + "storage_seal":receipt.map(|r|hex::encode(r.intent.storage_seal)), + "root_generation":receipt.map(|r|r.intent.root_generation), + "attestation_id":receipt.map(|r|r.attestation_id.to_string()), + "attestation_digest":receipt.map(|r|hex::encode(r.attestation_digest)), + "certificate_digest":receipt.map(|r|hex::encode(r.certificate_digest)), + }); + let row = txn.query_one_raw(sql("SELECT * FROM mst2_metadata_handoff($1::jsonb)", [request.into()])) + .await.map_err(database_error)?.ok_or_else(|| integrity("qualified handoff returned no lease"))?; + let lease: String = row.try_get("", "lease_id").map_err(internal)?; + let row = self.session_row(&txn, &built.snapshot_id, &lease, &expected.instance_id).await?; + let context = context(&row, &built.snapshot_id, &lease, &expected.instance_id)?; + if context.built.descriptor != built.descriptor || context.commit_oid != expected.root.commit + || context.root_tree_oid != expected.root.tree { return Err(integrity("qualified handoff fixed source changed")); } + Ok(Some(context)) + }.await; + finish(txn, result).await + } + + pub(super) async fn session_row( + &self, + db: &C, + sid: &str, + lease: &str, + instance: &str, + ) -> Result { + db.query_one_raw(sql( + "SELECT * FROM mst2_metadata_session_row($1,$2,$3)", + [sid.into(), lease.into(), instance.into()], + )) + .await + .map_err(database_error)? + .ok_or_else(gone) + } + + pub(crate) async fn context( + &self, + sid: &str, + lease: &str, + instance: &str, + ) -> Result { + // Read-only revalidation grants no mutable authority. The helper checks + // the exact route, incarnation, publication, LIVE root and owned roots. + let txn = self.read_transaction().await?; + let result = async { + let row = self.session_row(&txn, sid, lease, instance).await?; + context(&row, sid, lease, instance) + } + .await; + finish(txn, result).await + } + + pub(crate) async fn renew( + &self, + lease: &str, + seconds: u64, + instance: &str, + ) -> Result { + let txn = self.transaction().await?; + let result = async { + let row = txn + .query_one_raw(sql( + "SELECT * FROM mst2_metadata_renew_lease($1,$2,$3)", + [ + lease.into(), + (seconds.clamp(1, 3600) as i64).into(), + instance.into(), + ], + )) + .await + .map_err(database_error)? + .ok_or_else(gone)?; + Ok(LeaseRenewed { + lease_id: lease.into(), + snapshot_id: row.try_get("", "snapshot_id").map_err(internal)?, + expires_at_unix: u64::try_from( + row.try_get::("", "expires_at_unix") + .map_err(internal)?, + ) + .map_err(internal)?, + }) + } + .await; + finish(txn, result).await + } + + pub(crate) async fn release(&self, lease: &str) -> Result { + let txn = self.transaction().await?; + let result = async { + txn.query_one_raw(sql( + "SELECT mst2_metadata_release_lease($1) AS released", + [lease.into()], + )) + .await + .map_err(database_error)? + .ok_or_else(|| integrity("qualified release result missing"))? + .try_get("", "released") + .map_err(internal) + } + .await; + finish(txn, result).await + } + + pub(super) async fn read_transaction(&self) -> Result { + let txn = self + .connection + .begin_with_config(Some(IsolationLevel::ReadCommitted), None) + .await + .map_err(database_error)?; + if registered( + &txn, + &(self.namespace.core_schema.clone(), self.namespace.core_oid), + ) + .await + .map_err(|error| match error { + crate::common::errors::MegaError::Db(error) => database_error(error), + error => internal(error), + })? + .as_ref() + != Some(&self.namespace) + { + return Err(integrity( + "qualified read physical namespace or authority catalog changed", + )); + } + let valid: bool = txn + .query_one_raw(sql( + "SELECT mst2_metadata_scope_matches($1) AS valid", + [self.primary_scope.clone().into()], + )) + .await + .map_err(database_error)? + .ok_or_else(|| integrity("qualified read scope missing"))? + .try_get("", "valid") + .map_err(internal)?; + if !valid { + return Err(integrity("qualified read left its captured primary scope")); + } + #[cfg(test)] + if super::reader::READER_TEMP_SOURCE_SHADOW + .try_with(|shadow| *shadow) + .unwrap_or(false) + { + txn.execute_unprepared("CREATE TEMP TABLE mega_tree(id bigint,tree_id text,sub_trees bytea) ON COMMIT DROP; + CREATE TEMP TABLE mst2_rooted_source_tree_revision(tree_id text,tree_row_id bigint,revision uuid,body_digest bytea,valid boolean) ON COMMIT DROP") + .await.map_err(database_error)?; + } + Ok(txn) + } +} diff --git a/src/jupiter/storage/qualified_metadata_source_read.sql b/src/jupiter/storage/qualified_metadata_source_read.sql new file mode 100644 index 00000000..232a0e6d --- /dev/null +++ b/src/jupiter/storage/qualified_metadata_source_read.sql @@ -0,0 +1,132 @@ +-- Indexed fixed-source names are independently derived by the attestation +-- proof. Selected windows revalidate current file facts without source scans. +CREATE TABLE mst2_metadata_source_entry_reference ( + attestation_id uuid NOT NULL REFERENCES mst2_metadata_source_root_attestation(attestation_id), + name bytea NOT NULL CHECK(octet_length(name) BETWEEN 1 AND 255), + git_oid text NOT NULL CHECK(git_oid ~ '^(sha1:[0-9a-f]{40}|sha256:[0-9a-f]{64}|blake3:[0-9a-f]{64})$'), + kind smallint NOT NULL CHECK(kind BETWEEN 1 AND 4), + byte_size bigint,content_digest bytea,child_root bytea,child_generation bigint,child_certificate_digest bytea, + PRIMARY KEY(attestation_id,name), + CHECK(((kind=4 AND byte_size IS NULL AND content_digest IS NULL AND octet_length(child_root)=32 + AND child_generation>0 AND octet_length(child_certificate_digest)=32) + OR (kind<>4 AND byte_size BETWEEN 0 AND 8796093022208 AND octet_length(content_digest)=32 + AND child_root IS NULL AND child_generation IS NULL AND child_certificate_digest IS NULL)) IS TRUE), + CHECK((kind<>3 OR byte_size BETWEEN 1 AND 4095) IS TRUE) +); + +CREATE FUNCTION mst2_metadata_source_entry_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + RAISE EXCEPTION 'qualified fixed source entry history is immutable'; +END $$; +CREATE TRIGGER mst2_metadata_source_entry_guard BEFORE UPDATE OR DELETE ON mst2_metadata_source_entry_reference + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_source_entry_guard(); + +CREATE FUNCTION mst2_metadata_source_entries_proof() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE source_id uuid; expected jsonb; supplied jsonb; +BEGIN + FOR source_id IN SELECT DISTINCT attestation_id FROM added_source_entries LOOP + SELECT source_proof->'source_entries' INTO expected FROM mst2_metadata_source_root_attestation + WHERE attestation_id=source_id; + SELECT jsonb_object_agg(encode(reference.name,'hex'), + jsonb_build_object('kind',reference.kind,'git_oid',reference.git_oid)||CASE WHEN reference.kind=4 + THEN jsonb_build_object('child_root',encode(reference.child_root,'hex'), + 'child_generation',reference.child_generation,'child_certificate',encode(reference.child_certificate_digest,'hex')) + ELSE jsonb_build_object('size',reference.byte_size,'content_digest',encode(reference.content_digest,'hex')) END) + INTO supplied FROM added_source_entries reference WHERE reference.attestation_id=source_id; + IF expected IS NULL OR expected IS DISTINCT FROM supplied THEN + RAISE EXCEPTION 'fixed source entry batch differs from independently derived complete body and fact proof'; END IF; + END LOOP; + RETURN NULL; +END $$; +CREATE TRIGGER mst2_metadata_source_entries_proof AFTER INSERT ON mst2_metadata_source_entry_reference + REFERENCING NEW TABLE AS added_source_entries FOR EACH STATEMENT EXECUTE FUNCTION mst2_metadata_source_entries_proof(); + +CREATE FUNCTION mst2_metadata_source_entries_install() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + INSERT INTO mst2_metadata_source_entry_reference(attestation_id,name,git_oid,kind,byte_size,content_digest, + child_root,child_generation,child_certificate_digest) + SELECT NEW.attestation_id,decode(key,'hex'),value->>'git_oid',(value->>'kind')::smallint, + (value->>'size')::bigint,decode(value->>'content_digest','hex'),decode(value->>'child_root','hex'), + (value->>'child_generation')::bigint,decode(value->>'child_certificate','hex') + FROM jsonb_each(NEW.source_proof->'source_entries'); + RETURN NULL; +END $$; +CREATE TRIGGER mst2_metadata_source_entries_install AFTER INSERT ON mst2_metadata_source_root_attestation + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_source_entries_install(); + +CREATE FUNCTION mst2_metadata_source_entries_complete() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + IF (SELECT count(*) FROM mst2_metadata_source_entry_reference WHERE attestation_id=NEW.attestation_id) + IS DISTINCT FROM (NEW.source_proof->>'source_entry_count')::bigint THEN + RAISE EXCEPTION 'source attestation cannot commit with incomplete exact name references'; END IF; + RETURN NULL; +END $$; +CREATE CONSTRAINT TRIGGER mst2_metadata_source_entries_complete AFTER INSERT ON mst2_metadata_source_root_attestation + DEFERRABLE INITIALLY DEFERRED FOR EACH ROW EXECUTE FUNCTION mst2_metadata_source_entries_complete(); + +CREATE FUNCTION mst2_metadata_read_source_entries(op uuid,source_id uuid,p bytea,g bigint,c bytea,names jsonb) +RETURNS TABLE(name bytea,git_oid text,kind smallint,byte_size bigint,content_digest bytea,child_root bytea, + child_generation bigint,child_certificate_digest bytea,child_attestation_id uuid,fact_state text) +LANGUAGE plpgsql VOLATILE SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE a record; profile jsonb; +BEGIN + IF jsonb_typeof(names)<>'array' OR jsonb_array_length(names)>256 + OR EXISTS(SELECT 1 FROM jsonb_array_elements_text(names) wanted + WHERE wanted !~ '^([0-9a-f]{2}){1,255}$') THEN + RAISE EXCEPTION 'qualified source-name read exceeds its bounded exact request'; END IF; + SELECT source.namespace_uuid,source.source_profile,source.tagged_tree_oid,source.source_body_digest,source.source_revision, + source.root_page,source.root_generation,source.root_certificate_digest + INTO a FROM mst2_metadata_source_root_attestation source + JOIN mst2_metadata_prepare origin ON origin.prepare_id=source.origin_prepare_id + WHERE source.attestation_id=source_id AND origin.state='COMMITTED' + AND source.root_page=p AND source.root_generation=g AND source.root_certificate_digest=c + AND source.namespace_uuid='$NAMESPACE_UUID$'::uuid; + IF NOT FOUND THEN RAISE EXCEPTION 'qualified selected directory has no definitive source attestation'; END IF; + SELECT mst2_metadata_native_profile(session.prepare_id) INTO profile + FROM mst2_metadata_reader_operation reader JOIN mst2_qualified_session_incarnation session + ON session.snapshot_id=reader.snapshot_id AND session.session_incarnation=reader.session_incarnation + WHERE reader.operation_id=op AND reader.state='ACTIVE' + AND reader.hard_deadline_unix>floor(extract(epoch FROM clock_timestamp()))::bigint + AND EXISTS(SELECT 1 FROM mst2_metadata_root_anchor anchor WHERE anchor.reader_operation_id=reader.operation_id + AND anchor.anchor_kind='READER' AND anchor.root_page=reader.root_page AND anchor.root_generation=reader.root_generation); + IF NOT FOUND OR profile IS DISTINCT FROM a.source_profile + OR NOT mst2_metadata_root_live(a.root_page,a.root_generation,a.root_certificate_digest) THEN + RAISE EXCEPTION 'qualified source read lost its exact reader profile and current directory'; END IF; + IF NOT $CORE_SCHEMA$.mst2_route_source_tree_matches(split_part(a.tagged_tree_oid,':',2),a.source_revision,a.source_body_digest) THEN + RAISE EXCEPTION 'qualified selected directory source body changed'; END IF; + RETURN QUERY SELECT reference.name,reference.git_oid,reference.kind,reference.byte_size,reference.content_digest, + reference.child_root,reference.child_generation,reference.child_certificate_digest,child.attestation_id, + CASE WHEN reference.kind=4 THEN CASE WHEN child.attestation_id IS NULL THEN 'SOURCE_UNAVAILABLE' ELSE 'READY' END + WHEN fact.git_oid IS NULL THEN 'MISSING' + WHEN fact.state<>'VERIFIED' OR fact.verification_version NOT IN (1,2) + OR fact.size NOT BETWEEN 0 AND 8796093022208 OR octet_length(fact.raw_sha256)<>32 THEN 'INVALID' + WHEN fact.verification_version=1 THEN 'MISSING' + WHEN fact.size IS DISTINCT FROM reference.byte_size OR reference.kind=3 AND fact.size NOT BETWEEN 1 AND 4095 + OR CASE WHEN octet_length(fact.raw_sha256)=32 THEN fact.raw_sha256 ELSE NULL END + IS DISTINCT FROM reference.content_digest THEN 'INVALID' + ELSE 'READY' END + FROM (SELECT DISTINCT decode(value,'hex') AS name FROM jsonb_array_elements_text(names)) wanted + JOIN mst2_metadata_source_entry_reference reference ON reference.attestation_id=source_id AND reference.name=wanted.name + LEFT JOIN LATERAL ( + SELECT verified.git_oid,verified.state,verified.verification_version,verified.size,verified.raw_sha256 + FROM $CORE_SCHEMA$.mst2_verified_object verified WHERE reference.kind<>4 AND verified.storage_domain='git' + AND verified.object_kind='blob' AND verified.git_oid=split_part(reference.git_oid,':',2) + FOR SHARE OF verified NOWAIT + ) fact ON true + LEFT JOIN LATERAL ( + SELECT candidate.attestation_id FROM mst2_metadata_source_root_attestation candidate + JOIN mst2_metadata_prepare origin ON origin.prepare_id=candidate.origin_prepare_id + JOIN $CORE_SCHEMA$.mega_tree source_tree ON source_tree.tree_id=split_part(candidate.tagged_tree_oid,':',2) + WHERE reference.kind=4 AND candidate.namespace_uuid=a.namespace_uuid AND origin.state='COMMITTED' + AND candidate.tagged_tree_oid=reference.git_oid AND candidate.source_profile=a.source_profile + AND candidate.root_page=reference.child_root AND candidate.root_generation=reference.child_generation + AND candidate.root_certificate_digest=reference.child_certificate_digest + AND $CORE_SCHEMA$.mst2_route_source_tree_matches(split_part(candidate.tagged_tree_oid,':',2),candidate.source_revision,candidate.source_body_digest) + AND mst2_metadata_root_live(candidate.root_page,candidate.root_generation,candidate.root_certificate_digest) + ORDER BY candidate.attestation_id LIMIT 1 + ) child ON true ORDER BY reference.name; +END $$; diff --git a/src/jupiter/storage/qualified_metadata_source_revision_tests.rs b/src/jupiter/storage/qualified_metadata_source_revision_tests.rs new file mode 100644 index 00000000..9bbb6c54 --- /dev/null +++ b/src/jupiter/storage/qualified_metadata_source_revision_tests.rs @@ -0,0 +1,329 @@ +use super::{canonical_tests::seeded_rooted_plan, *}; +use crate::ceres::snapshot::rooted_metadata_projection::RootedReuseLookup; + +async fn revision(core: &DatabaseConnection, oid: &str) -> String { + core.query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT revision::text FROM mst2_rooted_source_tree_revision WHERE tree_id=$1", + [oid.into()], + )) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap() +} + +#[tokio::test] +async fn source_revision_rejects_forgery_preserves_unchanged_writes_and_requires_fresh_attestation() +{ + let (config, core, namespace, q, _guard) = fixture().await; + let (plan, payload) = seeded_rooted_plan(&core, 'a', "file", 1).await; + let writer = RootedQualifiedMetadataRepository::open(&core, &config) + .await + .unwrap(); + let intent = writer + .begin_intent("source-revision-first", &plan) + .await + .unwrap(); + writer + .install_pages(&intent, std::slice::from_ref(&payload)) + .await + .unwrap(); + writer.finalize(&intent).await.unwrap(); + let oid = "a".repeat(40); + let original = revision(&core, &oid).await; + for statement in [ + "UPDATE mst2_rooted_source_tree_revision SET revision=gen_random_uuid()", + "UPDATE mst2_rooted_source_tree_revision SET body_digest=decode(repeat('11',32),'hex')", + "UPDATE mst2_rooted_source_tree_revision SET valid=false", + "DELETE FROM mst2_rooted_source_tree_revision", + "TRUNCATE mst2_rooted_source_tree_revision", + ] { + assert!( + core.execute_unprepared(statement).await.is_err(), + "{statement}" + ); + } + assert_eq!(revision(&core, &oid).await, original); + core.execute_unprepared( + "CREATE TABLE source_revision_spoof(id integer); + CREATE FUNCTION source_revision_spoof_trigger() RETURNS trigger LANGUAGE plpgsql AS $$ + BEGIN UPDATE mst2_rooted_source_tree_revision SET valid=false; RETURN NEW; END $$; + CREATE TRIGGER source_revision_spoof AFTER INSERT ON source_revision_spoof + FOR EACH ROW EXECUTE FUNCTION source_revision_spoof_trigger()", + ) + .await + .unwrap(); + assert!( + core.execute_unprepared("INSERT INTO source_revision_spoof VALUES(1)") + .await + .is_err() + ); + assert_eq!(revision(&core, &oid).await, original); + core.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "UPDATE mega_tree SET sub_trees=sub_trees,pack_offset=pack_offset WHERE tree_id=$1", + [oid.clone().into()], + )) + .await + .unwrap(); + assert_eq!(revision(&core, &oid).await, original); + assert!( + writer + .lookup_reuse(&plan.identity.tagged_root_tree_oid, &plan.identity) + .await + .unwrap() + .is_some() + ); + core.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "UPDATE mega_tree SET sub_trees=set_byte(sub_trees,7,120) WHERE tree_id=$1", + [oid.clone().into()], + )) + .await + .unwrap(); + assert_ne!(revision(&core, &oid).await, original); + assert_eq!( + count( + &core, + "SELECT count(*) FROM mst2_rooted_source_tree_revision WHERE valid" + ) + .await, + 0 + ); + assert!( + writer + .lookup_reuse(&plan.identity.tagged_root_tree_oid, &plan.identity) + .await + .unwrap() + .is_none() + ); + let rejected=core.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "SELECT mst2_route_capture_source_tree($1,(SELECT source_body_digest FROM {q}.mst2_metadata_source_root_attestation LIMIT 1))" + .replace("{q}",&identifier(&namespace.schema)),[oid.clone().into()])).await; + assert!(rejected.is_err()); + // Restoring the same old bytes does not silently resurrect the old revision. + core.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "UPDATE mega_tree SET sub_trees=set_byte(sub_trees,7,102) WHERE tree_id=$1", + [oid.clone().into()], + )) + .await + .unwrap(); + assert_eq!( + count( + &core, + "SELECT count(*) FROM mst2_rooted_source_tree_revision WHERE valid" + ) + .await, + 0 + ); + let fresh = writer + .begin_intent("source-revision-reattest", &plan) + .await + .unwrap(); + writer.install_pages(&fresh, &[payload]).await.unwrap(); + writer.finalize(&fresh).await.unwrap(); + assert_eq!( + count( + &core, + "SELECT count(*) FROM mst2_rooted_source_tree_revision WHERE valid" + ) + .await, + 1 + ); + assert_ne!(revision(&core, &oid).await, original); + assert_eq!( + count( + &q, + "SELECT count(*) FROM mst2_metadata_source_root_attestation" + ) + .await, + 2 + ); + assert_eq!(count(&q,&format!("SELECT count(*) FROM mst2_metadata_source_root_attestation a WHERE NOT {}.mst2_route_source_tree_matches( + split_part(a.tagged_tree_oid,':',2),a.source_revision,a.source_body_digest)",identifier(&namespace.core_schema))).await,1); +} + +#[tokio::test] +async fn deleted_and_reinserted_source_oid_requires_new_independent_revision() { + let (config, core, namespace, q, _guard) = fixture().await; + let (plan, payload) = seeded_rooted_plan(&core, 'a', "file", 1).await; + let writer = RootedQualifiedMetadataRepository::open(&core, &config) + .await + .unwrap(); + let intent = writer + .begin_intent("source-delete-original", &plan) + .await + .unwrap(); + writer + .install_pages(&intent, std::slice::from_ref(&payload)) + .await + .unwrap(); + writer.finalize(&intent).await.unwrap(); + let oid = "a".repeat(40); + let original = revision(&core, &oid).await; + let saved: String = core + .query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT row_to_json(t)::text FROM mega_tree t WHERE tree_id=$1", + [oid.clone().into()], + )) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap(); + core.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "DELETE FROM mega_tree WHERE tree_id=$1", + [oid.clone().into()], + )) + .await + .unwrap(); + let deleted = revision(&core, &oid).await; + assert_ne!(deleted, original); + assert_eq!( + count( + &core, + "SELECT count(*) FROM mst2_rooted_source_tree_revision WHERE valid" + ) + .await, + 0 + ); + assert!( + writer + .lookup_reuse(&plan.identity.tagged_root_tree_oid, &plan.identity) + .await + .unwrap() + .is_none() + ); + // Even an identical OID, row ID, and byte body cannot revive the old stamp. + core.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "INSERT INTO mega_tree SELECT (json_populate_record(NULL::mega_tree,$1::json)).*", + [saved.into()], + )) + .await + .unwrap(); + assert_eq!(revision(&core, &oid).await, deleted); + assert_eq!( + count( + &core, + "SELECT count(*) FROM mst2_rooted_source_tree_revision WHERE valid" + ) + .await, + 0 + ); + assert!( + writer + .lookup_reuse(&plan.identity.tagged_root_tree_oid, &plan.identity) + .await + .unwrap() + .is_none() + ); + let fresh = writer + .begin_intent("source-delete-reattest", &plan) + .await + .unwrap(); + writer.install_pages(&fresh, &[payload]).await.unwrap(); + writer.finalize(&fresh).await.unwrap(); + assert_ne!(revision(&core, &oid).await, deleted); + assert_eq!( + count( + &core, + "SELECT count(*) FROM mst2_rooted_source_tree_revision WHERE valid" + ) + .await, + 1 + ); + assert_eq!(count(&q,&format!("SELECT count(*) FROM mst2_metadata_source_root_attestation a WHERE NOT {}.mst2_route_source_tree_matches( + split_part(a.tagged_tree_oid,':',2),a.source_revision,a.source_body_digest)",identifier(&namespace.core_schema))).await,1); +} + +#[tokio::test] +async fn current_source_revision_uses_captured_core_relations_under_temp_shadow() { + let (config, core, namespace, q, _guard) = fixture().await; + let (plan, payload) = seeded_rooted_plan(&core, 'a', "file", 1).await; + let writer = RootedQualifiedMetadataRepository::open(&core, &config) + .await + .unwrap(); + let intent = writer + .begin_intent("source-temp-shadow", &plan) + .await + .unwrap(); + writer.install_pages(&intent, &[payload]).await.unwrap(); + writer.finalize(&intent).await.unwrap(); + q.execute_unprepared("CREATE TEMP TABLE mega_tree(id bigint,tree_id text,sub_trees bytea); + CREATE TEMP TABLE mst2_rooted_source_tree_revision(tree_id text,tree_row_id bigint,revision uuid,body_digest bytea,valid boolean)") + .await.unwrap(); + assert_eq!(count(&q,&format!("SELECT {}.mst2_route_source_tree_matches(split_part(tagged_tree_oid,':',2),source_revision,source_body_digest)::bigint + FROM mst2_metadata_source_root_attestation",identifier(&namespace.core_schema))).await,1); +} + +#[tokio::test] +async fn current_source_revision_share_fence_orders_real_source_update_after_read() { + let (config, core, namespace, q, _guard) = fixture().await; + let (plan, payload) = seeded_rooted_plan(&core, 'a', "file", 1).await; + let writer = RootedQualifiedMetadataRepository::open(&core, &config) + .await + .unwrap(); + let intent = writer + .begin_intent("source-share-fence", &plan) + .await + .unwrap(); + writer.install_pages(&intent, &[payload]).await.unwrap(); + writer.finalize(&intent).await.unwrap(); + let read = q.begin().await.unwrap(); + assert_eq!(count(&read,&format!("SELECT {}.mst2_route_source_tree_matches(split_part(tagged_tree_oid,':',2),source_revision,source_body_digest)::bigint + FROM mst2_metadata_source_root_attestation",identifier(&namespace.core_schema))).await,1); + let core_writer = core.clone(); + let (ready, received) = tokio::sync::oneshot::channel(); + let pending = tokio::spawn(async move { + let txn = core_writer.begin().await.unwrap(); + let pid: i32 = txn + .query_one_raw(Statement::from_string( + DbBackend::Postgres, + "SELECT pg_backend_pid()", + )) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap(); + ready.send(pid).unwrap(); + txn.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "UPDATE mega_tree SET sub_trees=set_byte(sub_trees,7,120) WHERE tree_id=$1", + ["a".repeat(40).into()], + )) + .await + .unwrap(); + txn.commit().await + }); + let pid = received.await.unwrap(); + tokio::time::timeout(Duration::from_secs(4),async { + loop { + let waiting:bool=core.query_one_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "SELECT coalesce(wait_event_type='Lock',false) FROM pg_stat_activity WHERE pid=$1",[pid.into()])) + .await.unwrap().unwrap().try_get_by_index(0).unwrap(); + if waiting {break;} tokio::task::yield_now().await; + } + }).await.unwrap(); + assert!(!pending.is_finished()); + read.commit().await.unwrap(); + tokio::time::timeout(Duration::from_secs(4), pending) + .await + .unwrap() + .unwrap() + .unwrap(); + assert_eq!( + count( + &core, + "SELECT count(*) FROM mst2_rooted_source_tree_revision WHERE valid" + ) + .await, + 0 + ); +} diff --git a/src/jupiter/storage/qualified_source_revision.sql b/src/jupiter/storage/qualified_source_revision.sql new file mode 100644 index 00000000..855112ce --- /dev/null +++ b/src/jupiter/storage/qualified_source_revision.sql @@ -0,0 +1,124 @@ +-- This records only trees actually attested by Q, not the global Git inventory. +CREATE TABLE mst2_rooted_source_tree_revision ( + tree_id text PRIMARY KEY CHECK(tree_id ~ '^([0-9a-f]{40}|[0-9a-f]{64})$'), + tree_row_id bigint NOT NULL,revision uuid NOT NULL,body_digest bytea NOT NULL CHECK(octet_length(body_digest)=32), + valid boolean NOT NULL, + CHECK(substr(revision::text,15,1)='4' AND substr(revision::text,20,1) IN ('8','9','a','b')) +); + +CREATE FUNCTION mst2_route_source_tree_revision_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$CORE_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE source_id bigint; body bytea; +BEGIN + IF TG_OP='DELETE' OR TG_OP='TRUNCATE' THEN RAISE EXCEPTION 'rooted source revision watermarks are immutable'; END IF; + IF pg_is_in_recovery() OR current_setting('transaction_isolation')<>'read committed' + OR TG_RELID<>'mst2_rooted_source_tree_revision'::regclass THEN + RAISE EXCEPTION 'rooted source revision requires its captured primary relation'; END IF; + IF TG_OP='UPDATE' THEN + IF NEW.tree_id IS DISTINCT FROM OLD.tree_id THEN RAISE EXCEPTION 'rooted source revision cannot retarget its OID'; END IF; + IF NEW.valid IS FALSE THEN + -- A nested caller cannot invent invalidation: independently observe the + -- actual core mutation after it happened in this still-uncommitted writer. + IF pg_trigger_depth()<2 OR NOT OLD.valid OR NEW.tree_row_id IS DISTINCT FROM OLD.tree_row_id + OR NEW.body_digest IS DISTINCT FROM OLD.body_digest OR NEW.revision IS DISTINCT FROM OLD.revision THEN + RAISE EXCEPTION 'rooted source invalidation is outside its actual source writer'; END IF; + PERFORM mst2_route_enter($CORE_LITERAL$); + SELECT id,sub_trees INTO source_id,body FROM mega_tree WHERE tree_id=OLD.tree_id; + IF FOUND AND source_id=OLD.tree_row_id AND octet_length(body)<=67108864 + AND sha256(body)=OLD.body_digest THEN + RAISE EXCEPTION 'rooted source invalidation has no actual changed or deleted core bytes'; END IF; + NEW.revision:=gen_random_uuid(); RETURN NEW; + END IF; + IF OLD.valid OR NEW.revision IS DISTINCT FROM OLD.revision OR NEW.valid IS DISTINCT FROM true THEN + RAISE EXCEPTION 'ordinary DML cannot rewrite a valid rooted source revision'; END IF; + END IF; + PERFORM mst2_route_enter($CORE_LITERAL$); + SELECT id,sub_trees INTO source_id,body FROM mega_tree WHERE tree_id=NEW.tree_id FOR SHARE; + IF NOT FOUND OR octet_length(body)>67108864 THEN RAISE EXCEPTION 'rooted source capture has no bounded actual core tree'; END IF; + IF NEW.body_digest IS DISTINCT FROM sha256(body) THEN + RAISE EXCEPTION 'rooted source revision was not independently derived from its actual core bytes'; END IF; + NEW.tree_row_id:=source_id; NEW.revision:=gen_random_uuid(); NEW.valid:=true; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_route_source_tree_revision_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_rooted_source_tree_revision + FOR EACH ROW EXECUTE FUNCTION mst2_route_source_tree_revision_guard(); +CREATE TRIGGER mst2_route_source_tree_revision_truncate BEFORE TRUNCATE ON mst2_rooted_source_tree_revision + FOR EACH STATEMENT EXECUTE FUNCTION mst2_route_source_tree_revision_guard(); + +CREATE FUNCTION mst2_route_source_tree_inventory_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$CORE_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + IF (SELECT count(*) FROM (SELECT 1 FROM mst2_rooted_source_tree_revision LIMIT 65537) bounded)>65536 THEN + RAISE EXCEPTION 'rooted attested source inventory capacity is exceeded'; END IF; + RETURN NULL; +END $$; +CREATE TRIGGER mst2_route_source_tree_inventory_guard AFTER INSERT ON mst2_rooted_source_tree_revision + FOR EACH STATEMENT EXECUTE FUNCTION mst2_route_source_tree_inventory_guard(); + +CREATE FUNCTION mst2_route_source_tree_writer_enter() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$CORE_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + IF TG_RELID<>'mega_tree'::regclass THEN RAISE EXCEPTION 'rooted source writer left its captured core relation'; END IF; + IF TG_OP='TRUNCATE' THEN RAISE EXCEPTION 'core tree truncation cannot bypass rooted source revision invalidation'; END IF; + PERFORM mst2_route_enter($CORE_LITERAL$); + RETURN NULL; +END $$; +CREATE TRIGGER mst2_route_source_tree_writer_enter BEFORE INSERT OR UPDATE OR DELETE OR TRUNCATE ON mega_tree + FOR EACH STATEMENT EXECUTE FUNCTION mst2_route_source_tree_writer_enter(); + +CREATE FUNCTION mst2_route_source_tree_invalidate() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$CORE_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + IF TG_RELID<>'mega_tree'::regclass THEN RAISE EXCEPTION 'rooted invalidation left its captured core relation'; END IF; + IF TG_OP='UPDATE' AND NEW.tree_id IS NOT DISTINCT FROM OLD.tree_id AND NEW.id IS NOT DISTINCT FROM OLD.id + AND NEW.sub_trees IS NOT DISTINCT FROM OLD.sub_trees THEN RETURN NEW; END IF; + UPDATE mst2_rooted_source_tree_revision SET valid=false WHERE tree_id=OLD.tree_id AND valid; + IF TG_OP='DELETE' THEN RETURN OLD; END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_route_source_tree_invalidate AFTER UPDATE OR DELETE ON mega_tree + FOR EACH ROW EXECUTE FUNCTION mst2_route_source_tree_invalidate(); +CREATE FUNCTION mst2_route_source_tree_inserted() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$CORE_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + IF TG_RELID<>'mega_tree'::regclass THEN RAISE EXCEPTION 'rooted insertion left its captured core relation'; END IF; + UPDATE mst2_rooted_source_tree_revision SET valid=false WHERE tree_id=NEW.tree_id AND valid; + RETURN NULL; +END $$; +CREATE TRIGGER mst2_route_source_tree_inserted AFTER INSERT ON mega_tree + FOR EACH ROW EXECUTE FUNCTION mst2_route_source_tree_inserted(); + +CREATE FUNCTION mst2_route_capture_source_tree(oid_text text,expected_digest bytea) RETURNS uuid LANGUAGE plpgsql VOLATILE +SET search_path=$CORE_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE captured mst2_rooted_source_tree_revision%ROWTYPE; actual_id bigint; +BEGIN + PERFORM mst2_route_enter($CORE_LITERAL$); + SELECT id INTO actual_id FROM mega_tree WHERE tree_id=oid_text FOR SHARE; + IF NOT FOUND OR expected_digest IS NULL OR octet_length(expected_digest)<>32 THEN + RAISE EXCEPTION 'rooted source capture has no exact actual row and body digest'; END IF; + SELECT * INTO captured FROM mst2_rooted_source_tree_revision WHERE tree_id=oid_text FOR UPDATE; + IF FOUND THEN + IF captured.valid THEN + IF captured.tree_row_id<>actual_id OR captured.body_digest IS DISTINCT FROM expected_digest THEN + RAISE EXCEPTION 'rooted source bytes changed behind their exact immutable revision'; END IF; + RETURN captured.revision; + END IF; + UPDATE mst2_rooted_source_tree_revision SET valid=true,body_digest=expected_digest WHERE tree_id=oid_text + RETURNING revision INTO captured.revision; + ELSE + INSERT INTO mst2_rooted_source_tree_revision(tree_id,tree_row_id,revision,body_digest,valid) + VALUES(oid_text,actual_id,gen_random_uuid(),expected_digest,true) RETURNING revision INTO captured.revision; + END IF; + RETURN captured.revision; +END $$; + +CREATE FUNCTION mst2_route_source_tree_matches(oid_text text,exact_revision uuid,exact_digest bytea) +RETURNS boolean LANGUAGE plpgsql VOLATILE STRICT SET search_path=$CORE_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE captured mst2_rooted_source_tree_revision%ROWTYPE; +BEGIN + IF pg_is_in_recovery() OR current_setting('transaction_isolation')<>'read committed' THEN RETURN false; END IF; + SELECT revision.* INTO captured FROM mst2_rooted_source_tree_revision revision + JOIN mega_tree source ON source.id=revision.tree_row_id AND source.tree_id=revision.tree_id + WHERE revision.tree_id=oid_text FOR SHARE OF revision NOWAIT; + RETURN FOUND AND captured.valid AND captured.revision=exact_revision AND captured.body_digest=exact_digest; +END $$; diff --git a/src/jupiter/tests.rs b/src/jupiter/tests.rs index 4f39f1ad..aa3dd08d 100644 --- a/src/jupiter/tests.rs +++ b/src/jupiter/tests.rs @@ -223,6 +223,45 @@ fn drop_test_schema(admin_url: &str, schema: &str) { [Value::from(schema)], )) .await?; + // A bootstrap-created physical Q family belongs to this exact test + // core schema. Drop the pair even on panic, before its core FK targets. + let registry = admin + .query_one_raw(Statement::from_sql_and_values( + DatabaseBackend::Postgres, + "SELECT pg_catalog.to_regclass($1) IS NOT NULL AS present", + [format!("{schema}.mst2_metadata_namespace").into()], + )) + .await? + .unwrap() + .try_get::("", "present")?; + if registry { + let families = admin + .query_all_raw(Statement::from_string( + DatabaseBackend::Postgres, + format!( + "SELECT n.nspname FROM {schema}.mst2_metadata_namespace q + JOIN pg_catalog.pg_namespace n ON n.oid=q.metadata_schema_oid + JOIN pg_catalog.pg_namespace c ON c.oid=q.core_schema_oid + WHERE q.graph_domain='qualified-v1' AND c.nspname='{schema}' + AND q.core_schema='{schema}' AND q.family_identity='v3-rooted-qualified-1'" + ), + )) + .await?; + for family in families { + let q_schema: String = family.try_get("", "nspname")?; + if !q_schema.starts_with("mst2q_") { + return Err(sea_orm::DbErr::Custom( + "test Q schema escaped its generated prefix".into(), + )); + } + let quoted = format!("\"{}\"", q_schema.replace('"', "\"\"")); + admin + .execute_unprepared(&format!( + "SET lock_timeout='120s'; DROP SCHEMA {quoted} CASCADE" + )) + .await?; + } + } // The timeout turns a lock held by anything else into a leaked schema // and a message, not a hung test run. admin @@ -380,6 +419,8 @@ pub async fn test_storage_with_config(temp_dir: impl AsRef, config: Config native_projection_cache: Arc::default(), native_snapshot_sessions: Arc::default(), native_chunk_maps: Arc::default(), + shadow_qualified_metadata: Arc::default(), + rooted_qualified_metadata: Arc::default(), projection_observation_sink: None, cl_service: CLService::mock(), push_queue_service: PushQueueService::new( From e7ddee0e0e707e6b3079d9d8cbfe81e292e16175 Mon Sep 17 00:00:00 2001 From: Xiaoyang Han Date: Thu, 8 Oct 2026 06:57:30 +0800 Subject: [PATCH 33/50] test(v3): account for cold raw proof and instrument retry source opens (#84) The exact native suite exposed two test accounting errors. Cold raw opens one source to earn its receipt and one for authenticated delivery; its revocation regression now requires both passes while retaining rejected-tail, zero range IO and exact byte accounting. The admission/cancellation source-builder regression now increments its existing loader counter on its real successful retry, preserving the required two source opens, credit refund, producer drop and chunk verification. Production behavior is unchanged. --- src/api/router/snapshot_session_tests.rs | 10 ++++++++-- src/ceres/snapshot/chunks_stream_tests.rs | 5 ++++- 2 files changed, 12 insertions(+), 3 deletions(-) diff --git a/src/api/router/snapshot_session_tests.rs b/src/api/router/snapshot_session_tests.rs index 36101a05..68feb23a 100644 --- a/src/api/router/snapshot_session_tests.rs +++ b/src/api/router/snapshot_session_tests.rs @@ -861,9 +861,15 @@ async fn mst2_durable_http_frame_and_raw_delivery_recheck_after_release() { "revocation must suppress the next raw block or END frame" ); assert!(stream.next().await.is_none()); + // Cold raw first earns an immutable receipt, then opens the separate + // authenticated delivery stream. Revocation still suppresses its tail. fixture.counts.assert( - if raw { 1 } else { 0 }, - if raw { fixture.raw.len() } else { 0 }, + if raw { 2 } else { 0 }, + if raw { + fixture.raw.len() + first.len() + } else { + 0 + }, ); } } diff --git a/src/ceres/snapshot/chunks_stream_tests.rs b/src/ceres/snapshot/chunks_stream_tests.rs index 2cf7fb53..61de6453 100644 --- a/src/ceres/snapshot/chunks_stream_tests.rs +++ b/src/ceres/snapshot/chunks_stream_tests.rs @@ -104,7 +104,10 @@ async fn exact_source_builder_admits_all_install_workspace_before_open_and_cance let projection = build_source_with_resources( digest, size, - || async { Ok(stream(vec![Ok(raw.clone())])) }, + || async { + opens.fetch_add(1, Ordering::SeqCst); + Ok(stream(vec![Ok(raw.clone())])) + }, &budget, &builders, ) From 918a7db5187147ac57c46c6ea484d7ee25b7512e Mon Sep 17 00:00:00 2001 From: Xiaoyang Han Date: Thu, 8 Oct 2026 07:01:05 +0800 Subject: [PATCH 34/50] fix(v3): read metadata primary scope from its captured permanent schema (#85) A TEMP table with the same name could poison native metadata repository initialization. Capture the actual permanent schema through pg_catalog, qualify the storage identity read during initialization and mutation/recovery barriers, and continue observing and comparing the actual current database/schema/OIDs/server/replica identity. A matching fake TEMP UUID cannot authorize another primary schema. Add a single-connection regression with TEMP present before construction, primary schema switching and restoration. Runtime verification query counts and existing integrity assertions are preserved. --- .../storage/native_metadata_install.rs | 48 +++++++-- .../storage/native_metadata_install_tests.rs | 101 ++++++++++++++++++ 2 files changed, 140 insertions(+), 9 deletions(-) diff --git a/src/jupiter/storage/native_metadata_install.rs b/src/jupiter/storage/native_metadata_install.rs index 6ecf72c3..7d978fa0 100644 --- a/src/jupiter/storage/native_metadata_install.rs +++ b/src/jupiter/storage/native_metadata_install.rs @@ -156,7 +156,13 @@ struct PrimaryStorageScope { impl PostgresMetadataInstallRepository { /// The connection must target the deployment's primary PostgreSQL database. pub async fn new(connection: DatabaseConnection) -> Result { - let storage_scope = read_storage_scope(&connection).await?; + let schema = capture_storage_schema(&connection).await?; + let storage_scope = read_storage_scope(&connection, &schema).await?; + if storage_scope.schema != schema { + return Err(internal( + "metadata primary schema changed during storage scope capture", + )); + } Ok(Self { connection, barrier_timeout: Duration::from_secs(5), @@ -204,7 +210,7 @@ impl PostgresMetadataInstallRepository { &self, connection: &C, ) -> Result<(), SnapshotError> { - if read_storage_scope(connection).await? != self.storage_scope { + if read_storage_scope(connection, self.captured_schema()).await? != self.storage_scope { return Err(internal( "session mutation no longer targets its captured primary storage scope", )); @@ -659,7 +665,7 @@ impl PostgresMetadataInstallRepository { if let Some(family) = &self.qualified_family { family.enter(txn).await?; } - if read_storage_scope(txn).await? != self.storage_scope { + if read_storage_scope(txn, self.captured_schema()).await? != self.storage_scope { return Err(internal( "metadata recovery connection is outside the captured primary storage scope", )); @@ -703,20 +709,44 @@ impl PostgresMetadataInstallRepository { } } +async fn capture_storage_schema( + connection: &C, +) -> Result { + if connection.get_database_backend() != DbBackend::Postgres { + return Err(internal("metadata storage scope requires PostgreSQL")); + } + let row = connection + .query_one_raw(statement( + "SELECT n.nspname AS schema FROM pg_catalog.pg_namespace n + JOIN pg_catalog.pg_class c ON c.relnamespace=n.oid + AND c.relname='mst2_metadata_storage_scope' AND c.relkind='r' + WHERE n.nspname=pg_catalog.current_schema() AND n.nspname NOT LIKE 'pg_temp_%'", + [], + )) + .await + .map_err(internal)? + .ok_or_else(|| internal("metadata actual primary storage relation is missing"))?; + row.try_get("", "schema").map_err(internal) +} + async fn read_storage_scope( connection: &C, + schema: &str, ) -> Result { if connection.get_database_backend() != DbBackend::Postgres { return Err(internal("metadata storage scope requires PostgreSQL")); } let row = connection .query_one_raw(statement( - "SELECT s.storage_uuid, current_database() AS database, d.oid::bigint AS database_oid, - current_schema() AS schema, n.oid::bigint AS schema_oid, - inet_server_addr()::text AS server_address, inet_server_port() AS server_port, - pg_is_in_recovery() AS replica FROM mst2_metadata_storage_scope s - JOIN pg_catalog.pg_database d ON d.datname=current_database() - JOIN pg_catalog.pg_namespace n ON n.nspname=current_schema() WHERE s.singleton=1", + &format!( + "SELECT s.storage_uuid, pg_catalog.current_database() AS database, d.oid::bigint AS database_oid, + pg_catalog.current_schema() AS schema, n.oid::bigint AS schema_oid, + pg_catalog.inet_server_addr()::text AS server_address, pg_catalog.inet_server_port() AS server_port, + pg_catalog.pg_is_in_recovery() AS replica FROM \"{}\".mst2_metadata_storage_scope s + JOIN pg_catalog.pg_database d ON d.datname=pg_catalog.current_database() + JOIN pg_catalog.pg_namespace n ON n.nspname=pg_catalog.current_schema() WHERE s.singleton=1", + schema.replace('"', "\"\"") + ), [], )) .await diff --git a/src/jupiter/storage/native_metadata_install_tests.rs b/src/jupiter/storage/native_metadata_install_tests.rs index 4145afd8..4cb387dc 100644 --- a/src/jupiter/storage/native_metadata_install_tests.rs +++ b/src/jupiter/storage/native_metadata_install_tests.rs @@ -70,6 +70,107 @@ fn rejected(error: MetadataInstallError) -> SnapshotErrorCode { } } +#[tokio::test] +async fn native_metadata_scope_ignores_poisoned_temp_table_and_rejects_changed_primary_schema() { + let (first, _second, _schema, url) = fixture().await; + let (_other, _unused, other_schema, _other_url) = fixture().await; + let expected = PostgresMetadataInstallRepository::new(first) + .await + .unwrap() + .storage_scope; + let mut options = sea_orm::ConnectOptions::new(url); + options.max_connections(1).min_connections(1); + let connection = Database::connect(options).await.unwrap(); + connection + .execute_unprepared( + "CREATE TEMP TABLE mst2_metadata_storage_scope(singleton integer,storage_uuid text); + INSERT INTO pg_temp.mst2_metadata_storage_scope VALUES(1,'temp-poison')", + ) + .await + .unwrap(); + let repository = PostgresMetadataInstallRepository::new(connection.clone()) + .await + .unwrap(); + assert_eq!(repository.storage_scope, expected); + assert_eq!(repository.captured_schema(), _schema.schema()); + repository + .verify_primary_connection(&connection) + .await + .unwrap(); + let txn = connection.begin().await.unwrap(); + repository.barrier(&txn).await.unwrap(); + txn.rollback().await.unwrap(); + connection + .execute_raw(statement( + "UPDATE pg_temp.mst2_metadata_storage_scope SET storage_uuid=$1", + [repository.storage_scope.storage_uuid.clone().into()], + )) + .await + .unwrap(); + let txn = connection.begin().await.unwrap(); + txn.execute_unprepared(&format!( + "SET LOCAL search_path=\"{}\",pg_catalog,pg_temp", + other_schema.schema().replace('"', "\"\"") + )) + .await + .unwrap(); + let error = repository + .verify_primary_connection(&txn) + .await + .unwrap_err(); + assert_eq!(error.code, SnapshotErrorCode::Internal); + assert!( + error + .message + .contains("session mutation no longer targets its captured primary storage scope") + ); + let error = repository.barrier(&txn).await.unwrap_err(); + assert_eq!(error.code, SnapshotErrorCode::Internal); + assert!( + error + .message + .contains("metadata recovery connection is outside the captured primary storage scope") + ); + txn.rollback().await.unwrap(); + connection + .execute_unprepared("SET search_path=pg_temp,pg_catalog") + .await + .unwrap(); + let error = PostgresMetadataInstallRepository::new(connection.clone()) + .await + .err() + .unwrap(); + assert_eq!(error.code, SnapshotErrorCode::Internal); + assert_eq!( + error.message, + "metadata actual primary storage relation is missing" + ); + connection + .execute_unprepared(&format!( + "SET search_path=\"{}\",pg_catalog,pg_temp", + repository.captured_schema().replace('"', "\"\"") + )) + .await + .unwrap(); + repository + .verify_primary_connection(&connection) + .await + .unwrap(); + assert_eq!( + connection + .query_one_raw(statement( + "SELECT storage_uuid FROM pg_temp.mst2_metadata_storage_scope WHERE singleton=1", + [], + )) + .await + .unwrap() + .unwrap() + .try_get::("", "storage_uuid") + .unwrap(), + expected.storage_uuid + ); +} + #[tokio::test] async fn native_metadata_two_primary_connections_replay_one_receipt_without_recounting() { let (first, second, _schema, _url) = fixture().await; From 96dcea0db5294d5f70f1b83cb427c78967baef1a Mon Sep 17 00:00:00 2001 From: Xiaoyang Han Date: Thu, 8 Oct 2026 07:07:50 +0800 Subject: [PATCH 35/50] fix(v3): repair rooted metadata native type and lint gates (#86) The exact rooted metadata native check exposed unused private type reexports, a u64/u32 END assertion and two strict Clippy findings. Remove unused exports, compare the same exact count after lossless conversion, return existing typed SnapshotError from JSON/cursor helpers and name the unchanged proof page type. Response conversion keeps the same MST/2 HTTP error mapper at the caller boundary; proof bounds and validation are unchanged. --- src/api/router/snapshot_content_tests.rs | 2 +- src/api/router/snapshot_rooted_metadata.rs | 12 ++++++++---- src/jupiter/storage/qualified_metadata_family.rs | 4 +--- src/jupiter/storage/qualified_metadata_reader.rs | 6 ++++-- 4 files changed, 14 insertions(+), 10 deletions(-) diff --git a/src/api/router/snapshot_content_tests.rs b/src/api/router/snapshot_content_tests.rs index 65234d8b..2dc14015 100644 --- a/src/api/router/snapshot_content_tests.rs +++ b/src/api/router/snapshot_content_tests.rs @@ -1094,7 +1094,7 @@ async fn mst2_publication_disabled_resolve_preserves_head_body_object_map_page_a assert_eq!(chunk.file_content_id, fixture.digest); assert_eq!(chunk.chunk_index, 0); assert_eq!(end.request_item_count, 1); - assert_eq!(end.logical_bytes, CHUNK_SIZE); + assert_eq!(end.logical_bytes, u64::from(CHUNK_SIZE)); } #[tokio::test] diff --git a/src/api/router/snapshot_rooted_metadata.rs b/src/api/router/snapshot_rooted_metadata.rs index a35c7bd1..159685d4 100644 --- a/src/api/router/snapshot_rooted_metadata.rs +++ b/src/api/router/snapshot_rooted_metadata.rs @@ -8,7 +8,7 @@ use crate::{ jupiter::storage::qualified_metadata_family::RootedLookupStatus, }; -fn entry_json(entry: &Entry) -> Result { +fn entry_json(entry: &Entry) -> Result { let name = std::str::from_utf8(&entry.name).map_err(internal)?; let kind = match entry.kind { EntryKind::Regular => "regular", @@ -28,15 +28,19 @@ fn entry_json(entry: &Entry) -> Result { Ok(value) } -fn cursor_name(sid: &str, path: &str, query: &DirectoryQuery) -> Result, Response> { +fn cursor_name( + sid: &str, + path: &str, + query: &DirectoryQuery, +) -> Result, SnapshotError> { let Some(cursor) = query.cursor.as_deref() else { return Ok(None); }; let invalid = || { - mst2_error_response(SnapshotError::new( + SnapshotError::new( SnapshotErrorCode::CursorInvalid, "cursor bound to different parameters or invalid", - )) + ) }; let (payload, signature) = cursor.rsplit_once('.').ok_or_else(invalid)?; if runtime().sign_cursor(payload) != signature { diff --git a/src/jupiter/storage/qualified_metadata_family.rs b/src/jupiter/storage/qualified_metadata_family.rs index 52594dd3..024d325b 100644 --- a/src/jupiter/storage/qualified_metadata_family.rs +++ b/src/jupiter/storage/qualified_metadata_family.rs @@ -35,9 +35,7 @@ const GC_SQL: &str = include_str!("qualified_metadata_gc.sql"); #[path = "qualified_metadata_rooted.rs"] mod rooted; -pub(crate) use rooted::{ - RootedDirectoryWindow, RootedLookupBatch, RootedLookupStatus, RootedQualifiedMetadataRepository, -}; +pub(crate) use rooted::{RootedLookupStatus, RootedQualifiedMetadataRepository}; #[cfg(test)] pub(crate) use rooted::{ RootedPrepareIntent, with_rooted_reader_barriers, with_rooted_source_fact_barriers, diff --git a/src/jupiter/storage/qualified_metadata_reader.rs b/src/jupiter/storage/qualified_metadata_reader.rs index 63ccbdcd..6e28360a 100644 --- a/src/jupiter/storage/qualified_metadata_reader.rs +++ b/src/jupiter/storage/qualified_metadata_reader.rs @@ -16,6 +16,8 @@ use crate::{ }, }; +type ProofPages = Vec<([u8; 32], Vec)>; + #[derive(Debug, Clone, Copy, PartialEq, Eq)] struct Binding { generation: i64, @@ -98,7 +100,7 @@ pub(crate) struct RootedDirectoryWindow { pub entry_count: u64, pub entries: Vec, pub has_more: bool, - pub proof_pages: Vec<([u8; 32], Vec)>, + pub proof_pages: ProofPages, pub ancestors: Vec<(String, [u8; 32])>, } @@ -541,7 +543,7 @@ impl Reader<'_> { Ok(children) } - fn proofs(&self) -> Result)>, SnapshotError> { + fn proofs(&self) -> Result { let bytes: usize = self.cache.values().map(|page| page.bytes.len()).sum(); if bytes > 1_048_576 { return Err(SnapshotError::new( From 8178f86a3a12cd721a32a42d843bfd7f248515d3 Mon Sep 17 00:00:00 2001 From: Xiaoyang Han Date: Thu, 8 Oct 2026 07:23:13 +0800 Subject: [PATCH 36/50] fix(v3): remove unused rooted reader type reexports (#88) The exact native check exposed the remaining unused directory-window and lookup-batch reexports in the intermediate rooted module after the outer family reexports were removed. Remove those two unused names and retain the used lookup status reexport. Reader type definitions, methods, proof bounds, HTTP behavior and every test assertion are unchanged. --- src/jupiter/storage/qualified_metadata_rooted.rs | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/src/jupiter/storage/qualified_metadata_rooted.rs b/src/jupiter/storage/qualified_metadata_rooted.rs index 5dc2f4be..05527117 100644 --- a/src/jupiter/storage/qualified_metadata_rooted.rs +++ b/src/jupiter/storage/qualified_metadata_rooted.rs @@ -18,7 +18,7 @@ mod sessions; mod gc; #[path = "qualified_metadata_reader.rs"] mod reader; -pub(crate) use reader::{RootedDirectoryWindow, RootedLookupBatch, RootedLookupStatus}; +pub(crate) use reader::RootedLookupStatus; #[cfg(test)] pub(crate) use reader::{ with_rooted_reader_barriers, with_rooted_source_fact_barriers, From 6312095cef76e9845afbc4413b2fd7ba328d35ed Mon Sep 17 00:00:00 2001 From: Xiaoyang Han Date: Thu, 8 Oct 2026 07:37:21 +0800 Subject: [PATCH 37/50] test(v3): borrow the canonical donor payload without cloning (#89) Strict native Clippy identified an unnecessary cloned single-item slice in the rooted metadata donor fixture. Pass the same immutable payload with std::slice::from_ref. Preserve every donor certificate, source verification, delta reuse and fixed-root integrity assertion; production code is unchanged. --- src/jupiter/storage/qualified_metadata_canonical_tests.rs | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/src/jupiter/storage/qualified_metadata_canonical_tests.rs b/src/jupiter/storage/qualified_metadata_canonical_tests.rs index cccf4eef..1ff9c2fc 100644 --- a/src/jupiter/storage/qualified_metadata_canonical_tests.rs +++ b/src/jupiter/storage/qualified_metadata_canonical_tests.rs @@ -662,7 +662,7 @@ async fn late_raw_delta_membership_is_rejected_after_intent_commit() { .await .unwrap(); writer - .install_pages(&donor_intent, &[donor_payload.clone()]) + .install_pages(&donor_intent, std::slice::from_ref(&donor_payload)) .await .unwrap(); writer.finalize(&donor_intent).await.unwrap(); From f431064ed761dc0e30432abe898a309001eb94e5 Mon Sep 17 00:00:00 2001 From: Xiaoyang Han Date: Thu, 8 Oct 2026 08:41:12 +0800 Subject: [PATCH 38/50] Bound v3 chunk-map ownership and cancel stalled backend waits at the actual deadline (#91) Bound persisted v3 chunk maps and exact receipt generations with primary database owners, capacity admission, resumable collection and final transport-clone retention. Pending cold and warm backend opens, input polls and receipt creates now terminate under the actual remaining owner deadline; empty fragments grant no renewal. Add actual HTTP/storage/database test sources for expiry, withheld final bytes, refund and safe replay. Owned nightly format, canonical candidate whitespace and SQL lexical checks pass; native build/tests, database execution and commit-update performance measurements remain unrun. Truthful backend capabilities are a separately reviewed source-only follow-up dependency. --- .../snapshot_chunk_map_retention_tests.rs | 1405 +++++++++++++++++ src/api/router/snapshot_content.rs | 180 ++- src/api/router/snapshot_content_tests.rs | 104 ++ .../snapshot_persisted_chunk_map_tests.rs | 4 +- src/api/router/snapshot_raw_blob.rs | 44 +- src/api/router/snapshot_raw_blob_tests.rs | 2 +- src/ceres/snapshot/chunks.rs | 2 + src/ceres/snapshot/chunks_stream.rs | 40 +- src/ceres/snapshot/chunks_stream_tests.rs | 6 +- src/ceres/snapshot/content_budget.rs | 4 - ...008_000300_add_mst2_chunk_map_retention.rs | 26 + .../m20261008_000300_chunk_map_retention.sql | 262 +++ src/jupiter/migration/mod.rs | 22 +- src/jupiter/storage/native_chunk_map.rs | 285 ++-- .../storage/native_chunk_map/retention.rs | 1220 ++++++++++++++ src/jupiter/storage/object_storage.rs | 50 + src/orbit/adapter/object.rs | 105 ++ src/orbit_api/error.rs | 6 + src/orbit_api/object_storage.rs | 49 + src/server/http_server.rs | 103 +- 20 files changed, 3753 insertions(+), 166 deletions(-) create mode 100644 src/api/router/snapshot_chunk_map_retention_tests.rs create mode 100644 src/jupiter/migration/m20261008_000300_add_mst2_chunk_map_retention.rs create mode 100644 src/jupiter/migration/m20261008_000300_chunk_map_retention.sql create mode 100644 src/jupiter/storage/native_chunk_map/retention.rs diff --git a/src/api/router/snapshot_chunk_map_retention_tests.rs b/src/api/router/snapshot_chunk_map_retention_tests.rs new file mode 100644 index 00000000..587e4196 --- /dev/null +++ b/src/api/router/snapshot_chunk_map_retention_tests.rs @@ -0,0 +1,1405 @@ +use sea_orm::{DatabaseConnection, DbBackend, Statement}; +use tokio::{sync::Notify, time::timeout}; + +use super::*; +use crate::{ + ceres::snapshot::{ + chunks::{ChunkMapSource, VerifiedSourceChunkMap}, + error::SnapshotErrorCode, + }, + jupiter::storage::{ + native_chunk_map::PostgresChunkMapRepository, object_storage::mock_object_storage, + }, +}; + +fn statement(sql: &str, values: [sea_orm::Value; N]) -> Statement { + Statement::from_sql_and_values(DbBackend::Postgres, sql, values) +} + +fn router(state: &MonoApiServiceState) -> Router { + Router::new().nest("/api/v2", routers(state.clone()).with_state(state.clone())) +} + +fn budgeted_chunks_app( + fixture: &Fixture, + response_budget: &Arc, + scratch_budget: &Arc, +) -> Router { + use axum::{ + extract::{Path as AxumPath, State}, + routing::post, + }; + + use crate::api::router::snapshot_router::{ + content, request::Mst2Bytes, snapshot_auth_middleware, + }; + + let response_budget = response_budget.clone(); + let scratch_budget = scratch_budget.clone(); + Router::new() + .route( + "/api/v2/snapshots/{snapshot_id}/chunks", + post( + move |state: State, + path: AxumPath, + body: Mst2Bytes| { + content::chunks_with_budgets( + state, + path, + body, + content::ChunksBudgets { + response: response_budget.clone(), + scratch: scratch_budget.clone(), + }, + ) + }, + ), + ) + .route_layer(axum::middleware::from_fn_with_state( + fixture.state.clone(), + snapshot_auth_middleware, + )) + .with_state(fixture.state.clone()) +} + +async fn count(db: &DatabaseConnection, table: &str) -> i64 { + db.query_one_raw(statement( + &format!("SELECT count(*) AS count FROM {table}"), + [], + )) + .await + .unwrap() + .unwrap() + .try_get("", "count") + .unwrap() +} + +async fn wait_count(db: &DatabaseConnection, table: &str, expected: i64) { + timeout(Duration::from_secs(10), async { + loop { + if count(db, table).await == expected { + return; + } + tokio::time::sleep(Duration::from_millis(5)).await; + } + }) + .await + .expect("actual durable owner release did not complete"); +} + +async fn age_unowned_candidates(db: &DatabaseConnection) { + db.execute_unprepared("UPDATE mst2_chunk_receipt_generation SET last_progress=pg_catalog.clock_timestamp()-interval '2 hours' WHERE state='LIVE'; UPDATE mst2_chunk_map_lifetime SET last_used=pg_catalog.clock_timestamp()-interval '2 hours' WHERE state='LIVE'").await.unwrap(); +} + +async fn oid_for(fixture: &Fixture, path: &str) -> String { + let handler = MonoApiService::from(&fixture.state); + let main = fixture + .state + .storage + .mono_storage() + .get_main_ref("/project") + .await + .unwrap() + .unwrap(); + let tree = handler.get_tree_by_hash(&main.ref_tree_hash).await.unwrap(); + match resolve_abs_metadata(&handler, &tree, path).await.unwrap() { + MetadataWalkOutcome::FoundFile { oid, .. } => oid, + other => panic!("retention fixture did not resolve: {other:?}"), + } +} + +#[tokio::test] +async fn shared_map_survives_other_source_retirement_while_actual_reader_is_live() { + let fixture = + Fixture::new_in_metadata_family(false, 0, &[("other".to_string(), vec![19; 4096])], true) + .await; + let original = fixture.map("/file").await; + let other = oid_for(&fixture, "/other").await; + let db = fixture.state.storage.mono_storage().get_connection(); + // This G fixture isolates independent source receipts without changing a + // Q-certified file tuple. Two exact facts consume the same actual raw + // bytes independently; map sharing never admits the second source. + db.execute_raw(statement( + "UPDATE mst2_verified_object SET size=$1,raw_sha256=$2 WHERE git_oid=$3", + [ + (fixture.raw.len() as i64).into(), + fixture.digest.to_vec().into(), + other.clone().into(), + ], + )) + .await + .unwrap(); + *fixture.counts.object_size_override.lock().unwrap() = Some(fixture.raw.len() as i64); + *fixture.counts.object_fault.lock().unwrap() = Some(bounded_objects::StreamFault { + oid: other.clone(), + kind: bounded_objects::FaultKind::Parts( + fixture + .raw + .chunks(CHUNK_SIZE as usize) + .map(Bytes::copy_from_slice) + .collect(), + ), + }); + fixture.counts.reset(); + assert_eq!(fixture.map("/other").await["map"], original["map"]); + fixture.counts.assert(1, fixture.raw.len()); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 1); + wait_count(db, "mst2_chunk_reader", 0).await; + assert_eq!(count(db, "mst2_chunk_map_source").await, 2); + assert_eq!(count(db, "mst2_chunk_map").await, 1); + assert_eq!(count(db, "mst2_chunk_map_lifetime").await, 1); + let fact = fixture + .state + .storage + .mono_storage() + .get_verified_blobs(vec![other.clone()]) + .await + .unwrap() + .remove(&other) + .unwrap(); + let source = ChunkMapSource::from_fact(fact, &other).unwrap(); + let repository = fixture.state.storage.chunk_maps().await.unwrap(); + let objects = &fixture.state.storage.git_service.obj_storage; + let reader = repository.read(&source, objects).await.unwrap().unwrap(); + age_unowned_candidates(db).await; + repository.maintain(objects, 64).await.unwrap(); + assert_eq!(count(db, "mst2_chunk_map_source").await, 1); + assert_eq!(count(db, "mst2_chunk_map").await, 1); + assert_eq!(count(db, "mst2_chunk_map_leaf").await, 1); + assert_eq!(count(db, "mst2_chunk_map_node").await, 1); + assert_eq!(fixture.counts.receipt_deletes.load(Ordering::SeqCst), 1); + repository + .selected_page(&reader, 0) + .await + .unwrap() + .verify_chunk(&reader.map, 0, &fixture.raw[..CHUNK_SIZE as usize]) + .unwrap(); + drop(reader); + wait_count(db, "mst2_chunk_reader", 0).await; + age_unowned_candidates(db).await; + repository.maintain(objects, 64).await.unwrap(); + assert_eq!(count(db, "mst2_chunk_map_source").await, 0); + assert_eq!(count(db, "mst2_chunk_map").await, 0); + assert_eq!(fixture.counts.receipt_deletes.load(Ordering::SeqCst), 2); +} + +#[tokio::test] +async fn pending_receipt_is_replayed_before_another_aged_live_victim() { + let fixture = Fixture::new_with_pg_config_directories_and_objects( + false, + 0, + &[("other".to_string(), vec![19; 4096])], + ) + .await; + fixture.map("/file").await; + let other = fixture.map("/other").await; + let db = fixture.state.storage.mono_storage().get_connection(); + wait_count(db, "mst2_chunk_reader", 0).await; + age_unowned_candidates(db).await; + db.execute_raw(statement( + "UPDATE mst2_chunk_receipt_generation g SET state='DELETING' WHERE EXISTS(SELECT 1 FROM mst2_chunk_map_source s WHERE s.receipt_key=g.receipt_key AND s.git_oid=$1)", + [fixture.oid.clone().into()], + )) + .await + .unwrap(); + let repository = fixture.state.storage.chunk_maps().await.unwrap(); + repository + .maintain(&fixture.state.storage.git_service.obj_storage, 1) + .await + .unwrap(); + assert_eq!(count(db, "mst2_chunk_map_source").await, 1); + assert_eq!(count(db, "mst2_chunk_map").await, 1); + assert_eq!(fixture.counts.receipt_deletes.load(Ordering::SeqCst), 1); + fixture.counts.reset(); + assert_eq!(fixture.map("/other").await["map"], other["map"]); + fixture.counts.assert(0, 0); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 0); +} + +#[tokio::test] +async fn actual_http_replays_a_retiring_source_row_before_rebuilding_its_next_generation() { + let fixture = Fixture::new().await; + let original = fixture.map("/file").await; + let db = fixture.state.storage.mono_storage().get_connection(); + wait_count(db, "mst2_chunk_reader", 0).await; + let old_key: String = db + .query_one_raw(statement( + "SELECT receipt_key FROM mst2_chunk_map_source", + [], + )) + .await + .unwrap() + .unwrap() + .try_get("", "receipt_key") + .unwrap(); + db.execute_unprepared( + "UPDATE mst2_chunk_receipt_generation SET state='DELETING' WHERE state='LIVE'", + ) + .await + .unwrap(); + // The source row is deliberately still present; inline replay cannot + // depend on the source-index delete having already finished. + assert_eq!(count(db, "mst2_chunk_map_source").await, 1); + fixture.counts.reset(); + assert_eq!(fixture.map("/file").await["map"], original["map"]); + fixture.counts.assert(1, fixture.raw.len()); + assert_eq!(fixture.counts.receipt_deletes.load(Ordering::SeqCst), 1); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 1); + let new_key: String = db + .query_one_raw(statement( + "SELECT receipt_key FROM mst2_chunk_map_source", + [], + )) + .await + .unwrap() + .unwrap() + .try_get("", "receipt_key") + .unwrap(); + assert_ne!(old_key, new_key); + assert_eq!(count(db, "mst2_chunk_map").await, 1); + fixture.counts.reset(); + fixture.map("/alias").await; + fixture.counts.assert(0, 0); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 0); +} + +#[tokio::test] +async fn actual_warm_body_owner_cancels_never_returning_open_and_next_without_backend_release() { + use crate::ceres::snapshot::content_budget::{MemoryBudget, RANGE_WORK_BYTES}; + + for raw_route in [true, false] { + for stage in 0..3 { + let held_open = stage == 0; + let complete_without_eof = stage == 2; + let fixture = Fixture::new().await; + let map = fixture.map("/file").await; + let db = fixture.state.storage.mono_storage().get_connection(); + wait_count(db, "mst2_chunk_reader", 0).await; + fixture.counts.reset(); + let entered = Arc::new(Notify::new()); + let release = Arc::new(Notify::new()); + let drops = Arc::new(AtomicUsize::new(0)); + let tail_polls = Arc::new(AtomicUsize::new(0)); + if held_open { + let holds = Some((entered.clone(), release.clone(), drops.clone())); + if raw_route { + *fixture.counts.whole_open_holds.lock().unwrap() = holds; + } else { + *fixture.counts.range_open_holds.lock().unwrap() = holds; + } + } else { + *fixture.counts.object_fault.lock().unwrap() = Some(bounded_objects::StreamFault { + oid: fixture.oid.clone(), + kind: bounded_objects::FaultKind::HeldFragment { + prefix: if complete_without_eof { + Bytes::copy_from_slice(if raw_route { + &fixture.raw + } else { + &fixture.raw[..CHUNK_SIZE as usize] + }) + } else { + Bytes::new() + }, + fragment: None, + entered: entered.clone(), + release: release.clone(), + drops: drops.clone(), + tail_polls: tail_polls.clone(), + }, + }); + } + let response_bytes = if raw_route { + CHUNK_SIZE as usize + } else { + CHUNK_SIZE as usize + 2048 + }; + let response_budget = MemoryBudget::new(response_bytes); + let scratch_budget = MemoryBudget::new(RANGE_WORK_BYTES); + let app = if raw_route { + super::raw_blob::budgeted_app(&fixture, &response_budget, &scratch_budget) + } else { + budgeted_chunks_app(&fixture, &response_budget, &scratch_budget) + }; + let request = if raw_route { + fixture.request("GET", "blob?path=/file", Body::empty()) + } else { + fixture.request( + "POST", + "chunks", + Body::from( + fixture + .chunk_body("/file", map["map"]["map_id"].as_str().unwrap(), "0") + .to_string(), + ), + ) + }; + let expected_raw = fixture.raw.clone(); + let mut task = tokio::spawn(async move { + let response = app.oneshot(request).await.unwrap(); + if raw_route && !held_open { + assert_eq!(response.status(), 200); + let mut body = response.into_body().into_data_stream(); + if complete_without_eof { + let first = body.next().await.unwrap().unwrap(); + assert_eq!(first.as_ref(), &expected_raw[..CHUNK_SIZE as usize]); + drop(first); + } + assert!(body.next().await.unwrap().is_err()); + assert!(body.next().await.is_none()); + } else { + error(response, 410, "LEASE_EXPIRED", false).await; + } + }); + timeout(Duration::from_secs(10), entered.notified()) + .await + .unwrap(); + assert_eq!(count(db, "mst2_chunk_reader").await, 1); + assert_eq!(scratch_budget.used(), RANGE_WORK_BYTES); + assert_eq!( + response_budget.used(), + if raw_route && complete_without_eof { + fixture.raw.len() - CHUNK_SIZE as usize + } else { + response_bytes + } + ); + db.execute_unprepared("UPDATE mst2_chunk_reader SET deadline=pg_catalog.clock_timestamp()+interval '20 milliseconds'").await.unwrap(); + // Never release the backend and never cancel the caller. The + // request must observe actual expiry during its pending await. + let finished = timeout(Duration::from_secs(12), &mut task).await; + if finished.is_err() { + task.abort(); + let _ = task.await; + panic!("actual body owner did not cancel its permanently held backend"); + } + finished.unwrap().unwrap(); + assert_eq!(drops.load(Ordering::SeqCst), 1); + assert_eq!(tail_polls.load(Ordering::SeqCst), 0); + let consumed = if complete_without_eof { + if raw_route { + fixture.raw.len() + } else { + CHUNK_SIZE as usize + } + } else { + 0 + }; + assert_eq!(fixture.counts.bytes.load(Ordering::SeqCst), consumed); + assert_eq!(scratch_budget.used(), 0); + assert_eq!(response_budget.used(), 0); + if raw_route { + fixture.counts.assert(1, consumed); + } else { + assert_eq!(fixture.counts.whole.load(Ordering::SeqCst), 0); + assert_eq!(fixture.counts.range.load(Ordering::SeqCst), 1); + } + wait_count(db, "mst2_chunk_reader", 0).await; + assert_eq!(count(db, "mst2_chunk_map_source").await, 1); + drop(release); + } + } +} + +#[tokio::test] +async fn actual_cold_builder_uses_remaining_database_deadline_for_open_and_next_and_refunds_credit() +{ + use crate::ceres::snapshot::content_budget::MemoryBudget; + + for stage in 0..3 { + let held_open = stage == 0; + let complete_without_eof = stage == 2; + let fixture = Fixture::new().await; + let repository = fixture.state.storage.chunk_maps().await.unwrap(); + let source = ChunkMapSource::from_fact(fixture.fact().await, &fixture.oid).unwrap(); + let admission = repository + .admit_install(&source, &fixture.state.storage.git_service.obj_storage) + .await + .unwrap(); + let db = fixture.state.storage.mono_storage().get_connection(); + db.execute_unprepared("UPDATE mst2_chunk_receipt_generation SET deadline=pg_catalog.clock_timestamp()+interval '3 seconds' WHERE state='RESERVED'").await.unwrap(); + admission.test_check_next_owner_operation().await; + fixture.counts.reset(); + let entered = Arc::new(Notify::new()); + let release = Arc::new(Notify::new()); + let drops = Arc::new(AtomicUsize::new(0)); + let tail_polls = Arc::new(AtomicUsize::new(0)); + if held_open { + *fixture.counts.whole_open_holds.lock().unwrap() = + Some((entered.clone(), release.clone(), drops.clone())); + } else { + *fixture.counts.object_fault.lock().unwrap() = Some(bounded_objects::StreamFault { + oid: fixture.oid.clone(), + kind: bounded_objects::FaultKind::HeldFragment { + prefix: if complete_without_eof { + Bytes::copy_from_slice(&fixture.raw) + } else { + Bytes::new() + }, + fragment: None, + entered: entered.clone(), + release: release.clone(), + drops: drops.clone(), + tail_polls: tail_polls.clone(), + }, + }); + } + let budget = MemoryBudget::new(512 * 1024 * 1024); + let task_budget = budget.clone(); + let handler = MonoApiService::from(&fixture.state); + let mut task = tokio::spawn(async move { + VerifiedSourceChunkMap::verify(&handler, source, &task_budget, &admission).await + }); + timeout(Duration::from_secs(2), entered.notified()) + .await + .unwrap(); + assert!(budget.used() > 0); + let result = timeout(Duration::from_secs(5), &mut task).await; + if result.is_err() { + task.abort(); + let _ = task.await; + panic!("cold source ignored its actual remaining database deadline"); + } + let result = result.unwrap().unwrap(); + let error = match result { + Ok(_) => panic!("incomplete cold source produced trust"), + Err(error) => error, + }; + assert_eq!(error.code, SnapshotErrorCode::LeaseExpired); + assert_eq!(budget.used(), 0); + assert_eq!(drops.load(Ordering::SeqCst), 1); + assert_eq!(tail_polls.load(Ordering::SeqCst), 0); + fixture.counts.assert( + 1, + if complete_without_eof { + fixture.raw.len() + } else { + 0 + }, + ); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 0); + assert_eq!(count(db, "mst2_chunk_map").await, 0); + assert_eq!(count(db, "mst2_chunk_map_source").await, 0); + drop(release); + } +} + +#[tokio::test] +async fn actual_receipt_create_and_read_waits_cancel_without_backend_release_or_source_publication() +{ + use crate::ceres::snapshot::content_budget::MemoryBudget; + + for creating in [true, false] { + let fixture = Fixture::new().await; + let repository = fixture.state.storage.chunk_maps().await.unwrap(); + let source = ChunkMapSource::from_fact(fixture.fact().await, &fixture.oid).unwrap(); + let objects = fixture.state.storage.git_service.obj_storage.clone(); + let admission = repository.admit_install(&source, &objects).await.unwrap(); + let budget = MemoryBudget::new(512 * 1024 * 1024); + let handler = MonoApiService::from(&fixture.state); + let verified = VerifiedSourceChunkMap::verify(&handler, source, &budget, &admission) + .await + .unwrap(); + assert!(budget.used() > 0); + let entered = Arc::new(Notify::new()); + let release = Arc::new(Notify::new()); + let drops = if creating { + *fixture.counts.receipt_write_holds.lock().unwrap() = + Some((entered.clone(), release.clone())); + fixture.counts.receipt_write_wait_drops.clone() + } else { + let drops = Arc::new(AtomicUsize::new(0)); + *fixture.counts.receipt_open_holds.lock().unwrap() = + Some((entered.clone(), release.clone(), drops.clone())); + drops + }; + let mut task = + tokio::spawn(async move { repository.install(verified, &objects, &admission).await }); + timeout(Duration::from_secs(10), entered.notified()) + .await + .unwrap(); + let db = fixture.state.storage.mono_storage().get_connection(); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 1); + if creating { + db.execute_unprepared("UPDATE mst2_chunk_receipt_generation SET deadline=pg_catalog.clock_timestamp()+interval '20 milliseconds' WHERE state='CREATING'").await.unwrap(); + } + // Atomic create uses the actual owner expiry; receipt input also keeps + // its stricter five-second cap. Neither wait needs backend release. + let finished = timeout( + Duration::from_secs(if creating { 12 } else { 7 }), + &mut task, + ) + .await; + if finished.is_err() { + task.abort(); + let _ = task.await; + panic!("receipt wait retained its install workspace indefinitely"); + } + let error = finished.unwrap().unwrap().unwrap_err(); + assert_eq!( + error.code, + if creating { + SnapshotErrorCode::LeaseExpired + } else { + SnapshotErrorCode::TemporaryUnavailable + } + ); + assert_eq!(drops.load(Ordering::SeqCst), 1); + assert_eq!(budget.used(), 0); + fixture.counts.assert(1, fixture.raw.len()); + assert_eq!( + fixture.counts.receipt_reads.load(Ordering::SeqCst), + usize::from(!creating) + ); + assert_eq!(count(db, "mst2_chunk_reader").await, 0); + assert_eq!(count(db, "mst2_chunk_map_source").await, 0); + assert_eq!(count(db, "mst2_chunk_map").await, 0); + *fixture.counts.receipt_open_holds.lock().unwrap() = None; + let repository = fixture.state.storage.chunk_maps().await.unwrap(); + repository + .maintain(&fixture.state.storage.git_service.obj_storage, 64) + .await + .unwrap(); + assert_eq!(fixture.counts.receipt_deletes.load(Ordering::SeqCst), 1); + drop(release); + } +} + +#[tokio::test] +async fn held_raw_and_exact_range_do_not_resume_after_actual_reader_expiry() { + for raw_route in [true, false] { + let fixture = Fixture::new().await; + let map = fixture.map("/file").await; + let db = fixture.state.storage.mono_storage().get_connection(); + wait_count(db, "mst2_chunk_reader", 0).await; + fixture.counts.reset(); + let entered = Arc::new(Notify::new()); + let release = Arc::new(Notify::new()); + let drops = Arc::new(AtomicUsize::new(0)); + let tail_polls = Arc::new(AtomicUsize::new(0)); + *fixture.counts.object_fault.lock().unwrap() = Some(bounded_objects::StreamFault { + oid: fixture.oid.clone(), + kind: bounded_objects::FaultKind::HeldFragment { + prefix: Bytes::new(), + fragment: Some(Bytes::new()), + entered: entered.clone(), + release: release.clone(), + tail_polls: tail_polls.clone(), + drops: drops.clone(), + }, + }); + let app = fixture.app.clone(); + let request = if raw_route { + fixture.request("GET", "blob?path=/file", Body::empty()) + } else { + fixture.request( + "POST", + "chunks", + Body::from( + fixture + .chunk_body("/file", map["map"]["map_id"].as_str().unwrap(), "0") + .to_string(), + ), + ) + }; + let mut task = tokio::spawn(async move { + let response = app.oneshot(request).await.unwrap(); + if raw_route { + assert_eq!(response.status(), 200); + let mut body = response.into_body().into_data_stream(); + assert!(body.next().await.unwrap().is_err()); + assert!(body.next().await.is_none()); + } else { + error(response, 410, "LEASE_EXPIRED", false).await; + } + }); + timeout(Duration::from_secs(10), entered.notified()) + .await + .unwrap(); + assert_eq!(count(db, "mst2_chunk_reader").await, 1); + db.execute_unprepared("UPDATE mst2_chunk_reader SET deadline=pg_catalog.clock_timestamp()+interval '20 milliseconds'").await.unwrap(); + // Exercise the real request-held reader without reaching into it: + // let the existing ten-second local check interval elapse too. + tokio::time::sleep(Duration::from_secs(11)).await; + release.notify_one(); + let completed = timeout(Duration::from_secs(5), &mut task).await; + if completed.is_err() { + task.abort(); + let _ = task.await; + panic!("expired reader continued into another held backend poll"); + } + completed.unwrap().unwrap(); + assert_eq!(tail_polls.load(Ordering::SeqCst), 0); + assert_eq!(drops.load(Ordering::SeqCst), 1); + assert_eq!(fixture.counts.bytes.load(Ordering::SeqCst), 0); + wait_count(db, "mst2_chunk_reader", 0).await; + assert_eq!(count(db, "mst2_chunk_map_source").await, 1); + } +} + +#[tokio::test] +async fn production_q_bootstrap_after_retention_preserves_catalog_and_reconstructed_warm_routes() { + use crate::jupiter::storage::qualified_metadata_family::{ + SnapshotMetadataFamily, provision_or_verify_rooted_qualified_family, + }; + + let fixture = Fixture::new_with_pg_config(true).await; + let db = fixture.state.storage.mono_storage().get_connection(); + let namespace = provision_or_verify_rooted_qualified_family(db) + .await + .unwrap(); + assert_eq!( + fixture + .state + .storage + .snapshot_metadata_family(&fixture.lease, true) + .await + .unwrap(), + Some(SnapshotMetadataFamily::Rooted) + ); + let applied: i64 = db + .query_one_raw(statement( + "SELECT count(*) AS count FROM seaql_migrations WHERE version='m20261008_000300_add_mst2_chunk_map_retention'", + [], + )) + .await + .unwrap() + .unwrap() + .try_get("", "count") + .unwrap(); + assert_eq!(applied, 1); + let original = fixture.map("/file").await; + fixture.counts.assert(1, fixture.raw.len()); + wait_count(db, "mst2_chunk_reader", 0).await; + + // This is the production connection path: it reapplies migration checks + // and verifies the complete existing Q authority catalog without a test + // exemption or a refreshed registration fingerprint. + let config = fixture.state.storage.config(); + let connection = crate::jupiter::storage::init::database_connection(&config.database) + .await + .unwrap(); + assert_eq!( + provision_or_verify_rooted_qualified_family(&connection) + .await + .unwrap(), + namespace + ); + let storage = crate::jupiter::storage::Storage::new_with_connection( + config, + Arc::new(connection), + fixture.state.storage.git_service.obj_storage.clone(), + ) + .await + .unwrap(); + let state = MonoApiServiceState { + storage, + ..fixture.state.clone() + }; + assert_eq!( + state + .storage + .snapshot_metadata_family(&fixture.lease, true) + .await + .unwrap(), + Some(SnapshotMetadataFamily::Rooted) + ); + let app = router(&state); + fixture.counts.reset(); + let warm = success_json( + app.clone() + .oneshot(fixture.request("GET", "chunk-map?path=/alias", Body::empty())) + .await + .unwrap(), + ) + .await; + assert_eq!(warm["map"], original["map"]); + fixture.counts.assert(0, 0); + assert_eq!(fixture.counts.receipt_reads.load(Ordering::SeqCst), 1); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 0); + + fixture.counts.reset(); + let response = app + .oneshot( + fixture.request( + "POST", + "chunks", + Body::from( + fixture + .chunk_body("/alias", original["map"]["map_id"].as_str().unwrap(), "0") + .to_string(), + ), + ), + ) + .await + .unwrap(); + assert_eq!(response.status(), 200); + let wire = to_bytes(response.into_body(), 2 * 1024 * 1024) + .await + .unwrap(); + let frames = parse_stream(&wire).unwrap(); + let [Frame::Chunk(chunk), Frame::End(end)] = frames.as_slice() else { + panic!("reconstructed rooted route must emit CHUNK and terminal END"); + }; + assert_eq!(chunk.chunk_bytes, &fixture.raw[..CHUNK_SIZE as usize]); + assert_eq!(chunk.file_content_id, fixture.digest); + assert_eq!(chunk.chunk_index, 0); + assert_eq!(end.request_item_count, 1); + assert_eq!(end.logical_bytes, u64::from(CHUNK_SIZE)); + assert_eq!(fixture.counts.whole.load(Ordering::SeqCst), 0); + assert_eq!(fixture.counts.range.load(Ordering::SeqCst), 1); + assert_eq!( + fixture.counts.bytes.load(Ordering::SeqCst), + CHUNK_SIZE as usize + ); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 0); + drop(wire); + wait_count(db, "mst2_chunk_reader", 0).await; + assert_eq!(count(db, "mst2_chunk_map_source").await, 1); + assert_eq!(count(db, "mst2_chunk_map").await, 1); +} + +#[tokio::test] +async fn one_connection_install_commits_bounded_stages_and_reconstructed_warm_reads() { + let fixture = Fixture::new_with_pg_config(true).await; + let mut config = fixture.state.storage.config().database.clone(); + config.max_connection = 1; + config.min_connection = 1; + let connection = crate::jupiter::storage::init::database_connection(&config) + .await + .unwrap(); + let repository = PostgresChunkMapRepository::new(connection.clone()) + .await + .unwrap(); + assert!( + fixture + .state + .storage + .native_chunk_maps + .set(repository) + .is_ok() + ); + connection.execute_unprepared("CREATE TABLE chunk_stage_audit(relation text NOT NULL, writer_xid bigint NOT NULL); CREATE FUNCTION chunk_stage_audit() RETURNS trigger LANGUAGE plpgsql AS $audit$ BEGIN INSERT INTO chunk_stage_audit VALUES(TG_TABLE_NAME,pg_catalog.txid_current()); RETURN NEW; END $audit$; CREATE TRIGGER chunk_stage_audit AFTER INSERT ON mst2_chunk_map FOR EACH ROW EXECUTE FUNCTION chunk_stage_audit(); CREATE TRIGGER chunk_stage_audit AFTER INSERT ON mst2_chunk_map_leaf FOR EACH ROW EXECUTE FUNCTION chunk_stage_audit(); CREATE TRIGGER chunk_stage_audit AFTER INSERT ON mst2_chunk_map_node FOR EACH ROW EXECUTE FUNCTION chunk_stage_audit(); CREATE TRIGGER chunk_stage_audit AFTER INSERT ON mst2_chunk_map_source FOR EACH ROW EXECUTE FUNCTION chunk_stage_audit()").await.unwrap(); + let map = timeout(Duration::from_secs(10), fixture.map("/file")) + .await + .expect("single-connection staging waited for its own held connection"); + fixture.counts.assert(1, fixture.raw.len()); + let transactions: i64 = connection + .query_one_raw(statement( + "SELECT count(DISTINCT writer_xid) AS count FROM chunk_stage_audit", + [], + )) + .await + .unwrap() + .unwrap() + .try_get("", "count") + .unwrap(); + assert!(transactions >= 4); + for table in [ + "mst2_chunk_map", + "mst2_chunk_map_leaf", + "mst2_chunk_map_node", + "mst2_chunk_map_source", + ] { + assert_eq!(count(&connection, table).await, 1); + } + fixture.counts.reset(); + let reconstructed = PostgresChunkMapRepository::new(connection.clone()) + .await + .unwrap(); + let source = ChunkMapSource::from_fact(fixture.fact().await, &fixture.oid).unwrap(); + let persisted = timeout( + Duration::from_secs(10), + reconstructed.read(&source, &fixture.state.storage.git_service.obj_storage), + ) + .await + .unwrap() + .unwrap() + .unwrap(); + assert_eq!( + format!("sha256:{}", hex_of(&persisted.map_id)), + map["map"]["map_id"] + ); + let page = reconstructed.selected_page(&persisted, 0).await.unwrap(); + page.verify_chunk(&persisted.map, 0, &fixture.raw[..CHUNK_SIZE as usize]) + .unwrap(); + fixture.counts.assert(0, 0); + assert_eq!(fixture.counts.receipt_reads.load(Ordering::SeqCst), 1); +} + +#[tokio::test] +async fn cached_inventory_is_coalesced_and_bound_to_the_actual_backend_arc() { + let fixture = Fixture::new().await; + let repository = fixture.state.storage.chunk_maps().await.unwrap(); + let objects = &fixture.state.storage.git_service.obj_storage; + repository.maintain(objects, 8).await.unwrap(); + let first = fixture + .counts + .receipt_inventory_calls + .load(Ordering::SeqCst); + repository.maintain(objects, 8).await.unwrap(); + assert_eq!( + fixture + .counts + .receipt_inventory_calls + .load(Ordering::SeqCst), + first + ); + let counts = Arc::new(ReadCounts::default()); + counts + .receipt_retention_unsupported + .store(true, Ordering::SeqCst); + let other = MegaObjectStorageWrapper::new(Arc::new(CountingStorage { + inner: objects.clone(), + counts: counts.clone(), + })); + let error = repository.maintain(&other, 8).await.unwrap_err(); + assert_eq!(error.code, SnapshotErrorCode::TemporaryUnavailable); + assert_eq!(counts.receipt_inventory_calls.load(Ordering::SeqCst), 1); + fixture.counts.assert(0, 0); + assert_eq!( + count( + fixture.state.storage.mono_storage().get_connection(), + "mst2_chunk_receipt_generation" + ) + .await, + 0 + ); +} + +#[tokio::test] +async fn completion_during_real_inventory_keeps_backing_credit_after_pending_install_disappears() { + let mut fixture = Fixture::new().await; + let backing = mock_object_storage(); + backing + .inner + .put_stream( + &ObjectKey { + namespace: ObjectNamespace::Git, + key: fixture.oid.clone(), + }, + Box::pin(futures::stream::iter([Ok(Bytes::copy_from_slice( + &fixture.raw, + ))])), + ObjectMeta::default(), + ) + .await + .unwrap(); + fixture.state.storage.git_service = GitService { + obj_storage: MegaObjectStorageWrapper::new(Arc::new(CountingStorage { + inner: backing.clone(), + counts: fixture.counts.clone(), + })), + }; + fixture.app = router(&fixture.state); + let db = fixture.state.storage.mono_storage().get_connection(); + let create_entered = Arc::new(Notify::new()); + let create_release = Arc::new(Notify::new()); + *fixture.counts.receipt_late_create_holds.lock().unwrap() = + Some((create_entered.clone(), create_release.clone())); + let app = fixture.app.clone(); + let request = fixture.request("GET", "chunk-map?path=/file", Body::empty()); + let leader = tokio::spawn(async move { app.oneshot(request).await.unwrap() }); + timeout(Duration::from_secs(10), create_entered.notified()) + .await + .unwrap(); + *fixture.counts.receipt_late_create_holds.lock().unwrap() = None; + for index in 0..crate::orbit_api::object_storage::MAX_CHUNK_MAP_RECEIPTS - 1 { + backing + .inner + .put_metadata_atomic_create( + &ObjectKey { + namespace: ObjectNamespace::ChunkMapReceipt, + key: format!("{index:064x}"), + }, + Bytes::from_static(&[1; 129]), + ObjectMeta::default(), + ) + .await + .unwrap(); + } + // Synthetic occupants test physical capacity; they confer no source trust. + let actual = &fixture.state.storage.git_service.obj_storage; + let inventory_entered = Arc::new(Notify::new()); + let inventory_release = Arc::new(Notify::new()); + let observer_counts = Arc::new(ReadCounts::default()); + *observer_counts.receipt_inventory_holds.lock().unwrap() = + Some((inventory_entered.clone(), inventory_release.clone())); + let observer = PostgresChunkMapRepository::new(db.clone()).await.unwrap(); + let observed = MegaObjectStorageWrapper::new(Arc::new(CountingStorage { + inner: actual.clone(), + counts: observer_counts, + })); + let capacity = tokio::spawn(async move { observer.test_new_install_capacity(&observed).await }); + timeout(Duration::from_secs(30), inventory_entered.notified()) + .await + .unwrap(); + create_release.notify_one(); + success_json( + timeout(Duration::from_secs(30), leader) + .await + .unwrap() + .unwrap(), + ) + .await; + assert_eq!(count(db, "mst2_chunk_map_source").await, 1); + let pending:i64=db.query_one_raw(statement("SELECT count(*) AS count FROM mst2_chunk_receipt_generation WHERE state IN ('RESERVED','CREATING')",[])).await.unwrap().unwrap().try_get("","count").unwrap(); + assert_eq!(pending, 0); + inventory_release.notify_one(); + let error = timeout(Duration::from_secs(10), capacity) + .await + .unwrap() + .unwrap() + .unwrap_err(); + assert_eq!(error.code, SnapshotErrorCode::LimitExceeded); + fixture.counts.assert(1, fixture.raw.len()); + assert_eq!( + actual + .inner + .chunk_map_receipt_inventory() + .await + .unwrap() + .objects + .len(), + crate::orbit_api::object_storage::MAX_CHUNK_MAP_RECEIPTS + ); +} + +#[tokio::test] +async fn json_chunk_and_raw_last_transport_clones_block_actual_collection() { + use crate::ceres::snapshot::content_budget::{MemoryBudget, RANGE_WORK_BYTES}; + + for route in 0..3 { + let fixture = Fixture::new().await; + let map = fixture.map("/file").await; + let repository = fixture.state.storage.chunk_maps().await.unwrap(); + let db = fixture.state.storage.mono_storage().get_connection(); + wait_count(db, "mst2_chunk_reader", 0).await; + fixture.counts.reset(); + let response_budget = MemoryBudget::new(CHUNK_SIZE as usize + 2048); + let scratch_budget = MemoryBudget::new(RANGE_WORK_BYTES); + let response = match route { + 0 => { + fixture + .send("GET", "chunk-map?path=/file", Body::empty()) + .await + } + 1 => budgeted_chunks_app(&fixture, &response_budget, &scratch_budget) + .oneshot( + fixture.request( + "POST", + "chunks", + Body::from( + json!({"items":[{ + "path":"/file", "expected_digest":fixture.digest_string(), + "map_id":map["map"]["map_id"], "chunk_index":"0" + }],"encoding":"identity"}) + .to_string(), + ), + ), + ) + .await + .unwrap(), + _ => super::raw_blob::budgeted_app(&fixture, &response_budget, &scratch_budget) + .oneshot(fixture.request("GET", "blob?path=/file", Body::empty())) + .await + .unwrap(), + }; + assert_eq!(response.status(), 200); + let mut body = response.into_body().into_data_stream(); + let mut wire = Vec::new(); + let mut transport = Vec::new(); + while let Some(frame) = body.next().await { + let bytes = frame.unwrap(); + wire.extend_from_slice(&bytes); + transport.push(bytes.clone()); + } + drop(body); + if route == 2 { + assert_eq!(wire, fixture.raw); + } + if route == 1 { + let frames = parse_stream(&wire).unwrap(); + assert!(frames.iter().any(|frame| matches!(frame, Frame::End(_)))); + } + assert!(!transport.is_empty()); + let last = transport.pop().unwrap(); + let last_transport = last.clone(); + drop(last); + drop(transport); + assert_eq!(scratch_budget.used(), 0); + if route == 1 { + assert_eq!(response_budget.used(), CHUNK_SIZE as usize + 2048); + } else if route == 2 { + assert_eq!( + response_budget.used(), + fixture.raw.len() - CHUNK_SIZE as usize + ); + } + assert_eq!(count(db, "mst2_chunk_reader").await, 1); + age_unowned_candidates(db).await; + repository + .maintain(&fixture.state.storage.git_service.obj_storage, 64) + .await + .unwrap(); + assert_eq!(count(db, "mst2_chunk_map_source").await, 1); + assert_eq!(count(db, "mst2_chunk_map").await, 1); + assert_eq!(fixture.counts.receipt_deletes.load(Ordering::SeqCst), 0); + drop(last_transport); + assert_eq!(response_budget.used(), 0); + wait_count(db, "mst2_chunk_reader", 0).await; + repository + .maintain(&fixture.state.storage.git_service.obj_storage, 64) + .await + .unwrap(); + for table in [ + "mst2_chunk_map_source", + "mst2_chunk_map", + "mst2_chunk_map_leaf", + "mst2_chunk_map_node", + "mst2_chunk_map_lifetime", + ] { + assert_eq!(count(db, table).await, 0); + } + assert_eq!(fixture.counts.receipt_deletes.load(Ordering::SeqCst), 1); + assert_eq!(count(db, "mst2_chunk_receipt_generation").await, 1); + assert_eq!(count(db, "mst2_chunk_map_gc").await, 1); + } +} + +#[tokio::test] +async fn expired_reader_arc_cannot_resurrect_or_authenticate_a_page() { + let fixture = Fixture::new().await; + fixture.map("/file").await; + let db = fixture.state.storage.mono_storage().get_connection(); + wait_count(db, "mst2_chunk_reader", 0).await; + let repository = fixture.state.storage.chunk_maps().await.unwrap(); + let source = ChunkMapSource::from_fact(fixture.fact().await, &fixture.oid).unwrap(); + let map = repository + .read(&source, &fixture.state.storage.git_service.obj_storage) + .await + .unwrap() + .unwrap(); + let old_generation: i64 = db + .query_one_raw(statement( + "SELECT generation FROM mst2_chunk_map_lifetime WHERE map_id=$1", + [map.map_id.to_vec().into()], + )) + .await + .unwrap() + .unwrap() + .try_get("", "generation") + .unwrap(); + db.execute_unprepared("UPDATE mst2_chunk_reader SET deadline=pg_catalog.clock_timestamp()+interval '20 milliseconds'").await.unwrap(); + tokio::time::sleep(Duration::from_millis(50)).await; + map.test_check_next_owner_operation().await; + assert_eq!( + map.ensure_live().await.unwrap_err().code, + SnapshotErrorCode::LeaseExpired + ); + assert_eq!( + map.record_progress().await.unwrap_err().code, + SnapshotErrorCode::LeaseExpired + ); + assert_eq!( + repository.selected_page(&map, 0).await.err().unwrap().code, + SnapshotErrorCode::LeaseExpired + ); + assert!(db.execute_unprepared("UPDATE mst2_chunk_reader SET deadline=pg_catalog.clock_timestamp()+interval '59 seconds'").await.is_err()); + age_unowned_candidates(db).await; + repository + .maintain(&fixture.state.storage.git_service.obj_storage, 64) + .await + .unwrap(); + assert_eq!(count(db, "mst2_chunk_map").await, 0); + assert_eq!( + map.ensure_live().await.unwrap_err().code, + SnapshotErrorCode::LeaseExpired + ); + fixture.counts.reset(); + assert_eq!( + fixture.map("/file").await["map"]["file_size"], + fixture.raw.len().to_string() + ); + fixture.counts.assert(1, fixture.raw.len()); + let new_generation: i64 = db + .query_one_raw(statement( + "SELECT generation FROM mst2_chunk_map_lifetime WHERE map_id=$1", + [map.map_id.to_vec().into()], + )) + .await + .unwrap() + .unwrap() + .try_get("", "generation") + .unwrap(); + assert!(new_generation > old_generation); + assert!( + db.execute_raw(statement( + "UPDATE mst2_chunk_map_lifetime SET state='DELETING' WHERE map_id=$1", + [map.map_id.to_vec().into()] + )) + .await + .is_err() + ); + assert!( + db.execute_raw(statement( + "DELETE FROM mst2_chunk_map_leaf WHERE map_id=$1", + [map.map_id.to_vec().into()] + )) + .await + .is_err() + ); + fixture.counts.reset(); + fixture.map("/alias").await; + fixture.counts.assert(0, 0); + assert_eq!( + map.ensure_live().await.unwrap_err().code, + SnapshotErrorCode::LeaseExpired + ); +} + +#[tokio::test] +async fn late_cancelled_create_is_reserved_and_old_key_replay_preserves_new_generation() { + let fixture = Fixture::new().await; + let db = fixture.state.storage.mono_storage().get_connection(); + let entered = Arc::new(Notify::new()); + let release = Arc::new(Notify::new()); + *fixture.counts.receipt_late_create_holds.lock().unwrap() = + Some((entered.clone(), release.clone())); + let app = fixture.app.clone(); + let request = fixture.request("GET", "chunk-map?path=/file", Body::empty()); + let cancelled = tokio::spawn(async move { app.oneshot(request).await }); + timeout(Duration::from_secs(10), entered.notified()) + .await + .unwrap(); + let old_key: String = db + .query_one_raw(statement( + "SELECT receipt_key FROM mst2_chunk_receipt_generation", + [], + )) + .await + .unwrap() + .unwrap() + .try_get("", "receipt_key") + .unwrap(); + assert!( + !fixture + .state + .storage + .git_service + .obj_storage + .inner + .exists(&ObjectKey { + namespace: ObjectNamespace::ChunkMapReceipt, + key: old_key.clone() + }) + .await + .unwrap() + ); + cancelled.abort(); + assert!(cancelled.await.err().unwrap().is_cancelled()); + *fixture.counts.receipt_late_create_holds.lock().unwrap() = None; + let current = fixture.map("/file").await; + let repository = fixture.state.storage.chunk_maps().await.unwrap(); + wait_count(db, "mst2_chunk_reader", 0).await; + let history = db + .query_one_raw(statement( + "SELECT state,create_completed FROM mst2_chunk_receipt_generation WHERE receipt_key=$1", + [old_key.clone().into()], + )) + .await + .unwrap() + .unwrap(); + assert_eq!(history.try_get::("", "state").unwrap(), "APPLIED"); + assert!(!history.try_get::("", "create_completed").unwrap()); + let new_key: String = db + .query_one_raw(statement( + "SELECT receipt_key FROM mst2_chunk_map_source", + [], + )) + .await + .unwrap() + .unwrap() + .try_get("", "receipt_key") + .unwrap(); + assert_ne!(old_key, new_key); + release.notify_one(); + timeout(Duration::from_secs(10), async { + loop { + if fixture + .state + .storage + .git_service + .obj_storage + .inner + .exists(&ObjectKey { + namespace: ObjectNamespace::ChunkMapReceipt, + key: old_key.clone(), + }) + .await + .unwrap() + && fixture.counts.receipt_writes.load(Ordering::SeqCst) == 2 + { + break; + } + tokio::task::yield_now().await; + } + }) + .await + .unwrap(); + repository + .maintain(&fixture.state.storage.git_service.obj_storage, 64) + .await + .unwrap(); + assert!( + !fixture + .state + .storage + .git_service + .obj_storage + .inner + .exists(&ObjectKey { + namespace: ObjectNamespace::ChunkMapReceipt, + key: old_key.clone() + }) + .await + .unwrap() + ); + assert!( + fixture + .state + .storage + .git_service + .obj_storage + .inner + .exists(&ObjectKey { + namespace: ObjectNamespace::ChunkMapReceipt, + key: new_key + }) + .await + .unwrap() + ); + fixture.counts.reset(); + assert_eq!(fixture.map("/file").await["map"], current["map"]); + fixture.counts.assert(0, 0); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 0); + let history = db + .query_one_raw(statement( + "SELECT create_completed FROM mst2_chunk_receipt_generation WHERE receipt_key=$1", + [old_key.into()], + )) + .await + .unwrap() + .unwrap(); + assert!(!history.try_get::("", "create_completed").unwrap()); +} + +#[tokio::test] +async fn actual_backing_quota_and_unsupported_capability_reject_before_source_open() { + for unsupported in [true, false] { + let fixture = Fixture::new().await; + let counts = Arc::new(ReadCounts::default()); + counts + .receipt_retention_unsupported + .store(unsupported, Ordering::SeqCst); + let backend = mock_object_storage(); + if !unsupported { + for index in 0..=crate::orbit_api::object_storage::MAX_CHUNK_MAP_RECEIPTS { + backend + .inner + .put_metadata_atomic_create( + &ObjectKey { + namespace: ObjectNamespace::ChunkMapReceipt, + key: format!("{index:064x}"), + }, + Bytes::from_static(&[1; 129]), + ObjectMeta::default(), + ) + .await + .unwrap(); + } + } + let mut state = fixture.state.clone(); + state.storage.git_service = GitService { + obj_storage: MegaObjectStorageWrapper::new(Arc::new(CountingStorage { + inner: backend, + counts: counts.clone(), + })), + }; + let response = router(&state) + .oneshot(fixture.request("GET", "chunk-map?path=/file", Body::empty())) + .await + .unwrap(); + if unsupported { + error(response, 503, "TEMPORARY_UNAVAILABLE", true).await; + } else { + error(response, 413, "LIMIT_EXCEEDED", false).await; + } + counts.assert(0, 0); + assert_eq!(counts.receipt_writes.load(Ordering::SeqCst), 0); + let db = fixture.state.storage.mono_storage().get_connection(); + assert_eq!(count(db, "mst2_chunk_receipt_generation").await, 0); + assert_eq!(count(db, "mst2_chunk_map").await, 0); + } +} + +#[tokio::test] +async fn empty_source_fragment_after_deadline_does_not_renew_install_or_publish() { + let fixture = Fixture::new().await; + let repository = fixture.state.storage.chunk_maps().await.unwrap(); + let source = ChunkMapSource::from_fact(fixture.fact().await, &fixture.oid).unwrap(); + let admission = Arc::new( + repository + .admit_install(&source, &fixture.state.storage.git_service.obj_storage) + .await + .unwrap(), + ); + let entered = Arc::new(Notify::new()); + let release = Arc::new(Notify::new()); + let drops = Arc::new(AtomicUsize::new(0)); + let tail_polls = Arc::new(AtomicUsize::new(0)); + *fixture.counts.object_fault.lock().unwrap() = Some(bounded_objects::StreamFault { + oid: fixture.oid.clone(), + kind: bounded_objects::FaultKind::HeldFragment { + prefix: Bytes::new(), + fragment: Some(Bytes::new()), + entered: entered.clone(), + release: release.clone(), + tail_polls: tail_polls.clone(), + drops: drops.clone(), + }, + }); + let handler = MonoApiService::from(&fixture.state); + let budget = crate::ceres::snapshot::content_budget::MemoryBudget::new(8 * 1024 * 1024); + let verifier_budget = budget.clone(); + let owner = admission.clone(); + let task = tokio::spawn(async move { + VerifiedSourceChunkMap::verify(&handler, source, &verifier_budget, &owner).await + }); + timeout(Duration::from_secs(10), entered.notified()) + .await + .unwrap(); + let db = fixture.state.storage.mono_storage().get_connection(); + db.execute_unprepared("UPDATE mst2_chunk_receipt_generation SET deadline=pg_catalog.clock_timestamp()+interval '20 milliseconds' WHERE state='RESERVED'").await.unwrap(); + tokio::time::sleep(Duration::from_millis(50)).await; + admission.test_check_next_owner_operation().await; + release.notify_one(); + let error = timeout(Duration::from_secs(5), task) + .await + .unwrap() + .unwrap() + .err() + .unwrap(); + assert_eq!(error.code, SnapshotErrorCode::LeaseExpired); + assert_eq!(drops.load(Ordering::SeqCst), 1); + assert_eq!(tail_polls.load(Ordering::SeqCst), 0); + assert_eq!(fixture.counts.bytes.load(Ordering::SeqCst), 0); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 0); + assert_eq!(count(db, "mst2_chunk_map_source").await, 0); + assert_eq!(budget.used(), 0); + drop(admission); + repository + .maintain(&fixture.state.storage.git_service.obj_storage, 64) + .await + .unwrap(); + assert_eq!(count(db, "mst2_chunk_map").await, 0); +} diff --git a/src/api/router/snapshot_content.rs b/src/api/router/snapshot_content.rs index 275f7dd7..961d1b00 100644 --- a/src/api/router/snapshot_content.rs +++ b/src/api/router/snapshot_content.rs @@ -19,7 +19,7 @@ use super::{ }; use crate::ceres::snapshot::{ chunks::{ChunkMapSource, VerifiedSourceChunkMap, map_build_reservation_bytes}, - content_budget::{BudgetedFrame, MemoryLease, reserve_range_work, reserve_response}, + content_budget::{BudgetedFrame, MemoryLease, reserve_response}, error::{SnapshotError, SnapshotErrorCode}, pages::{MetadataWalkOutcome, base64_of, hex_of, resolve_abs_metadata}, resolver::FsKind, @@ -640,21 +640,29 @@ pub(super) async fn project_resolved, ) -> Result { let bytes = map_json_bytes(value, memory)?; let headers = super::REQUEST_HEADERS.try_with(Clone::clone).map_err(|_| { @@ -865,11 +874,17 @@ async fn guarded_map_json_response( ) })?; super::revalidate_access(state, context, &headers).await?; + map.ensure_live().await?; let state = state.clone(); let context = context.clone(); let stream = futures::stream::once(async move { super::revalidate_access(&state, &context, &headers).await?; - Ok::<_, SnapshotError>(bytes) + map.ensure_live().await?; + map.record_progress().await?; + Ok::<_, SnapshotError>(bytes::Bytes::from_owner(RetainedChunkFrame { + bytes, + _maps: std::sync::Arc::new(vec![map]), + })) }); Response::builder() .header("content-type", "application/json") @@ -879,6 +894,43 @@ async fn guarded_map_json_response( .map_err(|_| internal("chunk map JSON response build failed")) } +struct RetainedChunkFrame { + bytes: bytes::Bytes, + _maps: std::sync::Arc< + Vec>, + >, +} +impl AsRef<[u8]> for RetainedChunkFrame { + fn as_ref(&self) -> &[u8] { + &self.bytes + } +} + +fn retain_chunk_maps( + response: Response, + maps: Vec>, +) -> Response { + let maps = std::sync::Arc::new(maps); + let (parts, body) = response.into_parts(); + let stream = body.into_data_stream().then(move |frame| { + let maps = maps.clone(); + async move { + let bytes = frame.map_err(std::io::Error::other)?; + for map in maps.iter() { + map.ensure_live().await.map_err(std::io::Error::other)?; + if !bytes.is_empty() { + map.record_progress().await.map_err(std::io::Error::other)?; + } + } + Ok::<_, std::io::Error>(bytes::Bytes::from_owner(RetainedChunkFrame { + bytes, + _maps: maps, + })) + } + }); + Response::from_parts(parts, axum::body::Body::from_stream(stream)) +} + #[derive(Deserialize, Debug)] #[serde(deny_unknown_fields)] struct ChunksRequest { @@ -899,6 +951,20 @@ struct ChunkItem { const CHUNKS_MAX_ITEMS: usize = 128; const CHUNKS_TOTAL_MAX: u64 = 128 * 1024 * 1024; +pub(super) struct ChunksBudgets { + pub(super) response: std::sync::Arc, + pub(super) scratch: std::sync::Arc, +} + +impl Default for ChunksBudgets { + fn default() -> Self { + Self { + response: crate::ceres::snapshot::content_budget::response_budget().clone(), + scratch: crate::ceres::snapshot::content_budget::range_budget().clone(), + } + } +} + struct Planned { projection: std::sync::Arc, page: std::sync::Arc, @@ -932,6 +998,10 @@ async fn read_chunk_range( planned: &Planned, ) -> Result, Response> { let projection = &planned.projection; + projection + .ensure_live() + .await + .map_err(mst2_error_response)?; let len = projection.map.chunk_len(planned.index).map_err(|error| { mst2_error_response(internal(format!("invalid admitted chunk: {error}"))) })?; @@ -942,9 +1012,10 @@ async fn read_chunk_range( let end = start .checked_add(len) .ok_or_else(|| mst2_error_response(internal("chunk end overflow")))?; - let (mut input, meta) = handler - .get_raw_blob_range_stream_exact(&planned.oid, start, end) + let (mut input, meta) = projection + .await_backend(handler.get_raw_blob_range_stream_exact(&planned.oid, start, end)) .await + .map_err(mst2_error_response)? .map_err(|error| mst2_error_response(content_read_error(error)))? .ok_or_else(|| { mst2_error_response(SnapshotError::new( @@ -952,6 +1023,10 @@ async fn read_chunk_range( "fixed source does not support exact raw ranges", )) })?; + projection + .ensure_live() + .await + .map_err(mst2_error_response)?; if u64::try_from(meta.size).ok() != Some(projection.map.file_size) { return Err(mst2_error_response(SnapshotError::new( SnapshotErrorCode::IntegrityError, @@ -965,7 +1040,21 @@ async fn read_chunk_range( "range allocation could not be admitted", )) })?; - while let Some(part) = input.next().await { + let mut empty_parts = 0usize; + loop { + projection + .ensure_live() + .await + .map_err(mst2_error_response)?; + let part = projection + .await_backend(input.next()) + .await + .map_err(mst2_error_response)?; + projection + .ensure_live() + .await + .map_err(mst2_error_response)?; + let Some(part) = part else { break }; let bytes = part.map_err(|error| { tracing::warn!(error = %error, "fixed-view range stream failed"); mst2_error_response(SnapshotError::new( @@ -980,19 +1069,53 @@ async fn read_chunk_range( ))); } raw.extend_from_slice(&bytes); + if !bytes.is_empty() { + projection + .record_progress() + .await + .map_err(mst2_error_response)?; + } else { + empty_parts += 1; + if empty_parts == 32 { + tokio::task::yield_now().await; + projection + .ensure_live() + .await + .map_err(mst2_error_response)?; + empty_parts = 0; + } + } } planned .page .verify_chunk(&projection.map, planned.index, &raw) .map_err(mst2_error_response)?; + projection + .ensure_live() + .await + .map_err(mst2_error_response)?; + projection + .record_progress() + .await + .map_err(mst2_error_response)?; Ok(raw) } #[allow(clippy::result_large_err)] pub(super) async fn chunks( + state: State, + path: AxumPath, + body: Mst2Bytes, +) -> Result { + chunks_with_budgets(state, path, body, ChunksBudgets::default()).await +} + +#[allow(clippy::result_large_err)] +pub(super) async fn chunks_with_budgets( state: State, AxumPath(snapshot_id): AxumPath, Mst2Bytes(body): Mst2Bytes, + budgets: ChunksBudgets, ) -> Result { ensure(&state)?; let ctx = super::request_context(&state, &snapshot_id).map_err(mst2_error_response)?; @@ -1090,7 +1213,10 @@ pub(super) async fn chunks( .ok() .and_then(|bytes| bytes.checked_add(req.items.len() * 1024 + 1024)) .ok_or_else(|| mst2_error_response(internal("chunk response memory overflow")))?; - let response_memory = reserve_response(response_bytes).map_err(mst2_error_response)?; + let response_memory = budgets + .response + .reserve(response_bytes) + .map_err(mst2_error_response)?; let mut maps = std::collections::HashMap::new(); let mut pages = std::collections::HashMap::new(); for item in resolved { @@ -1141,7 +1267,10 @@ pub(super) async fn chunks( let mut stream = FrameStream::new(1, encoding); let mut out: Vec> = Vec::new(); for p in planned.iter() { - let _work_memory = reserve_range_work().map_err(mst2_error_response)?; + let _work_memory = budgets + .scratch + .reserve(crate::ceres::snapshot::content_budget::RANGE_WORK_BYTES) + .map_err(mst2_error_response)?; // The source is the current request's fixed OID, never a cached // handler/backend/credential from a different scope. let bytes = read_chunk_range(handler.as_ref(), p).await?; @@ -1163,7 +1292,7 @@ pub(super) async fn chunks( ); out.push(end); - guarded_treeframe_response_with_budget( + let response = guarded_treeframe_response_with_budget( &state, &ctx, &snapshot_id, @@ -1171,7 +1300,8 @@ pub(super) async fn chunks( out, Some(response_memory), ) - .map_err(mst2_error_response) + .map_err(mst2_error_response)?; + Ok(retain_chunk_maps(response, maps.into_values().collect())) } /// Strict decimal-string parse for unsigned counts (spec 04 §1: no leading diff --git a/src/api/router/snapshot_content_tests.rs b/src/api/router/snapshot_content_tests.rs index 2dc14015..052c8369 100644 --- a/src/api/router/snapshot_content_tests.rs +++ b/src/api/router/snapshot_content_tests.rs @@ -88,6 +88,9 @@ mod bounded_chunks; #[path = "snapshot_persisted_chunk_map_tests.rs"] mod persisted_chunk_maps; +#[path = "snapshot_chunk_map_retention_tests.rs"] +mod chunk_map_retention; + #[path = "snapshot_raw_blob_tests.rs"] mod raw_blob; @@ -107,6 +110,27 @@ mod generation_qualified_fixture; mod install_capability_fixture; type ReceiptWriteHold = (Arc, Arc); +type ReadOpenHold = ( + Arc, + Arc, + Arc, +); + +struct ReadOpenDrop(Arc); + +impl Drop for ReadOpenDrop { + fn drop(&mut self) { + self.0.fetch_add(1, Ordering::SeqCst); + } +} + +async fn await_read_open_hold(hold: Option) { + if let Some((entered, release, drops)) = hold { + let _owner = ReadOpenDrop(drops); + entered.notify_one(); + release.notified().await; + } +} #[path = "snapshot_rooted_metadata_tests.rs"] mod rooted_metadata; @@ -126,6 +150,15 @@ struct ReadCounts { receipt_read_meta_size: std::sync::atomic::AtomicI64, receipt_read_late_error: AtomicBool, receipt_write_holds: std::sync::Mutex>, + receipt_write_wait_drops: Arc, + receipt_late_create_holds: std::sync::Mutex>, + receipt_inventory_holds: std::sync::Mutex>, + receipt_inventory_calls: AtomicUsize, + receipt_retention_unsupported: AtomicBool, + receipt_deletes: AtomicUsize, + whole_open_holds: std::sync::Mutex>, + range_open_holds: std::sync::Mutex>, + receipt_open_holds: std::sync::Mutex>, } impl ReadCounts { @@ -151,12 +184,71 @@ struct CountingStorage { #[async_trait::async_trait] impl MegaObjectStorage for CountingStorage { + async fn chunk_map_receipt_inventory( + &self, + ) -> OrbitResult { + self.counts + .receipt_inventory_calls + .fetch_add(1, Ordering::SeqCst); + if self + .counts + .receipt_retention_unsupported + .load(Ordering::SeqCst) + { + return Err(crate::orbit_api::error::IoOrbitError::ChunkMapRetentionUnsupported); + } + let inventory = self.inner.inner.chunk_map_receipt_inventory().await?; + let holds = self.counts.receipt_inventory_holds.lock().unwrap().clone(); + if let Some((entered, release)) = holds { + entered.notify_one(); + release.notified().await; + } + Ok(inventory) + } + + async fn delete_chunk_map_receipt( + &self, + authority: &crate::orbit_api::object_storage::ChunkMapReceiptDeletion, + ) -> OrbitResult { + let deleted = self.inner.inner.delete_chunk_map_receipt(authority).await?; + if deleted { + self.counts.receipt_deletes.fetch_add(1, Ordering::SeqCst); + } + Ok(deleted) + } + async fn put_metadata_atomic_create( &self, key: &ObjectKey, bytes: Bytes, meta: ObjectMeta, ) -> OrbitResult<()> { + let late = if key.namespace == ObjectNamespace::ChunkMapReceipt { + self.counts + .receipt_late_create_holds + .lock() + .unwrap() + .clone() + } else { + None + }; + if let Some((entered, release)) = late { + let inner = self.inner.clone(); + let key = key.clone(); + let counts = self.counts.clone(); + return tokio::spawn(async move { + entered.notify_one(); + release.notified().await; + inner + .inner + .put_metadata_atomic_create(&key, bytes, meta) + .await?; + counts.receipt_writes.fetch_add(1, Ordering::SeqCst); + Ok(()) + }) + .await + .unwrap(); + } self.inner .inner .put_metadata_atomic_create(key, bytes, meta) @@ -165,6 +257,7 @@ impl MegaObjectStorage for CountingStorage { self.counts.receipt_writes.fetch_add(1, Ordering::SeqCst); let holds = self.counts.receipt_write_holds.lock().unwrap().clone(); if let Some((entered, release)) = holds { + let _hold = ReadOpenDrop(self.counts.receipt_write_wait_drops.clone()); entered.notify_one(); release.notified().await; } @@ -193,6 +286,8 @@ impl MegaObjectStorage for CountingStorage { async fn get_stream(&self, key: &ObjectKey) -> OrbitResult<(ObjectByteStream, ObjectMeta)> { if key.namespace == ObjectNamespace::ChunkMapReceipt { self.counts.receipt_reads.fetch_add(1, Ordering::SeqCst); + let holds = self.counts.receipt_open_holds.lock().unwrap().clone(); + await_read_open_hold(holds).await; if self.counts.receipt_read_failure.load(Ordering::SeqCst) { return Err( crate::orbit_api::error::IoOrbitError::object_store_not_found( @@ -225,6 +320,8 @@ impl MegaObjectStorage for CountingStorage { return self.inner.inner.get_stream(key).await; } self.counts.whole.fetch_add(1, Ordering::SeqCst); + let holds = self.counts.whole_open_holds.lock().unwrap().clone(); + await_read_open_hold(holds).await; let chunk_fault = self .counts .chunk_faults @@ -286,6 +383,8 @@ impl MegaObjectStorage for CountingStorage { end: u64, ) -> OrbitResult> { self.counts.range.fetch_add(1, Ordering::SeqCst); + let holds = self.counts.range_open_holds.lock().unwrap().clone(); + await_read_open_hold(holds).await; let fault = self .counts .chunk_faults @@ -312,7 +411,12 @@ impl MegaObjectStorage for CountingStorage { .inner .get_range_stream_exact(key, start, end) .await?; + let fault = self.counts.object_fault.lock().unwrap().clone(); Ok(result.map(|(stream, meta)| { + let stream = match fault { + Some(fault) if fault.oid == key.key => fault.stream(), + _ => stream, + }; let counts = self.counts.clone(); let stream = stream.map(move |part| { if let Ok(bytes) = &part { diff --git a/src/api/router/snapshot_persisted_chunk_map_tests.rs b/src/api/router/snapshot_persisted_chunk_map_tests.rs index 6b261115..23e1808b 100644 --- a/src/api/router/snapshot_persisted_chunk_map_tests.rs +++ b/src/api/router/snapshot_persisted_chunk_map_tests.rs @@ -832,7 +832,7 @@ pub(super) async fn assert_three_page_proofs_and_selected_sibling_faults( .await .unwrap(); } else { - db.execute_raw(statement("UPDATE mst2_chunk_map_node SET digest=$2 WHERE map_id=$1 AND first_page=2 AND page_count=1", [id.clone().into(), vec![9; 32].into()])).await.unwrap(); + db.execute_raw(statement("UPDATE mst2_chunk_map_node SET digest=$2 WHERE map_id=$1 AND first_page=2 AND page_count=1", [id.clone().into(), vec![9u8; 32].into()])).await.unwrap(); } db.execute_unprepared("ALTER TABLE mst2_chunk_map_node ENABLE TRIGGER USER") .await @@ -1160,7 +1160,7 @@ async fn ordinary_incomplete_source_dml_cannot_commit_an_admitted_partial_map() )) .await .unwrap(); - txn.execute_raw(statement("INSERT INTO mst2_chunk_map_source(storage_domain,git_oid,object_kind,fact_id,source_id,source_bytes,primary_scope,map_id,receipt_digest) VALUES('git',$1,'blob',$2,$3,$4,$5,$6,$7)", [fixture.oid.clone().into(), source.fact().id.into(), repository.source_identity(&source).unwrap().to_vec().into(), source.canonical_bytes().unwrap().into(), scope.to_vec().into(), map.map_id.to_vec().into(), vec![9;32].into()])).await.unwrap(); + txn.execute_raw(statement("INSERT INTO mst2_chunk_map_source(storage_domain,git_oid,object_kind,fact_id,source_id,source_bytes,primary_scope,map_id,receipt_digest) VALUES('git',$1,'blob',$2,$3,$4,$5,$6,$7)", [fixture.oid.clone().into(), source.fact().id.into(), repository.source_identity(&source).unwrap().to_vec().into(), source.canonical_bytes().unwrap().into(), scope.to_vec().into(), map.map_id.to_vec().into(), vec![9u8;32].into()])).await.unwrap(); assert!(txn.commit().await.is_err()); for table in [ "mst2_chunk_map", diff --git a/src/api/router/snapshot_raw_blob.rs b/src/api/router/snapshot_raw_blob.rs index ea438ec1..016310cd 100644 --- a/src/api/router/snapshot_raw_blob.rs +++ b/src/api/router/snapshot_raw_blob.rs @@ -133,10 +133,20 @@ pub(super) async fn blob_with_budgets( } else { None }; - let (input, meta) = handler - .get_raw_blob_stream_with_meta(&file.oid) - .await - .map_err(|error| mst2_error_response(content::content_read_error(error)))?; + if let Some(map) = map.as_ref() { + map.ensure_live().await.map_err(mst2_error_response)?; + } + let open = handler.get_raw_blob_stream_with_meta(&file.oid); + let opened = if let Some(map) = map.as_ref() { + map.await_backend(open).await.map_err(mst2_error_response)? + } else { + open.await + }; + let (input, meta) = + opened.map_err(|error| mst2_error_response(content::content_read_error(error)))?; + if let Some(map) = map.as_ref() { + map.ensure_live().await.map_err(mst2_error_response)?; + } if u64::try_from(meta.size).ok() != Some(file.size) { return Err(mst2_error_response(integrity( "raw source physical size disagrees with its fixed verified fact", @@ -231,7 +241,11 @@ impl AsRef<[u8]> for RawBlobChunk { impl RawBlobReader { async fn validate(&mut self) -> Result<(), SnapshotError> { - revalidate_access(&self.state, &self.context, &self.headers).await + revalidate_access(&self.state, &self.context, &self.headers).await?; + if let Some(map) = self.map.as_ref() { + map.ensure_live().await?; + } + Ok(()) } async fn next_chunk(&mut self) -> Result, SnapshotError> { @@ -327,6 +341,9 @@ impl RawBlobReader { } } self.validate().await?; + if let Some(map) = self.map.as_ref() { + map.record_progress().await?; + } if raw.capacity() > credit.bytes { return Err(integrity("raw chunk allocation exceeds its owned credit")); } @@ -342,19 +359,30 @@ impl RawBlobReader { } async fn poll_source(&mut self) -> Result, SnapshotError> { + self.validate().await?; let input = self .input .as_mut() .ok_or_else(|| integrity("raw source stream was already closed"))?; - let result = input.next().await; + let result = if let Some(map) = self.map.as_ref() { + map.await_backend(input.next()).await? + } else { + input.next().await + }; self.validate().await?; - result.transpose().map_err(|error| { + let part = result.transpose().map_err(|error| { tracing::warn!(%error, "fixed-view raw stream failed"); SnapshotError::new( SnapshotErrorCode::ObjectUnavailable, "fixed-view raw stream failed", ) - }) + })?; + if part.as_ref().is_some_and(|bytes| !bytes.is_empty()) { + if let Some(map) = self.map.as_ref() { + map.record_progress().await?; + } + } + Ok(part) } async fn require_eof(&mut self) -> Result<(), SnapshotError> { diff --git a/src/api/router/snapshot_raw_blob_tests.rs b/src/api/router/snapshot_raw_blob_tests.rs index 1c4e70f9..79a38c79 100644 --- a/src/api/router/snapshot_raw_blob_tests.rs +++ b/src/api/router/snapshot_raw_blob_tests.rs @@ -11,7 +11,7 @@ use crate::{ ceres::snapshot::content_budget::{MemoryBudget, RANGE_WORK_BYTES}, }; -fn budgeted_app( +pub(super) fn budgeted_app( fixture: &Fixture, response_budget: &Arc, scratch_budget: &Arc, diff --git a/src/ceres/snapshot/chunks.rs b/src/ceres/snapshot/chunks.rs index 8346cec2..8018c245 100644 --- a/src/ceres/snapshot/chunks.rs +++ b/src/ceres/snapshot/chunks.rs @@ -78,6 +78,7 @@ impl VerifiedSourceChunkMap { handler: &T, source: ChunkMapSource, budget: &Arc, + admission: &crate::jupiter::storage::native_chunk_map::retention::ChunkMapInstall, ) -> Result { let digest: [u8; 32] = source .fact @@ -106,6 +107,7 @@ impl VerifiedSourceChunkMap { }) }, budget, + admission, ) .await?; Ok(Self { source, projection }) diff --git a/src/ceres/snapshot/chunks_stream.rs b/src/ceres/snapshot/chunks_stream.rs index ab19e5b5..35e33e97 100644 --- a/src/ceres/snapshot/chunks_stream.rs +++ b/src/ceres/snapshot/chunks_stream.rs @@ -63,6 +63,7 @@ async fn build_stream( input, lease, size <= STAGED_CAP_BYTES as u64, + None, ) .await } @@ -72,6 +73,7 @@ pub(super) async fn build_source_stream( size: u64, open: F, budget: &Arc, + admission: &crate::jupiter::storage::native_chunk_map::retention::ChunkMapInstall, ) -> Result where F: FnOnce() -> Fut, @@ -84,6 +86,7 @@ where open, budget, BUILDERS.get_or_init(|| Arc::new(Semaphore::new(MAX_BUILDERS))), + Some(admission), ) .await } @@ -94,6 +97,7 @@ async fn build_source_with_resources( open: F, budget: &Arc, builders: &Arc, + admission: Option<&crate::jupiter::storage::native_chunk_map::retention::ChunkMapInstall>, ) -> Result where F: FnOnce() -> Fut, @@ -106,8 +110,15 @@ where ) })?; let lease = budget.reserve(source_reservation_bytes(size)?)?; - let input = open().await?; - build_stream_with_inline(content_id, size, input, lease, false).await + let input = if let Some(admission) = admission { + admission.ensure_live().await?; + let result = admission.await_backend(open()).await?; + admission.ensure_live().await?; + result? + } else { + open().await? + }; + build_stream_with_inline(content_id, size, input, lease, false, admission).await } pub(super) fn source_reservation_bytes(size: u64) -> Result { @@ -135,6 +146,7 @@ async fn build_stream_with_inline( mut input: ObjectByteStream, lease: MemoryLease, inline: bool, + admission: Option<&crate::jupiter::storage::native_chunk_map::retention::ChunkMapInstall>, ) -> Result { let chunk_count = size.div_ceil(CHUNK_SIZE as u64); let page_count = chunk_count.div_ceil(CHUNKS_PER_PAGE as u64); @@ -161,7 +173,17 @@ async fn build_stream_with_inline( let mut chunk_bytes = 0usize; let mut digests = 0u64; let mut since_yield = 0usize; - while let Some(part) = input.next().await { + let mut empty_parts = 0usize; + loop { + let next = if let Some(admission) = admission { + admission.ensure_live().await?; + let next = admission.await_backend(input.next()).await?; + admission.ensure_live().await?; + next + } else { + input.next().await + }; + let Some(part) = next else { break }; let bytes = part.map_err(|error| { tracing::warn!(error = %error, "fixed-view chunk projection stream failed"); SnapshotError::new( @@ -185,6 +207,18 @@ async fn build_stream_with_inline( raw.extend_from_slice(&bytes); } received += bytes.len() as u64; + if let Some(admission) = admission { + if !bytes.is_empty() { + admission.record_progress(received).await?; + } else { + empty_parts += 1; + if empty_parts == 32 { + tokio::task::yield_now().await; + admission.ensure_live().await?; + empty_parts = 0; + } + } + } let mut rest = bytes.as_ref(); while !rest.is_empty() { let len = rest.len().min(CHUNK_SIZE as usize - chunk_bytes); diff --git a/src/ceres/snapshot/chunks_stream_tests.rs b/src/ceres/snapshot/chunks_stream_tests.rs index 61de6453..d6f7e042 100644 --- a/src/ceres/snapshot/chunks_stream_tests.rs +++ b/src/ceres/snapshot/chunks_stream_tests.rs @@ -34,6 +34,7 @@ async fn exact_source_builder_admits_all_install_workspace_before_open_and_cance }, &budget, &builders, + None, ) .await .err() @@ -52,7 +53,8 @@ async fn exact_source_builder_admits_all_install_workspace_before_open_and_cance Ok(stream(vec![Ok(raw.clone())])) }, &budget, - &builders + &builders, + None ) .await .err() @@ -88,6 +90,7 @@ async fn exact_source_builder_admits_all_install_workspace_before_open_and_cance }, &task_budget, &task_builders, + None, ) .await }); @@ -110,6 +113,7 @@ async fn exact_source_builder_admits_all_install_workspace_before_open_and_cance }, &budget, &builders, + None, ) .await .unwrap(); diff --git a/src/ceres/snapshot/content_budget.rs b/src/ceres/snapshot/content_budget.rs index 6bb8d32b..18e16a51 100644 --- a/src/ceres/snapshot/content_budget.rs +++ b/src/ceres/snapshot/content_budget.rs @@ -89,10 +89,6 @@ pub(crate) fn response_budget() -> &'static Arc { BUDGET.get_or_init(|| MemoryBudget::new(RESPONSE_LIVE_BYTES)) } -pub(crate) fn reserve_range_work() -> Result { - range_budget().reserve(RANGE_WORK_BYTES) -} - pub(crate) fn range_budget() -> &'static Arc { static BUDGET: OnceLock> = OnceLock::new(); BUDGET.get_or_init(|| MemoryBudget::new(RANGE_SCRATCH_BYTES)) diff --git a/src/jupiter/migration/m20261008_000300_add_mst2_chunk_map_retention.rs b/src/jupiter/migration/m20261008_000300_add_mst2_chunk_map_retention.rs new file mode 100644 index 00000000..81e58229 --- /dev/null +++ b/src/jupiter/migration/m20261008_000300_add_mst2_chunk_map_retention.rs @@ -0,0 +1,26 @@ +use sea_orm::{ConnectionTrait, DbBackend}; +use sea_orm_migration::prelude::*; + +#[derive(DeriveMigrationName)] +pub struct Migration; + +#[async_trait::async_trait] +impl MigrationTrait for Migration { + async fn up(&self, manager: &SchemaManager) -> Result<(), DbErr> { + if manager.get_database_backend() != DbBackend::Postgres { + return Err(DbErr::Custom( + "chunk-map retention requires primary PostgreSQL".into(), + )); + } + manager + .get_connection() + .execute_unprepared(include_str!("m20261008_000300_chunk_map_retention.sql")) + .await?; + Ok(()) + } + + async fn down(&self, _manager: &SchemaManager) -> Result<(), DbErr> { + // Never discard generation fences or late-create reconciliation history. + Ok(()) + } +} diff --git a/src/jupiter/migration/m20261008_000300_chunk_map_retention.sql b/src/jupiter/migration/m20261008_000300_chunk_map_retention.sql new file mode 100644 index 00000000..46a75aa3 --- /dev/null +++ b/src/jupiter/migration/m20261008_000300_chunk_map_retention.sql @@ -0,0 +1,262 @@ +-- Physical receipt generations are never reused. APPLIED history remains +-- bounded and is rechecked: a timed-out backend create can complete late. +CREATE SEQUENCE mst2_chunk_generation_sequence START 2; +ALTER TABLE mst2_chunk_map_source ADD COLUMN receipt_generation bigint NOT NULL DEFAULT 1 CHECK(receipt_generation>0); +ALTER TABLE mst2_chunk_map_source ADD COLUMN receipt_key text GENERATED ALWAYS AS + (CASE WHEN receipt_generation=1 THEN pg_catalog.encode(source_id,'hex') + ELSE pg_catalog.encode(source_id,'hex')||'-'||pg_catalog.lpad(pg_catalog.to_hex(receipt_generation),16,'0') END) STORED; +CREATE UNIQUE INDEX mst2_chunk_map_source_receipt_key ON mst2_chunk_map_source(receipt_key); + +CREATE TABLE mst2_chunk_receipt_generation ( + receipt_key text PRIMARY KEY CHECK(receipt_key ~ '^[0-9a-f]{64}(-[0-9a-f]{16})?$'), + source_id bytea NOT NULL CHECK(pg_catalog.octet_length(source_id)=32), + generation bigint NOT NULL CHECK(generation>0), + source_bytes bytea NOT NULL CHECK(pg_catalog.octet_length(source_bytes) BETWEEN 1 AND 2048), + primary_scope bytea NOT NULL CHECK(pg_catalog.octet_length(primary_scope) BETWEEN 1 AND 1024), + map_id bytea CHECK(map_id IS NULL OR pg_catalog.octet_length(map_id)=32), + map_generation bigint CHECK(map_generation IS NULL OR map_generation>0), + receipt_bytes bytea CHECK(receipt_bytes IS NULL OR pg_catalog.octet_length(receipt_bytes) BETWEEN 129 AND 4096), + create_completed boolean NOT NULL DEFAULT false, + create_completed_at timestamptz, + owner uuid, + state text NOT NULL CHECK(state IN ('RESERVED','CREATING','LIVE','DELETING','APPLIED')), + reserved_pages integer NOT NULL CHECK(reserved_pages BETWEEN 0 AND 32768), + reserved_nodes integer NOT NULL CHECK(reserved_nodes BETWEEN 0 AND 65535), + reserved_bytes bigint NOT NULL CHECK(reserved_bytes BETWEEN 0 AND 536870912), + deadline timestamptz, + created_at timestamptz NOT NULL DEFAULT pg_catalog.clock_timestamp(), + last_progress timestamptz NOT NULL DEFAULT pg_catalog.clock_timestamp(), + checked_at timestamptz NOT NULL DEFAULT '-infinity', + UNIQUE(source_id,generation), + CHECK(receipt_key=CASE WHEN generation=1 THEN pg_catalog.encode(source_id,'hex') ELSE pg_catalog.encode(source_id,'hex')||'-'||pg_catalog.lpad(pg_catalog.to_hex(generation),16,'0') END), + CHECK(state NOT IN ('CREATING','LIVE') OR (map_id IS NOT NULL AND receipt_bytes IS NOT NULL)), + CHECK(state NOT IN ('RESERVED','CREATING') OR (owner IS NOT NULL AND deadline IS NOT NULL)) +); +CREATE INDEX mst2_chunk_receipt_generation_maintenance ON mst2_chunk_receipt_generation(state,checked_at,receipt_key); +CREATE INDEX mst2_chunk_receipt_generation_created ON mst2_chunk_receipt_generation(created_at) WHERE receipt_bytes IS NOT NULL; +CREATE INDEX mst2_chunk_receipt_generation_completed ON mst2_chunk_receipt_generation(create_completed_at) WHERE receipt_bytes IS NOT NULL; +CREATE INDEX mst2_chunk_receipt_generation_install ON mst2_chunk_receipt_generation(map_id,map_generation,deadline) WHERE state='CREATING'; +CREATE TABLE mst2_chunk_map_lifetime ( + map_id bytea PRIMARY KEY REFERENCES mst2_chunk_map(map_id) DEFERRABLE INITIALLY DEFERRED, + generation bigint NOT NULL CHECK(generation>0), + state text NOT NULL CHECK(state IN ('LIVE','DELETING')), + last_used timestamptz NOT NULL DEFAULT pg_catalog.clock_timestamp(), + UNIQUE(map_id,generation) +); +CREATE TABLE mst2_chunk_reader ( + owner uuid PRIMARY KEY, + receipt_key text NOT NULL REFERENCES mst2_chunk_receipt_generation(receipt_key), + source_generation bigint NOT NULL CHECK(source_generation>0), + map_id bytea NOT NULL CHECK(pg_catalog.octet_length(map_id)=32), + map_generation bigint NOT NULL CHECK(map_generation>0), + deadline timestamptz NOT NULL, + last_progress timestamptz NOT NULL DEFAULT pg_catalog.clock_timestamp() +); +CREATE INDEX mst2_chunk_reader_receipt ON mst2_chunk_reader(receipt_key,source_generation,deadline); +CREATE INDEX mst2_chunk_reader_map ON mst2_chunk_reader(map_id,map_generation,deadline); +CREATE TABLE mst2_chunk_map_gc ( + map_id bytea NOT NULL CHECK(pg_catalog.octet_length(map_id)=32), + generation bigint NOT NULL CHECK(generation>0), + state text NOT NULL CHECK(state IN ('PENDING','APPLIED')), + created_at timestamptz NOT NULL DEFAULT pg_catalog.clock_timestamp(), + applied_at timestamptz, + PRIMARY KEY(map_id,generation) +); + +-- Bootstrap existing #77 data from actual bounded rows. SQL does not infer +-- backing receipt presence; repository admission separately inventories it. +DO $$ DECLARE maps bigint; sources bigint; leaves bigint; nodes bigint; bytes bigint; BEGIN + SELECT pg_catalog.count(*) INTO maps FROM mst2_chunk_map; + SELECT pg_catalog.count(*) INTO sources FROM mst2_chunk_map_source; + SELECT pg_catalog.count(*) INTO leaves FROM mst2_chunk_map_leaf; + SELECT pg_catalog.count(*) INTO nodes FROM mst2_chunk_map_node; + SELECT COALESCE(pg_catalog.sum(pg_catalog.pg_total_relation_size(c.oid)),0) INTO bytes + FROM pg_catalog.pg_class c JOIN pg_catalog.pg_namespace n ON n.oid=c.relnamespace + WHERE n.nspname=pg_catalog.current_schema() AND c.relkind='r' + AND c.relname IN ('mst2_chunk_map','mst2_chunk_map_source','mst2_chunk_map_leaf','mst2_chunk_map_node'); + IF maps>4096 OR sources>16384 OR leaves>65536 OR nodes>131072 OR bytes>536870912 THEN + RAISE EXCEPTION 'existing chunk-map indexes exceed bounded retention bootstrap'; + END IF; +END $$; + +INSERT INTO mst2_chunk_map_lifetime(map_id,generation,state) SELECT map_id,1,'LIVE' FROM mst2_chunk_map; +INSERT INTO mst2_chunk_receipt_generation(receipt_key,source_id,generation,source_bytes,primary_scope,map_id,map_generation,receipt_bytes,create_completed,create_completed_at,state,reserved_pages,reserved_nodes,reserved_bytes) + SELECT s.receipt_key,s.source_id,1,s.source_bytes,s.primary_scope,s.map_id,1, + pg_catalog.convert_to('MST2-CHUNK-MAP-RECEIPT','UTF8')||pg_catalog.decode('00','hex')|| + pg_catalog.int4send(pg_catalog.octet_length(s.primary_scope))||s.primary_scope|| + pg_catalog.int4send(pg_catalog.octet_length(s.source_bytes))||s.source_bytes||m.descriptor, + true,pg_catalog.clock_timestamp(),'LIVE',0,0,0 FROM mst2_chunk_map_source s JOIN mst2_chunk_map m ON m.map_id=s.map_id; + +CREATE FUNCTION mst2_chunk_retention_barrier() RETURNS void LANGUAGE plpgsql AS $$ +BEGIN + IF pg_catalog.pg_is_in_recovery() OR pg_catalog.current_setting('transaction_isolation')<>'read committed' THEN + RAISE EXCEPTION 'chunk retention requires primary READ COMMITTED'; + END IF; + PERFORM pg_catalog.pg_advisory_xact_lock(1296717363,pg_catalog.hashtext(pg_catalog.current_schema())); +END $$; + +CREATE FUNCTION mst2_chunk_retention_bytes() RETURNS bigint LANGUAGE sql AS $$ + SELECT COALESCE(pg_catalog.sum(pg_catalog.pg_total_relation_size(c.oid)),0)::bigint + FROM pg_catalog.pg_class c JOIN pg_catalog.pg_namespace n ON n.oid=c.relnamespace + WHERE n.nspname=pg_catalog.current_schema() AND c.relkind='r' + AND c.relname IN ('mst2_chunk_map','mst2_chunk_map_source','mst2_chunk_map_leaf','mst2_chunk_map_node', + 'mst2_chunk_receipt_generation','mst2_chunk_map_lifetime','mst2_chunk_reader','mst2_chunk_map_gc') +$$; +DO $$ BEGIN + IF mst2_chunk_retention_bytes()>536870912 THEN RAISE EXCEPTION 'bounded chunk retention bootstrap exceeds actual durable byte quota'; END IF; +END $$; + +CREATE OR REPLACE FUNCTION mst2_chunk_map_primary() RETURNS trigger LANGUAGE plpgsql AS $$ +BEGIN + IF pg_catalog.current_schema() IS DISTINCT FROM TG_TABLE_SCHEMA THEN + RAISE EXCEPTION 'chunk resident insertion requires its actual primary schema'; + END IF; + PERFORM mst2_chunk_retention_barrier(); + IF mst2_chunk_retention_bytes()+1048576>536870912 THEN + RAISE EXCEPTION 'chunk resident insertion lacks actual durable byte headroom'; + END IF; + RETURN NULL; +END $$; + +-- Preserve the original immutable/source-completeness oracles. Only exact +-- terminal generations with no live owners may lose their resident indexes. +CREATE OR REPLACE FUNCTION mst2_chunk_map_immutable() RETURNS trigger LANGUAGE plpgsql AS $$ +DECLARE permitted boolean:=false; +BEGIN + IF TG_OP<>'DELETE' OR pg_catalog.current_schema() IS DISTINCT FROM TG_TABLE_SCHEMA THEN + RAISE EXCEPTION 'chunk map indexes and source receipts are immutable'; + END IF; + PERFORM mst2_chunk_retention_barrier(); + IF TG_TABLE_NAME='mst2_chunk_map_source' THEN + EXECUTE pg_catalog.format('SELECT EXISTS(SELECT 1 FROM %I.mst2_chunk_receipt_generation g WHERE g.receipt_key=$1 AND g.source_id=$2 AND g.generation=$3 AND g.map_id=$4 AND g.state=''APPLIED'') AND NOT EXISTS(SELECT 1 FROM %I.mst2_chunk_reader r WHERE r.receipt_key=$1 AND r.source_generation=$3 AND r.deadline>pg_catalog.clock_timestamp())',TG_TABLE_SCHEMA,TG_TABLE_SCHEMA) + INTO permitted USING OLD.receipt_key,OLD.source_id,OLD.receipt_generation,OLD.map_id; + ELSE + EXECUTE pg_catalog.format('SELECT EXISTS(SELECT 1 FROM %I.mst2_chunk_map_lifetime l JOIN %I.mst2_chunk_map_gc g ON g.map_id=l.map_id AND g.generation=l.generation WHERE l.map_id=$1 AND l.state=''DELETING'' AND g.state=''PENDING'') AND NOT EXISTS(SELECT 1 FROM %I.mst2_chunk_map_source s WHERE s.map_id=$1) AND NOT EXISTS(SELECT 1 FROM %I.mst2_chunk_reader r JOIN %I.mst2_chunk_map_lifetime l ON l.map_id=r.map_id AND l.generation=r.map_generation WHERE r.map_id=$1 AND r.deadline>pg_catalog.clock_timestamp())',TG_TABLE_SCHEMA,TG_TABLE_SCHEMA,TG_TABLE_SCHEMA,TG_TABLE_SCHEMA,TG_TABLE_SCHEMA) + INTO permitted USING OLD.map_id; + END IF; + IF NOT permitted THEN RAISE EXCEPTION 'chunk map deletion lacks an exact retired generation'; END IF; + RETURN OLD; +END $$; + +CREATE FUNCTION mst2_chunk_generation_guard() RETURNS trigger LANGUAGE plpgsql AS $$ +DECLARE owned boolean; +BEGIN + IF pg_catalog.current_schema() IS DISTINCT FROM TG_TABLE_SCHEMA THEN + RAISE EXCEPTION 'chunk retention mutation requires its actual primary schema'; + END IF; + PERFORM mst2_chunk_retention_barrier(); + IF TG_OP='INSERT' THEN + EXECUTE pg_catalog.format('SELECT pg_catalog.count(*)>=131072 FROM %I.mst2_chunk_receipt_generation',TG_TABLE_SCHEMA) INTO owned; + IF owned THEN RAISE EXCEPTION 'chunk receipt history quota exhausted'; END IF; + END IF; + IF TG_OP='DELETE' OR TG_OP='TRUNCATE' THEN + RAISE EXCEPTION 'chunk receipt generations and collection history are retained'; + END IF; + IF mst2_chunk_retention_bytes()+CASE WHEN NEW.state IN ('DELETING','APPLIED') THEN 262144 ELSE 1048576 END>536870912 THEN + RAISE EXCEPTION 'chunk receipt mutation lacks actual durable byte headroom'; + END IF; + IF TG_OP='UPDATE' AND (NEW.receipt_key IS DISTINCT FROM OLD.receipt_key OR NEW.source_id IS DISTINCT FROM OLD.source_id + OR NEW.generation IS DISTINCT FROM OLD.generation OR NEW.source_bytes IS DISTINCT FROM OLD.source_bytes + OR NEW.primary_scope IS DISTINCT FROM OLD.primary_scope OR NEW.created_at IS DISTINCT FROM OLD.created_at + OR (OLD.receipt_bytes IS NOT NULL AND NEW.receipt_bytes IS DISTINCT FROM OLD.receipt_bytes) + OR (OLD.map_id IS NOT NULL AND NEW.map_id IS DISTINCT FROM OLD.map_id) + OR (OLD.map_generation IS NOT NULL AND NEW.map_generation IS DISTINCT FROM OLD.map_generation) + OR (OLD.create_completed AND NOT NEW.create_completed) + OR (OLD.create_completed_at IS NOT NULL AND NEW.create_completed_at IS DISTINCT FROM OLD.create_completed_at) + OR (OLD.state='APPLIED' AND NEW.state<>'APPLIED') + OR (OLD.state='DELETING' AND NEW.state NOT IN ('DELETING','APPLIED')) + OR (OLD.state='LIVE' AND NEW.state NOT IN ('LIVE','DELETING')) + OR (OLD.state='CREATING' AND NEW.state NOT IN ('CREATING','LIVE','DELETING')) + OR (OLD.state='RESERVED' AND NEW.state NOT IN ('RESERVED','CREATING','DELETING','APPLIED'))) THEN + RAISE EXCEPTION 'chunk receipt generation identity or transition changed'; + END IF; + IF NEW.state='DELETING' AND (TG_OP='INSERT' OR OLD.state<>'DELETING') THEN + EXECUTE pg_catalog.format('SELECT EXISTS(SELECT 1 FROM %I.mst2_chunk_reader WHERE receipt_key=$1 AND source_generation=$2 AND deadline>pg_catalog.clock_timestamp())',TG_TABLE_SCHEMA) + INTO owned USING NEW.receipt_key,NEW.generation; + IF owned THEN RAISE EXCEPTION 'receipt retirement cannot cross a live reader'; END IF; + END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_chunk_generation_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_chunk_receipt_generation + FOR EACH ROW EXECUTE FUNCTION mst2_chunk_generation_guard(); +CREATE TRIGGER mst2_chunk_generation_no_truncate BEFORE TRUNCATE ON mst2_chunk_receipt_generation + FOR EACH STATEMENT EXECUTE FUNCTION mst2_chunk_generation_guard(); + +CREATE FUNCTION mst2_chunk_gc_history_guard() RETURNS trigger LANGUAGE plpgsql AS $$ +DECLARE full_history boolean; +BEGIN + IF pg_catalog.current_schema() IS DISTINCT FROM TG_TABLE_SCHEMA THEN RAISE EXCEPTION 'wrong chunk history schema'; END IF; + PERFORM mst2_chunk_retention_barrier(); + IF TG_OP='INSERT' THEN + EXECUTE pg_catalog.format('SELECT pg_catalog.count(*)>=131072 FROM %I.mst2_chunk_map_gc',TG_TABLE_SCHEMA) INTO full_history; + IF full_history THEN RAISE EXCEPTION 'chunk collection history quota exhausted'; END IF; + END IF; + IF TG_OP='DELETE' OR TG_OP='TRUNCATE' THEN RAISE EXCEPTION 'chunk collection history is retained'; END IF; + IF mst2_chunk_retention_bytes()+262144>536870912 THEN RAISE EXCEPTION 'chunk collection history lacks actual durable byte headroom'; END IF; + IF TG_OP='UPDATE' AND (NEW.map_id IS DISTINCT FROM OLD.map_id OR NEW.generation IS DISTINCT FROM OLD.generation + OR NEW.created_at IS DISTINCT FROM OLD.created_at OR OLD.state='APPLIED' OR NEW.state<>'APPLIED') THEN + RAISE EXCEPTION 'chunk collection history is immutable except terminal completion'; + END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_chunk_gc_history_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_chunk_map_gc + FOR EACH ROW EXECUTE FUNCTION mst2_chunk_gc_history_guard(); +CREATE TRIGGER mst2_chunk_gc_no_truncate BEFORE TRUNCATE ON mst2_chunk_map_gc + FOR EACH STATEMENT EXECUTE FUNCTION mst2_chunk_gc_history_guard(); + +CREATE FUNCTION mst2_chunk_reader_guard() RETURNS trigger LANGUAGE plpgsql AS $$ +DECLARE live boolean; +BEGIN + IF pg_catalog.current_schema() IS DISTINCT FROM TG_TABLE_SCHEMA THEN RAISE EXCEPTION 'wrong chunk reader schema'; END IF; + PERFORM mst2_chunk_retention_barrier(); + IF TG_OP='TRUNCATE' THEN RAISE EXCEPTION 'chunk readers cannot be truncated'; END IF; + IF TG_OP='DELETE' THEN RETURN OLD; END IF; + IF mst2_chunk_retention_bytes()+1048576>536870912 THEN RAISE EXCEPTION 'chunk reader mutation lacks actual durable byte headroom'; END IF; + IF TG_OP='INSERT' THEN + EXECUTE pg_catalog.format('SELECT pg_catalog.count(*)>=4096 FROM %I.mst2_chunk_reader',TG_TABLE_SCHEMA) INTO live; + IF live THEN RAISE EXCEPTION 'chunk reader quota exhausted'; END IF; + END IF; + IF NEW.deadline>pg_catalog.clock_timestamp()+interval '60 seconds' OR NEW.deadline<=pg_catalog.clock_timestamp() + OR (TG_OP='UPDATE' AND (OLD.deadline<=pg_catalog.clock_timestamp() OR NEW.owner IS DISTINCT FROM OLD.owner + OR NEW.receipt_key IS DISTINCT FROM OLD.receipt_key OR NEW.source_generation IS DISTINCT FROM OLD.source_generation + OR NEW.map_id IS DISTINCT FROM OLD.map_id OR NEW.map_generation IS DISTINCT FROM OLD.map_generation)) THEN + RAISE EXCEPTION 'chunk reader cannot renew an expired or different owner'; + END IF; + EXECUTE pg_catalog.format('SELECT EXISTS(SELECT 1 FROM %I.mst2_chunk_receipt_generation g JOIN %I.mst2_chunk_map_lifetime l ON l.map_id=g.map_id AND l.generation=g.map_generation JOIN %I.mst2_chunk_map_source s ON s.receipt_key=g.receipt_key AND s.receipt_generation=g.generation WHERE g.receipt_key=$1 AND g.generation=$2 AND g.map_id=$3 AND g.map_generation=$4 AND g.state=''LIVE'' AND l.state=''LIVE'')',TG_TABLE_SCHEMA,TG_TABLE_SCHEMA,TG_TABLE_SCHEMA) + INTO live USING NEW.receipt_key,NEW.source_generation,NEW.map_id,NEW.map_generation; + IF NOT live THEN RAISE EXCEPTION 'chunk reader lacks exact live source and map generations'; END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_chunk_reader_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_chunk_reader + FOR EACH ROW EXECUTE FUNCTION mst2_chunk_reader_guard(); +CREATE TRIGGER mst2_chunk_reader_no_truncate BEFORE TRUNCATE ON mst2_chunk_reader + FOR EACH STATEMENT EXECUTE FUNCTION mst2_chunk_reader_guard(); + +CREATE FUNCTION mst2_chunk_map_lifetime_guard() RETURNS trigger LANGUAGE plpgsql AS $$ +DECLARE permitted boolean; +BEGIN + IF pg_catalog.current_schema() IS DISTINCT FROM TG_TABLE_SCHEMA THEN RAISE EXCEPTION 'wrong chunk map lifetime schema'; END IF; + PERFORM mst2_chunk_retention_barrier(); + IF TG_OP='TRUNCATE' THEN RAISE EXCEPTION 'chunk map lifetimes cannot be truncated'; END IF; + IF TG_OP='DELETE' THEN + EXECUTE pg_catalog.format('SELECT $2=''DELETING'' AND EXISTS(SELECT 1 FROM %I.mst2_chunk_map_gc g WHERE g.map_id=$1 AND g.generation=$3 AND g.state=''PENDING'') AND NOT EXISTS(SELECT 1 FROM %I.mst2_chunk_map_source s WHERE s.map_id=$1) AND NOT EXISTS(SELECT 1 FROM %I.mst2_chunk_reader r WHERE r.map_id=$1 AND r.map_generation=$3 AND r.deadline>pg_catalog.clock_timestamp())',TG_TABLE_SCHEMA,TG_TABLE_SCHEMA,TG_TABLE_SCHEMA) + INTO permitted USING OLD.map_id,OLD.state,OLD.generation; + IF NOT permitted THEN RAISE EXCEPTION 'map lifetime deletion lacks exact reader-free collection claim'; END IF; + RETURN OLD; + END IF; + IF mst2_chunk_retention_bytes()+CASE WHEN NEW.state='DELETING' THEN 262144 ELSE 1048576 END>536870912 THEN + RAISE EXCEPTION 'chunk map lifetime mutation lacks actual durable byte headroom'; + END IF; + IF TG_OP='UPDATE' AND (NEW.map_id IS DISTINCT FROM OLD.map_id OR NEW.generation IS DISTINCT FROM OLD.generation + OR (OLD.state='DELETING' AND NEW.state<>'DELETING')) THEN RAISE EXCEPTION 'map lifetime cannot reincarnate in place'; END IF; + IF TG_OP='UPDATE' AND NEW.state='DELETING' AND OLD.state='LIVE' THEN + EXECUTE pg_catalog.format('SELECT EXISTS(SELECT 1 FROM %I.mst2_chunk_map_gc g WHERE g.map_id=$1 AND g.generation=$2 AND g.state=''PENDING'') AND NOT EXISTS(SELECT 1 FROM %I.mst2_chunk_map_source WHERE map_id=$1) AND NOT EXISTS(SELECT 1 FROM %I.mst2_chunk_reader WHERE map_id=$1 AND map_generation=$2 AND deadline>pg_catalog.clock_timestamp()) AND NOT EXISTS(SELECT 1 FROM %I.mst2_chunk_receipt_generation WHERE map_id=$1 AND map_generation=$2 AND state=''CREATING'' AND deadline>pg_catalog.clock_timestamp())',TG_TABLE_SCHEMA,TG_TABLE_SCHEMA,TG_TABLE_SCHEMA,TG_TABLE_SCHEMA) + INTO permitted USING OLD.map_id,OLD.generation; + IF NOT permitted THEN RAISE EXCEPTION 'map collection cannot cross a reader or active partial installation'; END IF; + END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_chunk_map_lifetime_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_chunk_map_lifetime + FOR EACH ROW EXECUTE FUNCTION mst2_chunk_map_lifetime_guard(); +CREATE TRIGGER mst2_chunk_map_lifetime_no_truncate BEFORE TRUNCATE ON mst2_chunk_map_lifetime + FOR EACH STATEMENT EXECUTE FUNCTION mst2_chunk_map_lifetime_guard(); diff --git a/src/jupiter/migration/mod.rs b/src/jupiter/migration/mod.rs index 2fbc20ce..018716bc 100644 --- a/src/jupiter/migration/mod.rs +++ b/src/jupiter/migration/mod.rs @@ -153,6 +153,7 @@ mod m20261007_000500_add_mst2_install_capability; mod m20261007_000600_add_mst2_storage_routes; mod m20261008_000100_add_mst2_chunk_maps; mod m20261008_000200_add_mst2_rooted_qualified_family; +mod m20261008_000300_add_mst2_chunk_map_retention; mod runner; pub use m20260905_000100_add_push_queue::ensure_queue_control_seed; pub use runner::apply_migrations; @@ -294,6 +295,7 @@ impl MigratorTrait for Migrator { Box::new(m20261007_000600_add_mst2_storage_routes::Migration), Box::new(m20261008_000100_add_mst2_chunk_maps::Migration), Box::new(m20261008_000200_add_mst2_rooted_qualified_family::Migration), + Box::new(m20261008_000300_add_mst2_chunk_map_retention::Migration), ] } } @@ -1204,7 +1206,7 @@ mod tests { async fn import_repo_alias_rows_canonicalized() { let names = migration_names(); assert_eq!( - &names[names.len() - 16..names.len() - 7], + &names[names.len() - 17..names.len() - 8], &[ "m20260923_000200_canonicalize_import_repo_paths".to_string(), "m20260925_000100_media_paging".to_string(), @@ -1219,34 +1221,38 @@ mod tests { "native retention, metadata installation and view tables follow media paging" ); assert_eq!( - &names[names.len() - 7], + &names[names.len() - 8], "m20261007_000200_add_mst2_metadata_generations" ); assert_eq!( - &names[names.len() - 6], + &names[names.len() - 7], "m20261007_000300_add_mst2_metadata_lifetime_history" ); assert_eq!( - &names[names.len() - 5], + &names[names.len() - 6], "m20261007_000400_add_mst2_qualified_metadata_gc" ); assert_eq!( - &names[names.len() - 4], + &names[names.len() - 5], "m20261007_000500_add_mst2_install_capability" ); assert_eq!( - &names[names.len() - 3], + &names[names.len() - 4], "m20261007_000600_add_mst2_storage_routes" ); assert_eq!( - &names[names.len() - 2], + &names[names.len() - 3], "m20261008_000100_add_mst2_chunk_maps" ); assert_eq!( - names.last().unwrap(), + &names[names.len() - 2], "m20261008_000200_add_mst2_rooted_qualified_family" ); + assert_eq!( + names.last().unwrap(), + "m20261008_000300_add_mst2_chunk_map_retention" + ); let db = alias_db().await; insert_repo(&db, 1, "/third-party//a").await; insert_repo(&db, 2, "/third-party/b/").await; diff --git a/src/jupiter/storage/native_chunk_map.rs b/src/jupiter/storage/native_chunk_map.rs index 5ce849a8..1d5e5666 100644 --- a/src/jupiter/storage/native_chunk_map.rs +++ b/src/jupiter/storage/native_chunk_map.rs @@ -2,8 +2,8 @@ //! //! The object writer is trusted like the verified-object writer. A database //! row, its checksum or a map's presence is never a full-source proof. Only -//! the opaque full-stream verifier can install a source receipt. This initial -//! store is append-only; bounded retention and collection remain separate work. +//! the opaque full-stream verifier can install a source receipt. Persistent +//! owners, reservations and physical generations bound its retention. use std::sync::Arc; @@ -33,11 +33,17 @@ const DESCRIPTOR_CREDIT: usize = 32 * 1024; const PAGE_CREDIT: usize = 64 * 1024; const RECEIPT_DOMAIN: &[u8] = b"MST2-CHUNK-MAP-RECEIPT\0"; +pub(crate) mod retention; + +#[derive(Clone)] pub(crate) struct PostgresChunkMapRepository { connection: DatabaseConnection, primary_scope: Vec, budget: Arc, schema: String, + receipt_observation: Arc>>>, + cancelled_installs: Arc>>, + admission_maintenance: Arc>>, } pub(crate) struct PersistedChunkMap { @@ -45,6 +51,7 @@ pub(crate) struct PersistedChunkMap { pub map_id: [u8; 32], source: ChunkMapSource, source_id: [u8; 32], + reader: retention::ChunkMapReader, _memory: MemoryLease, } @@ -52,6 +59,26 @@ impl PersistedChunkMap { pub(crate) fn source_id(&self) -> [u8; 32] { self.source_id } + + pub(crate) async fn ensure_live(&self) -> Result<(), SnapshotError> { + self.reader.ensure_live().await + } + + pub(crate) async fn record_progress(&self) -> Result<(), SnapshotError> { + self.reader.record_progress().await + } + + pub(crate) async fn await_backend( + &self, + future: F, + ) -> Result { + self.reader.await_backend(future).await + } + + #[cfg(test)] + pub(crate) async fn test_check_next_owner_operation(&self) { + self.reader.test_check_next_owner_operation().await; + } } pub(crate) struct AuthenticatedChunkPage { @@ -100,6 +127,9 @@ impl PostgresChunkMapRepository { primary_scope, budget: projection_budget().clone(), schema, + receipt_observation: Arc::default(), + cancelled_installs: Arc::default(), + admission_maintenance: Arc::default(), }) } @@ -108,27 +138,78 @@ impl PostgresChunkMapRepository { source: &ChunkMapSource, objects: &MegaObjectStorageWrapper, ) -> Result>, SnapshotError> { + for pass in 0..2 { + let (map, pending) = self.read_current(source, objects).await?; + if !pending { + return Ok(map); + } + if pass == 0 { + self.maintain(objects, 8).await?; + } + } + Err(SnapshotError::new( + SnapshotErrorCode::TemporaryUnavailable, + "previous source receipt deletion is still pending", + )) + } + + async fn read_current( + &self, + source: &ChunkMapSource, + objects: &MegaObjectStorageWrapper, + ) -> Result<(Option>, bool), SnapshotError> { let memory = self.budget.reserve(DESCRIPTOR_CREDIT)?; let txn = self.transaction().await?; let result = async { self.require_scope(&txn).await?; require_current_source(&txn, source, &self.schema).await?; let Some(row) = source_row(&txn, source, &self.schema).await? else { - return Ok(None); + let pending = self.require_missing_source_retired(&txn, source).await?; + return Ok((None, pending)); }; let (map, source_id) = self.validate_source_row(source, &row)?; + let generation_state: Option = + row.try_get("", "generation_state").map_err(db_error)?; + if generation_state.as_deref() == Some("DELETING") { + // A legitimate retirement can still retain its source row + // until the physical delete completes. Replay it just like + // a missing retired row, after checking every source fact; + // damaged LIVE rows never enter this cold-rebuild path. + return Ok((None, true)); + } let bytes = receipt(&self.primary_scope, source, &map)?; - read_receipt(objects, &receipt_key(source_id), &bytes).await?; - Ok(Some(Arc::new(PersistedChunkMap { - map_id: map.map_id(), - map, - source: source.clone(), - source_id, - _memory: memory, - }))) + let reader = self.admit_reader(&txn, source, &row, &map).await?; + let key = ObjectKey { + namespace: ObjectNamespace::ChunkMapReceipt, + key: row.try_get("", "receipt_key").map_err(db_error)?, + }; + Ok(( + Some(( + Arc::new(PersistedChunkMap { + map_id: map.map_id(), + map, + source: source.clone(), + source_id, + reader, + _memory: memory, + }), + key, + bytes, + )), + false, + )) } .await; - finish(txn, result).await + let (admitted, pending) = finish(txn, result).await?; + let Some((map, key, bytes)) = admitted else { + return Ok((None, pending)); + }; + // Admission is durable before I/O. No completion barrier or fact row + // lock is held while the independent backend receipt is consumed. + map.await_backend(read_receipt(objects, &key, &bytes)) + .await??; + map.reader.validate_after_receipt().await?; + Ok((Some(map), false)) } /// Publication accepts no caller-provided map or "verified" flag. @@ -136,75 +217,96 @@ impl PostgresChunkMapRepository { &self, verified: VerifiedSourceChunkMap, objects: &MegaObjectStorageWrapper, + admission: &retention::ChunkMapInstall, + ) -> Result<(), SnapshotError> { + let result = self.install_reserved(verified, objects, admission).await; + if result.is_err() { + if let Err(error) = admission.retire().await { + tracing::warn!( + ?error, + "failed chunk-map installation will retire by database deadline" + ); + } else if let Err(error) = self.collect_maps(64).await { + tracing::warn!( + ?error, + "retired partial chunk-map indexes await bounded collection replay" + ); + } + } + result + } + + async fn install_reserved( + &self, + verified: VerifiedSourceChunkMap, + objects: &MegaObjectStorageWrapper, + admission: &retention::ChunkMapInstall, ) -> Result<(), SnapshotError> { let source = verified.source(); let map = verified.map(); - // The verifier reserved index and bounded SQL parameter workspace - // before opening the body. It owns those credits through publication. let nodes = indexed_nodes(verified.leaf_hashes())?; - if nodes.last().map(|n| n.digest) != Some(map.pages_root) { + if nodes.last().map(|node| node.digest) != Some(map.pages_root) { return Err(integrity( "verified chunk map index disagrees with canonical root", )); } - let source_bytes = source.canonical_bytes()?; - let source_id = source_id(&self.primary_scope, &source_bytes); - let bytes = receipt(&self.primary_scope, source, map)?; - let key = receipt_key(source_id); + let bytes = admission.prepare(&verified).await?; + let key = admission.key(); + admission + .await_backend(objects.inner.put_metadata_atomic_create( + key, + Bytes::copy_from_slice(&bytes), + ObjectMeta { + size: bytes.len() as i64, + ..Default::default() + }, + )) + .await? + .map_err(storage_error)?; + admission.acknowledge_create().await?; + admission + .await_backend(read_receipt(objects, key, &bytes)) + .await??; + let observation = retention::inventory(self, objects).await?; + let map_generation = self + .stage_map_start(&verified, admission, &observation) + .await?; + for batch in verified.leaves().chunks(64) { + self.stage_map_batch(&verified, admission, map_generation, Some(batch), None) + .await?; + } + for batch in nodes.chunks(1024) { + self.stage_map_batch(&verified, admission, map_generation, None, Some(batch)) + .await?; + } + self.compare_staged_map(&verified, &nodes, admission, map_generation) + .await?; let txn = self.transaction().await?; - let result = async { + let result=async { self.require_scope(&txn).await?; - require_current_source(&txn, source, &self.schema).await?; - if let Some(row) = source_row(&txn, source, &self.schema).await? { - let (stored, _) = self.validate_source_row(source, &row)?; - if stored != *map { return Err(integrity("immutable source map conflicts with reverified content")); } - read_receipt(objects, &key, &bytes).await?; - compare_complete_map(&txn, &verified, &nodes, &self.schema).await?; - return Ok(()); - } - // The complete source pass happened before this atomic create. - // A lost DB commit leaves an orphan trusted receipt, never an - // admitted partial map; exact re-verification can replay it. - objects.inner.put_metadata_atomic_create(&key, Bytes::copy_from_slice(&bytes), ObjectMeta { - size: bytes.len() as i64, ..Default::default() - }).await.map_err(storage_error)?; - read_receipt(objects, &key, &bytes).await?; - txn.execute_raw(stmt(&self.schema, "INSERT INTO mst2_chunk_map(map_id,descriptor,page_count,pages_root) VALUES($1,$2,$3,$4) ON CONFLICT(map_id) DO NOTHING", - [map.map_id().to_vec().into(), map.encode().into(), (map.page_count as i32).into(), map.pages_root.to_vec().into()])).await.map_err(db_error)?; - // Existing indexes must be compared, never repaired or overwritten. - let present = txn.query_one_raw(stmt(&self.schema, "SELECT EXISTS(SELECT 1 FROM mst2_chunk_map_source WHERE map_id=$1) AS sealed", [map.map_id().to_vec().into()])) - .await.map_err(db_error)?.ok_or_else(|| integrity("chunk map seal observation is missing"))? - .try_get::("", "sealed").map_err(db_error)?; - if !present { - for batch in verified.leaves().chunks(64) { - let encoded = encode_leaves(batch)?; - txn.execute_raw(stmt(&self.schema, "INSERT INTO mst2_chunk_map_leaf(map_id,page_index,payload) SELECT $1,p.page_index,pg_catalog.decode(p.payload,'hex') FROM pg_catalog.jsonb_to_recordset($2::jsonb) AS p(page_index integer,payload text) ON CONFLICT(map_id,page_index) DO NOTHING", - [map.map_id().to_vec().into(), encoded.into()])).await.map_err(db_error)?; - } - for batch in nodes.chunks(1024) { - let encoded = encode_nodes(batch)?; - txn.execute_raw(stmt(&self.schema, "INSERT INTO mst2_chunk_map_node(map_id,first_page,page_count,digest) SELECT $1,p.first_page,p.page_count,pg_catalog.decode(p.digest,'hex') FROM pg_catalog.jsonb_to_recordset($2::jsonb) AS p(first_page integer,page_count integer,digest text) ON CONFLICT(map_id,first_page,page_count) DO NOTHING", - [map.map_id().to_vec().into(), encoded.into()])).await.map_err(db_error)?; - } - } - compare_complete_map(&txn, &verified, &nodes, &self.schema).await?; - let fact = source.fact(); - txn.execute_raw(stmt(&self.schema, "INSERT INTO mst2_chunk_map_source(storage_domain,git_oid,object_kind,fact_id,source_id,source_bytes,primary_scope,map_id,receipt_digest) VALUES($1,$2,$3,$4,$5,$6,$7,$8,$9) ON CONFLICT(storage_domain,git_oid,object_kind) DO NOTHING", - [fact.storage_domain.clone().into(), fact.git_oid.clone().into(), fact.object_kind.clone().into(), fact.id.into(), - source_id.to_vec().into(), source_bytes.into(), self.primary_scope.clone().into(), map.map_id().to_vec().into(), Sha256::digest(&bytes).to_vec().into()])).await.map_err(db_error)?; - let row = source_row(&txn, source, &self.schema).await?.ok_or_else(|| integrity("installed source receipt is missing"))?; - let (stored, _) = self.validate_source_row(source, &row)?; - if stored != *map { return Err(integrity("concurrent source installation conflicts")); } + require_current_source(&txn,source,&self.schema).await?; + retention::map_barrier(&txn,map.map_id()).await?; + retention::barrier(&txn,&self.schema).await?; + retention::ensure_quotas(&txn,&self.schema,&observation,0,0,0,Some(admission.owner())).await?; + let admitted=txn.query_one_raw(stmt(&self.schema,"UPDATE mst2_chunk_receipt_generation SET state='LIVE',owner=NULL,deadline=NULL,reserved_pages=0,reserved_nodes=0,reserved_bytes=0,last_progress=pg_catalog.clock_timestamp() WHERE receipt_key=$1 AND generation=$2 AND owner=$3::uuid AND state='CREATING' AND deadline>pg_catalog.clock_timestamp() AND create_completed AND map_id=$4 AND map_generation=$5 AND receipt_bytes=$6 AND EXISTS(SELECT 1 FROM mst2_chunk_map_lifetime WHERE map_id=$4 AND generation=$5 AND state='LIVE') RETURNING generation",[key.key.clone().into(),admission.generation().into(),admission.owner().to_string().into(),map.map_id().to_vec().into(),map_generation.into(),bytes.clone().into()])).await.map_err(db_error)?; + if admitted.is_none() {return Err(SnapshotError::new(SnapshotErrorCode::LeaseExpired,"chunk-map publication lost its exact live install reservation"));} + let fact=source.fact(); + let source_bytes=source.canonical_bytes()?; + let source_id=source_id(&self.primary_scope,&source_bytes); + txn.execute_raw(stmt(&self.schema,"INSERT INTO mst2_chunk_map_source(storage_domain,git_oid,object_kind,fact_id,source_id,source_bytes,primary_scope,map_id,receipt_digest,receipt_generation) VALUES($1,$2,$3,$4,$5,$6,$7,$8,$9,$10) ON CONFLICT(storage_domain,git_oid,object_kind) DO NOTHING",[fact.storage_domain.clone().into(),fact.git_oid.clone().into(),fact.object_kind.clone().into(),fact.id.into(),source_id.to_vec().into(),source_bytes.into(),self.primary_scope.clone().into(),map.map_id().to_vec().into(),Sha256::digest(&bytes).to_vec().into(),admission.generation().into()])).await.map_err(db_error)?; + let row=source_row(&txn,source,&self.schema).await?.ok_or_else(||integrity("installed source receipt is missing"))?; + let (stored,_)=self.validate_source_row(source,&row)?; + if stored!=*map || row.try_get::("","receipt_key").map_err(db_error)?!=key.key {return Err(integrity("concurrent source installation conflicts"));} Ok(()) }.await; finish(txn, result).await } - pub(crate) async fn selected_page( &self, map: &PersistedChunkMap, page_index: u64, ) -> Result, SnapshotError> { + map.ensure_live().await?; let intervals = proof_intervals(map.map.page_count, page_index)?; let memory = self.budget.reserve(PAGE_CREDIT)?; let txn = self.transaction().await?; @@ -240,7 +342,10 @@ impl PostgresChunkMapRepository { tracing::debug!(target:"mst2::chunk_map", selected_leaf_bytes = bytes.len(), selected_sibling_rows = nodes.len(), page_count = map.map.page_count, "authenticated persisted chunk map selected page"); Ok(Arc::new(AuthenticatedChunkPage { leaf, proof, _memory: memory })) }.await; - finish(txn, result).await + let page = finish(txn, result).await?; + map.ensure_live().await?; + map.record_progress().await?; + Ok(page) } async fn transaction(&self) -> Result { @@ -412,7 +517,7 @@ async fn source_row( source: &ChunkMapSource, schema: &str, ) -> Result, SnapshotError> { - connection.query_one_raw(stmt(schema, "SELECT s.fact_id,CASE WHEN pg_catalog.octet_length(s.source_id)=32 THEN s.source_id ELSE NULL END AS source_id,CASE WHEN pg_catalog.octet_length(s.source_bytes)<=2048 THEN s.source_bytes ELSE NULL END AS source_bytes,CASE WHEN pg_catalog.octet_length(s.primary_scope)<=1024 THEN s.primary_scope ELSE NULL END AS primary_scope,CASE WHEN pg_catalog.octet_length(s.map_id)=32 THEN s.map_id ELSE NULL END AS map_id,CASE WHEN pg_catalog.octet_length(s.receipt_digest)=32 THEN s.receipt_digest ELSE NULL END AS receipt_digest,CASE WHEN pg_catalog.octet_length(m.descriptor)=100 THEN m.descriptor ELSE NULL END AS descriptor,m.page_count,CASE WHEN pg_catalog.octet_length(m.pages_root)=32 THEN m.pages_root ELSE NULL END AS pages_root FROM mst2_chunk_map_source s LEFT JOIN mst2_chunk_map m ON m.map_id=s.map_id WHERE s.storage_domain=$1 AND s.git_oid=$2 AND s.object_kind=$3", + connection.query_one_raw(stmt(schema, "SELECT s.fact_id,s.receipt_key,s.receipt_generation,g.state AS generation_state,l.state AS map_state,l.generation AS map_generation,CASE WHEN pg_catalog.octet_length(g.receipt_bytes)<=4096 THEN g.receipt_bytes ELSE NULL END AS generation_receipt_bytes,CASE WHEN pg_catalog.octet_length(s.source_id)=32 THEN s.source_id ELSE NULL END AS source_id,CASE WHEN pg_catalog.octet_length(s.source_bytes)<=2048 THEN s.source_bytes ELSE NULL END AS source_bytes,CASE WHEN pg_catalog.octet_length(s.primary_scope)<=1024 THEN s.primary_scope ELSE NULL END AS primary_scope,CASE WHEN pg_catalog.octet_length(s.map_id)=32 THEN s.map_id ELSE NULL END AS map_id,CASE WHEN pg_catalog.octet_length(s.receipt_digest)=32 THEN s.receipt_digest ELSE NULL END AS receipt_digest,CASE WHEN pg_catalog.octet_length(m.descriptor)=100 THEN m.descriptor ELSE NULL END AS descriptor,m.page_count,CASE WHEN pg_catalog.octet_length(m.pages_root)=32 THEN m.pages_root ELSE NULL END AS pages_root FROM mst2_chunk_map_source s LEFT JOIN mst2_chunk_map m ON m.map_id=s.map_id LEFT JOIN mst2_chunk_receipt_generation g ON g.receipt_key=s.receipt_key AND g.generation=s.receipt_generation AND g.map_id=s.map_id LEFT JOIN mst2_chunk_map_lifetime l ON l.map_id=s.map_id AND l.generation=g.map_generation WHERE s.storage_domain=$1 AND s.git_oid=$2 AND s.object_kind=$3", [source.fact().storage_domain.clone().into(), source.fact().git_oid.clone().into(), source.fact().object_kind.clone().into()])).await.map_err(db_error) } @@ -447,13 +552,6 @@ fn receipt( Ok(bytes) } -fn receipt_key(source_id: [u8; 32]) -> ObjectKey { - ObjectKey { - namespace: ObjectNamespace::ChunkMapReceipt, - key: hex::encode(source_id), - } -} - async fn read_receipt( objects: &MegaObjectStorageWrapper, key: &ObjectKey, @@ -505,46 +603,6 @@ fn encode_nodes(nodes: &[ChunkMapNode]) -> Result { serde_json::to_string(&nodes.iter().map(|n| json!({"first_page":n.start,"page_count":n.pages,"digest":hex::encode(n.digest)})).collect::>()).map_err(db_error) } -async fn compare_complete_map( - txn: &DatabaseTransaction, - verified: &VerifiedSourceChunkMap, - nodes: &[ChunkMapNode], - schema: &str, -) -> Result<(), SnapshotError> { - let map = verified.map(); - let row = txn.query_one_raw(stmt(schema, "SELECT CASE WHEN pg_catalog.octet_length(descriptor)=100 THEN descriptor ELSE NULL END AS descriptor,page_count,CASE WHEN pg_catalog.octet_length(pages_root)=32 THEN pages_root ELSE NULL END AS pages_root,(SELECT pg_catalog.count(*) FROM mst2_chunk_map_leaf WHERE map_id=$1) AS leaves,(SELECT pg_catalog.count(*) FROM mst2_chunk_map_node WHERE map_id=$1) AS nodes FROM mst2_chunk_map WHERE map_id=$1", [map.map_id().to_vec().into()])) - .await.map_err(db_error)?.ok_or_else(|| integrity("chunk map installation descriptor is missing"))?; - if bounded_bytes(&row, "descriptor")? != map.encode() - || row.try_get::("", "page_count").map_err(db_error)? as u64 != map.page_count - || bounded_bytes(&row, "pages_root")? != map.pages_root - || row.try_get::("", "leaves").map_err(db_error)? as u64 != map.page_count - || row.try_get::("", "nodes").map_err(db_error)? as usize != nodes.len() - { - return Err(integrity( - "chunk map installation has conflicting descriptor or incomplete coverage", - )); - } - for batch in verified.leaves().chunks(64) { - let bad = txn.query_one_raw(stmt(schema, "SELECT p.page_index FROM pg_catalog.jsonb_to_recordset($2::jsonb) AS p(page_index integer,payload text) LEFT JOIN mst2_chunk_map_leaf l ON l.map_id=$1 AND l.page_index=p.page_index WHERE l.payload IS DISTINCT FROM pg_catalog.decode(p.payload,'hex') LIMIT 1", - [map.map_id().to_vec().into(), encode_leaves(batch)?.into()])).await.map_err(db_error)?; - if bad.is_some() { - return Err(integrity( - "immutable stored chunk map leaf conflicts with verified source", - )); - } - } - for batch in nodes.chunks(1024) { - let bad = txn.query_one_raw(stmt(schema, "SELECT p.first_page FROM pg_catalog.jsonb_to_recordset($2::jsonb) AS p(first_page integer,page_count integer,digest text) LEFT JOIN mst2_chunk_map_node n ON n.map_id=$1 AND n.first_page=p.first_page AND n.page_count=p.page_count WHERE n.digest IS DISTINCT FROM pg_catalog.decode(p.digest,'hex') LIMIT 1", - [map.map_id().to_vec().into(), encode_nodes(batch)?.into()])).await.map_err(db_error)?; - if bad.is_some() { - return Err(integrity( - "immutable stored chunk map node conflicts with verified source", - )); - } - } - Ok(()) -} - fn quoted(schema: &str) -> String { format!("\"{}\"", schema.replace('"', "\"\"")) } @@ -591,6 +649,11 @@ fn qualify_relations(schema: &str, sql: &str) -> String { "mst2_chunk_map_source", "mst2_chunk_map_leaf", "mst2_chunk_map_node", + "mst2_chunk_receipt_generation", + "mst2_chunk_map_lifetime", + "mst2_chunk_reader", + "mst2_chunk_map_gc", + "mst2_chunk_retention_barrier", ] .contains(&&sql[start..cursor]) { diff --git a/src/jupiter/storage/native_chunk_map/retention.rs b/src/jupiter/storage/native_chunk_map/retention.rs new file mode 100644 index 00000000..82a74da2 --- /dev/null +++ b/src/jupiter/storage/native_chunk_map/retention.rs @@ -0,0 +1,1220 @@ +//! Bounded primary owners and exact physical-generation reconciliation. + +use std::{ + sync::OnceLock, + time::{Duration, Instant}, +}; + +use tokio::sync::Mutex; +use uuid::Uuid; + +use super::*; +use crate::orbit_api::{ + error::IoOrbitError, + object_storage::{ChunkMapReceiptDeletion, ChunkMapReceiptInventory}, +}; + +const RENEW_INTERVAL: Duration = Duration::from_secs(10); +const HISTORY_CAP: i64 = 131_072; +const MAINTENANCE_MAX: usize = 64; +const OWNER_TTL: Duration = Duration::from_secs(59); + +struct OwnerDeadline(std::sync::Mutex); + +impl OwnerDeadline { + fn from_sql_start(started: Instant) -> Self { + // The SQL INSERT/renew sets clock_timestamp()+59s after this sample. + // Counting SQL and commit latency against the TTL is conservative. + Self(std::sync::Mutex::new(started + OWNER_TTL)) + } + + fn get(&self) -> Result { + let deadline = *self + .0 + .lock() + .map_err(|_| integrity("owner deadline is poisoned"))?; + if deadline <= Instant::now() { + return Err(expired("chunk-map ownership deadline elapsed")); + } + Ok(deadline) + } + + fn observe(&self, deadline: Instant, renewed: bool) -> Result<(), SnapshotError> { + let mut previous = self + .0 + .lock() + .map_err(|_| integrity("owner deadline is poisoned"))?; + *previous = if renewed { + deadline + } else { + (*previous).min(deadline) + }; + if *previous <= Instant::now() { + return Err(expired("chunk-map ownership deadline elapsed")); + } + Ok(()) + } + + async fn bound(&self, future: F) -> Result { + let deadline = self.get()?; + let result = tokio::time::timeout_at(tokio::time::Instant::from_std(deadline), future) + .await + .map_err(|_| expired("chunk-map ownership deadline elapsed"))?; + self.get()?; + Ok(result) + } +} + +fn remaining_deadline(started: Instant, row: &QueryResult) -> Result { + let remaining = row.try_get::("", "remaining_us").map_err(db_error)?; + let remaining = u64::try_from(remaining) + .ok() + .filter(|value| *value > 0) + .ok_or_else(|| expired("chunk-map database ownership deadline elapsed"))?; + started + .checked_add(Duration::from_micros(remaining)) + .ok_or_else(|| integrity("chunk-map ownership deadline is not representable")) +} + +fn retention_budget() -> &'static Arc { + // Inventory survives requests. Its separate process-wide credits bound + // independent repositories without retaining source or response credit. + static BUDGET: OnceLock> = OnceLock::new(); + BUDGET.get_or_init(|| MemoryBudget::new(128 * 1024 * 1024)) +} + +pub(super) struct BackingObservation { + inventory: ChunkMapReceiptInventory, + started_at: String, + captured: Instant, + backing: MegaObjectStorageWrapper, + _memory: MemoryLease, +} + +pub(crate) struct ChunkMapReader { + repository: PostgresChunkMapRepository, + source: ChunkMapSource, + owner: Uuid, + receipt_key: String, + generation: i64, + map_id: [u8; 32], + map_generation: i64, + confirmed: Mutex<(Instant, Instant)>, + deadline: OwnerDeadline, +} + +impl ChunkMapReader { + #[cfg(test)] + pub(crate) async fn test_check_next_owner_operation(&self) { + let before = Instant::now() - Duration::from_secs(11); + *self.confirmed.lock().await = (before, before); + } + pub(crate) async fn ensure_live(&self) -> Result<(), SnapshotError> { + self.deadline.get()?; + let mut confirmed = self.confirmed.lock().await; + self.deadline.get()?; + if confirmed.0.elapsed() < RENEW_INTERVAL { + return Ok(()); + } + let deadline = self.deadline.bound(self.check(false)).await??; + self.deadline.observe(deadline, false)?; + confirmed.0 = Instant::now(); + Ok(()) + } + + pub(super) async fn record_progress(&self) -> Result<(), SnapshotError> { + self.deadline.get()?; + let mut confirmed = self.confirmed.lock().await; + self.deadline.get()?; + if confirmed.1.elapsed() < RENEW_INTERVAL { + return Ok(()); + } + let deadline = self.deadline.bound(self.check(true)).await??; + self.deadline.observe(deadline, true)?; + *confirmed = (Instant::now(), Instant::now()); + Ok(()) + } + + pub(super) async fn validate_after_receipt(&self) -> Result<(), SnapshotError> { + let mut confirmed = self.confirmed.lock().await; + self.deadline.get()?; + let deadline = self.deadline.bound(self.check(true)).await??; + self.deadline.observe(deadline, true)?; + *confirmed = (Instant::now(), Instant::now()); + Ok(()) + } + + pub(crate) async fn await_backend( + &self, + future: F, + ) -> Result { + self.ensure_live().await?; + tokio::pin!(future); + loop { + let deadline = self.deadline.get()?; + let check_at = self.confirmed.lock().await.0 + RENEW_INTERVAL; + tokio::select! { + biased; + _ = tokio::time::sleep_until(tokio::time::Instant::from_std(deadline)) => { + return Err(expired("chunk-map reader backend wait outlived its ownership")); + } + _ = tokio::time::sleep_until(tokio::time::Instant::from_std(check_at)) => { + self.ensure_live().await?; + } + value = &mut future => { + self.deadline.get()?; + return Ok(value); + } + } + } + } + + async fn check(&self, renew: bool) -> Result { + let started = Instant::now(); + let txn = self.repository.transaction().await?; + let result = async { + self.repository.require_scope(&txn).await?; + require_current_source(&txn, &self.source, &self.repository.schema).await?; + barrier(&txn, &self.repository.schema).await?; + let row = if renew {txn.query_one_raw(stmt(&self.repository.schema, + "UPDATE mst2_chunk_reader SET deadline=pg_catalog.clock_timestamp()+interval '59 seconds',last_progress=pg_catalog.clock_timestamp() WHERE owner=$1::uuid AND receipt_key=$2 AND source_generation=$3 AND map_id=$4 AND map_generation=$5 AND deadline>pg_catalog.clock_timestamp() RETURNING owner", + [self.owner.to_string().into(),self.receipt_key.clone().into(),self.generation.into(),self.map_id.to_vec().into(),self.map_generation.into()])).await.map_err(db_error)?} else { + txn.query_one_raw(stmt(&self.repository.schema,"SELECT r.owner FROM mst2_chunk_reader r JOIN mst2_chunk_receipt_generation g ON g.receipt_key=r.receipt_key AND g.generation=r.source_generation JOIN mst2_chunk_map_lifetime l ON l.map_id=r.map_id AND l.generation=r.map_generation WHERE r.owner=$1::uuid AND r.receipt_key=$2 AND r.source_generation=$3 AND r.map_id=$4 AND r.map_generation=$5 AND r.deadline>pg_catalog.clock_timestamp() AND g.state='LIVE' AND l.state='LIVE'",[self.owner.to_string().into(),self.receipt_key.clone().into(),self.generation.into(),self.map_id.to_vec().into(),self.map_generation.into()])).await.map_err(db_error)? + }; + if row.is_none() { return Err(expired("chunk-map reader expired before progress resumed")); } + if renew {txn.execute_raw(stmt(&self.repository.schema, + "UPDATE mst2_chunk_map_lifetime SET last_used=pg_catalog.clock_timestamp() WHERE map_id=$1 AND generation=$2 AND state='LIVE'", + [self.map_id.to_vec().into(),self.map_generation.into()])).await.map_err(db_error)?; + txn.execute_raw(stmt(&self.repository.schema,"UPDATE mst2_chunk_receipt_generation SET last_progress=pg_catalog.clock_timestamp() WHERE receipt_key=$1 AND generation=$2 AND state='LIVE'",[self.receipt_key.clone().into(),self.generation.into()])).await.map_err(db_error)?; + } + let remaining = txn.query_one_raw(stmt(&self.repository.schema, + "SELECT pg_catalog.floor(EXTRACT(EPOCH FROM (deadline-pg_catalog.clock_timestamp()))*1000000)::bigint AS remaining_us FROM mst2_chunk_reader WHERE owner=$1::uuid", + [self.owner.to_string().into()])).await.map_err(db_error)? + .ok_or_else(|| expired("chunk-map reader deadline disappeared"))?; + remaining_deadline(started, &remaining) + }.await; + finish(txn, result).await + } +} + +impl Drop for ChunkMapReader { + fn drop(&mut self) { + let Ok(runtime) = tokio::runtime::Handle::try_current() else { + return; + }; + let repository = self.repository.clone(); + let owner = self.owner; + runtime.spawn(async move { + let result = async { + let txn = repository.transaction().await?; + let result = async { + repository.require_scope(&txn).await?; + barrier(&txn, &repository.schema).await?; + txn.execute_raw(stmt( + &repository.schema, + "DELETE FROM mst2_chunk_reader WHERE owner=$1::uuid", + [owner.to_string().into()], + )) + .await + .map_err(db_error)?; + Ok(()) + } + .await; + finish(txn, result).await + } + .await; + if let Err(error) = result { + tracing::warn!( + ?error, + "chunk-map reader release will expire by database deadline" + ); + } + }); + } +} + +/// A capacity reservation exists before the verifier opens the source body. +/// It cannot publish source trust; installation still consumes the opaque +/// complete-stream verifier and independently checks the backend receipt. +pub(crate) struct ChunkMapInstall { + repository: PostgresChunkMapRepository, + source: ChunkMapSource, + owner: Uuid, + key: ObjectKey, + generation: i64, + progress: Mutex<(Instant, Instant, u64)>, + deadline: OwnerDeadline, +} + +#[derive(Clone)] +pub(super) struct CancelledInstall { + owner: Uuid, + key: String, + generation: i64, +} + +impl ChunkMapInstall { + #[cfg(test)] + pub(crate) async fn test_check_next_owner_operation(&self) { + let mut progress = self.progress.lock().await; + let before = Instant::now() - Duration::from_secs(11); + progress.0 = before; + } + pub(crate) async fn ensure_live(&self) -> Result<(), SnapshotError> { + self.renew(false, 0).await + } + + pub(crate) async fn record_progress(&self, completed: u64) -> Result<(), SnapshotError> { + self.renew(true, completed).await + } + + pub(crate) async fn await_backend( + &self, + future: F, + ) -> Result { + self.ensure_live().await?; + tokio::pin!(future); + loop { + let deadline = self.deadline.get()?; + let check_at = self.progress.lock().await.0 + RENEW_INTERVAL; + tokio::select! { + biased; + _ = tokio::time::sleep_until(tokio::time::Instant::from_std(deadline)) => { + return Err(expired("chunk-map install backend wait outlived its ownership")); + } + _ = tokio::time::sleep_until(tokio::time::Instant::from_std(check_at)) => { + self.ensure_live().await?; + } + value = &mut future => { + self.deadline.get()?; + return Ok(value); + } + } + } + } + + async fn renew(&self, progress: bool, completed: u64) -> Result<(), SnapshotError> { + self.deadline.get()?; + let mut confirmed = self.progress.lock().await; + self.deadline.get()?; + if progress && completed <= confirmed.2 { + return Ok(()); + } + if (if progress { + confirmed.1.elapsed() + } else { + confirmed.0.elapsed() + }) < RENEW_INTERVAL + { + if progress { + confirmed.2 = completed; + } + return Ok(()); + } + // Calling ensure_live after a long stalled await checks the old + // deadline without extending it. Only new source progress renews it. + let started = Instant::now(); + let deadline = self.deadline.bound(async { + let txn = self.repository.transaction().await?; + let result=async { + self.repository.require_scope(&txn).await?; + require_current_source(&txn,&self.source,&self.repository.schema).await?; + barrier(&txn,&self.repository.schema).await?; + let row=if progress { + txn.query_one_raw(stmt(&self.repository.schema,"UPDATE mst2_chunk_receipt_generation SET deadline=pg_catalog.clock_timestamp()+interval '59 seconds',last_progress=pg_catalog.clock_timestamp() WHERE receipt_key=$1 AND generation=$2 AND owner=$3::uuid AND state IN ('RESERVED','CREATING') AND deadline>pg_catalog.clock_timestamp() RETURNING generation",[self.key.key.clone().into(),self.generation.into(),self.owner.to_string().into()])).await.map_err(db_error)? + } else { + txn.query_one_raw(stmt(&self.repository.schema,"SELECT generation FROM mst2_chunk_receipt_generation WHERE receipt_key=$1 AND generation=$2 AND owner=$3::uuid AND state IN ('RESERVED','CREATING') AND deadline>pg_catalog.clock_timestamp()",[self.key.key.clone().into(),self.generation.into(),self.owner.to_string().into()])).await.map_err(db_error)? + }; + if row.is_none() { return Err(expired("chunk-map install reservation expired or was retired")); } + let remaining = txn.query_one_raw(stmt(&self.repository.schema, + "SELECT pg_catalog.floor(EXTRACT(EPOCH FROM (deadline-pg_catalog.clock_timestamp()))*1000000)::bigint AS remaining_us FROM mst2_chunk_receipt_generation WHERE receipt_key=$1 AND generation=$2 AND owner=$3::uuid AND state IN ('RESERVED','CREATING')", + [self.key.key.clone().into(),self.generation.into(),self.owner.to_string().into()])).await.map_err(db_error)? + .ok_or_else(|| expired("chunk-map install deadline disappeared"))?; + remaining_deadline(started, &remaining) + }.await; + finish(txn, result).await + }).await??; + self.deadline.observe(deadline, progress)?; + if progress { + *confirmed = (Instant::now(), Instant::now(), completed); + } else { + confirmed.0 = Instant::now(); + } + Ok(()) + } + + pub(super) async fn prepare( + &self, + verified: &VerifiedSourceChunkMap, + ) -> Result, SnapshotError> { + if verified.source() != &self.source { + return Err(integrity( + "verified source differs from install reservation", + )); + } + let bytes = receipt(&self.repository.primary_scope, &self.source, verified.map())?; + let started = Instant::now(); + self.deadline.bound(async { + let txn = self.repository.transaction().await?; + let result=async { + self.repository.require_scope(&txn).await?; + require_current_source(&txn,&self.source,&self.repository.schema).await?; + barrier(&txn,&self.repository.schema).await?; + let row=txn.query_one_raw(stmt(&self.repository.schema,"UPDATE mst2_chunk_receipt_generation SET state='CREATING',map_id=$4,receipt_bytes=$5,deadline=pg_catalog.clock_timestamp()+interval '59 seconds',last_progress=pg_catalog.clock_timestamp() WHERE receipt_key=$1 AND generation=$2 AND owner=$3::uuid AND state='RESERVED' AND deadline>pg_catalog.clock_timestamp() RETURNING generation",[self.key.key.clone().into(),self.generation.into(),self.owner.to_string().into(),verified.map().map_id().to_vec().into(),bytes.clone().into()])).await.map_err(db_error)?; + if row.is_none() { return Err(expired("chunk-map install lost its exact reservation before receipt creation")); } + Ok(()) + }.await; + finish(txn, result).await?; + Ok::<_, SnapshotError>(()) + }).await??; + self.deadline.observe(started + OWNER_TTL, true)?; + Ok(bytes) + } + + pub(super) fn key(&self) -> &ObjectKey { + &self.key + } + pub(super) fn generation(&self) -> i64 { + self.generation + } + pub(super) fn owner(&self) -> Uuid { + self.owner + } + + pub(super) async fn acknowledge_create(&self) -> Result<(), SnapshotError> { + // This method is called only after the actual atomic create future + // returns success. Absence, a checksum or inventory cannot call it. + let started = Instant::now(); + let renewed = self.deadline.bound(async { + let txn = self.repository.transaction().await?; + let result=async { + self.repository.require_scope(&txn).await?; + barrier(&txn,&self.repository.schema).await?; + let row=txn.query_one_raw(stmt(&self.repository.schema,"UPDATE mst2_chunk_receipt_generation SET create_completed=true,create_completed_at=COALESCE(create_completed_at,pg_catalog.clock_timestamp()),deadline=CASE WHEN state='CREATING' AND owner=$3::uuid AND deadline>pg_catalog.clock_timestamp() THEN pg_catalog.clock_timestamp()+interval '59 seconds' ELSE deadline END WHERE receipt_key=$1 AND generation=$2 AND receipt_bytes IS NOT NULL RETURNING generation,CASE WHEN state='CREATING' AND owner=$3::uuid AND deadline>pg_catalog.clock_timestamp() THEN true ELSE false END AS deadline_renewed",[self.key.key.clone().into(),self.generation.into(),self.owner.to_string().into()])).await.map_err(db_error)? + .ok_or_else(|| integrity("completed create lost its persistent generation fence"))?; + row.try_get::("", "deadline_renewed").map_err(db_error) + }.await; + finish(txn, result).await + }).await??; + if renewed { + self.deadline.observe(started + OWNER_TTL, true)?; + } + Ok(()) + } + + pub(super) async fn retire(&self) -> Result<(), SnapshotError> { + retire_install(&self.repository, self.owner, &self.key.key, self.generation).await + } +} + +impl Drop for ChunkMapInstall { + fn drop(&mut self) { + if let Ok(mut cancelled) = self.repository.cancelled_installs.lock() { + if cancelled.len() < 64 { + cancelled.push(CancelledInstall { + owner: self.owner, + key: self.key.key.clone(), + generation: self.generation, + }); + } + } + let Ok(runtime) = tokio::runtime::Handle::try_current() else { + return; + }; + let repository = self.repository.clone(); + let owner = self.owner; + let key = self.key.key.clone(); + let generation = self.generation; + runtime.spawn(async move { + let result = retire_install(&repository, owner, &key, generation).await; + if result.is_ok() { + forget_cancelled_install(&repository, owner); + } + if let Err(error) = result { + tracing::warn!( + ?error, + "chunk-map install cancellation will reconcile by database deadline" + ); + } + }); + } +} + +fn forget_cancelled_install(repository: &PostgresChunkMapRepository, owner: Uuid) { + if let Ok(mut cancelled) = repository.cancelled_installs.lock() { + cancelled.retain(|install| install.owner != owner); + } +} + +async fn retire_install( + repository: &PostgresChunkMapRepository, + owner: Uuid, + key: &str, + generation: i64, +) -> Result<(), SnapshotError> { + let txn = repository.transaction().await?; + let result = async { + repository.require_scope(&txn).await?; + barrier(&txn, &repository.schema).await?; + txn.execute_raw(stmt(&repository.schema,"UPDATE mst2_chunk_receipt_generation SET state=CASE WHEN receipt_bytes IS NULL THEN 'APPLIED' ELSE 'DELETING' END,owner=NULL,deadline=NULL,reserved_pages=0,reserved_nodes=0,reserved_bytes=0 WHERE receipt_key=$1 AND generation=$2 AND owner=$3::uuid AND state IN ('RESERVED','CREATING')",[key.into(),generation.into(),owner.to_string().into()])).await.map_err(db_error)?; + Ok(()) + }.await; + finish(txn, result).await +} + +/// Constructor stays in the collector module. Its private fields carry a +/// real primary claim and the independently read exact backend body. +pub(crate) struct VerifiedReceiptDeletionClaim { + key: ObjectKey, + expected: Vec, + _generation: i64, +} +impl VerifiedReceiptDeletionClaim { + pub(crate) fn key(&self) -> &ObjectKey { + &self.key + } + pub(crate) fn expected_bytes(&self) -> &[u8] { + &self.expected + } +} + +impl PostgresChunkMapRepository { + async fn replay_cancelled_installs(&self) -> Result<(), SnapshotError> { + let cancelled = self + .cancelled_installs + .lock() + .map_err(|_| integrity("cancelled install replay ownership is poisoned"))? + .clone(); + for install in cancelled { + retire_install(self, install.owner, &install.key, install.generation).await?; + forget_cancelled_install(self, install.owner); + } + Ok(()) + } + + async fn maintain_before_admission( + &self, + objects: &MegaObjectStorageWrapper, + ) -> Result<(), SnapshotError> { + let mut completed = self.admission_maintenance.lock().await; + self.replay_cancelled_installs().await?; + if completed + .as_ref() + .is_some_and(|last| last.elapsed() < RENEW_INTERVAL) + { + let txn = self.transaction().await?; + self.require_scope(&txn).await?; + let pending=txn.query_one_raw(stmt(&self.schema,"SELECT EXISTS(SELECT 1 FROM mst2_chunk_receipt_generation WHERE state='DELETING' OR (state IN ('RESERVED','CREATING') AND deadline<=pg_catalog.clock_timestamp())) AS pending",[])).await.map_err(db_error)?.ok_or_else(||integrity("pending chunk-map admission replay observation missing"))?.try_get::("","pending").map_err(db_error)?; + txn.commit().await.map_err(db_error)?; + if !pending { + return Ok(()); + } + } + self.maintain(objects, 8).await?; + *completed = Some(Instant::now()); + Ok(()) + } + #[cfg(test)] + pub(crate) async fn test_new_install_capacity( + &self, + objects: &MegaObjectStorageWrapper, + ) -> Result<(), SnapshotError> { + let observation = inventory(self, objects).await?; + let txn = self.transaction().await?; + let result = async { + self.require_scope(&txn).await?; + barrier(&txn, &self.schema).await?; + ensure_quotas(&txn, &self.schema, &observation, 1, 1, 1024 * 1024, None).await + } + .await; + finish(txn, result).await + } + pub(super) async fn stage_map_start( + &self, + verified: &VerifiedSourceChunkMap, + admission: &ChunkMapInstall, + observation: &BackingObservation, + ) -> Result { + let map = verified.map(); + let txn = self.transaction().await?; + let result=async { + self.require_scope(&txn).await?; + require_current_source(&txn,verified.source(),&self.schema).await?; + map_barrier(&txn,map.map_id()).await?; + barrier(&txn,&self.schema).await?; + let inserted=txn.query_one_raw(stmt(&self.schema,"INSERT INTO mst2_chunk_map(map_id,descriptor,page_count,pages_root) VALUES($1,$2,$3,$4) ON CONFLICT(map_id) DO NOTHING RETURNING map_id",[map.map_id().to_vec().into(),map.encode().into(),(map.page_count as i32).into(),map.pages_root.to_vec().into()])).await.map_err(db_error)?.is_some(); + let lifetime=txn.query_one_raw(stmt(&self.schema,"SELECT generation,state FROM mst2_chunk_map_lifetime WHERE map_id=$1",[map.map_id().to_vec().into()])).await.map_err(db_error)?; + let generation=if let Some(row)=lifetime { + if row.try_get::("","state").map_err(db_error)?!="LIVE" {return Err(unavailable("shared chunk map collection is pending"));} + row.try_get::("","generation").map_err(db_error)? + } else if inserted { + let generation=next_generation(&txn,&self.schema).await?; + txn.execute_raw(stmt(&self.schema,"INSERT INTO mst2_chunk_map_lifetime(map_id,generation,state) VALUES($1,$2,'LIVE')",[map.map_id().to_vec().into(),generation.into()])).await.map_err(db_error)?; + generation + } else {return Err(integrity("existing chunk map has no exact lifetime; never repair it"));}; + let row=txn.query_one_raw(stmt(&self.schema,"SELECT CASE WHEN pg_catalog.octet_length(descriptor)=100 THEN descriptor ELSE NULL END AS descriptor,page_count,CASE WHEN pg_catalog.octet_length(pages_root)=32 THEN pages_root ELSE NULL END AS pages_root FROM mst2_chunk_map WHERE map_id=$1",[map.map_id().to_vec().into()])).await.map_err(db_error)?.ok_or_else(||integrity("staged map descriptor missing"))?; + if bounded_bytes(&row,"descriptor")?!=map.encode() || row.try_get::("","page_count").map_err(db_error)? as u64!=map.page_count || bounded_bytes(&row,"pages_root")?!=map.pages_root {return Err(integrity("immutable staged map descriptor conflicts with verified source"));} + let assigned=txn.query_one_raw(stmt(&self.schema,"UPDATE mst2_chunk_receipt_generation SET map_generation=$4,deadline=pg_catalog.clock_timestamp()+interval '59 seconds',last_progress=pg_catalog.clock_timestamp() WHERE receipt_key=$1 AND generation=$2 AND owner=$3::uuid AND state='CREATING' AND deadline>pg_catalog.clock_timestamp() AND map_id=$5 AND create_completed RETURNING generation",[admission.key.key.clone().into(),admission.generation.into(),admission.owner.to_string().into(),generation.into(),map.map_id().to_vec().into()])).await.map_err(db_error)?; + if assigned.is_none() {return Err(expired("map staging lost the exact completed-create reservation"));} + ensure_quotas(&txn,&self.schema,observation,0,0,0,Some(admission.owner)).await?; + Ok(generation) + }.await; + finish(txn, result).await + } + + pub(super) async fn stage_map_batch( + &self, + verified: &VerifiedSourceChunkMap, + admission: &ChunkMapInstall, + generation: i64, + leaves: Option<&[ChunkLeaf]>, + nodes: Option<&[ChunkMapNode]>, + ) -> Result<(), SnapshotError> { + let map_id = verified.map().map_id(); + let txn = self.transaction().await?; + let result=async { + self.require_scope(&txn).await?; + require_current_source(&txn,verified.source(),&self.schema).await?; + map_barrier(&txn,map_id).await?; + let sealed=txn.query_one_raw(stmt(&self.schema,"SELECT EXISTS(SELECT 1 FROM mst2_chunk_map_source WHERE map_id=$1) AS sealed",[map_id.to_vec().into()])).await.map_err(db_error)?.ok_or_else(||integrity("staged map seal observation missing"))?.try_get::("","sealed").map_err(db_error)?; + if !sealed { + if let Some(batch)=leaves { + txn.execute_raw(stmt(&self.schema,"INSERT INTO mst2_chunk_map_leaf(map_id,page_index,payload) SELECT $1,p.page_index,pg_catalog.decode(p.payload,'hex') FROM pg_catalog.jsonb_to_recordset($2::jsonb) AS p(page_index integer,payload text) ON CONFLICT(map_id,page_index) DO NOTHING",[map_id.to_vec().into(),encode_leaves(batch)?.into()])).await.map_err(db_error)?; + } + if let Some(batch)=nodes { + txn.execute_raw(stmt(&self.schema,"INSERT INTO mst2_chunk_map_node(map_id,first_page,page_count,digest) SELECT $1,p.first_page,p.page_count,pg_catalog.decode(p.digest,'hex') FROM pg_catalog.jsonb_to_recordset($2::jsonb) AS p(first_page integer,page_count integer,digest text) ON CONFLICT(map_id,first_page,page_count) DO NOTHING",[map_id.to_vec().into(),encode_nodes(batch)?.into()])).await.map_err(db_error)?; + } + } + self.checkpoint_staging(&txn,admission,map_id,generation).await + }.await; + finish(txn, result).await + } + + async fn checkpoint_staging( + &self, + txn: &DatabaseTransaction, + admission: &ChunkMapInstall, + map_id: [u8; 32], + generation: i64, + ) -> Result<(), SnapshotError> { + barrier(txn, &self.schema).await?; + let progress=txn.query_one_raw(stmt(&self.schema,"UPDATE mst2_chunk_receipt_generation SET deadline=pg_catalog.clock_timestamp()+interval '59 seconds',last_progress=pg_catalog.clock_timestamp() WHERE receipt_key=$1 AND generation=$2 AND owner=$3::uuid AND state='CREATING' AND deadline>pg_catalog.clock_timestamp() AND map_id=$4 AND map_generation=$5 AND EXISTS(SELECT 1 FROM mst2_chunk_map_lifetime WHERE map_id=$4 AND generation=$5 AND state='LIVE') RETURNING generation",[admission.key.key.clone().into(),admission.generation.into(),admission.owner.to_string().into(),map_id.to_vec().into(),generation.into()])).await.map_err(db_error)?; + if progress.is_none() { + return Err(expired( + "bounded map publication batch lost its exact install generation", + )); + } + Ok(()) + } + + pub(super) async fn compare_staged_map( + &self, + verified: &VerifiedSourceChunkMap, + nodes: &[ChunkMapNode], + admission: &ChunkMapInstall, + generation: i64, + ) -> Result<(), SnapshotError> { + let map = verified.map(); + let txn = self.transaction().await?; + let result=async { + self.require_scope(&txn).await?; + require_current_source(&txn,verified.source(),&self.schema).await?; + map_barrier(&txn,map.map_id()).await?; + let row=txn.query_one_raw(stmt(&self.schema,"SELECT CASE WHEN pg_catalog.octet_length(descriptor)=100 THEN descriptor ELSE NULL END AS descriptor,page_count,CASE WHEN pg_catalog.octet_length(pages_root)=32 THEN pages_root ELSE NULL END AS pages_root,(SELECT pg_catalog.count(*) FROM mst2_chunk_map_leaf WHERE map_id=$1) AS leaves,(SELECT pg_catalog.count(*) FROM mst2_chunk_map_node WHERE map_id=$1) AS nodes FROM mst2_chunk_map WHERE map_id=$1",[map.map_id().to_vec().into()])).await.map_err(db_error)?.ok_or_else(||integrity("staged map descriptor missing"))?; + if bounded_bytes(&row,"descriptor")?!=map.encode() || row.try_get::("","page_count").map_err(db_error)? as u64!=map.page_count || bounded_bytes(&row,"pages_root")?!=map.pages_root || row.try_get::("","leaves").map_err(db_error)? as u64!=map.page_count || row.try_get::("","nodes").map_err(db_error)? as usize!=nodes.len() {return Err(integrity("staged map has conflicting descriptor or incomplete coverage"));} + self.checkpoint_staging(&txn,admission,map.map_id(),generation).await + }.await; + finish(txn, result).await?; + for batch in verified.leaves().chunks(64) { + let txn = self.transaction().await?; + let result=async { + self.require_scope(&txn).await?; + require_current_source(&txn,verified.source(),&self.schema).await?; + map_barrier(&txn,map.map_id()).await?; + let bad=txn.query_one_raw(stmt(&self.schema,"SELECT p.page_index FROM pg_catalog.jsonb_to_recordset($2::jsonb) AS p(page_index integer,payload text) LEFT JOIN mst2_chunk_map_leaf l ON l.map_id=$1 AND l.page_index=p.page_index WHERE l.payload IS DISTINCT FROM pg_catalog.decode(p.payload,'hex') LIMIT 1",[map.map_id().to_vec().into(),encode_leaves(batch)?.into()])).await.map_err(db_error)?; + if bad.is_some() {return Err(integrity("immutable stored chunk map leaf conflicts with verified source"));} + self.checkpoint_staging(&txn,admission,map.map_id(),generation).await + }.await; + finish(txn, result).await?; + } + for batch in nodes.chunks(1024) { + let txn = self.transaction().await?; + let result=async { + self.require_scope(&txn).await?; + require_current_source(&txn,verified.source(),&self.schema).await?; + map_barrier(&txn,map.map_id()).await?; + let bad=txn.query_one_raw(stmt(&self.schema,"SELECT p.first_page FROM pg_catalog.jsonb_to_recordset($2::jsonb) AS p(first_page integer,page_count integer,digest text) LEFT JOIN mst2_chunk_map_node n ON n.map_id=$1 AND n.first_page=p.first_page AND n.page_count=p.page_count WHERE n.digest IS DISTINCT FROM pg_catalog.decode(p.digest,'hex') LIMIT 1",[map.map_id().to_vec().into(),encode_nodes(batch)?.into()])).await.map_err(db_error)?; + if bad.is_some() {return Err(integrity("immutable stored chunk map node conflicts with verified source"));} + self.checkpoint_staging(&txn,admission,map.map_id(),generation).await + }.await; + finish(txn, result).await?; + } + Ok(()) + } + + pub(crate) async fn maintain( + &self, + objects: &MegaObjectStorageWrapper, + limit: usize, + ) -> Result<(), SnapshotError> { + if !(1..=MAINTENANCE_MAX).contains(&limit) { + return Err(SnapshotError::new( + SnapshotErrorCode::InvalidRequest, + "chunk-map maintenance limit must be 1..64", + )); + } + // Unsupported enumeration must fail before changing any resident + // generation. Overquota stores may still replay known deletions. + match inventory(self, objects).await { + Ok(_) => {} + Err(error) if error.code == SnapshotErrorCode::LimitExceeded => {} + Err(error) => return Err(error), + } + self.replay_cancelled_installs().await?; + let _memory = retention_budget().reserve(8 * 1024 * 1024)?; + let txn = self.transaction().await?; + let result=async { + self.require_scope(&txn).await?; + barrier(&txn,&self.schema).await?; + txn.execute_raw(stmt(&self.schema,"DELETE FROM mst2_chunk_reader WHERE owner IN (SELECT owner FROM mst2_chunk_reader WHERE deadline<=pg_catalog.clock_timestamp() ORDER BY deadline,owner LIMIT $1)",[(limit as i64).into()])).await.map_err(db_error)?; + txn.execute_raw(stmt(&self.schema,"UPDATE mst2_chunk_receipt_generation SET state=CASE WHEN receipt_bytes IS NULL THEN 'APPLIED' ELSE 'DELETING' END,owner=NULL,deadline=NULL,reserved_pages=0,reserved_nodes=0,reserved_bytes=0 WHERE receipt_key IN (SELECT receipt_key FROM mst2_chunk_receipt_generation WHERE state IN ('RESERVED','CREATING') AND deadline<=pg_catalog.clock_timestamp() ORDER BY deadline,receipt_key LIMIT $1)",[(limit as i64).into()])).await.map_err(db_error)?; + // Replay unfinished receipt claims before admitting new victims. + let pending=txn.query_one_raw(stmt(&self.schema,"SELECT EXISTS(SELECT 1 FROM mst2_chunk_receipt_generation WHERE state='DELETING') AS pending",[])).await.map_err(db_error)?.ok_or_else(||integrity("chunk receipt replay observation missing"))?.try_get::("","pending").map_err(db_error)?; + if !pending { + txn.execute_raw(stmt(&self.schema,"UPDATE mst2_chunk_receipt_generation SET state='DELETING' WHERE receipt_key IN (SELECT g.receipt_key FROM mst2_chunk_receipt_generation g WHERE g.state='LIVE' AND g.last_progress<=pg_catalog.clock_timestamp()-interval '1 hour' AND NOT EXISTS(SELECT 1 FROM mst2_chunk_reader r WHERE r.receipt_key=g.receipt_key AND r.source_generation=g.generation AND r.deadline>pg_catalog.clock_timestamp()) ORDER BY g.last_progress,g.receipt_key LIMIT $1)",[(limit as i64).into()])).await.map_err(db_error)?; + } + // Rotate APPLIED claims too: absence is never a promise that a + // cancelled remote create cannot restore this retired key later. + txn.query_all_raw(stmt(&self.schema,"SELECT receipt_key,generation,CASE WHEN pg_catalog.octet_length(receipt_bytes)<=4096 THEN receipt_bytes ELSE NULL END AS receipt_bytes,CASE WHEN pg_catalog.octet_length(primary_scope)<=1024 THEN primary_scope ELSE NULL END AS primary_scope,map_id,map_generation,state FROM mst2_chunk_receipt_generation WHERE state IN ('DELETING','APPLIED') ORDER BY (state='DELETING') DESC,checked_at,receipt_key LIMIT $1",[(limit as i64).into()])).await.map_err(db_error) + }.await; + let claims = finish(txn, result).await?; + for row in claims { + self.reconcile_receipt(objects, row).await?; + } + self.collect_maps(limit).await?; + let observation = inventory(self, objects).await?; + let requested = serde_json::to_string( + &observation + .inventory + .objects + .iter() + .map(|(key, _)| &key.key) + .collect::>(), + ) + .map_err(db_error)?; + let txn = self.transaction().await?; + self.require_scope(&txn).await?; + let unknown=txn.query_all_raw(stmt(&self.schema,"SELECT p.key FROM pg_catalog.jsonb_array_elements_text($1::jsonb) AS p(key) LEFT JOIN mst2_chunk_receipt_generation g ON g.receipt_key=p.key WHERE g.receipt_key IS NULL ORDER BY p.key LIMIT $2",[requested.into(),(limit as i64).into()])).await.map_err(db_error)?; + txn.commit().await.map_err(db_error)?; + for row in unknown { + let key = ObjectKey { + namespace: ObjectNamespace::ChunkMapReceipt, + key: row.try_get("", "key").map_err(db_error)?, + }; + self.claim_orphan(objects, &key).await?; + } + tracing::debug!(target:"mst2::chunk_map",maintenance_limit=limit,actual_backing_receipts=observation.inventory.objects.len(),actual_backing_bytes=observation.inventory.bytes,"bounded chunk-map retention tick"); + Ok(()) + } + + async fn reconcile_receipt( + &self, + objects: &MegaObjectStorageWrapper, + row: QueryResult, + ) -> Result<(), SnapshotError> { + if bounded_bytes(&row, "primary_scope")? != self.primary_scope { + return Err(integrity( + "retired receipt belongs to a different primary scope", + )); + } + let key = ObjectKey { + namespace: ObjectNamespace::ChunkMapReceipt, + key: row.try_get("", "receipt_key").map_err(db_error)?, + }; + let generation: i64 = row.try_get("", "generation").map_err(db_error)?; + let expected: Option> = row.try_get("", "receipt_bytes").map_err(db_error)?; + let txn = self.transaction().await?; + let result=async { + self.require_scope(&txn).await?; + barrier(&txn,&self.schema).await?; + txn.query_one_raw(stmt(&self.schema,"SELECT generation FROM mst2_chunk_receipt_generation WHERE receipt_key=$1 AND generation=$2 AND receipt_bytes IS NOT DISTINCT FROM $3::bytea AND primary_scope=$4 AND state IN ('DELETING','APPLIED') AND NOT EXISTS(SELECT 1 FROM mst2_chunk_reader WHERE receipt_key=$1 AND source_generation=$2 AND deadline>pg_catalog.clock_timestamp())",[key.key.clone().into(),generation.into(),expected.clone().into(),self.primary_scope.clone().into()])).await.map_err(db_error) + }.await; + if finish(txn, result).await?.is_none() { + return Ok(()); + } + if let Some(bytes) = expected.as_ref() { + let (scope, _, map, physical_generation) = decode_orphan(&key, bytes)?; + let stored_map: Option> = row.try_get("", "map_id").map_err(db_error)?; + if scope != self.primary_scope + || physical_generation != generation + || stored_map.as_deref() != Some(map.map_id().as_slice()) + { + return Err(integrity( + "receipt deletion claim disagrees with the independent physical generation", + )); + } + } + if let Some(expected) = expected { + let present = tokio::time::timeout(Duration::from_secs(5), objects.inner.exists(&key)) + .await + .map_err(|_| unavailable("retired receipt existence check timed out"))? + .map_err(storage_error)?; + if present { + read_receipt(objects, &key, &expected).await?; + let authority = ChunkMapReceiptDeletion::from_claim(VerifiedReceiptDeletionClaim { + key: key.clone(), + expected, + _generation: generation, + }); + tokio::time::timeout( + Duration::from_secs(5), + objects.inner.delete_chunk_map_receipt(&authority), + ) + .await + .map_err(|_| unavailable("retired receipt deletion timed out"))? + .map_err(storage_error)?; + } + } else if tokio::time::timeout(Duration::from_secs(5), objects.inner.exists(&key)) + .await + .map_err(|_| unavailable("unused receipt check timed out"))? + .map_err(storage_error)? + { + return Err(integrity( + "receipt exists for a generation that never prepared a create", + )); + } + let txn = self.transaction().await?; + let result=async { + self.require_scope(&txn).await?; + barrier(&txn,&self.schema).await?; + let applied=txn.query_one_raw(stmt(&self.schema,"UPDATE mst2_chunk_receipt_generation SET state='APPLIED',owner=NULL,deadline=NULL,reserved_pages=0,reserved_nodes=0,reserved_bytes=0,checked_at=pg_catalog.clock_timestamp() WHERE receipt_key=$1 AND generation=$2 AND state IN ('DELETING','APPLIED') AND NOT EXISTS(SELECT 1 FROM mst2_chunk_reader r WHERE r.receipt_key=$1 AND r.source_generation=$2 AND r.deadline>pg_catalog.clock_timestamp()) RETURNING generation",[key.key.clone().into(),generation.into()])).await.map_err(db_error)?; + if applied.is_none() { return Err(integrity("retired receipt lost its exact reader-free completion claim")); } + txn.execute_raw(stmt(&self.schema,"DELETE FROM mst2_chunk_map_source WHERE receipt_key=$1 AND receipt_generation=$2",[key.key.into(),generation.into()])).await.map_err(db_error)?; + Ok(()) + }.await; + finish(txn, result).await + } + + async fn claim_orphan( + &self, + objects: &MegaObjectStorageWrapper, + key: &ObjectKey, + ) -> Result<(), SnapshotError> { + let actual = read_bounded_orphan(objects, key).await?; + let Some(actual) = actual else { return Ok(()) }; + let (scope, source_bytes, map, generation) = decode_orphan(key, &actual)?; + if scope != self.primary_scope { + return Err(integrity( + "orphan receipt belongs to a different captured primary", + )); + } + let identity = source_id(&scope, &source_bytes); + let txn = self.transaction().await?; + let result=async { + self.require_scope(&txn).await?; + barrier(&txn,&self.schema).await?; + let count=txn.query_one_raw(stmt(&self.schema,"SELECT pg_catalog.count(*) AS history FROM mst2_chunk_receipt_generation",[])).await.map_err(db_error)?.ok_or_else(||integrity("orphan receipt history count missing"))?.try_get::("","history").map_err(db_error)?; + if count>=HISTORY_CAP { return Err(SnapshotError::new(SnapshotErrorCode::LimitExceeded,"chunk receipt reconciliation history quota exhausted")); } + // An orphan never admits a source or map. Its actual body only + // identifies a retired physical key for collection. + txn.execute_raw(stmt(&self.schema,"INSERT INTO mst2_chunk_receipt_generation(receipt_key,source_id,generation,source_bytes,primary_scope,map_id,receipt_bytes,state,reserved_pages,reserved_nodes,reserved_bytes) VALUES($1,$2,$3,$4,$5,$6,$7,'DELETING',0,0,0) ON CONFLICT(receipt_key) DO NOTHING",[key.key.clone().into(),identity.to_vec().into(),generation.into(),source_bytes.into(),scope.into(),map.map_id().to_vec().into(),actual.into()])).await.map_err(db_error)?; + Ok(()) + }.await; + finish(txn, result).await + } + + pub(super) async fn collect_maps(&self, limit: usize) -> Result<(), SnapshotError> { + let txn = self.transaction().await?; + self.require_scope(&txn).await?; + let rows=txn.query_all_raw(stmt(&self.schema,"SELECT l.map_id,l.generation,l.state FROM mst2_chunk_map_lifetime l WHERE l.state='DELETING' OR (NOT EXISTS(SELECT 1 FROM mst2_chunk_map_lifetime WHERE state='DELETING') AND (l.last_used<=pg_catalog.clock_timestamp()-interval '1 hour' OR EXISTS(SELECT 1 FROM mst2_chunk_receipt_generation g WHERE g.map_id=l.map_id AND g.map_generation=l.generation AND g.state IN ('DELETING','APPLIED'))) AND NOT EXISTS(SELECT 1 FROM mst2_chunk_map_source s WHERE s.map_id=l.map_id) AND NOT EXISTS(SELECT 1 FROM mst2_chunk_receipt_generation g WHERE g.map_id=l.map_id AND g.map_generation=l.generation AND g.state='CREATING' AND g.deadline>pg_catalog.clock_timestamp())) ORDER BY (l.state='DELETING') DESC,l.last_used,l.map_id LIMIT $1",[(limit as i64).into()])).await.map_err(db_error)?; + txn.commit().await.map_err(db_error)?; + for row in rows { + let map_id: Vec = row.try_get("", "map_id").map_err(db_error)?; + let generation: i64 = row.try_get("", "generation").map_err(db_error)?; + let txn = self.transaction().await?; + let result=async { + self.require_scope(&txn).await?; + let acquired=txn.query_one_raw(Statement::from_sql_and_values(DbBackend::Postgres,"SELECT pg_catalog.pg_try_advisory_xact_lock(1296717364,pg_catalog.hashtext($1)) AS acquired",[hex::encode(&map_id).into()])).await.map_err(db_error)?.ok_or_else(||integrity("map collection lock observation missing"))?.try_get::("","acquired").map_err(db_error)?; + if !acquired { return Ok(()); } + barrier(&txn,&self.schema).await?; + let eligible=txn.query_one_raw(stmt(&self.schema,"SELECT state FROM mst2_chunk_map_lifetime l WHERE map_id=$1 AND generation=$2 AND (state='DELETING' OR last_used<=pg_catalog.clock_timestamp()-interval '1 hour' OR EXISTS(SELECT 1 FROM mst2_chunk_receipt_generation g WHERE g.map_id=l.map_id AND g.map_generation=l.generation AND g.state IN ('DELETING','APPLIED'))) AND NOT EXISTS(SELECT 1 FROM mst2_chunk_map_source s WHERE s.map_id=$1) AND NOT EXISTS(SELECT 1 FROM mst2_chunk_reader r WHERE r.map_id=$1 AND r.map_generation=$2 AND r.deadline>pg_catalog.clock_timestamp()) AND NOT EXISTS(SELECT 1 FROM mst2_chunk_receipt_generation g WHERE g.map_id=$1 AND g.map_generation=$2 AND g.state='CREATING' AND g.deadline>pg_catalog.clock_timestamp())",[map_id.clone().into(),generation.into()])).await.map_err(db_error)?; + let Some(eligible)=eligible else { return Ok(()) }; + if eligible.try_get::("","state").map_err(db_error)?=="LIVE" { + let count=txn.query_one_raw(stmt(&self.schema,"SELECT pg_catalog.count(*) AS history FROM mst2_chunk_map_gc",[])).await.map_err(db_error)?.ok_or_else(||integrity("map collection history count missing"))?.try_get::("","history").map_err(db_error)?; + if count>=HISTORY_CAP { return Err(SnapshotError::new(SnapshotErrorCode::LimitExceeded,"chunk map collection history quota exhausted")); } + txn.execute_raw(stmt(&self.schema,"INSERT INTO mst2_chunk_map_gc(map_id,generation,state) VALUES($1,$2,'PENDING') ON CONFLICT(map_id,generation) DO NOTHING",[map_id.clone().into(),generation.into()])).await.map_err(db_error)?; + txn.execute_raw(stmt(&self.schema,"UPDATE mst2_chunk_map_lifetime SET state='DELETING' WHERE map_id=$1 AND generation=$2 AND state='LIVE'",[map_id.clone().into(),generation.into()])).await.map_err(db_error)?; + } + // One bounded resident batch per exact map generation. Large + // maps stay PENDING and are replayed before new map victims. + txn.execute_raw(stmt(&self.schema,"DELETE FROM mst2_chunk_map_leaf WHERE map_id=$1 AND page_index IN (SELECT page_index FROM mst2_chunk_map_leaf WHERE map_id=$1 ORDER BY page_index LIMIT 1024)",[map_id.clone().into()])).await.map_err(db_error)?; + txn.execute_raw(stmt(&self.schema,"DELETE FROM mst2_chunk_map_node WHERE (map_id,first_page,page_count) IN (SELECT map_id,first_page,page_count FROM mst2_chunk_map_node WHERE map_id=$1 ORDER BY first_page,page_count LIMIT 1024)",[map_id.clone().into()])).await.map_err(db_error)?; + let empty=txn.query_one_raw(stmt(&self.schema,"SELECT NOT EXISTS(SELECT 1 FROM mst2_chunk_map_leaf WHERE map_id=$1) AND NOT EXISTS(SELECT 1 FROM mst2_chunk_map_node WHERE map_id=$1) AS empty",[map_id.clone().into()])).await.map_err(db_error)?.ok_or_else(||integrity("map collection completion observation missing"))?.try_get::("","empty").map_err(db_error)?; + if empty { + // Deferred lifetime FK permits map deletion while its + // exact PENDING generation is still available to guards. + txn.execute_raw(stmt(&self.schema,"DELETE FROM mst2_chunk_map WHERE map_id=$1",[map_id.clone().into()])).await.map_err(db_error)?; + txn.execute_raw(stmt(&self.schema,"DELETE FROM mst2_chunk_map_lifetime WHERE map_id=$1 AND generation=$2 AND state='DELETING'",[map_id.clone().into(),generation.into()])).await.map_err(db_error)?; + txn.execute_raw(stmt(&self.schema,"UPDATE mst2_chunk_map_gc SET state='APPLIED',applied_at=pg_catalog.clock_timestamp() WHERE map_id=$1 AND generation=$2 AND state='PENDING'",[map_id.clone().into(),generation.into()])).await.map_err(db_error)?; + } + Ok(()) + }.await; + finish(txn, result).await?; + } + Ok(()) + } + + pub(crate) async fn admit_install( + &self, + source: &ChunkMapSource, + objects: &MegaObjectStorageWrapper, + ) -> Result { + // Replay first. Every backend await happens outside the primary + // completion barrier; active reservations count concurrent creates. + self.maintain_before_admission(objects).await?; + let inventory = inventory(self, objects).await?; + let source_bytes = source.canonical_bytes()?; + let identity = source_id(&self.primary_scope, &source_bytes); + let pages = (source.fact().size as u64) + .div_ceil(mst2_codec::chunkmap::CHUNK_SIZE as u64) + .div_ceil(CHUNKS_PER_PAGE as u64) as i32; + let nodes = pages * 2 - 1; + let reserved_bytes = pages as i64 * 12288 + nodes as i64 * 512 + 1024 * 1024; + let owner = Uuid::new_v4(); + let txn = self.transaction().await?; + let result=async { + self.require_scope(&txn).await?; + require_current_source(&txn,source,&self.schema).await?; + barrier(&txn,&self.schema).await?; + ensure_quotas(&txn,&self.schema,&inventory,pages,nodes,reserved_bytes,None).await?; + let row=txn.query_one_raw(stmt(&self.schema,"SELECT EXISTS(SELECT 1 FROM mst2_chunk_receipt_generation WHERE source_id=$1) AS used,EXISTS(SELECT 1 FROM mst2_chunk_receipt_generation WHERE source_id=$1 AND state IN ('RESERVED','CREATING','LIVE','DELETING')) AS occupied",[identity.to_vec().into()])).await.map_err(db_error)?.ok_or_else(|| integrity("chunk-map reservation observation missing"))?; + if row.try_get::("","occupied").map_err(db_error)? { return Err(unavailable("exact source already has a live or pending chunk-map generation")); } + let generation=if row.try_get::("","used").map_err(db_error)? { next_generation(&txn,&self.schema).await? } else { 1 }; + let key=physical_key(identity,generation); + let owner_started=Instant::now(); + txn.execute_raw(stmt(&self.schema,"INSERT INTO mst2_chunk_receipt_generation(receipt_key,source_id,generation,source_bytes,primary_scope,owner,state,reserved_pages,reserved_nodes,reserved_bytes,deadline) VALUES($1,$2,$3,$4,$5,$6::uuid,'RESERVED',$7,$8,$9,pg_catalog.clock_timestamp()+interval '59 seconds')",[key.key.clone().into(),identity.to_vec().into(),generation.into(),source_bytes.clone().into(),self.primary_scope.clone().into(),owner.to_string().into(),pages.into(),nodes.into(),reserved_bytes.into()])).await.map_err(db_error)?; + Ok(ChunkMapInstall{repository:self.clone(),source:source.clone(),owner,key,generation,progress:Mutex::new((Instant::now(),Instant::now(),0)),deadline:OwnerDeadline::from_sql_start(owner_started)}) + }.await; + // An uncertain commit is an error. Its UUID is never reused to mint + // another reservation; the durable deadline resolves any lost reply. + finish(txn, result).await + } + + pub(super) async fn admit_reader( + &self, + txn: &DatabaseTransaction, + source: &ChunkMapSource, + row: &QueryResult, + map: &ChunkMap, + ) -> Result { + barrier(txn, &self.schema).await?; + let key: String = row.try_get("", "receipt_key").map_err(db_error)?; + let generation: i64 = row.try_get("", "receipt_generation").map_err(db_error)?; + let map_generation: i64 = row + .try_get::>("", "map_generation") + .map_err(db_error)? + .ok_or_else(|| integrity("admitted chunk map lifetime is missing"))?; + let state: Option = row.try_get("", "generation_state").map_err(db_error)?; + let map_state: Option = row.try_get("", "map_state").map_err(db_error)?; + if state.as_deref() == Some("DELETING") { + return Err(unavailable("admitted chunk map generation is retiring")); + } + if state.as_deref() != Some("LIVE") || map_state.as_deref() != Some("LIVE") { + return Err(integrity( + "admitted source has a missing or terminal map lifetime", + )); + } + let identity = self.source_identity(source)?; + if key != physical_key(identity, generation).key + || bounded_bytes(row, "generation_receipt_bytes")? + != receipt(&self.primary_scope, source, map)? + { + return Err(integrity( + "chunk map lifetime and independent receipt identity disagree", + )); + } + txn.execute_raw(stmt(&self.schema,"DELETE FROM mst2_chunk_reader WHERE owner IN (SELECT owner FROM mst2_chunk_reader WHERE deadline<=pg_catalog.clock_timestamp() ORDER BY deadline,owner LIMIT 64)",[])).await.map_err(db_error)?; + let count = txn + .query_one_raw(stmt( + &self.schema, + "SELECT pg_catalog.count(*) AS readers FROM mst2_chunk_reader", + [], + )) + .await + .map_err(db_error)? + .ok_or_else(|| integrity("chunk reader count missing"))? + .try_get::("", "readers") + .map_err(db_error)?; + if count >= 4096 { + return Err(unavailable("chunk-map reader quota is occupied")); + } + let owner = Uuid::new_v4(); + let owner_started = Instant::now(); + txn.execute_raw(stmt(&self.schema,"INSERT INTO mst2_chunk_reader(owner,receipt_key,source_generation,map_id,map_generation,deadline) VALUES($1::uuid,$2,$3,$4,$5,pg_catalog.clock_timestamp()+interval '59 seconds')",[owner.to_string().into(),key.clone().into(),generation.into(),map.map_id().to_vec().into(),map_generation.into()])).await.map_err(db_error)?; + txn.execute_raw(stmt(&self.schema,"UPDATE mst2_chunk_map_lifetime SET last_used=pg_catalog.clock_timestamp() WHERE map_id=$1 AND generation=$2 AND state='LIVE'",[map.map_id().to_vec().into(),map_generation.into()])).await.map_err(db_error)?; + txn.execute_raw(stmt(&self.schema,"UPDATE mst2_chunk_receipt_generation SET last_progress=pg_catalog.clock_timestamp() WHERE receipt_key=$1 AND generation=$2 AND state='LIVE'",[key.clone().into(),generation.into()])).await.map_err(db_error)?; + Ok(ChunkMapReader { + repository: self.clone(), + source: source.clone(), + owner, + receipt_key: key, + generation, + map_id: map.map_id(), + map_generation, + confirmed: Mutex::new((Instant::now(), Instant::now())), + deadline: OwnerDeadline::from_sql_start(owner_started), + }) + } + + pub(super) async fn require_missing_source_retired( + &self, + txn: &DatabaseTransaction, + source: &ChunkMapSource, + ) -> Result { + let identity = self.source_identity(source)?; + let row=txn.query_one_raw(stmt(&self.schema,"SELECT state FROM mst2_chunk_receipt_generation WHERE source_id=$1 AND state IN ('LIVE','DELETING') ORDER BY generation DESC LIMIT 1",[identity.to_vec().into()])).await.map_err(db_error)?; + if let Some(row) = row { + if row.try_get::("", "state").map_err(db_error)? == "LIVE" { + return Err(integrity( + "live chunk-map source row is missing; never fall back to the cold body", + )); + } + return Ok(true); + } + Ok(false) + } +} + +pub(super) async fn barrier(txn: &DatabaseTransaction, schema: &str) -> Result<(), SnapshotError> { + txn.query_one_raw(stmt(schema, "SELECT mst2_chunk_retention_barrier()", [])) + .await + .map_err(db_error)?; + Ok(()) +} + +pub(super) async fn map_barrier( + txn: &DatabaseTransaction, + map_id: [u8; 32], +) -> Result<(), SnapshotError> { + // Per-map lock precedes the short global completion barrier. Collection + // never holds the global barrier while waiting for a map writer. + txn.query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT pg_catalog.pg_advisory_xact_lock(1296717364,pg_catalog.hashtext($1))", + [hex::encode(map_id).into()], + )) + .await + .map_err(db_error)?; + Ok(()) +} + +pub(super) async fn next_generation( + txn: &DatabaseTransaction, + schema: &str, +) -> Result { + txn.query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT pg_catalog.nextval($1::regclass) AS generation", + [format!("{}.mst2_chunk_generation_sequence", quoted(schema)).into()], + )) + .await + .map_err(db_error)? + .ok_or_else(|| integrity("chunk generation sequence missing"))? + .try_get("", "generation") + .map_err(db_error) +} + +fn physical_key(source: [u8; 32], generation: i64) -> ObjectKey { + ObjectKey { + namespace: ObjectNamespace::ChunkMapReceipt, + key: if generation == 1 { + hex::encode(source) + } else { + format!("{}-{generation:016x}", hex::encode(source)) + }, + } +} + +pub(super) async fn inventory( + repository: &PostgresChunkMapRepository, + objects: &MegaObjectStorageWrapper, +) -> Result, SnapshotError> { + let mut cached = repository.receipt_observation.lock().await; + if let Some(observation) = cached.as_ref().filter(|observation| { + Arc::ptr_eq(&observation.backing.inner, &objects.inner) + && observation.captured.elapsed() < RENEW_INTERVAL + }) { + return Ok(observation.clone()); + } + let memory = retention_budget().reserve(4 * 1024 * 1024)?; + let txn = repository.transaction().await?; + repository.require_scope(&txn).await?; + let started_at = txn + .query_one_raw(Statement::from_string( + DbBackend::Postgres, + "SELECT pg_catalog.clock_timestamp()::text AS started_at", + )) + .await + .map_err(db_error)? + .ok_or_else(|| integrity("backing inventory database clock missing"))? + .try_get("", "started_at") + .map_err(db_error)?; + txn.commit().await.map_err(db_error)?; + let inventory = tokio::time::timeout( + Duration::from_secs(30), + objects.inner.chunk_map_receipt_inventory(), + ) + .await + .map_err(|_| unavailable("bounded chunk receipt inventory timed out"))? + .map_err(|error| match error { + IoOrbitError::ChunkMapRetentionUnsupported => SnapshotError::new( + SnapshotErrorCode::TemporaryUnavailable, + "backing store does not support bounded chunk-map retention", + ), + IoOrbitError::ChunkMapRetentionCapacityExceeded => SnapshotError::new( + SnapshotErrorCode::LimitExceeded, + "actual backing chunk-map receipts exceed retention quota", + ), + error => storage_error(error), + })?; + let observation = Arc::new(BackingObservation { + inventory, + started_at, + captured: Instant::now(), + backing: objects.clone(), + _memory: memory, + }); + *cached = Some(observation.clone()); + Ok(observation) +} + +pub(super) async fn ensure_quotas( + txn: &DatabaseTransaction, + schema: &str, + observation: &BackingObservation, + pages: i32, + nodes: i32, + bytes: i64, + exclude: Option, +) -> Result<(), SnapshotError> { + let inventory = &observation.inventory; + let row=txn.query_one_raw(stmt(schema,"SELECT (SELECT pg_catalog.count(*) FROM mst2_chunk_map) AS maps,(SELECT pg_catalog.count(*) FROM mst2_chunk_map_source) AS sources,(SELECT pg_catalog.count(*) FROM mst2_chunk_map_leaf) AS leaves,(SELECT pg_catalog.count(*) FROM mst2_chunk_map_node) AS nodes,(SELECT pg_catalog.count(*) FROM mst2_chunk_receipt_generation) AS history,(SELECT pg_catalog.count(*) FROM mst2_chunk_map_gc) AS map_history,(SELECT pg_catalog.count(*) FROM mst2_chunk_receipt_generation WHERE receipt_bytes IS NOT NULL AND NOT create_completed) AS uncertain,(SELECT pg_catalog.count(*) FROM mst2_chunk_receipt_generation WHERE receipt_bytes IS NOT NULL AND (created_at>=$2::timestamptz OR create_completed_at>=$2::timestamptz)) AS recent,pg_catalog.count(*) AS pending,COALESCE(pg_catalog.sum(reserved_pages),0)::bigint AS pending_pages,COALESCE(pg_catalog.sum(reserved_nodes),0)::bigint AS pending_nodes,COALESCE(pg_catalog.sum(reserved_bytes),0)::bigint AS pending_bytes FROM mst2_chunk_receipt_generation WHERE state IN ('RESERVED','CREATING') AND ($1::uuid IS NULL OR owner IS DISTINCT FROM $1::uuid)",[exclude.map(|id|id.to_string()).into(),observation.started_at.clone().into()])).await.map_err(db_error)?.ok_or_else(|| integrity("chunk-map actual quota observation missing"))?; + let get = |name: &str| row.try_get::("", name).map_err(db_error); + let pending = get("pending")?; + let relation_bytes=txn.query_one_raw(Statement::from_sql_and_values(DbBackend::Postgres,"SELECT COALESCE(pg_catalog.sum(pg_catalog.pg_total_relation_size(c.oid)),0)::bigint AS bytes FROM pg_catalog.pg_class c JOIN pg_catalog.pg_namespace n ON n.oid=c.relnamespace WHERE n.nspname=$1 AND c.relkind='r' AND c.relname IN ('mst2_chunk_map','mst2_chunk_map_source','mst2_chunk_map_leaf','mst2_chunk_map_node','mst2_chunk_receipt_generation','mst2_chunk_map_lifetime','mst2_chunk_reader','mst2_chunk_map_gc')",[schema.into()])).await.map_err(db_error)?.ok_or_else(||integrity("chunk relation byte observation missing"))?.try_get::("","bytes").map_err(db_error)?; + let extra = if exclude.is_none() { 1 } else { 0 }; + if get("maps")? + pending + extra > 4096 + || get("sources")? + pending + extra > 16384 + || get("leaves")? + get("pending_pages")? + pages as i64 > 65536 + || get("nodes")? + get("pending_nodes")? + nodes as i64 > 131072 + || relation_bytes + get("pending_bytes")? + bytes + 1024 * 1024 > 536870912 + || get("history")? + extra > HISTORY_CAP + || get("map_history")? > HISTORY_CAP + || pending + extra > 64 + || inventory.objects.len() as i64 + pending + get("uncertain")? + get("recent")? + extra + > 16384 + || inventory.bytes + ((pending + get("uncertain")? + get("recent")? + extra) as u64) * 4096 + > 64 * 1024 * 1024 + { + return Err(SnapshotError::new( + SnapshotErrorCode::LimitExceeded, + "actual chunk-map rows, history, backing or reserved capacity exceed retention quota", + )); + } + Ok(()) +} + +fn unavailable(message: &str) -> SnapshotError { + SnapshotError::new(SnapshotErrorCode::TemporaryUnavailable, message) +} +fn expired(message: &str) -> SnapshotError { + SnapshotError::new(SnapshotErrorCode::LeaseExpired, message) +} + +async fn read_bounded_orphan( + objects: &MegaObjectStorageWrapper, + key: &ObjectKey, +) -> Result>, SnapshotError> { + tokio::time::timeout(Duration::from_secs(5), async { + let (mut stream, meta) = match objects.inner.get_stream(key).await { + Ok(value) => value, + Err(error) if error.is_not_found() => return Ok(None), + Err(error) => return Err(storage_error(error)), + }; + if !(129..=4096).contains(&meta.size) { + return Err(integrity("orphan receipt exceeds its byte profile")); + } + let mut bytes = Vec::with_capacity(meta.size as usize); + while let Some(part) = stream.next().await { + let part = part.map_err(storage_error)?; + if part.len() > meta.size as usize - bytes.len() { + return Err(integrity( + "orphan receipt exceeds its advertised exact size", + )); + } + bytes.extend_from_slice(&part); + } + if bytes.len() != meta.size as usize { + return Err(integrity("orphan receipt is truncated")); + } + Ok(Some(bytes)) + }) + .await + .map_err(|_| unavailable("orphan receipt read timed out"))? +} + +fn decode_orphan( + key: &ObjectKey, + bytes: &[u8], +) -> Result<(Vec, Vec, ChunkMap, i64), SnapshotError> { + if !bytes.starts_with(RECEIPT_DOMAIN) { + return Err(integrity("orphan is not a canonical trusted receipt")); + } + let mut offset = RECEIPT_DOMAIN.len(); + let take = |offset: &mut usize, max: usize| -> Result, SnapshotError> { + let len_bytes = bytes + .get(*offset..*offset + 4) + .ok_or_else(|| integrity("orphan receipt length is truncated"))?; + let len = u32::from_be_bytes( + len_bytes + .try_into() + .map_err(|_| integrity("invalid orphan receipt length"))?, + ) as usize; + *offset += 4; + if len == 0 || len > max { + return Err(integrity("orphan receipt field exceeds its profile")); + } + let field = bytes + .get(*offset..*offset + len) + .ok_or_else(|| integrity("orphan receipt field is truncated"))? + .to_vec(); + *offset += len; + Ok(field) + }; + let scope = take(&mut offset, 1024)?; + let source = take(&mut offset, 2048)?; + let descriptor = bytes + .get(offset..) + .ok_or_else(|| integrity("orphan map descriptor is missing"))?; + let map = ChunkMap::decode(descriptor) + .map_err(|_| integrity("orphan map descriptor is not canonical MCM2"))?; + if map.encode() != descriptor { + return Err(integrity("orphan receipt has noncanonical map bytes")); + } + let generation = if key.key.len() == 64 { + 1 + } else if key.key.len() == 81 && key.key.as_bytes()[64] == b'-' { + i64::from_str_radix(&key.key[65..], 16) + .map_err(|_| integrity("orphan physical generation is invalid"))? + } else { + return Err(integrity( + "orphan receipt key has invalid generation profile", + )); + }; + if generation <= 0 || physical_key(source_id(&scope, &source), generation) != *key { + return Err(integrity( + "orphan physical receipt key disagrees with its exact body", + )); + } + Ok((scope, source, map, generation)) +} diff --git a/src/jupiter/storage/object_storage.rs b/src/jupiter/storage/object_storage.rs index f65a33ab..3fbe45d8 100644 --- a/src/jupiter/storage/object_storage.rs +++ b/src/jupiter/storage/object_storage.rs @@ -278,6 +278,56 @@ impl MegaObjectStorage for InMemoryObjectStorage { Ok((Self::stream_bytes(bytes), meta)) } + async fn chunk_map_receipt_inventory( + &self, + ) -> OrbitResult { + use crate::orbit_api::object_storage::{ + ChunkMapReceiptInventory, MAX_CHUNK_MAP_RECEIPT_BYTES, MAX_CHUNK_MAP_RECEIPTS, + ObjectNamespace, + }; + let objects = self + .objects + .lock() + .map_err(|_| IoOrbitError::Other("object storage lock poisoned".into()))?; + let mut inventory = ChunkMapReceiptInventory { + objects: Vec::new(), + bytes: 0, + }; + for (key, (bytes, _)) in objects + .iter() + .filter(|(key, _)| key.namespace == ObjectNamespace::ChunkMapReceipt) + { + if inventory.objects.len() == MAX_CHUNK_MAP_RECEIPTS { + return Err(IoOrbitError::ChunkMapRetentionCapacityExceeded); + } + inventory.bytes = inventory + .bytes + .checked_add(bytes.len() as u64) + .filter(|bytes| *bytes <= MAX_CHUNK_MAP_RECEIPT_BYTES) + .ok_or(IoOrbitError::ChunkMapRetentionCapacityExceeded)?; + inventory.objects.push((key.clone(), bytes.len() as u64)); + } + Ok(inventory) + } + + async fn delete_chunk_map_receipt( + &self, + authority: &crate::orbit_api::object_storage::ChunkMapReceiptDeletion, + ) -> OrbitResult { + let mut objects = self + .objects + .lock() + .map_err(|_| IoOrbitError::Other("object storage lock poisoned".into()))?; + let Some((bytes, _)) = objects.get(authority.key()) else { + return Ok(false); + }; + if bytes.as_ref() != authority.expected_bytes() { + return Err(IoOrbitError::Other("retired receipt body changed".into())); + } + objects.remove(authority.key()); + Ok(true) + } + async fn get_range_stream( &self, key: &ObjectKey, diff --git a/src/orbit/adapter/object.rs b/src/orbit/adapter/object.rs index 87327a39..4ae923b0 100644 --- a/src/orbit/adapter/object.rs +++ b/src/orbit/adapter/object.rs @@ -104,6 +104,111 @@ impl MegaObjectStorage for ObjectStoreAdapter { Ok((Box::pin(stream), meta)) } + async fn chunk_map_receipt_inventory( + &self, + ) -> OrbitResult { + use crate::orbit_api::object_storage::{ + ChunkMapReceiptInventory, MAX_CHUNK_MAP_RECEIPT_BYTES, MAX_CHUNK_MAP_RECEIPTS, + }; + // object_store 0.14 cannot inventory/delete retained cloud versions. + // A current-object listing is not a physical backing-byte quota. + if !matches!(&self.store, BackendStore::Local(_)) { + return Err(IoOrbitError::ChunkMapRetentionUnsupported); + } + let prefix = object_store::path::Path::from("chunk-map-receipt"); + let mut listing = self.to_store().list(Some(&prefix)); + let mut inventory = ChunkMapReceiptInventory { + objects: Vec::new(), + bytes: 0, + }; + while let Some(entry) = listing.next().await { + let meta = entry.map_err(IoOrbitError::from)?; + if inventory.objects.len() == MAX_CHUNK_MAP_RECEIPTS { + return Err(IoOrbitError::ChunkMapRetentionCapacityExceeded); + } + inventory.bytes = inventory + .bytes + .checked_add(meta.size) + .filter(|bytes| *bytes <= MAX_CHUNK_MAP_RECEIPT_BYTES) + .ok_or(IoOrbitError::ChunkMapRetentionCapacityExceeded)?; + let location = meta.location.as_ref(); + let parts: Vec<_> = location.split('/').collect(); + if parts.len() != 5 + || parts[0] != "chunk-map-receipt" + || parts[1..4].iter().any(|part| part.len() != 2) + || parts[4].len() > 128 + { + return Err(IoOrbitError::Other( + "invalid physical chunk-map receipt path".into(), + )); + } + let key = ObjectKey { + namespace: ObjectNamespace::ChunkMapReceipt, + key: parts[1..].concat(), + }; + key.validate()?; + if key.default_sharding() != location { + return Err(IoOrbitError::Other( + "noncanonical physical chunk-map receipt path".into(), + )); + } + inventory.objects.push((key, meta.size)); + } + Ok(inventory) + } + + async fn delete_chunk_map_receipt( + &self, + authority: &crate::orbit_api::object_storage::ChunkMapReceiptDeletion, + ) -> OrbitResult { + if !matches!(&self.store, BackendStore::Local(_)) { + return Err(IoOrbitError::ChunkMapRetentionUnsupported); + } + let key = authority.key(); + if key.namespace != ObjectNamespace::ChunkMapReceipt { + return Err(IoOrbitError::Other( + "invalid sealed receipt namespace".into(), + )); + } + let (mut stream, meta) = match self.get_stream(key).await { + Ok(object) => object, + Err(error) if error.is_not_found() => return Ok(false), + Err(error) => return Err(error), + }; + let expected = authority.expected_bytes(); + if meta.size != expected.len() as i64 { + return Err(IoOrbitError::Other( + "retired receipt body size changed".into(), + )); + } + let mut offset: usize = 0; + while let Some(part) = stream.next().await { + let bytes = part?; + let end = offset + .checked_add(bytes.len()) + .filter(|end| *end <= expected.len()) + .ok_or_else(|| { + IoOrbitError::Other("retired receipt body exceeds its exact profile".into()) + })?; + if expected[offset..end] != bytes[..] { + return Err(IoOrbitError::Other("retired receipt body changed".into())); + } + offset = end; + } + if offset != expected.len() { + return Err(IoOrbitError::Other( + "retired receipt body is truncated".into(), + )); + } + // Physical generation keys are never reused. A late create can only + // restore this same retired body; persistent history will reconcile it. + match self.to_store().delete(&Self::checked_path(key)?).await { + Ok(()) => Ok(true), + Err(object_store::Error::NotFound { .. }) => Ok(false), + Err(error) => Err(IoOrbitError::from(error)), + } + } + async fn get_range_stream( &self, key: &ObjectKey, diff --git a/src/orbit_api/error.rs b/src/orbit_api/error.rs index 365d2e4d..9de51a0c 100644 --- a/src/orbit_api/error.rs +++ b/src/orbit_api/error.rs @@ -17,6 +17,12 @@ pub enum IoOrbitError { #[error("write manifest precondition failed")] WriteManifestPreconditionFailed, + #[error("chunk-map retention is not supported by this storage backend")] + ChunkMapRetentionUnsupported, + + #[error("chunk-map backing receipts exceed the fixed retention quota")] + ChunkMapRetentionCapacityExceeded, + #[error("other error: {0}")] Other(String), } diff --git a/src/orbit_api/object_storage.rs b/src/orbit_api/object_storage.rs index 90539315..9264d944 100644 --- a/src/orbit_api/object_storage.rs +++ b/src/orbit_api/object_storage.rs @@ -171,6 +171,39 @@ pub type ObjectByteStream = Pin, + pub bytes: u64, +} + +/// Sealed by the primary retention repository after claiming one exact +/// retired generation and independently reading its immutable backend body. +/// Ordinary object callers cannot construct a receipt deletion authority. +pub struct ChunkMapReceiptDeletion { + claim: crate::jupiter::storage::native_chunk_map::retention::VerifiedReceiptDeletionClaim, +} + +impl ChunkMapReceiptDeletion { + pub(crate) fn from_claim( + claim: crate::jupiter::storage::native_chunk_map::retention::VerifiedReceiptDeletionClaim, + ) -> Self { + Self { claim } + } + + pub(crate) fn key(&self) -> &ObjectKey { + self.claim.key() + } + + pub(crate) fn expected_bytes(&self) -> &[u8] { + self.claim.expected_bytes() + } +} + /// A streaming source of multiple objects. /// /// Each item yields: @@ -278,6 +311,22 @@ pub trait MegaObjectStorage: Send + Sync { )) } + /// Enumerate the real receipt namespace with fixed count/byte bounds. + /// Unsupported stores fail before a cold source body can be opened. + async fn chunk_map_receipt_inventory(&self) -> OrbitResult { + Err(IoOrbitError::ChunkMapRetentionUnsupported) + } + + /// Delete only the independently checked body of a sealed retired + /// physical generation. The key is never reused by a later installation. + /// Missing is idempotent; transport/authentication errors remain errors. + async fn delete_chunk_map_receipt( + &self, + _authority: &ChunkMapReceiptDeletion, + ) -> OrbitResult { + Err(IoOrbitError::ChunkMapRetentionUnsupported) + } + /// Retrieve a single object from the storage backend. /// /// # Returns diff --git a/src/server/http_server.rs b/src/server/http_server.rs index 04be7f77..777e732a 100644 --- a/src/server/http_server.rs +++ b/src/server/http_server.rs @@ -380,6 +380,35 @@ fn spawn_view_worker_task( spawn_view_worker_with_round(storage, token, production_round()) } +fn spawn_chunk_map_retention_task( + storage: Storage, + token: CancellationToken, + available: bool, +) -> Option> { + if !available { + return None; + } + Some(tokio::spawn(async move { + let mut interval = tokio::time::interval(std::time::Duration::from_secs(30)); + interval.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Skip); + loop { + tokio::select! { + _=token.cancelled()=>break, + _=interval.tick()=>{ + // Cancellation drops the real maintenance future. Claims + // are durable, so interrupted backend operations replay. + tokio::select! { + _=token.cancelled()=>break, + result=async { + storage.chunk_maps().await?.maintain(&storage.git_service.obj_storage,64).await + }=>if let Err(error)=result {tracing::warn!(?error,"bounded chunk-map retention tick failed");} + } + } + } + } + })) +} + /// Returns a future that completes when the cancellation token is triggered. async fn shutdown_signal(token: CancellationToken) { token.cancelled().await; @@ -450,6 +479,46 @@ pub async fn start_http(ctx: AppContext, options: CommonHttpOptions) -> MegaResu // TP-15 / 4.1 ②③⑥: open CLs, last_policy vs non-terminal rows, watermark reset. ctx.storage.prepare_push_policy_startup().await?; ctx.storage.check_view_reserved_startup().await?; + let chunk_map_retention_available = if ctx.storage.config().mst2.enabled { + // Detect the actual backend capability. Unsupported CHUNK retention + // must not disable other snapshot objects or generic HTTP serving. + match tokio::time::timeout( + std::time::Duration::from_secs(30), + ctx.storage + .git_service + .obj_storage + .inner + .chunk_map_receipt_inventory(), + ) + .await + { + Ok(Err(crate::orbit_api::error::IoOrbitError::ChunkMapRetentionUnsupported)) => { + tracing::info!( + "CHUNK cold admission and maintenance unavailable for this backing store; other snapshot routes remain available" + ); + false + } + _ => { + let result = async { + ctx.storage + .chunk_maps() + .await? + .maintain(&ctx.storage.git_service.obj_storage, 64) + .await + } + .await; + if let Err(error) = result { + tracing::warn!( + ?error, + "initial CHUNK retention check failed; cold admission remains fail closed" + ); + } + true + } + } + } else { + false + }; // First-build the shared authorization snapshot before the listener binds // (UN-02). `off` is a no-op; a failed first build fails startup. @@ -479,6 +548,11 @@ pub async fn start_http(ctx: AppContext, options: CommonHttpOptions) -> MegaResu spawn_artifact_gc_task(shutdown_context.clone(), shutdown_token.clone())?; let view_worker_handle = spawn_view_worker_task(shutdown_context.storage.clone(), shutdown_token.clone())?; + let chunk_map_retention_handle = spawn_chunk_map_retention_task( + shutdown_context.storage.clone(), + shutdown_token.clone(), + chunk_map_retention_available, + ); let notification_shutdown = shutdown_context.notification_shutdown.clone(); let server_token = shutdown_token.clone(); @@ -541,7 +615,7 @@ pub async fn start_http(ctx: AppContext, options: CommonHttpOptions) -> MegaResu }, }; - let (cleanup_result, artifact_gc_result, view_worker_result) = tokio::join!( + let (cleanup_result, artifact_gc_result, view_worker_result, chunk_map_retention_result) = tokio::join!( async { if let Some(handle) = cleanup_handle { match tokio::time::timeout(std::time::Duration::from_secs(30), handle).await { @@ -612,11 +686,34 @@ pub async fn start_http(ctx: AppContext, options: CommonHttpOptions) -> MegaResu } else { Ok(()) } + }, + async { + if let Some(mut handle) = chunk_map_retention_handle { + match tokio::time::timeout(std::time::Duration::from_secs(30), &mut handle).await { + Ok(Ok(())) => Ok(()), + Ok(Err(error)) => { + tracing::error!(%error,"chunk-map retention task panicked"); + Err(()) + } + Err(_) => { + handle.abort(); + let _ = handle.await; + tracing::error!( + "chunk-map retention task exceeded shutdown deadline and was aborted" + ); + Err(()) + } + } + } else { + Ok(()) + } } ); - let shutdown_failed = - cleanup_result.is_err() || artifact_gc_result.is_err() || view_worker_result.is_err(); + let shutdown_failed = cleanup_result.is_err() + || artifact_gc_result.is_err() + || view_worker_result.is_err() + || chunk_map_retention_result.is_err(); match (shutdown_failed, &server_result) { (false, Ok(())) => { tracing::info!("Graceful shutdown completed successfully"); From d89434c87fbf875a56922d24ca9144d8691ba850 Mon Sep 17 00:00:00 2001 From: Xiaoyang Han Date: Thu, 8 Oct 2026 08:52:42 +0800 Subject: [PATCH 39/50] Stream v3 rooted facts and report actual backend hydration capabilities (#92) Missing rooted blob facts now hash a complete source stream under fixed consumer credit, removing full-file materialization while preserving exact size, digest and EOF checks before persistence. Local and memory backend streams split visible items lazily without copying; capability discovery reports raw, chunk and full hydration support from the actual backend without IO. Eleven added test sources retain previous assertions and the actual merged chunk-owner deadline work. Seven owned Rust format/parse and canonical whitespace checks pass; native/database/performance execution remains unrun. --- src/api/router/snapshot_content_tests.rs | 260 +++++++++++ src/api/router/snapshot_router.rs | 27 +- .../snapshot/rooted_metadata_projection.rs | 415 +++++++++++++++++- src/jupiter/storage/object_storage.rs | 8 +- src/orbit/adapter/object.rs | 98 ++++- src/orbit_api/factory.rs | 4 + src/orbit_api/object_storage.rs | 105 +++++ 7 files changed, 879 insertions(+), 38 deletions(-) diff --git a/src/api/router/snapshot_content_tests.rs b/src/api/router/snapshot_content_tests.rs index 052c8369..aa8f92bc 100644 --- a/src/api/router/snapshot_content_tests.rs +++ b/src/api/router/snapshot_content_tests.rs @@ -131,6 +131,22 @@ async fn await_read_open_hold(hold: Option) { release.notified().await; } } + +struct NoRootedReuse; + +#[async_trait::async_trait] +impl crate::ceres::snapshot::rooted_metadata_projection::RootedReuseLookup for NoRootedReuse { + async fn lookup_reuse( + &self, + _tree_oid: &str, + _identity: &crate::ceres::snapshot::metadata_install::MetadataInstallIdentity, + ) -> Result< + Option, + crate::ceres::snapshot::error::SnapshotError, + > { + Ok(None) + } +} #[path = "snapshot_rooted_metadata_tests.rs"] mod rooted_metadata; @@ -184,6 +200,14 @@ struct CountingStorage { #[async_trait::async_trait] impl MegaObjectStorage for CountingStorage { + fn supports_chunk_map_retention(&self) -> bool { + !self + .counts + .receipt_retention_unsupported + .load(Ordering::SeqCst) + && self.inner.inner.supports_chunk_map_retention() + } + async fn chunk_map_receipt_inventory( &self, ) -> OrbitResult { @@ -1018,6 +1042,242 @@ async fn error(response: Response, status: u16, code: &str, retryable: bool) { assert_eq!(value["error"]["request_id"], request_id); } +#[tokio::test] +async fn mst2_rooted_streamed_fact_persists_one_source_pass_and_reuses_warm_alias_facts() { + use crate::ceres::snapshot::rooted_metadata_projection::prepare_rooted_native_metadata; + + let fixture = Fixture::new().await; + let handler = MonoApiService::from(&fixture.state); + let mut raw = vec![0xff; crate::orbit_api::object_storage::OBJECT_STREAM_ITEM_BYTES + 113]; + let prefix = format!("blob 3\0abc\0{}", uuid::Uuid::new_v4()); + raw[..prefix.len()].copy_from_slice(prefix.as_bytes()); + let oid = fixture + .state + .storage + .git_service + .save_object_from_raw(Bytes::copy_from_slice(&raw)) + .await + .unwrap(); + let source = ObjectHash::from_hex_for_kind(HashKind::Sha1, &oid).unwrap(); + let root = tree(vec![ + item(TreeItemMode::Blob, source, "alias"), + item(TreeItemMode::Blob, source, "file"), + ]); + let mono = fixture.state.storage.mono_storage(); + assert!( + mono.get_verified_blobs(vec![oid.clone()]) + .await + .unwrap() + .is_empty() + ); + fixture.counts.reset(); + let first = prepare_rooted_native_metadata(&handler, &root, "/", &NoRootedReuse) + .await + .unwrap(); + fixture.counts.assert(1, raw.len()); + assert_eq!(first.work.blob_fetches, 1); + assert_eq!(first.work.raw_bytes_fetched, raw.len() as u64); + assert_eq!(first.work.raw_bytes_hashed, raw.len() as u64); + assert_eq!(first.work.verified_blob_persistence_batches, 1); + let stored = mono.get_verified_blobs(vec![oid.clone()]).await.unwrap(); + assert_eq!(stored[&oid].size, raw.len() as i64); + assert_eq!(stored[&oid].raw_sha256, Sha256::digest(&raw).to_vec()); + fixture.counts.reset(); + let warm = prepare_rooted_native_metadata(&handler, &root, "/", &NoRootedReuse) + .await + .unwrap(); + fixture.counts.assert(0, 0); + assert_eq!(warm.plan.identity, first.plan.identity); + assert_eq!(warm.plan.root, first.plan.root); + assert_eq!(warm.work.blob_fetches, 0); + assert_eq!(warm.work.raw_bytes_fetched, 0); + assert_eq!(warm.work.raw_bytes_hashed, 0); + assert_eq!(warm.work.verified_blob_persistence_batches, 0); + raw.push(b'2'); + let next_oid = fixture + .state + .storage + .git_service + .save_object_from_raw(Bytes::copy_from_slice(&raw)) + .await + .unwrap(); + let next_source = ObjectHash::from_hex_for_kind(HashKind::Sha1, &next_oid).unwrap(); + let next_root = tree(vec![ + item(TreeItemMode::Blob, source, "alias"), + item(TreeItemMode::Blob, next_source, "file"), + ]); + fixture.counts.reset(); + let next = prepare_rooted_native_metadata(&handler, &next_root, "/", &NoRootedReuse) + .await + .unwrap(); + fixture.counts.assert(1, raw.len()); + assert_ne!( + next.plan.identity.tagged_root_tree_oid, + first.plan.identity.tagged_root_tree_oid + ); + assert_ne!(next.plan.root, first.plan.root); + assert_eq!(next.work.blob_fetches, 1); + assert_eq!(next.work.raw_bytes_fetched, raw.len() as u64); + assert_eq!(next.work.raw_bytes_hashed, raw.len() as u64); + assert_eq!(next.work.verified_blob_persistence_batches, 1); + assert_eq!(next.work.verified_blob_facts_loaded, 2); +} + +#[tokio::test] +async fn mst2_rooted_streamed_fact_never_persists_truncated_or_late_failed_sources() { + use crate::ceres::snapshot::rooted_metadata_projection::prepare_rooted_native_metadata; + + let fixture = Fixture::new().await; + let handler = MonoApiService::from(&fixture.state); + for late_error in [false, true] { + let raw = Bytes::from(format!("raw\0{}", uuid::Uuid::new_v4())); + let oid = fixture + .state + .storage + .git_service + .save_object_from_raw(raw.clone()) + .await + .unwrap(); + let source = ObjectHash::from_hex_for_kind(HashKind::Sha1, &oid).unwrap(); + let root = tree(vec![item(TreeItemMode::Blob, source, "file")]); + *fixture.counts.object_fault.lock().unwrap() = Some(bounded_objects::StreamFault { + oid: oid.clone(), + kind: if late_error { + bounded_objects::FaultKind::LateError(raw.clone()) + } else { + bounded_objects::FaultKind::Parts(vec![raw.slice(..raw.len() - 1)]) + }, + }); + fixture.counts.reset(); + let error = prepare_rooted_native_metadata(&handler, &root, "/", &NoRootedReuse) + .await + .unwrap_err(); + assert_eq!( + error.code, + if late_error { + SnapshotErrorCode::ObjectUnavailable + } else { + SnapshotErrorCode::IntegrityError + } + ); + fixture + .counts + .assert(1, raw.len() - usize::from(!late_error)); + assert!( + fixture + .state + .storage + .mono_storage() + .get_verified_blobs(vec![oid]) + .await + .unwrap() + .is_empty() + ); + } +} + +#[tokio::test] +async fn mst2_capabilities_use_actual_backend_without_inventory_and_preserve_metadata_and_objects() +{ + let fixture = Fixture::new().await; + let inventory_before = fixture + .counts + .receipt_inventory_calls + .load(Ordering::SeqCst); + let get_capabilities = || async { + success_json( + fixture + .app + .clone() + .oneshot( + Request::builder() + .uri("/api/v2/snapshots/capabilities") + .body(Body::empty()) + .unwrap(), + ) + .await + .unwrap(), + ) + .await + }; + let supported = get_capabilities().await; + for feature in ["raw_blob", "small_objects", "chunk_reads", "full_hydration"] { + assert_eq!(supported["features"][feature], true); + } + fixture + .counts + .receipt_retention_unsupported + .store(true, Ordering::SeqCst); + let mut unsupported = supported.clone(); + for feature in ["raw_blob", "chunk_reads", "full_hydration"] { + unsupported["features"][feature] = json!(false); + } + for _ in 0..3 { + assert_eq!(get_capabilities().await, unsupported); + } + fixture.counts.assert(0, 0); + assert_eq!( + fixture + .counts + .receipt_inventory_calls + .load(Ordering::SeqCst), + inventory_before + ); + assert_eq!(fixture.counts.receipt_reads.load(Ordering::SeqCst), 0); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 0); + let head = fixture.send("HEAD", "blob?path=/file", Body::empty()).await; + assert_eq!(head.status(), 200); + assert_eq!( + head.headers()["content-length"], + fixture.raw.len().to_string() + ); + assert_eq!( + head.headers()["etag"], + format!("\"{}\"", fixture.digest_string()) + ); + assert!(to_bytes(head.into_body(), 1024).await.unwrap().is_empty()); + success_json(fixture.send("GET", "directory?path=/", Body::empty()).await).await; + fixture.counts.assert(0, 0); + for path in ["blob?path=/file", "chunk-map?path=/file"] { + error( + fixture.send("GET", path, Body::empty()).await, + 503, + "TEMPORARY_UNAVAILABLE", + true, + ) + .await; + } + fixture.counts.assert(0, 0); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 0); + let link_digest = digest(b"file"); + let request = json!({"items":[{"path":"/link", "expected_digest":format!("sha256:{}", hex_of(&link_digest))}], "encoding":"identity"}).to_string(); + let response = fixture + .send("POST", "objects", Body::from(request.clone())) + .await; + assert_eq!(response.status(), 200); + let wire = to_bytes(response.into_body(), 2 * 1024 * 1024) + .await + .unwrap(); + let frames = parse_stream(&wire).unwrap(); + let [Frame::Object(object), Frame::End(end)] = frames.as_slice() else { + panic!("expected OBJECT and terminal END"); + }; + assert_eq!(object.objects, vec![(link_digest, b"file".to_vec())]); + assert_eq!(end.request_item_count, 1); + assert_eq!(end.logical_bytes, 4); + assert_eq!( + end.request_body_sha256, + <[u8; 32]>::from(Sha256::digest(request.as_bytes())) + ); + fixture.counts.assert(1, 4); + fixture + .counts + .receipt_retention_unsupported + .store(false, Ordering::SeqCst); + assert_eq!(get_capabilities().await, supported); + fixture.counts.assert(1, 4); +} + #[tokio::test] async fn mst2_fixed_head_uses_verified_facts_without_body_reads_and_preserves_raw_bytes() { let fixture = Fixture::new().await; diff --git a/src/api/router/snapshot_router.rs b/src/api/router/snapshot_router.rs index 4b568aeb..cb593e72 100644 --- a/src/api/router/snapshot_router.rs +++ b/src/api/router/snapshot_router.rs @@ -35,6 +35,7 @@ use crate::{ runtime::{now_unix, runtime}, view::{SnapshotView, validate_scope_relative_path}, }, + orbit_api::factory::MegaObjectStorageWrapper, }; pub fn routers(api_state: MonoApiServiceState) -> Router { @@ -534,11 +535,14 @@ fn internal(e: E) -> SnapshotError { SnapshotError::new(SnapshotErrorCode::Internal, e.to_string()) } -async fn capabilities() -> Json { - // Keep this document on the canonical discovery contract consumed by the - // v3 reader. The full delivery path serves the complete metadata/object - // closure, so advertising `full_hydration` as false would make a typed - // resolve reject before it reaches this router. +async fn capabilities(State(state): State) -> Json { + capabilities_document(&state.storage.git_service.obj_storage) +} + +fn capabilities_document(backend: &MegaObjectStorageWrapper) -> Json { + // Full delivery must work from a cold source, including receipt admission. + // Existing warm receipts do not establish this backend-wide contract. + let full_delivery = backend.supports_chunk_map_retention(); Json(json!({ "protocol_versions": [2], "metadata_codecs": [1], @@ -548,10 +552,10 @@ async fn capabilities() -> Json { "directory": true, "lookup": true, "metadata_pages": true, - "raw_blob": true, + "raw_blob": full_delivery, "small_objects": true, - "chunk_reads": true, - "full_hydration": true, + "chunk_reads": full_delivery, + "full_hydration": full_delivery, "region_hints": false, "offline_export": false, }, @@ -1640,9 +1644,10 @@ async fn metadata_pages( mod tests { use super::*; - #[tokio::test] - async fn capabilities_advertise_the_canonical_full_delivery_contract() { - let Json(value) = capabilities().await; + #[test] + fn capabilities_advertise_the_canonical_full_delivery_contract() { + let backend = crate::jupiter::storage::object_storage::mock_object_storage(); + let Json(value) = capabilities_document(&backend); assert_eq!( value, json!({ diff --git a/src/ceres/snapshot/rooted_metadata_projection.rs b/src/ceres/snapshot/rooted_metadata_projection.rs index 086db637..b4b1a517 100644 --- a/src/ceres/snapshot/rooted_metadata_projection.rs +++ b/src/ceres/snapshot/rooted_metadata_projection.rs @@ -4,9 +4,12 @@ use std::{ borrow::Cow, collections::{BTreeMap, BTreeSet, VecDeque}, + sync::Arc, + time::Duration, }; use async_trait::async_trait; +use futures::StreamExt; use git_internal::{hash::ObjectHash, internal::object::tree::Tree}; use mst2_codec::{ descriptor::{ @@ -19,6 +22,7 @@ use sea_orm::ActiveValue::Set; use sha2::{Digest, Sha256}; use super::{ + content_budget::{MemoryBudget, RANGE_WORK_BYTES, projection_budget}, error::{SnapshotError, SnapshotErrorCode}, metadata_install::MetadataInstallIdentity, projection_observation::NATIVE_PROJECTION_REVISION, @@ -31,8 +35,12 @@ use crate::{ ceres::api_service::ApiHandler, common::errors::MegaError, jupiter::storage::{Storage, mono_storage::MST2_VERIFICATION_VERSION}, + orbit_api::object_storage::{ObjectByteStream, ObjectMeta}, }; +const MAX_VERIFIED_FILE_BYTES: u64 = 8_796_093_022_208; +const SOURCE_PROGRESS_TIMEOUT: Duration = Duration::from_secs(30); + #[async_trait] pub(crate) trait RootedReuseLookup: Send + Sync { async fn lookup_reuse( @@ -732,28 +740,12 @@ async fn project_directory 8_796_093_022_208 { - return Err(integrity( - "raw file exceeds the native verified-object size profile", - )); - } - let digest: [u8; 32] = Sha256::digest(&raw).into(); - state.work.raw_bytes_hashed = state - .work - .raw_bytes_hashed - .checked_add(size) - .ok_or_else(budget_limit)?; - drop(raw); - let fact = BlobFact { size, digest }; + let fact = stream_blob_fact_with_resources( + || handler.get_raw_blob_stream_with_meta(&raw_oid), + projection_budget(), + &mut state.work, + ) + .await?; state.blob_facts.insert(raw_oid.clone(), fact); pending_facts.push((raw_oid.clone(), fact)); if pending_facts.len() == 64 { @@ -795,12 +787,107 @@ async fn project_directory( + open: F, + budget: &Arc, + work: &mut RootedProjectionWork, +) -> Result +where + F: FnOnce() -> Fut, + Fut: std::future::Future>, +{ + // Credit bounds consumer-visible processing, not the backend's backing + // allocation or work surviving cancellation. No source bytes accumulate. + let _lease = budget.reserve(RANGE_WORK_BYTES)?; + work.blob_fetches = work.blob_fetches.checked_add(1).ok_or_else(budget_limit)?; + let (mut input, meta) = tokio::time::timeout(SOURCE_PROGRESS_TIMEOUT, open()) + .await + .map_err(|_| source_progress_timeout())? + .map_err(source_error)?; + let size = u64::try_from(meta.size) + .ok() + .filter(|size| *size <= MAX_VERIFIED_FILE_BYTES) + .ok_or_else(|| integrity("raw file exceeds the native verified-object size profile"))?; + let mut digest = Sha256::new(); + let mut received = 0u64; + let mut since_yield = 0usize; + let mut empty_parts = 0usize; + let mut progress_deadline = tokio::time::Instant::now() + SOURCE_PROGRESS_TIMEOUT; + loop { + if tokio::time::Instant::now() >= progress_deadline { + return Err(source_progress_timeout()); + } + let part = tokio::time::timeout_at(progress_deadline, input.next()) + .await + .map_err(|_| source_progress_timeout())?; + let Some(part) = part else { break }; + let bytes = part.map_err(|error| { + tracing::warn!(error = %error, "rooted metadata raw source stream failed"); + SnapshotError::new( + SnapshotErrorCode::ObjectUnavailable, + "raw source stream failed", + ) + })?; + if bytes.len() as u64 > size - received { + return Err(integrity( + "raw source length disagrees with its physical size", + )); + } + if bytes.len() > RANGE_WORK_BYTES { + return Err(SnapshotError::new( + SnapshotErrorCode::TemporaryUnavailable, + "raw source producer item exceeds the consumer processing limit", + )); + } + digest.update(&bytes); + received += bytes.len() as u64; + since_yield += bytes.len(); + if bytes.is_empty() { + empty_parts += 1; + } else { + empty_parts = 0; + progress_deadline = tokio::time::Instant::now() + SOURCE_PROGRESS_TIMEOUT; + } + if since_yield >= RANGE_WORK_BYTES || empty_parts == 32 { + tokio::task::yield_now().await; + since_yield = 0; + empty_parts = 0; + } + } + if received != size { + return Err(integrity( + "raw source length disagrees with its physical size", + )); + } + let fetched = work + .raw_bytes_fetched + .checked_add(size) + .ok_or_else(budget_limit)?; + let hashed = work + .raw_bytes_hashed + .checked_add(size) + .ok_or_else(budget_limit)?; + work.raw_bytes_fetched = fetched; + work.raw_bytes_hashed = hashed; + Ok(BlobFact { + size, + digest: digest.finalize().into(), + }) +} + +fn source_progress_timeout() -> SnapshotError { + SnapshotError::new( + SnapshotErrorCode::TemporaryUnavailable, + "rooted metadata raw source made no progress", + ) +} + fn verified_fact( fact: &crate::callisto::mst2_verified_object::Model, ) -> Result { if fact.state != "VERIFIED" || fact.verification_version != MST2_VERIFICATION_VERSION - || !(0..=8_796_093_022_208).contains(&fact.size) + || !(0..=MAX_VERIFIED_FILE_BYTES as i64).contains(&fact.size) { return Err(integrity( "native projection received an invalid current verified blob fact", @@ -894,11 +981,293 @@ fn path_limit() -> SnapshotError { #[cfg(test)] mod tests { + use std::{ + pin::Pin, + sync::atomic::{AtomicUsize, Ordering}, + task::{Context, Poll}, + }; + + use bytes::Bytes; use git_internal::hash::HashKind; + use tokio::sync::Notify; use uuid::Uuid; use super::*; + struct FactInput { + parts: VecDeque>, + polls: Arc, + drops: Arc, + hold_eof: Option>, + } + + impl futures::Stream for FactInput { + type Item = std::io::Result; + + fn poll_next(self: Pin<&mut Self>, _cx: &mut Context<'_>) -> Poll> { + let input = self.get_mut(); + input.polls.fetch_add(1, Ordering::SeqCst); + if let Some(part) = input.parts.pop_front() { + Poll::Ready(Some(part)) + } else if let Some(entered) = &input.hold_eof { + entered.notify_one(); + Poll::Pending + } else { + Poll::Ready(None) + } + } + } + + impl Drop for FactInput { + fn drop(&mut self) { + self.drops.fetch_add(1, Ordering::SeqCst); + } + } + + fn fact_input( + parts: Vec>, + hold_eof: Option>, + ) -> (ObjectByteStream, Arc, Arc) { + let polls = Arc::new(AtomicUsize::new(0)); + let drops = Arc::new(AtomicUsize::new(0)); + ( + Box::pin(FactInput { + parts: parts.into(), + polls: polls.clone(), + drops: drops.clone(), + hold_eof, + }), + polls, + drops, + ) + } + + #[tokio::test] + async fn streamed_cold_fact_keeps_large_memory_source_and_empty_raw_costs_exact() { + use crate::orbit_api::object_storage::{ObjectKey, ObjectNamespace}; + + for size in [0, RANGE_WORK_BYTES + 113] { + let backend = crate::jupiter::storage::object_storage::mock_object_storage(); + let key = ObjectKey { + namespace: ObjectNamespace::Git, + key: "ab".repeat(20), + }; + let mut raw = vec![0xff; size]; + if !raw.is_empty() { + raw[..11].copy_from_slice(b"blob 3\0abc\0"); + } + let expected: [u8; 32] = Sha256::digest(&raw).into(); + backend + .inner + .put_stream( + &key, + Box::pin(futures::stream::iter([Ok(Bytes::from(raw))])), + ObjectMeta::default(), + ) + .await + .unwrap(); + let budget = MemoryBudget::new(RANGE_WORK_BYTES); + let opens = AtomicUsize::new(0); + let mut work = RootedProjectionWork::default(); + let fact = stream_blob_fact_with_resources( + || async { + opens.fetch_add(1, Ordering::SeqCst); + backend + .inner + .get_stream(&key) + .await + .map_err(MegaError::from) + }, + &budget, + &mut work, + ) + .await + .unwrap(); + assert_eq!( + fact, + BlobFact { + size: size as u64, + digest: expected + } + ); + assert_eq!(opens.load(Ordering::SeqCst), 1); + assert_eq!(work.blob_fetches, 1); + assert_eq!(work.raw_bytes_fetched, size as u64); + assert_eq!(work.raw_bytes_hashed, size as u64); + assert_eq!(work.verified_blob_persistence_batches, 0); + assert_eq!(budget.used(), 0); + } + } + + #[tokio::test] + async fn streamed_cold_fact_rejects_physical_mismatch_and_late_error_before_fact_return() { + let raw = Bytes::from_static(b"raw"); + for (size, parts, code, polls_expected) in [ + ( + -1, + vec![Ok(raw.clone())], + SnapshotErrorCode::IntegrityError, + 0, + ), + ( + MAX_VERIFIED_FILE_BYTES as i64 + 1, + vec![Ok(raw.clone())], + SnapshotErrorCode::IntegrityError, + 0, + ), + ( + MAX_VERIFIED_FILE_BYTES as i64, + vec![], + SnapshotErrorCode::IntegrityError, + 1, + ), + ( + RANGE_WORK_BYTES as i64 + 1, + vec![Ok(Bytes::from(vec![7; RANGE_WORK_BYTES + 1]))], + SnapshotErrorCode::TemporaryUnavailable, + 1, + ), + ( + 2, + vec![Ok(raw.clone())], + SnapshotErrorCode::IntegrityError, + 1, + ), + ( + 4, + vec![Ok(raw.clone())], + SnapshotErrorCode::IntegrityError, + 2, + ), + ( + 3, + vec![Ok(raw.clone()), Ok(Bytes::from_static(b"x"))], + SnapshotErrorCode::IntegrityError, + 2, + ), + ( + 3, + vec![Ok(raw), Err(std::io::Error::other("late source error"))], + SnapshotErrorCode::ObjectUnavailable, + 2, + ), + ( + 0, + vec![Err(std::io::Error::other("empty source error"))], + SnapshotErrorCode::ObjectUnavailable, + 1, + ), + ] { + let (input, polls, drops) = fact_input(parts, None); + let budget = MemoryBudget::new(RANGE_WORK_BYTES); + let mut work = RootedProjectionWork::default(); + let error = stream_blob_fact_with_resources( + || async move { + Ok(( + input, + ObjectMeta { + size, + ..Default::default() + }, + )) + }, + &budget, + &mut work, + ) + .await + .unwrap_err(); + assert_eq!(error.code, code); + assert_eq!(polls.load(Ordering::SeqCst), polls_expected); + assert_eq!(drops.load(Ordering::SeqCst), 1); + assert_eq!(work.blob_fetches, 1); + assert_eq!(work.raw_bytes_fetched, 0); + assert_eq!(work.raw_bytes_hashed, 0); + assert_eq!(budget.used(), 0); + } + } + + #[tokio::test] + async fn streamed_cold_fact_cancellation_at_exact_size_drops_source_and_refunds_credit() { + let entered = Arc::new(Notify::new()); + let (input, polls, drops) = + fact_input(vec![Ok(Bytes::from_static(b"raw"))], Some(entered.clone())); + let budget = MemoryBudget::new(RANGE_WORK_BYTES); + let task_budget = budget.clone(); + let task = tokio::spawn(async move { + let mut work = RootedProjectionWork::default(); + stream_blob_fact_with_resources( + || async move { + Ok(( + input, + ObjectMeta { + size: 3, + ..Default::default() + }, + )) + }, + &task_budget, + &mut work, + ) + .await + }); + tokio::time::timeout(Duration::from_secs(2), entered.notified()) + .await + .unwrap(); + assert_eq!(polls.load(Ordering::SeqCst), 2); + assert_eq!(budget.used(), RANGE_WORK_BYTES); + assert_eq!(drops.load(Ordering::SeqCst), 0); + task.abort(); + assert!(task.await.unwrap_err().is_cancelled()); + assert_eq!(drops.load(Ordering::SeqCst), 1); + assert_eq!(budget.used(), 0); + } + + #[tokio::test] + async fn streamed_cold_fact_reserves_fixed_credit_before_open_and_preserves_open_errors() { + let budget = MemoryBudget::new(RANGE_WORK_BYTES); + let occupied = budget.reserve(RANGE_WORK_BYTES).unwrap(); + let mut work = RootedProjectionWork::default(); + let opens = AtomicUsize::new(0); + let error = stream_blob_fact_with_resources( + || async { + opens.fetch_add(1, Ordering::SeqCst); + Err(MegaError::Other("unexpected open".into())) + }, + &budget, + &mut work, + ) + .await + .unwrap_err(); + assert_eq!(error.code, SnapshotErrorCode::TemporaryUnavailable); + assert_eq!(opens.load(Ordering::SeqCst), 0); + assert_eq!(work, RootedProjectionWork::default()); + drop(occupied); + for (error, code) in [ + ( + MegaError::ObjStorageNotFound("missing".into()), + SnapshotErrorCode::ObjectUnavailable, + ), + ( + MegaError::ObjStorageInconsistent("inconsistent".into()), + SnapshotErrorCode::IntegrityError, + ), + ( + MegaError::Other("transport".into()), + SnapshotErrorCode::Internal, + ), + ] { + let mut work = RootedProjectionWork::default(); + let got = stream_blob_fact_with_resources(|| async { Err(error) }, &budget, &mut work) + .await + .unwrap_err(); + assert_eq!(got.code, code); + assert_eq!(work.blob_fetches, 1); + assert_eq!(work.raw_bytes_fetched, 0); + assert_eq!(work.raw_bytes_hashed, 0); + assert_eq!(budget.used(), 0); + } + } + fn state() -> ProjectionState { ProjectionState::new(MetadataInstallIdentity { source_domain: "native-git".into(), diff --git a/src/jupiter/storage/object_storage.rs b/src/jupiter/storage/object_storage.rs index 3fbe45d8..891232cf 100644 --- a/src/jupiter/storage/object_storage.rs +++ b/src/jupiter/storage/object_storage.rs @@ -199,7 +199,9 @@ impl InMemoryObjectStorage { } fn stream_bytes(bytes: Bytes) -> ObjectByteStream { - Box::pin(futures::stream::once(async move { Ok(bytes) })) + crate::orbit_api::object_storage::fragment_object_stream(Box::pin(futures::stream::once( + async move { Ok(bytes) }, + ))) } } @@ -209,6 +211,10 @@ pub fn mock_object_storage() -> MegaObjectStorageWrapper { #[async_trait::async_trait] impl MegaObjectStorage for InMemoryObjectStorage { + fn supports_chunk_map_retention(&self) -> bool { + true + } + async fn put_stream( &self, key: &ObjectKey, diff --git a/src/orbit/adapter/object.rs b/src/orbit/adapter/object.rs index 4ae923b0..14c3d175 100644 --- a/src/orbit/adapter/object.rs +++ b/src/orbit/adapter/object.rs @@ -6,6 +6,10 @@ impl MegaObjectStorage for ObjectStoreAdapter { matches!(&self.store, BackendStore::S3(_) | BackendStore::Gcs(_)) } + fn supports_chunk_map_retention(&self) -> bool { + matches!(&self.store, BackendStore::Local(_)) + } + async fn put_stream( &self, key: &ObjectKey, @@ -101,7 +105,10 @@ impl MegaObjectStorage for ObjectStoreAdapter { let meta = build_object_meta(&res.meta); let stream = res.into_stream().map_err(std::io::Error::other); - Ok((Box::pin(stream), meta)) + Ok(( + crate::orbit_api::object_storage::fragment_object_stream(Box::pin(stream)), + meta, + )) } async fn chunk_map_receipt_inventory( @@ -112,7 +119,7 @@ impl MegaObjectStorage for ObjectStoreAdapter { }; // object_store 0.14 cannot inventory/delete retained cloud versions. // A current-object listing is not a physical backing-byte quota. - if !matches!(&self.store, BackendStore::Local(_)) { + if !self.supports_chunk_map_retention() { return Err(IoOrbitError::ChunkMapRetentionUnsupported); } let prefix = object_store::path::Path::from("chunk-map-receipt"); @@ -161,7 +168,7 @@ impl MegaObjectStorage for ObjectStoreAdapter { &self, authority: &crate::orbit_api::object_storage::ChunkMapReceiptDeletion, ) -> OrbitResult { - if !matches!(&self.store, BackendStore::Local(_)) { + if !self.supports_chunk_map_retention() { return Err(IoOrbitError::ChunkMapRetentionUnsupported); } let key = authority.key(); @@ -341,6 +348,91 @@ pub(super) fn reject_receipt_mutation(key: &ObjectKey) -> OrbitResult<()> { mod exact_range_tests { use super::*; + #[tokio::test] + async fn backend_chunk_retention_capabilities_match_cold_admission_without_cloud_io() { + let s3 = object_store::aws::AmazonS3Builder::new() + .with_bucket_name("capabilities-test") + .with_region("us-east-1") + .with_access_key_id("fixture-access") + .with_secret_access_key("fixture-secret") + .with_skip_signature(true) + .build() + .unwrap(); + let gcs = object_store::gcp::GoogleCloudStorageBuilder::new() + .with_bucket_name("capabilities-test") + .with_credentials(Arc::new(object_store::StaticCredentialProvider::new( + object_store::gcp::GcpCredential { + bearer: "fixture-bearer".into(), + }, + ))) + .with_skip_signature(true) + .build() + .unwrap(); + for store in [ + BackendStore::S3(Arc::new(s3)), + BackendStore::Gcs(Arc::new(gcs)), + ] { + let backend = crate::orbit_api::factory::MegaObjectStorageWrapper::new(Arc::new( + ObjectStoreAdapter { + store, + upload_strategy: UploadStrategy::SinglePut, + presign_store: None, + }, + )); + assert!(!backend.supports_chunk_map_retention()); + assert!(matches!( + backend.inner.chunk_map_receipt_inventory().await, + Err(IoOrbitError::ChunkMapRetentionUnsupported) + )); + } + } + + #[tokio::test] + async fn local_whole_stream_keeps_large_file_size_and_digest_with_bounded_fragments() { + use sha2::{Digest, Sha256}; + + let directory = tempfile::tempdir().unwrap(); + let adapter = ObjectStoreAdapter { + store: BackendStore::Local(Arc::new( + LocalFileSystem::new_with_prefix(directory.path()).unwrap(), + )), + upload_strategy: UploadStrategy::SinglePut, + presign_store: None, + }; + let key = ObjectKey { + namespace: ObjectNamespace::Git, + key: "ab".repeat(20), + }; + let raw = Bytes::from(vec![ + 0xff; + crate::orbit_api::object_storage::OBJECT_STREAM_ITEM_BYTES + + 113 + ]); + let expected: [u8; 32] = Sha256::digest(&raw).into(); + let size = raw.len(); + adapter + .put_stream( + &key, + Box::pin(stream::iter([Ok(raw)])), + ObjectMeta::default(), + ) + .await + .unwrap(); + let (mut input, meta) = adapter.get_stream(&key).await.unwrap(); + assert_eq!(meta.size, size as i64); + assert!(adapter.supports_chunk_map_retention()); + let mut received = 0; + let mut hashed = Sha256::new(); + while let Some(part) = input.next().await { + let part = part.unwrap(); + assert!(part.len() <= crate::orbit_api::object_storage::OBJECT_STREAM_ITEM_BYTES); + received += part.len(); + hashed.update(&part); + } + assert_eq!(received, size); + assert_eq!(<[u8; 32]>::from(hashed.finalize()), expected); + } + #[tokio::test] async fn local_chunk_receipts_reject_all_mutation_routes() { let directory = tempfile::tempdir().unwrap(); diff --git a/src/orbit_api/factory.rs b/src/orbit_api/factory.rs index c1f8e86e..f6891a18 100644 --- a/src/orbit_api/factory.rs +++ b/src/orbit_api/factory.rs @@ -135,6 +135,10 @@ impl MegaObjectStorageWrapper { pub fn supports_presigned_urls(&self) -> bool { MegaObjectStorage::supports_presigned_urls(&*self.inner) } + + pub fn supports_chunk_map_retention(&self) -> bool { + MegaObjectStorage::supports_chunk_map_retention(&*self.inner) + } } #[cfg(test)] diff --git a/src/orbit_api/object_storage.rs b/src/orbit_api/object_storage.rs index 9264d944..e3898424 100644 --- a/src/orbit_api/object_storage.rs +++ b/src/orbit_api/object_storage.rs @@ -168,6 +168,30 @@ pub struct ObjectMeta { /// - The stream must be fully consumed by the caller. pub type ObjectByteStream = Pin> + Send>>; +pub(crate) const OBJECT_STREAM_ITEM_BYTES: usize = 8 * 1024 * 1024; + +/// Split visible items without copying or prebuilding a fragment list. Bytes +/// retain their original owner; this does not bound backend backing buffers. +pub(crate) fn fragment_object_stream(input: ObjectByteStream) -> ObjectByteStream { + Box::pin(futures::stream::unfold( + (input, Bytes::new()), + |(mut input, pending)| async move { + let next = if pending.is_empty() { + input.next().await? + } else { + Ok(pending) + }; + match next { + Ok(mut bytes) => { + let part = bytes.split_to(bytes.len().min(OBJECT_STREAM_ITEM_BYTES)); + Some((Ok(part), (input, bytes))) + } + Err(error) => Some((Err(error), (input, Bytes::new()))), + } + }, + )) +} + /// Upper bound for [`MegaObjectStorage::put_metadata_atomic`] (ADR-MF-05). pub const MAX_METADATA_ATOMIC_BYTES: usize = 1024 * 1024; @@ -232,6 +256,13 @@ pub trait MegaObjectStorage: Send + Sync { false } + /// Complete receipt inventory and sealed deletion are available without + /// uncounted retained versions. This is a static backend contract, not an + /// I/O health or quota check; cold requests still perform real admission. + fn supports_chunk_map_retention(&self) -> bool { + false + } + /// Upload a single object to the storage backend. /// /// # Parameters @@ -676,6 +707,80 @@ mod tests { struct UnsupportedBoundedStore; + struct SharedStreamOwner { + bytes: Vec, + drops: std::sync::Arc, + } + + impl AsRef<[u8]> for SharedStreamOwner { + fn as_ref(&self) -> &[u8] { + &self.bytes + } + } + + impl Drop for SharedStreamOwner { + fn drop(&mut self) { + self.drops.fetch_add(1, std::sync::atomic::Ordering::SeqCst); + } + } + + #[tokio::test] + async fn fragmented_source_is_lazy_zero_copy_and_keeps_original_owner_until_last_clone() { + use std::sync::{ + Arc, + atomic::{AtomicUsize, Ordering}, + }; + + let drops = Arc::new(AtomicUsize::new(0)); + let input_polls = Arc::new(AtomicUsize::new(0)); + let bytes = Bytes::from_owner(SharedStreamOwner { + bytes: vec![0xa5; 2 * OBJECT_STREAM_ITEM_BYTES + 17], + drops: drops.clone(), + }); + let pointer = bytes.as_ptr(); + let polls = input_polls.clone(); + let input = + futures::stream::iter([Ok(bytes), Err(std::io::Error::other("late backend error"))]) + .inspect(move |_| { + polls.fetch_add(1, Ordering::SeqCst); + }); + let mut fragmented = fragment_object_stream(Box::pin(input)); + assert_eq!(input_polls.load(Ordering::SeqCst), 0); + let first = fragmented.next().await.unwrap().unwrap(); + assert_eq!(first.len(), OBJECT_STREAM_ITEM_BYTES); + assert_eq!(first.as_ptr(), pointer); + let transport = first.clone(); + drop(first); + let second = fragmented.next().await.unwrap().unwrap(); + assert_eq!(second.len(), OBJECT_STREAM_ITEM_BYTES); + assert_eq!( + second.as_ptr() as usize, + pointer as usize + OBJECT_STREAM_ITEM_BYTES + ); + drop(second); + let tail = fragmented.next().await.unwrap().unwrap(); + assert_eq!(tail.as_ref(), &[0xa5; 17]); + drop(tail); + assert_eq!(input_polls.load(Ordering::SeqCst), 1); + assert!(fragmented.next().await.unwrap().is_err()); + assert_eq!(input_polls.load(Ordering::SeqCst), 2); + assert!(fragmented.next().await.is_none()); + drop(fragmented); + assert_eq!(drops.load(Ordering::SeqCst), 0); + drop(transport); + assert_eq!(drops.load(Ordering::SeqCst), 1); + } + + #[tokio::test] + async fn default_backend_does_not_advertise_or_infer_chunk_retention() { + let backend = UnsupportedBoundedStore; + assert!(!backend.supports_chunk_map_retention()); + assert!(matches!( + backend.chunk_map_receipt_inventory().await, + Err(IoOrbitError::ChunkMapRetentionUnsupported) + )); + } + #[async_trait::async_trait] impl MegaObjectStorage for UnsupportedBoundedStore { async fn put_stream( From 950423be73fcee9ba953fd31707dca5a9c5c3e8b Mon Sep 17 00:00:00 2001 From: Xiaoyang Han Date: Thu, 8 Oct 2026 09:05:59 +0800 Subject: [PATCH 40/50] Fix rooted plan SQL column bindings and exact large-integer byte encoding (#93) Rooted cold and zero-delta preparation referenced value from unnamed unnest outputs in both graph-member checks, causing the exact native column-value failures. Explicitly bind the four array output columns. MTP2 integer serialization also divided numeric values before truncation, rounding u64MAX into wrong bytes; use the exact integer quotient instead. Enhance existing real database test sources for persisted rooted plan/payload reuse and independent Rust byte encoding at integer boundaries, retaining all prior full-byte/page-id and rejection assertions. Owned nightly format/parse, canonical whitespace and exact SQL lexical preservation checks pass; PostgreSQL/native/performance execution remains pending. Other CI failure groups are separate. --- .../storage/qualified_metadata_canonical.sql | 2 +- .../qualified_metadata_canonical_tests.rs | 88 +++++++++++++++++++ .../storage/qualified_metadata_rooted.sql | 4 +- 3 files changed, 91 insertions(+), 3 deletions(-) diff --git a/src/jupiter/storage/qualified_metadata_canonical.sql b/src/jupiter/storage/qualified_metadata_canonical.sql index 2ce0c283..bc04dbf3 100644 --- a/src/jupiter/storage/qualified_metadata_canonical.sql +++ b/src/jupiter/storage/qualified_metadata_canonical.sql @@ -19,7 +19,7 @@ BEGIN RAISE EXCEPTION 'MTP2 output integer is out of range'; END IF; result:=decode(repeat('00',w),'hex'); - FOR i IN 0..w-1 LOOP result:=set_byte(result,i,mod(v,256)::integer); v:=trunc(v/256); END LOOP; + FOR i IN 0..w-1 LOOP result:=set_byte(result,i,mod(v,256)::integer); v:=div(v,256); END LOOP; RETURN result; END $$; diff --git a/src/jupiter/storage/qualified_metadata_canonical_tests.rs b/src/jupiter/storage/qualified_metadata_canonical_tests.rs index 1ff9c2fc..714148ba 100644 --- a/src/jupiter/storage/qualified_metadata_canonical_tests.rs +++ b/src/jupiter/storage/qualified_metadata_canonical_tests.rs @@ -32,6 +32,51 @@ fn files(names: impl IntoIterator) -> Vec { #[tokio::test] async fn database_canonical_builder_matches_pinned_codec_at_split_and_name_boundaries() { let (_config, _core, _namespace, q, _guard) = fixture().await; + for (width, value) in [ + (2i32, 0u64), + (2, 255), + (2, 256), + (2, u16::MAX as u64), + (4, u32::MAX as u64), + (8, (1u64 << 53) + 1), + (8, (1u64 << 56) - 1), + (8, i64::MAX as u64), + (8, (1u64 << 63) + 1), + (8, u64::MAX - 1), + (8, u64::MAX), + ] { + let row = q.query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT mst2_metadata_write_le($1::numeric,$2) AS encoded,mst2_metadata_read_le(mst2_metadata_write_le($1::numeric,$2),0,$2)::text AS decoded", + [value.to_string().into(), width.into()], + )).await.unwrap().unwrap(); + assert_eq!( + row.try_get::>("", "encoded").unwrap(), + value.to_le_bytes()[..width as usize] + ); + assert_eq!( + row.try_get::("", "decoded").unwrap(), + value.to_string() + ); + } + for (value, width) in [ + ("-1", 8i32), + ("0.5", 8), + ("65536", 2), + ("4294967296", 4), + ("18446744073709551616", 8), + ("1", 1), + ] { + assert!( + q.query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT mst2_metadata_write_le($1::numeric,$2)", + [value.into(), width.into()], + )) + .await + .is_err() + ); + } let terminal_names = std::iter::once("a".to_owned()).chain((0..129).map(|index| format!("a{index:03}"))); let cases = [ @@ -438,6 +483,7 @@ pub(super) async fn seeded_rooted_plan( async fn actual_rooted_cold_and_zero_delta_reuse_keep_one_canonical_graph() { let (config, core, _namespace, q, _guard) = fixture().await; let (cold, payload) = seeded_rooted_plan(&core, 'a', "file", 1).await; + let expected_payload = payload.bytes.clone(); let writer = RootedQualifiedMetadataRepository::open(&core, &config) .await .unwrap(); @@ -506,6 +552,48 @@ async fn actual_rooted_cold_and_zero_delta_reuse_keep_one_canonical_graph() { 1 ); assert_eq!(count(&q,"SELECT count(*) FROM mst2_metadata_prepare WHERE plan_kind='ROOTED' AND state='COMMITTED' AND node_count=0").await,1); + for (operation, plan, delta_count, reused_count) in [ + ("actual-rooted-cold", &cold, 1usize, 0usize), + ("actual-zero-delta-root", &warm, 0usize, 1usize), + ] { + let row = q.query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT canonical_plan,mst2_metadata_decode_rooted_plan(canonical_plan) AS decoded FROM mst2_metadata_prepare WHERE operation_id=$1 AND state='COMMITTED'", + [operation.into()], + )).await.unwrap().unwrap(); + assert_eq!( + row.try_get::>("", "canonical_plan").unwrap(), + plan.encode().unwrap() + ); + let decoded: Value = row.try_get("", "decoded").unwrap(); + assert_eq!(decoded["root"], hex::encode(cold.root)); + assert_eq!(decoded["delta"].as_array().unwrap().len(), delta_count); + assert_eq!(decoded["reused"].as_array().unwrap().len(), reused_count); + assert_eq!(decoded["source_roots"].as_array().unwrap().len(), 1); + let member = if delta_count == 1 { + &decoded["delta"][0] + } else { + &decoded["reused"][0] + }; + assert_eq!(member["page"], hex::encode(cold.root)); + } + let resident = q + .query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT generation,payload FROM mst2_metadata_payload WHERE page_id=$1", + [cold.root.to_vec().into()], + )) + .await + .unwrap() + .unwrap(); + assert_eq!( + resident.try_get::>("", "payload").unwrap(), + expected_payload + ); + assert_eq!( + resident.try_get::("", "generation").unwrap(), + intent.root_generation() + ); let error = q .execute_unprepared("DELETE FROM mst2_metadata_root_anchor WHERE anchor_kind='REUSE'") .await diff --git a/src/jupiter/storage/qualified_metadata_rooted.sql b/src/jupiter/storage/qualified_metadata_rooted.sql index 8116bc61..e95655db 100644 --- a/src/jupiter/storage/qualified_metadata_rooted.sql +++ b/src/jupiter/storage/qualified_metadata_rooted.sql @@ -105,7 +105,7 @@ BEGIN IF cursor_pos<>octet_length(b) THEN RAISE EXCEPTION 'rooted preparation has trailing bytes'; END IF; SELECT coalesce(jsonb_object_agg(value->>'page',true),'{}'::jsonb) INTO delta_index FROM unnest(delta_rows) d(value); SELECT coalesce(jsonb_object_agg(value->>'page',true),'{}'::jsonb) INTO node_index FROM ( - SELECT value FROM unnest(delta_rows) UNION ALL SELECT value FROM unnest(reuse_rows)) nodes; + SELECT d.value FROM unnest(delta_rows) d(value) UNION ALL SELECT r.value FROM unnest(reuse_rows) r(value)) nodes; IF coalesce(array_length(delta_rows,1),0)+coalesce(array_length(reuse_rows,1),0)=0 OR EXISTS(SELECT 1 FROM unnest(reuse_rows) r(value) WHERE delta_index ? (r.value->>'page')) OR NOT node_index ? encode(root,'hex') @@ -120,7 +120,7 @@ BEGIN GROUP BY value->>'parent') grouped; IF EXISTS(WITH RECURSIVE reached(page) AS (SELECT encode(root,'hex') UNION SELECT child FROM reached r CROSS JOIN LATERAL jsonb_array_elements_text(adjacency->r.page) children(child)) - SELECT 1 FROM (SELECT value FROM unnest(delta_rows) UNION ALL SELECT value FROM unnest(reuse_rows)) nodes + SELECT 1 FROM (SELECT d.value FROM unnest(delta_rows) d(value) UNION ALL SELECT r.value FROM unnest(reuse_rows) r(value)) nodes WHERE NOT EXISTS(SELECT 1 FROM reached r WHERE r.page=nodes.value->>'page')) THEN RAISE EXCEPTION 'rooted plan includes members outside its bounded delta and boundary closure'; END IF; From b00b7e57740b24cc0a6befa84484483c8f53f7cc Mon Sep 17 00:00:00 2001 From: Xiaoyang Han Date: Thu, 8 Oct 2026 09:23:08 +0800 Subject: [PATCH 41/50] Bootstrap rooted metadata fixtures through their actual database configuration (#95) Use production rooted metadata provisioning for ordinary snapshot, view operations, and native push fixtures. Retain schema ownership, generic history behavior, native queue timing, and all existing test assertions. Validation is source-only: owned nightly format checks and Git whitespace checks; native builds, database tests, and performance runs are deferred. --- src/api/router/snapshot_content_tests.rs | 2 +- .../service/native_publication_push_tests.rs | 26 +++++++++++++++++-- src/jupiter/storage/view_ops_tests.rs | 13 ++++++---- 3 files changed, 33 insertions(+), 8 deletions(-) diff --git a/src/api/router/snapshot_content_tests.rs b/src/api/router/snapshot_content_tests.rs index aa8f92bc..906752ff 100644 --- a/src/api/router/snapshot_content_tests.rs +++ b/src/api/router/snapshot_content_tests.rs @@ -679,7 +679,7 @@ impl Fixture { config.mst2.auth_token = Some(TOKEN.to_string()); let backend = build_object_storage(&config.object_storage).await.unwrap(); let counts = Arc::new(ReadCounts::default()); - let (mut storage, schema) = if rebuildable { + let (mut storage, schema) = if rebuildable || !generic_history { let (database, schema) = test_db_config(temp.path()).await; config.database = database; let assembly = async { diff --git a/src/jupiter/service/native_publication_push_tests.rs b/src/jupiter/service/native_publication_push_tests.rs index ce3febbb..eb746d1c 100644 --- a/src/jupiter/service/native_publication_push_tests.rs +++ b/src/jupiter/service/native_publication_push_tests.rs @@ -8,14 +8,36 @@ use crate::jupiter::utils::converter::FromMegaModel; const NATIVE_INSTANCE: &str = "6ab219b0-4275-45ba-9d7b-7b0b633018cd"; -async fn native_fixture() -> (tempfile::TempDir, crate::jupiter::storage::Storage, git_internal::internal::object::commit::Commit, String) { +async fn native_fixture() -> ( + tempfile::TempDir, + crate::jupiter::storage::Storage, + git_internal::internal::object::commit::Commit, + String, +) { let temp = tempfile::tempdir().unwrap(); let mut config = crate::config::testing::isolated_config(temp.path().join("config")); config.monorepo.push_policy = PushPolicy::Trunk; config.mst2.enabled = true; config.mst2.publication_enabled = true; config.mst2.instance_uuid = Some(NATIVE_INSTANCE.to_owned()); - let storage = crate::jupiter::tests::test_storage_with_config(temp.path(), config).await; + let (database, schema) = crate::jupiter::tests::test_db_config(temp.path()).await; + config.database = database; + let mut connection = crate::jupiter::storage::init::database_connection(&config.database) + .await + .unwrap(); + connection.set_metric_callback(move |_| { + let _held_by_the_connection = &schema; + }); + let mut storage = crate::jupiter::storage::Storage::new_with_connection( + Arc::new(config), + Arc::new(connection), + crate::jupiter::storage::object_storage::mock_object_storage(), + ) + .await + .unwrap(); + storage.push_queue_service = storage + .push_queue_service + .with_timeouts(Duration::from_secs(30), Duration::from_millis(20)); let storage = crate::jupiter::tests::with_test_vault(storage, temp.path()).await; let name = format!("native-{}", uuid::Uuid::new_v4().simple()); use git_internal::internal::object::{ diff --git a/src/jupiter/storage/view_ops_tests.rs b/src/jupiter/storage/view_ops_tests.rs index e0cfa446..bc691179 100644 --- a/src/jupiter/storage/view_ops_tests.rs +++ b/src/jupiter/storage/view_ops_tests.rs @@ -15,7 +15,6 @@ use crate::{ ceres::view::filter::parse_for_registration, config::testing::isolated_config, jupiter::{ - migration::apply_migrations, service::{ view_metrics::ViewMetrics, view_projection_service::{CatchUpOutcome, ViewProjectionService}, @@ -23,6 +22,7 @@ use crate::{ storage::{ Storage, git_db_storage::fu18_support::single_connection, + init::database_connection, object_storage::mock_object_storage, view_root_chain::RootChainOutcome, view_storage::ViewLockMode, @@ -31,7 +31,7 @@ use crate::{ seed_unrelated_root_history, }, }, - tests::test_db_connection, + tests::{TestSchemaGuard, test_db_config}, }, }; @@ -58,6 +58,7 @@ struct OpsFixture { filter_a: i64, filter_b: i64, filter_a_id: String, + _schema: TestSchemaGuard, } fn fixed_time() -> chrono::NaiveDateTime { @@ -222,9 +223,10 @@ async fn insert_warming_filter( async fn ops_fixture() -> OpsFixture { let temp = tempfile::tempdir().unwrap(); - let db = test_db_connection(temp.path()).await; - apply_migrations(&db, true).await.unwrap(); - let config = isolated_config(temp.path().join("config")); + let (database, schema) = test_db_config(temp.path()).await; + let mut config = isolated_config(temp.path().join("config")); + config.database = database; + let db = database_connection(&config.database).await.unwrap(); let storage = Storage::new_with_connection( Arc::new(config), Arc::new(db.clone()), @@ -406,6 +408,7 @@ async fn ops_fixture() -> OpsFixture { filter_a: filter_a.id, filter_b: filter_b.id, filter_a_id: filter_a.filter_id, + _schema: schema, } } From d870ce94cc38cc5530f32208a795bbed0e92407f Mon Sep 17 00:00:00 2001 From: Xiaoyang Han Date: Thu, 8 Oct 2026 09:27:40 +0800 Subject: [PATCH 42/50] Align native fixtures with captured database scope and exact guards (#96) The UN30 facade fixture now supplies its actual per-test database config to full storage assembly, so the rooted Q pool shares the correct database. The forged namespace insertion must match its exact BEFORE INSERT registration guard and roll back, preserving every route/source state assertion. The durable scope-drift test now expects the shared family-selection contract: 502 INTEGRITY_ERROR with retryable=false for both descriptor and renew, before a G/Q repository is selected. It retains stream error/no-END, scope restoration and no-IO assertions. Three owned nightly format/parse and whitespace source checks pass; native execution remains unrun. --- src/api/router/snapshot_session_tests.rs | 13 +++++++------ src/api/router/snapshot_storage_route_tests.rs | 12 ++++++++++-- src/context/un30_readonly.rs | 6 +++--- 3 files changed, 20 insertions(+), 11 deletions(-) diff --git a/src/api/router/snapshot_session_tests.rs b/src/api/router/snapshot_session_tests.rs index 68feb23a..e7f4f734 100644 --- a/src/api/router/snapshot_session_tests.rs +++ b/src/api/router/snapshot_session_tests.rs @@ -1293,11 +1293,12 @@ async fn mst2_durable_http_warm_reads_renew_and_frame_delivery_reject_primary_sc .try_get_by_index(0) .unwrap(); overwrite_primary_scope_for_test(db, uuid::Uuid::new_v4().to_string()).await; + // Family selection rejects physical authority drift as INTEGRITY_ERROR. error( fixture.send("GET", "descriptor", Body::empty()).await, - 500, - "INTERNAL", - true, + 502, + "INTEGRITY_ERROR", + false, ) .await; error( @@ -1314,9 +1315,9 @@ async fn mst2_durable_http_warm_reads_renew_and_frame_delivery_reject_primary_sc ) .await .unwrap(), - 500, - "INTERNAL", - true, + 502, + "INTEGRITY_ERROR", + false, ) .await; assert!( diff --git a/src/api/router/snapshot_storage_route_tests.rs b/src/api/router/snapshot_storage_route_tests.rs index 63b59fc6..e5a0ad42 100644 --- a/src/api/router/snapshot_storage_route_tests.rs +++ b/src/api/router/snapshot_storage_route_tests.rs @@ -483,11 +483,19 @@ async fn mst2_generic_storage_routes_raw_lease_derives_atomically_and_half_rows_ SELECT (jsonb_populate_record(NULL::mst2_lease_storage_route, to_jsonb(r)||jsonb_build_object('lease_id','{lease}'))).* FROM mst2_lease_storage_route r", )).await; - rejected(db, + let txn = db.begin().await.unwrap(); + let error = txn.execute_unprepared( "INSERT INTO mst2_metadata_namespace SELECT (jsonb_populate_record(NULL::mst2_metadata_namespace, to_jsonb(n)||jsonb_build_object('namespace_uuid','00000000-0000-4000-8000-000000000001', 'graph_domain','qualified-v1'))).* FROM mst2_metadata_namespace n", - ).await; + ).await.expect_err("unadmitted qualified namespace must be rejected before commit"); + assert!( + error.to_string().contains( + "qualified namespace registration has no exact admitted rooted family scope and lock set" + ), + "wrong qualified namespace rejection: {error}" + ); + txn.rollback().await.unwrap(); assert_eq!(routes(db).await, original); assert_eq!(sources(db).await, source); } diff --git a/src/context/un30_readonly.rs b/src/context/un30_readonly.rs index ad829a0f..d95a03b6 100644 --- a/src/context/un30_readonly.rs +++ b/src/context/un30_readonly.rs @@ -244,9 +244,9 @@ async fn un30_building_the_read_facade_writes_no_sidebar() { .await .expect("writable connection"), ); - let config = Arc::new(crate::config::testing::isolated_config( - temp.path().join("config"), - )); + let mut config = crate::config::testing::isolated_config(temp.path().join("config")); + config.database = db_config.clone(); + let config = Arc::new(config); assert!( !sidebar_table_exists(&writable).await, From c1ff280281440891e732ecfaed4732ee59c41a90 Mon Sep 17 00:00:00 2001 From: Xiaoyang Han Date: Thu, 8 Oct 2026 09:30:16 +0800 Subject: [PATCH 43/50] Fix chunk retention error conversion and test owner lifetimes (#97) The exact PR91 repository gate stopped at compilation: common IoOrbitError conversion missed the two retention variants, fourteen test connection borrows outlived temporary MonoStorage owners, and one spawned install borrowed fixture storage. Explicitly convert both concrete retention errors through existing MegaError::ObjStorage without changing their specialized Snapshot classifications, bind each connection owner, and move an owned Storage clone sharing the original Arc repository cache into the install task before borrowing that same cached repository. All original assertions, deadlines, cancellation and final Bytes retention checks remain byte-for-byte preserved outside these lifetime edits. Owned nightly format and exact source/tree checks pass; native build, Clippy and repository tests remain pending. --- .../snapshot_chunk_map_retention_tests.rs | 49 +++++++++++++------ src/common/errors/mod.rs | 6 +++ 2 files changed, 39 insertions(+), 16 deletions(-) diff --git a/src/api/router/snapshot_chunk_map_retention_tests.rs b/src/api/router/snapshot_chunk_map_retention_tests.rs index 587e4196..6f326b57 100644 --- a/src/api/router/snapshot_chunk_map_retention_tests.rs +++ b/src/api/router/snapshot_chunk_map_retention_tests.rs @@ -115,7 +115,8 @@ async fn shared_map_survives_other_source_retirement_while_actual_reader_is_live .await; let original = fixture.map("/file").await; let other = oid_for(&fixture, "/other").await; - let db = fixture.state.storage.mono_storage().get_connection(); + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); // This G fixture isolates independent source receipts without changing a // Q-certified file tuple. Two exact facts consume the same actual raw // bytes independently; map sharing never admits the second source. @@ -193,7 +194,8 @@ async fn pending_receipt_is_replayed_before_another_aged_live_victim() { .await; fixture.map("/file").await; let other = fixture.map("/other").await; - let db = fixture.state.storage.mono_storage().get_connection(); + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); wait_count(db, "mst2_chunk_reader", 0).await; age_unowned_candidates(db).await; db.execute_raw(statement( @@ -220,7 +222,8 @@ async fn pending_receipt_is_replayed_before_another_aged_live_victim() { async fn actual_http_replays_a_retiring_source_row_before_rebuilding_its_next_generation() { let fixture = Fixture::new().await; let original = fixture.map("/file").await; - let db = fixture.state.storage.mono_storage().get_connection(); + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); wait_count(db, "mst2_chunk_reader", 0).await; let old_key: String = db .query_one_raw(statement( @@ -273,7 +276,8 @@ async fn actual_warm_body_owner_cancels_never_returning_open_and_next_without_ba let complete_without_eof = stage == 2; let fixture = Fixture::new().await; let map = fixture.map("/file").await; - let db = fixture.state.storage.mono_storage().get_connection(); + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); wait_count(db, "mst2_chunk_reader", 0).await; fixture.counts.reset(); let entered = Arc::new(Notify::new()); @@ -415,7 +419,8 @@ async fn actual_cold_builder_uses_remaining_database_deadline_for_open_and_next_ .admit_install(&source, &fixture.state.storage.git_service.obj_storage) .await .unwrap(); - let db = fixture.state.storage.mono_storage().get_connection(); + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); db.execute_unprepared("UPDATE mst2_chunk_receipt_generation SET deadline=pg_catalog.clock_timestamp()+interval '3 seconds' WHERE state='RESERVED'").await.unwrap(); admission.test_check_next_owner_operation().await; fixture.counts.reset(); @@ -512,12 +517,16 @@ async fn actual_receipt_create_and_read_waits_cancel_without_backend_release_or_ Some((entered.clone(), release.clone(), drops.clone())); drops }; - let mut task = - tokio::spawn(async move { repository.install(verified, &objects, &admission).await }); + let task_storage = fixture.state.storage.clone(); + let mut task = tokio::spawn(async move { + let repository = task_storage.chunk_maps().await.unwrap(); + repository.install(verified, &objects, &admission).await + }); timeout(Duration::from_secs(10), entered.notified()) .await .unwrap(); - let db = fixture.state.storage.mono_storage().get_connection(); + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 1); if creating { db.execute_unprepared("UPDATE mst2_chunk_receipt_generation SET deadline=pg_catalog.clock_timestamp()+interval '20 milliseconds' WHERE state='CREATING'").await.unwrap(); @@ -569,7 +578,8 @@ async fn held_raw_and_exact_range_do_not_resume_after_actual_reader_expiry() { for raw_route in [true, false] { let fixture = Fixture::new().await; let map = fixture.map("/file").await; - let db = fixture.state.storage.mono_storage().get_connection(); + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); wait_count(db, "mst2_chunk_reader", 0).await; fixture.counts.reset(); let entered = Arc::new(Notify::new()); @@ -643,7 +653,8 @@ async fn production_q_bootstrap_after_retention_preserves_catalog_and_reconstruc }; let fixture = Fixture::new_with_pg_config(true).await; - let db = fixture.state.storage.mono_storage().get_connection(); + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); let namespace = provision_or_verify_rooted_qualified_family(db) .await .unwrap(); @@ -891,7 +902,8 @@ async fn completion_during_real_inventory_keeps_backing_credit_after_pending_ins })), }; fixture.app = router(&fixture.state); - let db = fixture.state.storage.mono_storage().get_connection(); + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); let create_entered = Arc::new(Notify::new()); let create_release = Arc::new(Notify::new()); *fixture.counts.receipt_late_create_holds.lock().unwrap() = @@ -972,7 +984,8 @@ async fn json_chunk_and_raw_last_transport_clones_block_actual_collection() { let fixture = Fixture::new().await; let map = fixture.map("/file").await; let repository = fixture.state.storage.chunk_maps().await.unwrap(); - let db = fixture.state.storage.mono_storage().get_connection(); + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); wait_count(db, "mst2_chunk_reader", 0).await; fixture.counts.reset(); let response_budget = MemoryBudget::new(CHUNK_SIZE as usize + 2048); @@ -1070,7 +1083,8 @@ async fn json_chunk_and_raw_last_transport_clones_block_actual_collection() { async fn expired_reader_arc_cannot_resurrect_or_authenticate_a_page() { let fixture = Fixture::new().await; fixture.map("/file").await; - let db = fixture.state.storage.mono_storage().get_connection(); + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); wait_count(db, "mst2_chunk_reader", 0).await; let repository = fixture.state.storage.chunk_maps().await.unwrap(); let source = ChunkMapSource::from_fact(fixture.fact().await, &fixture.oid).unwrap(); @@ -1160,7 +1174,8 @@ async fn expired_reader_arc_cannot_resurrect_or_authenticate_a_page() { #[tokio::test] async fn late_cancelled_create_is_reserved_and_old_key_replay_preserves_new_generation() { let fixture = Fixture::new().await; - let db = fixture.state.storage.mono_storage().get_connection(); + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); let entered = Arc::new(Notify::new()); let release = Arc::new(Notify::new()); *fixture.counts.receipt_late_create_holds.lock().unwrap() = @@ -1336,7 +1351,8 @@ async fn actual_backing_quota_and_unsupported_capability_reject_before_source_op } counts.assert(0, 0); assert_eq!(counts.receipt_writes.load(Ordering::SeqCst), 0); - let db = fixture.state.storage.mono_storage().get_connection(); + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); assert_eq!(count(db, "mst2_chunk_receipt_generation").await, 0); assert_eq!(count(db, "mst2_chunk_map").await, 0); } @@ -1378,7 +1394,8 @@ async fn empty_source_fragment_after_deadline_does_not_renew_install_or_publish( timeout(Duration::from_secs(10), entered.notified()) .await .unwrap(); - let db = fixture.state.storage.mono_storage().get_connection(); + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); db.execute_unprepared("UPDATE mst2_chunk_receipt_generation SET deadline=pg_catalog.clock_timestamp()+interval '20 milliseconds' WHERE state='RESERVED'").await.unwrap(); tokio::time::sleep(Duration::from_millis(50)).await; admission.test_check_next_owner_operation().await; diff --git a/src/common/errors/mod.rs b/src/common/errors/mod.rs index d1872272..5d5ef9b1 100644 --- a/src/common/errors/mod.rs +++ b/src/common/errors/mod.rs @@ -184,6 +184,12 @@ impl From for MegaError { IoOrbitError::WriteManifestPreconditionFailed => { MegaError::Other("write manifest precondition failed".to_string()) } + error @ IoOrbitError::ChunkMapRetentionUnsupported => { + MegaError::ObjStorage(error.to_string()) + } + error @ IoOrbitError::ChunkMapRetentionCapacityExceeded => { + MegaError::ObjStorage(error.to_string()) + } IoOrbitError::Other(e) => MegaError::Other(e), } } From 7f675c41089b930e1679830bd64ad2938137ddb8 Mon Sep 17 00:00:00 2001 From: Xiaoyang Han Date: Thu, 8 Oct 2026 09:47:07 +0800 Subject: [PATCH 44/50] Use equivalent let chains for raw progress and cancelled installs (#98) The exact PR97 and integrated c1ff repository gates both stop at two Clippy collapsible_if errors. Collapse only those nested conditions: nonempty raw bytes followed by an existing chunk map, and a successful cancelled-install mutex lock followed by its strict 64-entry cap. Preserve short-circuit order, lock scope, ownership, progress error propagation and every cancellation/stream assertion. Two owned nightly format/parse checks, exact whole-file forward/inverse reconstruction and the complete unchanged-outside-paths tree proof pass. Native Clippy, build and repository tests remain unrun for this candidate. --- src/api/router/snapshot_raw_blob.rs | 8 ++++---- .../storage/native_chunk_map/retention.rs | 16 ++++++++-------- 2 files changed, 12 insertions(+), 12 deletions(-) diff --git a/src/api/router/snapshot_raw_blob.rs b/src/api/router/snapshot_raw_blob.rs index 016310cd..fa8fa3c7 100644 --- a/src/api/router/snapshot_raw_blob.rs +++ b/src/api/router/snapshot_raw_blob.rs @@ -377,10 +377,10 @@ impl RawBlobReader { "fixed-view raw stream failed", ) })?; - if part.as_ref().is_some_and(|bytes| !bytes.is_empty()) { - if let Some(map) = self.map.as_ref() { - map.record_progress().await?; - } + if part.as_ref().is_some_and(|bytes| !bytes.is_empty()) + && let Some(map) = self.map.as_ref() + { + map.record_progress().await?; } Ok(part) } diff --git a/src/jupiter/storage/native_chunk_map/retention.rs b/src/jupiter/storage/native_chunk_map/retention.rs index 82a74da2..235ec86f 100644 --- a/src/jupiter/storage/native_chunk_map/retention.rs +++ b/src/jupiter/storage/native_chunk_map/retention.rs @@ -409,14 +409,14 @@ impl ChunkMapInstall { impl Drop for ChunkMapInstall { fn drop(&mut self) { - if let Ok(mut cancelled) = self.repository.cancelled_installs.lock() { - if cancelled.len() < 64 { - cancelled.push(CancelledInstall { - owner: self.owner, - key: self.key.key.clone(), - generation: self.generation, - }); - } + if let Ok(mut cancelled) = self.repository.cancelled_installs.lock() + && cancelled.len() < 64 + { + cancelled.push(CancelledInstall { + owner: self.owner, + key: self.key.key.clone(), + generation: self.generation, + }); } let Ok(runtime) = tokio::runtime::Handle::try_current() else { return; From c67665d4b6b8ac767eb99d608c163588f852980d Mon Sep 17 00:00:00 2001 From: Xiaoyang Han Date: Thu, 8 Oct 2026 10:20:18 +0800 Subject: [PATCH 45/50] Parenthesize chunk retention CASE expressions in PL/pgSQL IF guards (#99) Database bootstrap currently fails with PostgreSQL 42601 at Original(10355): the first unparenthesized CASE in a PL/pgSQL IF exposes its internal THEN as the end of the IF condition. Parenthesize that CASE and the identical later map guard expression. Preserve both 512 MiB capacity checks, state-dependent 256 KiB/1 MiB reserves, error messages and every other migration byte. Exact original source-offset and whole-file inverse proofs plus whitespace and unchanged-tree checks pass. The cancelled native run has no completed test summary; this candidate has not executed SQL or native tests. --- .../migration/m20261008_000300_chunk_map_retention.sql | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/src/jupiter/migration/m20261008_000300_chunk_map_retention.sql b/src/jupiter/migration/m20261008_000300_chunk_map_retention.sql index 46a75aa3..f967d440 100644 --- a/src/jupiter/migration/m20261008_000300_chunk_map_retention.sql +++ b/src/jupiter/migration/m20261008_000300_chunk_map_retention.sql @@ -152,7 +152,7 @@ BEGIN IF TG_OP='DELETE' OR TG_OP='TRUNCATE' THEN RAISE EXCEPTION 'chunk receipt generations and collection history are retained'; END IF; - IF mst2_chunk_retention_bytes()+CASE WHEN NEW.state IN ('DELETING','APPLIED') THEN 262144 ELSE 1048576 END>536870912 THEN + IF mst2_chunk_retention_bytes()+(CASE WHEN NEW.state IN ('DELETING','APPLIED') THEN 262144 ELSE 1048576 END)>536870912 THEN RAISE EXCEPTION 'chunk receipt mutation lacks actual durable byte headroom'; END IF; IF TG_OP='UPDATE' AND (NEW.receipt_key IS DISTINCT FROM OLD.receipt_key OR NEW.source_id IS DISTINCT FROM OLD.source_id @@ -244,7 +244,7 @@ BEGIN IF NOT permitted THEN RAISE EXCEPTION 'map lifetime deletion lacks exact reader-free collection claim'; END IF; RETURN OLD; END IF; - IF mst2_chunk_retention_bytes()+CASE WHEN NEW.state='DELETING' THEN 262144 ELSE 1048576 END>536870912 THEN + IF mst2_chunk_retention_bytes()+(CASE WHEN NEW.state='DELETING' THEN 262144 ELSE 1048576 END)>536870912 THEN RAISE EXCEPTION 'chunk map lifetime mutation lacks actual durable byte headroom'; END IF; IF TG_OP='UPDATE' AND (NEW.map_id IS DISTINCT FROM OLD.map_id OR NEW.generation IS DISTINCT FROM OLD.generation From cc90da9947c6dd469dd74bb6a6596ef69725df75 Mon Sep 17 00:00:00 2001 From: Xiaoyang Han Date: Thu, 8 Oct 2026 10:32:38 +0800 Subject: [PATCH 46/50] fix(v3): bound qualified reader history with complete owner identity (#100) Qualified metadata requests previously retained every completed reader, so the 65,536-owner history cap eventually blocked new requests. Reclaim at most 64 previously committed terminal owners without roots before admission, and fence every page/source/finish operation with UUID plus a monotonically issued generation stored in one high-water row per Q. Same-transaction terminal owners remain until their deferred completion commits. Add a trusted transactional forward migration, remove UUID-only function overloads, and preserve permanent SID, source, certificate and lease history. Nine default database regression test sources cover fresh/old-Q/Q-less/tampered upgrades, real HTTP requests, stale actual-reader replay, rollback issuance, maximum issuance and the 64-owner deletion budget; a new ignored capacity soak issues 65,537 real HTTP requests. Owned nightly formatting, complete candidate whitespace, patch checks and offline source proofs pass. Native build/tests/Clippy, PostgreSQL execution and performance measurements remain pending; this package claims source validation only. --- .../router/snapshot_reader_retention_tests.rs | 431 ++++++++++++++++++ ...snapshot_reader_retention_upgrade_tests.rs | 199 ++++++++ .../router/snapshot_rooted_metadata_tests.rs | 3 + ...261008_000400_add_mst2_reader_retention.rs | 397 ++++++++++++++++ ...261008_000400_reader_retention_upgrade.sql | 46 ++ src/jupiter/migration/mod.rs | 44 +- .../storage/qualified_metadata_anchors.sql | 18 +- .../storage/qualified_metadata_family.rs | 9 + .../storage/qualified_metadata_family.sql | 2 +- src/jupiter/storage/qualified_metadata_gc.sql | 14 +- .../storage/qualified_metadata_reader.rs | 51 ++- .../qualified_metadata_reader_lifecycle.sql | 110 +++++ ...lified_metadata_reader_previous_fixture.rs | 295 ++++++++++++ ...ified_metadata_reader_previous_fixture.sql | 425 +++++++++++++++++ .../storage/qualified_metadata_serving.sql | 70 +-- .../qualified_metadata_source_read.sql | 6 +- 16 files changed, 2015 insertions(+), 105 deletions(-) create mode 100644 src/api/router/snapshot_reader_retention_tests.rs create mode 100644 src/api/router/snapshot_reader_retention_upgrade_tests.rs create mode 100644 src/jupiter/migration/m20261008_000400_add_mst2_reader_retention.rs create mode 100644 src/jupiter/migration/m20261008_000400_reader_retention_upgrade.sql create mode 100644 src/jupiter/storage/qualified_metadata_reader_lifecycle.sql create mode 100644 src/jupiter/storage/qualified_metadata_reader_previous_fixture.rs create mode 100644 src/jupiter/storage/qualified_metadata_reader_previous_fixture.sql diff --git a/src/api/router/snapshot_reader_retention_tests.rs b/src/api/router/snapshot_reader_retention_tests.rs new file mode 100644 index 00000000..546ece7a --- /dev/null +++ b/src/api/router/snapshot_reader_retention_tests.rs @@ -0,0 +1,431 @@ +use sea_orm::DatabaseTransaction; + +use super::*; + +#[path = "snapshot_reader_retention_upgrade_tests.rs"] +mod upgrade; + +async fn transaction(fixture: &Fixture) -> DatabaseTransaction { + let schema = q_schema(fixture).await; + let mono = fixture.state.storage.mono_storage(); + let txn = mono.get_connection().begin().await.unwrap(); + txn.execute_unprepared(&format!( + "SET LOCAL search_path=\"{}\",pg_catalog,pg_temp", + schema.replace('"', "\"\"") + )) + .await + .unwrap(); + txn +} + +async fn admission(txn: &DatabaseTransaction, fixture: &Fixture) -> (String, i64) { + let reader = txn.query_one_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "SELECT operation_id::text,reader_issuance FROM mst2_metadata_begin_reader($1,$2, + (SELECT instance_id FROM mst2_qualified_session_incarnation WHERE snapshot_id=$1 AND state='READY' LIMIT 1))", + [fixture.snapshot.clone().into(),fixture.lease.clone().into()])).await.unwrap().unwrap(); + ( + reader.try_get("", "operation_id").unwrap(), + reader.try_get("", "reader_issuance").unwrap(), + ) +} + +async fn finish(txn: &DatabaseTransaction, ticket: &(String, i64)) { + txn.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT mst2_metadata_finish_reader($1::uuid,$2::bigint)", + [ticket.0.clone().into(), ticket.1.into()], + )) + .await + .unwrap(); +} + +async fn prune(txn: &DatabaseTransaction, maximum: i32) -> i64 { + txn.query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT mst2_metadata_prune_readers($1)", + [maximum.into()], + )) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap() +} + +#[tokio::test] +async fn committed_metadata_requests_retire_owners_without_retiring_source_history() { + let fixture = Fixture::new_with_pg_config(true).await; + let initial = q_count( + &fixture, + "SELECT high_water FROM {q}.mst2_metadata_reader_issuance", + ) + .await; + let certificate_count = q_count( + &fixture, + "SELECT count(*) FROM {q}.mst2_metadata_page_certificate", + ) + .await; + let source_count = q_count( + &fixture, + "SELECT count(*) FROM {q}.mst2_metadata_source_root_attestation", + ) + .await; + for _ in 0..96 { + success_json( + fixture + .app + .clone() + .oneshot(fixture.request( + "POST", + "lookup", + Body::from(json!({"paths":["/file"]}).to_string()), + )) + .await + .unwrap(), + ) + .await; + } + assert_eq!( + q_count( + &fixture, + "SELECT high_water FROM {q}.mst2_metadata_reader_issuance" + ) + .await, + initial + 96 + ); + assert_eq!( + q_count( + &fixture, + "SELECT count(*) FROM {q}.mst2_metadata_reader_operation" + ) + .await, + 1 + ); + assert_eq!( + q_count( + &fixture, + "SELECT count(*) FROM {q}.mst2_metadata_reader_operation WHERE state='ACTIVE'" + ) + .await, + 0 + ); + assert_eq!(q_count(&fixture,"SELECT count(*) FROM {q}.mst2_metadata_root_anchor WHERE anchor_kind IN ('REQUEST','READER')").await,0); + assert_eq!( + q_count( + &fixture, + "SELECT count(*) FROM {q}.mst2_metadata_page_certificate" + ) + .await, + certificate_count + ); + assert_eq!( + q_count( + &fixture, + "SELECT count(*) FROM {q}.mst2_metadata_source_root_attestation" + ) + .await, + source_count + ); + success_json( + fixture + .app + .clone() + .oneshot(fixture.request("GET", "descriptor", Body::empty())) + .await + .unwrap(), + ) + .await; + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn terminal_owner_survives_its_deferred_completion_and_prunes_in_a_later_transaction() { + let fixture = Fixture::new_with_pg_config(true).await; + let txn = transaction(&fixture).await; + let ticket = admission(&txn, &fixture).await; + txn.commit().await.unwrap(); + for forbidden in [ + "DELETE FROM mst2_metadata_reader_operation", + "UPDATE mst2_metadata_reader_operation SET state='EXPIRED'", + "UPDATE mst2_metadata_reader_issuance SET high_water=0", + "DELETE FROM mst2_metadata_reader_issuance", + "TRUNCATE mst2_metadata_reader_issuance", + "SELECT mst2_metadata_prune_readers(65)", + ] { + let txn = transaction(&fixture).await; + assert!(txn.execute_unprepared(forbidden).await.is_err()); + txn.rollback().await.unwrap(); + } + let txn = transaction(&fixture).await; + finish(&txn, &ticket).await; + assert_eq!(prune(&txn, 64).await, 0); + txn.commit().await.unwrap(); + let txn = transaction(&fixture).await; + // A terminal no-op must not enqueue another deferred row lookup. + txn.execute_unprepared("UPDATE mst2_metadata_reader_operation SET state=state") + .await + .unwrap(); + assert_eq!(prune(&txn, 0).await, 0); + assert_eq!(prune(&txn, 64).await, 1); + txn.commit().await.unwrap(); + assert_eq!( + q_count( + &fixture, + "SELECT count(*) FROM {q}.mst2_metadata_reader_operation" + ) + .await, + 0 + ); + assert_eq!( + q_count( + &fixture, + "SELECT high_water FROM {q}.mst2_metadata_reader_issuance" + ) + .await, + ticket.1 + ); +} + +#[tokio::test] +async fn reader_pruning_respects_the_64_owner_budget_with_a_committed_backlog() { + let fixture = Fixture::new_with_pg_config(true).await; + let txn = transaction(&fixture).await; + let mut tickets = Vec::new(); + for _ in 0..65 { + tickets.push(admission(&txn, &fixture).await); + } + txn.commit().await.unwrap(); + let txn = transaction(&fixture).await; + for ticket in &tickets { + finish(&txn, ticket).await; + } + assert_eq!(prune(&txn, 64).await, 0); + txn.commit().await.unwrap(); + let txn = transaction(&fixture).await; + assert_eq!(prune(&txn, 64).await, 64); + txn.commit().await.unwrap(); + assert_eq!( + q_count( + &fixture, + "SELECT count(*) FROM {q}.mst2_metadata_reader_operation" + ) + .await, + 1 + ); + let txn = transaction(&fixture).await; + assert_eq!(prune(&txn, 64).await, 1); + txn.commit().await.unwrap(); + assert_eq!( + q_count( + &fixture, + "SELECT count(*) FROM {q}.mst2_metadata_reader_operation" + ) + .await, + 0 + ); + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn stale_actual_reader_cannot_read_or_finish_a_reissued_uuid() { + let fixture = Fixture::new_with_pg_config(true).await; + let admitted = Arc::new(Barrier::new(2)); + let resume = Arc::new(Barrier::new(2)); + let request = fixture.request( + "POST", + "metadata/pages", + Body::from( + json!({"encoding":"identity","items":[{"directory_path":"/nested"}]}).to_string(), + ), + ); + let app = fixture.app.clone(); + let pending = tokio::spawn(with_rooted_reader_barriers( + admitted.clone(), + resume.clone(), + async move { app.oneshot(request).await.unwrap() }, + )); + tokio::time::timeout(Duration::from_secs(4), admitted.wait()) + .await + .unwrap(); + let txn = transaction(&fixture).await; + let owner=txn.query_one_raw(Statement::from_string(DbBackend::Postgres, + "SELECT operation_id::text,reader_issuance,to_jsonb(r) AS owner FROM mst2_metadata_reader_operation r WHERE state='ACTIVE'")) + .await.unwrap().unwrap(); + let old = ( + owner.try_get::("", "operation_id").unwrap(), + owner.try_get::("", "reader_issuance").unwrap(), + ); + let old_row: Value = owner.try_get("", "owner").unwrap(); + let anchors:Value=txn.query_one_raw(Statement::from_string(DbBackend::Postgres, + "SELECT jsonb_agg(to_jsonb(a)) AS anchors FROM mst2_metadata_root_anchor a WHERE anchor_kind IN ('REQUEST','READER')")) + .await.unwrap().unwrap().try_get("","anchors").unwrap(); + finish(&txn, &old).await; + txn.commit().await.unwrap(); + let txn = transaction(&fixture).await; + assert_eq!(prune(&txn, 64).await, 1); + txn.commit().await.unwrap(); + let txn = transaction(&fixture).await; + assert!(txn.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "INSERT INTO mst2_metadata_reader_operation SELECT * FROM jsonb_populate_record(NULL::mst2_metadata_reader_operation,$1::jsonb)", + [old_row.clone().into()])).await.is_err()); + txn.rollback().await.unwrap(); + let next = old.1 + 1; + let txn = transaction(&fixture).await; + txn.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "INSERT INTO mst2_metadata_reader_operation SELECT * FROM jsonb_populate_record(NULL::mst2_metadata_reader_operation, + $1::jsonb||jsonb_build_object('reader_issuance',$2::bigint))",[old_row.into(),next.into()])).await.unwrap(); + txn.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "INSERT INTO mst2_metadata_root_anchor SELECT (jsonb_populate_record(NULL::mst2_metadata_root_anchor, + value||jsonb_build_object('reader_issuance',$2::bigint))).* FROM jsonb_array_elements($1::jsonb)", + [anchors.into(),next.into()])).await.unwrap(); + txn.commit().await.unwrap(); + let txn = transaction(&fixture).await; + let source = txn + .query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT a.attestation_id::text,a.root_page,a.root_generation,a.root_certificate_digest + FROM mst2_metadata_source_root_attestation a JOIN mst2_metadata_reader_operation r + ON r.root_page=a.root_page AND r.root_generation=a.root_generation + WHERE r.operation_id=$1::uuid AND r.reader_issuance=$2 LIMIT 1", + [old.0.clone().into(), next.into()], + )) + .await + .unwrap() + .unwrap(); + let source_id: String = source.try_get("", "attestation_id").unwrap(); + let source_root: Vec = source.try_get("", "root_page").unwrap(); + let source_generation: i64 = source.try_get("", "root_generation").unwrap(); + let source_certificate: Vec = source.try_get("", "root_certificate_digest").unwrap(); + assert!(txn.query_all_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "SELECT * FROM mst2_metadata_read_source_entries($1::uuid,$2::bigint,$3::uuid,$4,$5,$6,'[]'::jsonb)", + [old.0.clone().into(),old.1.into(),source_id.clone().into(),source_root.clone().into(),source_generation.into(),source_certificate.clone().into()])).await.is_err()); + txn.rollback().await.unwrap(); + let txn = transaction(&fixture).await; + assert!(txn.query_all_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "SELECT * FROM mst2_metadata_read_source_entries($1::uuid,$2::bigint,$3::uuid,$4,$5,$6,'[]'::jsonb)", + [old.0.clone().into(),next.into(),source_id.into(),source_root.into(),source_generation.into(),source_certificate.into()])).await.unwrap().is_empty()); + txn.commit().await.unwrap(); + resume.wait().await; + let response = tokio::time::timeout(Duration::from_secs(4), pending) + .await + .unwrap() + .unwrap(); + assert!(response.status().is_server_error()); + assert_eq!( + q_count( + &fixture, + "SELECT count(*) FROM {q}.mst2_metadata_reader_operation WHERE state='ACTIVE'" + ) + .await, + 1 + ); + assert_eq!(q_count(&fixture,"SELECT count(*) FROM {q}.mst2_metadata_root_anchor WHERE anchor_kind IN ('REQUEST','READER')").await,2); + let txn = transaction(&fixture).await; + finish(&txn, &(old.0, next)).await; + txn.commit().await.unwrap(); + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn rolled_back_reader_issuance_is_unpublished_and_extreme_issuance_fails_closed() { + let fixture = Fixture::new_with_pg_config(true).await; + let initial = q_count( + &fixture, + "SELECT high_water FROM {q}.mst2_metadata_reader_issuance", + ) + .await; + let txn = transaction(&fixture).await; + let unpublished = admission(&txn, &fixture).await; + txn.rollback().await.unwrap(); + assert_eq!( + q_count( + &fixture, + "SELECT high_water FROM {q}.mst2_metadata_reader_issuance" + ) + .await, + initial + ); + assert_eq!( + q_count( + &fixture, + "SELECT count(*) FROM {q}.mst2_metadata_reader_operation" + ) + .await, + 0 + ); + let txn = transaction(&fixture).await; + let published = admission(&txn, &fixture).await; + assert_eq!(published.1, unpublished.1); + txn.commit().await.unwrap(); + let txn = transaction(&fixture).await; + assert_eq!( + txn.query_one_raw(Statement::from_string( + DbBackend::Postgres, + "SELECT mst2_metadata_next_reader_issuance(9223372036854775806)" + )) + .await + .unwrap() + .unwrap() + .try_get_by_index::(0) + .unwrap(), + i64::MAX + ); + assert!( + txn.query_one_raw(Statement::from_string( + DbBackend::Postgres, + "SELECT mst2_metadata_next_reader_issuance(9223372036854775807)" + )) + .await + .is_err() + ); + txn.rollback().await.unwrap(); + assert_eq!( + q_count( + &fixture, + "SELECT high_water FROM {q}.mst2_metadata_reader_issuance" + ) + .await, + published.1 + ); + let txn = transaction(&fixture).await; + finish(&txn, &published).await; + txn.commit().await.unwrap(); +} + +#[tokio::test] +#[ignore = "capacity soak: run explicitly against dedicated PostgreSQL"] +async fn reader_history_remains_bounded_after_more_than_65536_committed_http_requests() { + let fixture = Fixture::new_with_pg_config(true).await; + for _ in 0..65_537 { + success_json( + fixture + .app + .clone() + .oneshot(fixture.request( + "POST", + "lookup", + Body::from(json!({"paths":["/file"]}).to_string()), + )) + .await + .unwrap(), + ) + .await; + } + assert_eq!( + q_count( + &fixture, + "SELECT count(*) FROM {q}.mst2_metadata_reader_operation" + ) + .await, + 1 + ); + assert!( + q_count( + &fixture, + "SELECT high_water FROM {q}.mst2_metadata_reader_issuance" + ) + .await + > 65_536 + ); + fixture.counts.assert(0, 0); +} diff --git a/src/api/router/snapshot_reader_retention_upgrade_tests.rs b/src/api/router/snapshot_reader_retention_upgrade_tests.rs new file mode 100644 index 00000000..1f798a8e --- /dev/null +++ b/src/api/router/snapshot_reader_retention_upgrade_tests.rs @@ -0,0 +1,199 @@ +use super::*; +use crate::jupiter::{ + migration::test_upgrade_reader_retention, + storage::qualified_metadata_family::{ + RootedLookupStatus, RootedQualifiedMetadataRepository, + provision_or_verify_rooted_qualified_family, reader_previous_fixture::restore_previous, + }, +}; + +async fn permanent_rows(fixture: &Fixture) -> Value { + let txn = transaction(fixture).await; + let rows=txn.query_one_raw(Statement::from_string(DbBackend::Postgres, + "SELECT jsonb_build_object( + 'sessions',(SELECT jsonb_agg(to_jsonb(x) ORDER BY snapshot_id,session_incarnation) FROM mst2_qualified_session_incarnation x), + 'leases',(SELECT jsonb_agg(to_jsonb(x) ORDER BY lease_id) FROM mst2_qualified_lease_binding x), + 'sources',(SELECT jsonb_agg(to_jsonb(x) ORDER BY attestation_id) FROM mst2_metadata_source_root_attestation x), + 'certificates',(SELECT jsonb_agg(to_jsonb(x) ORDER BY page_id,generation) FROM mst2_metadata_page_certificate x)) AS rows")) + .await.unwrap().unwrap().try_get("","rows").unwrap(); + txn.commit().await.unwrap(); + rows +} + +#[tokio::test] +async fn current_reader_retention_migration_verifies_fresh_family_without_rewriting_history() { + let fixture = Fixture::new_with_pg_config(true).await; + let before = permanent_rows(&fixture).await; + let high_water = q_count( + &fixture, + "SELECT high_water FROM {q}.mst2_metadata_reader_issuance", + ) + .await; + let mono = fixture.state.storage.mono_storage(); + let core = mono.get_connection(); + test_upgrade_reader_retention(core).await.unwrap(); + assert_eq!(permanent_rows(&fixture).await, before); + assert_eq!( + q_count( + &fixture, + "SELECT high_water FROM {q}.mst2_metadata_reader_issuance" + ) + .await, + high_water + ); + provision_or_verify_rooted_qualified_family(core) + .await + .unwrap(); + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn captured_previous_q_upgrade_preserves_old_sid_source_proofs_and_active_legacy_reader() { + let fixture = Fixture::new_with_pg_config(true).await; + let pinned = fixture + .state + .storage + .snapshot_context(&fixture.snapshot, &fixture.lease) + .await + .unwrap(); + let before = permanent_rows(&fixture).await; + let txn = transaction(&fixture).await; + let active = admission(&txn, &fixture).await; + txn.commit().await.unwrap(); + let schema = q_schema(&fixture).await; + let mono = fixture.state.storage.mono_storage(); + let core = mono.get_connection(); + restore_previous(core, &schema, true).await; + assert_eq!(permanent_rows(&fixture).await, before); + test_upgrade_reader_retention(core).await.unwrap(); + assert_eq!(permanent_rows(&fixture).await, before); + assert_eq!(q_count(&fixture,"SELECT count(*) FROM {q}.mst2_metadata_reader_operation WHERE reader_issuance=0 AND state='ACTIVE'").await,1); + assert_eq!(q_count(&fixture,"SELECT count(*) FROM {q}.mst2_metadata_root_anchor WHERE reader_issuance=0 AND anchor_kind IN ('REQUEST','READER')").await,2); + let restarted = + RootedQualifiedMetadataRepository::open(core, &fixture.state.storage.config().database) + .await + .unwrap(); + let fixed = restarted + .context(&fixture.snapshot, &fixture.lease, &pinned.built.instance_id) + .await + .unwrap(); + assert_eq!(fixed.built.descriptor, pinned.built.descriptor); + assert_eq!(fixed.commit_oid, pinned.commit_oid); + let lookup = restarted + .lookup_metadata(&fixed, &["/file".to_owned()]) + .await + .unwrap(); + assert!(matches!( + lookup.results.as_slice(), + [RootedLookupStatus::File { .. }] + )); + assert_eq!(permanent_rows(&fixture).await, before); + assert_eq!(q_count(&fixture,"SELECT count(*) FROM {q}.mst2_metadata_reader_operation WHERE reader_issuance=0 AND state='ACTIVE'").await,1); + let txn = transaction(&fixture).await; + finish(&txn, &(active.0.clone(), 0)).await; + txn.commit().await.unwrap(); + let txn = transaction(&fixture).await; + assert!(prune(&txn, 64).await >= 1); + txn.commit().await.unwrap(); + assert_eq!( + q_count( + &fixture, + "SELECT count(*) FROM {q}.mst2_metadata_reader_operation WHERE reader_issuance=0" + ) + .await, + 0 + ); + provision_or_verify_rooted_qualified_family(core) + .await + .unwrap(); + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn captured_previous_core_without_q_upgrades_before_new_provisioning() { + let temp = tempfile::tempdir().unwrap(); + let (config, _schema) = test_db_config(temp.path()).await; + let core = crate::jupiter::storage::init::database_connection(&config) + .await + .unwrap(); + let old_schema: String = core + .query_one_raw(Statement::from_string( + DbBackend::Postgres, + "SELECT metadata_schema FROM mst2_metadata_namespace WHERE graph_domain='qualified-v1'", + )) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap(); + restore_previous(&core, &old_schema, false).await; + test_upgrade_reader_retention(&core).await.unwrap(); + provision_or_verify_rooted_qualified_family(&core) + .await + .unwrap(); + let schema: String = core + .query_one_raw(Statement::from_string( + DbBackend::Postgres, + "SELECT metadata_schema FROM mst2_metadata_namespace WHERE graph_domain='qualified-v1'", + )) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap(); + assert_ne!(schema, old_schema); + let water: i64 = core + .query_one_raw(Statement::from_string( + DbBackend::Postgres, + format!( + "SELECT high_water FROM \"{}\".mst2_metadata_reader_issuance", + schema.replace('"', "\"\"") + ), + )) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap(); + assert_eq!(water, 0); + provision_or_verify_rooted_qualified_family(&core) + .await + .unwrap(); +} + +#[tokio::test] +async fn captured_previous_tampered_q_rejects_upgrade_without_partial_issuance_schema() { + let fixture = Fixture::new_with_pg_config(true).await; + let before = permanent_rows(&fixture).await; + let schema = q_schema(&fixture).await; + let mono = fixture.state.storage.mono_storage(); + let core = mono.get_connection(); + restore_previous(core, &schema, true).await; + core.execute_unprepared(&format!( + "ALTER TABLE \"{}\".mst2_metadata_reader_operation ADD COLUMN unexpected integer", + schema.replace('"', "\"\"") + )) + .await + .unwrap(); + assert!(test_upgrade_reader_retention(core).await.is_err()); + assert_eq!(permanent_rows(&fixture).await, before); + let exists: bool = core + .query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT to_regclass($1) IS NULL AS absent", + [format!("{schema}.mst2_metadata_reader_issuance").into()], + )) + .await + .unwrap() + .unwrap() + .try_get("", "absent") + .unwrap(); + assert!(exists); + let implementation:String=core.query_one_raw(Statement::from_string(DbBackend::Postgres, + "SELECT encode(implementation_fingerprint,'hex') FROM mst2_qualified_family_policy WHERE singleton=1")) + .await.unwrap().unwrap().try_get_by_index(0).unwrap(); + assert_eq!( + implementation, + "6d7095dc052e60bf6de21a987b72e23fdfbfe819355f0e847d2a7f3c1cbd3f27" + ); +} diff --git a/src/api/router/snapshot_rooted_metadata_tests.rs b/src/api/router/snapshot_rooted_metadata_tests.rs index 7c944913..46316703 100644 --- a/src/api/router/snapshot_rooted_metadata_tests.rs +++ b/src/api/router/snapshot_rooted_metadata_tests.rs @@ -7,6 +7,9 @@ use crate::jupiter::storage::qualified_metadata_family::{ with_rooted_source_temporary_shadow, }; +#[path = "snapshot_reader_retention_tests.rs"] +mod reader_retention; + async fn q_schema(fixture: &Fixture) -> String { fixture .state diff --git a/src/jupiter/migration/m20261008_000400_add_mst2_reader_retention.rs b/src/jupiter/migration/m20261008_000400_add_mst2_reader_retention.rs new file mode 100644 index 00000000..5be3fd26 --- /dev/null +++ b/src/jupiter/migration/m20261008_000400_add_mst2_reader_retention.rs @@ -0,0 +1,397 @@ +//! Upgrade only the trusted Q reader lifecycle, preserving fixed source history. + +use sea_orm::{ConnectionTrait, DbBackend, Statement, Value}; +use sea_orm_migration::prelude::*; + +use crate::jupiter::storage::qualified_metadata_family::{ + implementation_fingerprint, render_family, +}; + +const PREVIOUS_IMPLEMENTATION: &str = + "6d7095dc052e60bf6de21a987b72e23fdfbfe819355f0e847d2a7f3c1cbd3f27"; + +#[derive(DeriveMigrationName)] +pub struct Migration; + +fn rejected(message: &str) -> DbErr { + DbErr::Custom(message.into()) +} + +fn identifier(value: &str) -> String { + format!("\"{}\"", value.replace('"', "\"\"")) +} + +fn literal(value: &str) -> String { + format!("'{}'", value.replace('\'', "''")) +} + +fn trusted_function(source: &str, name: &str) -> Result { + let prefix = format!("CREATE FUNCTION {name}("); + let start = source + .find(&prefix) + .ok_or_else(|| rejected("reader migration trusted function is missing"))?; + let function = &source[start..]; + let body = function + .find("AS $") + .ok_or_else(|| rejected("reader migration function body is missing"))? + + 3; + let tag_end = function[body + 1..] + .find('$') + .ok_or_else(|| rejected("reader migration function delimiter is missing"))? + + body + + 2; + let tag = &function[body..tag_end]; + let end = function[tag_end..] + .find(tag) + .ok_or_else(|| rejected("reader migration function terminator is missing"))? + + tag_end + + tag.len(); + if function.as_bytes().get(end) != Some(&b';') { + return Err(rejected("reader migration function terminator is invalid")); + } + Ok(function[..=end].replacen("CREATE FUNCTION", "CREATE OR REPLACE FUNCTION", 1)) +} + +async fn digest( + connection: &C, + sql: String, + values: Vec, +) -> Result, DbErr> { + connection + .query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + sql, + values, + )) + .await? + .ok_or_else(|| rejected("reader migration physical fingerprint is missing"))? + .try_get("", "fingerprint") +} + +async fn catalog( + connection: &C, + core_oid: i64, + q_oid: i64, + exempt_oid: i64, +) -> Result, DbErr> { + let source = include_str!("../storage/qualified_family_catalog.sql") + .replace("$CORE_OID$", "$1::bigint::oid") + .replace("$Q_OID$", "$2::bigint::oid") + .replace("$EXEMPT_Q_OID$", "$3::bigint::oid"); + digest( + connection, + source, + vec![core_oid.into(), q_oid.into(), exempt_oid.into()], + ) + .await +} + +async fn shape( + connection: &C, + q_oid: i64, + namespace: &str, + storage: &str, +) -> Result, DbErr> { + let source = include_str!("../storage/qualified_family_shape.sql") + .replace("q_oid", "$1::bigint::oid") + .replace("n_uuid", "$2::uuid") + .replace("s_uuid", "$3::text"); + digest( + connection, + source, + vec![q_oid.into(), namespace.into(), storage.into()], + ) + .await +} + +struct Family { + schema: String, + oid: i64, + namespace: String, + storage: String, + catalog: Vec, +} + +#[async_trait::async_trait] +impl MigrationTrait for Migration { + async fn up(&self, manager: &SchemaManager) -> Result<(), DbErr> { + if manager.get_database_backend() != DbBackend::Postgres { + return Err(rejected( + "qualified reader retention requires primary PostgreSQL", + )); + } + let connection = manager.get_connection(); + let captured = connection.query_one_raw(Statement::from_string(DbBackend::Postgres, + "SELECT n.nspname,n.oid::bigint,registry.mono_lock_key2 FROM pg_catalog.pg_namespace n + JOIN mst2_metadata_namespace registry ON registry.singleton=1 AND registry.core_schema=n.nspname + AND registry.core_schema_oid=n.oid WHERE n.nspname=current_schema()")) + .await?.ok_or_else(|| rejected("reader migration requires its captured core schema"))?; + let core: String = captured.try_get("", "nspname")?; + let core_oid: i64 = captured.try_get("", "oid")?; + let mono: i32 = captured.try_get("", "mono_lock_key2")?; + let c = identifier(&core); + connection + .execute_unprepared("SELECT pg_catalog.set_config('lock_timeout','5000ms',true)") + .await?; + connection + .execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT pg_catalog.pg_advisory_xact_lock(1297043024,$1)", + [mono.into()], + )) + .await?; + connection + .execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT pg_catalog.pg_advisory_xact_lock(1296718001,pg_catalog.hashtext($1))", + [core.clone().into()], + )) + .await?; + let namespaces = connection.query_all_raw(Statement::from_string(DbBackend::Postgres, + format!("SELECT metadata_schema FROM {c}.mst2_metadata_namespace ORDER BY namespace_uuid"))).await?; + if !(1..=2).contains(&namespaces.len()) { + return Err(rejected( + "reader migration namespace registry is not bounded", + )); + } + for namespace in namespaces { + let schema: String = namespace.try_get("", "metadata_schema")?; + connection + .execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT pg_catalog.pg_advisory_xact_lock(1296717362,pg_catalog.hashtext($1))", + [schema.into()], + )) + .await?; + } + connection.execute_unprepared(&format!( + "LOCK TABLE {c}.mst2_metadata_namespace,{c}.mst2_qualified_family_policy IN ACCESS EXCLUSIVE MODE" + )).await?; + let policy=connection.query_one_raw(Statement::from_string(DbBackend::Postgres, + format!("SELECT implementation_fingerprint,expected_shape,authority_catalog FROM {c}.mst2_qualified_family_policy WHERE singleton=1"))) + .await?.ok_or_else(|| rejected("reader migration trusted policy is missing"))?; + let implementation: Vec = policy.try_get("", "implementation_fingerprint")?; + let current = implementation_fingerprint(); + let previous = + hex::decode(PREVIOUS_IMPLEMENTATION).map_err(|error| rejected(&error.to_string()))?; + if implementation != previous && implementation != current { + return Err(rejected( + "reader migration refuses an unsupported prior implementation", + )); + } + let rows=connection.query_all_raw(Statement::from_string(DbBackend::Postgres,format!( + "SELECT namespace_uuid::text,metadata_schema,metadata_schema_oid::bigint,metadata_storage_uuid, + implementation_fingerprint,catalog_fingerprint FROM {c}.mst2_metadata_namespace WHERE graph_domain='qualified-v1'"))).await?; + let family = match rows.as_slice() { + [] => None, + [row] => Some(Family { + schema: row.try_get("", "metadata_schema")?, + oid: row.try_get("", "metadata_schema_oid")?, + namespace: row.try_get("", "namespace_uuid")?, + storage: row.try_get("", "metadata_storage_uuid")?, + catalog: row.try_get("", "catalog_fingerprint")?, + }), + _ => return Err(rejected("reader migration requires at most one Q family")), + }; + let q_oid = family.as_ref().map_or(0, |family| family.oid); + if policy.try_get::>("", "authority_catalog")? + != catalog(connection, core_oid, 0, q_oid).await? + { + return Err(rejected( + "reader migration prior core authority catalog changed", + )); + } + let scope = connection + .query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + format!("SELECT {c}.mst2_route_scope_valid($1) AS valid"), + [core.clone().into()], + )) + .await? + .ok_or_else(|| rejected("reader migration core scope is missing"))?; + if !scope.try_get::("", "valid")? { + return Err(rejected("reader migration prior core scope changed")); + } + if let Some(family) = &family { + let q = identifier(&family.schema); + let identity=connection.query_one_raw(Statement::from_sql_and_values(DbBackend::Postgres,format!( + "SELECT EXISTS(SELECT 1 FROM {c}.mst2_metadata_namespace n + JOIN {c}.mst2_metadata_namespace g ON g.singleton=1 + JOIN {q}.mst2_metadata_family_identity i ON i.singleton=1 + JOIN {q}.mst2_metadata_storage_scope s ON s.singleton=1 + JOIN pg_catalog.pg_namespace physical ON physical.oid=n.metadata_schema_oid AND physical.nspname=n.metadata_schema + WHERE n.namespace_uuid=$1::uuid AND n.metadata_schema='mst2q_'||replace(n.namespace_uuid::text,'-','') + AND n.singleton IS NULL AND n.family_identity='v3-rooted-qualified-1' AND n.graph_domain='qualified-v1' + AND n.admission_state='ROOTED_Q_ADMITTED' AND n.collector_state='ENABLED' + AND ROW(n.core_schema,n.core_schema_oid,n.database_name,n.database_oid,n.storage_uuid,n.server_address,n.server_port,n.mono_lock_key2) + IS NOT DISTINCT FROM ROW(g.core_schema,g.core_schema_oid,g.database_name,g.database_oid,g.storage_uuid,g.server_address,g.server_port,g.mono_lock_key2) + AND n.metadata_storage_uuid<>g.storage_uuid AND n.implementation_fingerprint=$2 + AND ROW(i.namespace_uuid,i.storage_uuid,i.core_schema_oid,i.metadata_schema_oid,i.family_identity,i.implementation_fingerprint) + IS NOT DISTINCT FROM ROW(n.namespace_uuid,n.metadata_storage_uuid,n.core_schema_oid,n.metadata_schema_oid,n.family_identity,n.implementation_fingerprint) + AND s.storage_uuid=i.storage_uuid) AS valid"),[family.namespace.clone().into(),implementation.clone().into()])).await? + .ok_or_else(|| rejected("reader migration prior Q identity is missing"))?; + if !identity.try_get::("", "valid")? + || catalog(connection, core_oid, family.oid, 0).await? != family.catalog + || shape(connection, family.oid, &family.namespace, &family.storage).await? + != policy.try_get::>("", "expected_shape")? + { + return Err(rejected( + "reader migration prior Q physical stamp or shape changed", + )); + } + } + if implementation == current { + return Ok(()); + } + + let registration = include_str!("m20261008_000200_rooted_qualified_family.sql") + .replace("$CORE_SCHEMA$", &c) + .replace("$CORE_LITERAL$", &literal(&core)) + .replace("$IMPLEMENTATION_SHA$", &hex::encode(¤t)); + connection + .execute_unprepared(&trusted_function( + ®istration, + "mst2_route_family_registration_guard", + )?) + .await?; + + let template_namespace = uuid::Uuid::new_v4().to_string(); + let template_schema = format!("mst2q_{}", template_namespace.replace('-', "")); + let template_storage = uuid::Uuid::new_v4().to_string(); + let t = identifier(&template_schema); + connection + .execute_unprepared(&format!("CREATE SCHEMA {t}")) + .await?; + let template_oid: i64 = connection + .query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT oid::bigint FROM pg_catalog.pg_namespace WHERE nspname=$1", + [template_schema.clone().into()], + )) + .await? + .ok_or_else(|| rejected("reader migration trusted template schema is missing"))? + .try_get("", "oid")?; + connection + .execute_unprepared(&render_family( + &core, + core_oid, + &template_schema, + template_oid, + &template_namespace, + &template_storage, + )) + .await?; + let expected_shape = shape( + connection, + template_oid, + &template_namespace, + &template_storage, + ) + .await?; + connection + .execute_unprepared(&format!( + "SET LOCAL search_path={c},pg_catalog,pg_temp; DROP SCHEMA {t} CASCADE" + )) + .await?; + + if let Some(family) = &family { + let q = identifier(&family.schema); + connection.execute_unprepared(&format!("LOCK TABLE {q}.mst2_metadata_reader_operation,{q}.mst2_metadata_root_anchor IN ACCESS EXCLUSIVE MODE")).await?; + let rendered = render_family( + &core, + core_oid, + &family.schema, + family.oid, + &family.namespace, + &family.storage, + ); + let mut functions = String::new(); + for name in [ + "mst2_metadata_dml_barrier", + "mst2_metadata_gc_enabled", + "mst2_metadata_root_anchor_guard", + "mst2_metadata_serving_complete", + "mst2_metadata_next_reader_issuance", + "mst2_metadata_reader_issuance_guard", + "mst2_metadata_reader_guard", + "mst2_metadata_prune_readers", + "mst2_metadata_begin_reader", + "mst2_metadata_finish_reader", + "mst2_metadata_read_source_entries", + "mst2_metadata_cleanup_expired", + "mst2_metadata_gc_owner_cleanup", + ] { + functions.push_str(&trusted_function(&rendered, name)?); + functions.push('\n'); + } + let upgrade = include_str!("m20261008_000400_reader_retention_upgrade.sql") + .replace("$Q_SCHEMA$", &q) + .replace("$READER_FUNCTIONS_SQL$", &functions); + connection.execute_unprepared(&upgrade).await?; + if shape(connection, family.oid, &family.namespace, &family.storage).await? + != expected_shape + { + return Err(rejected( + "reader migration upgraded Q differs from the trusted complete shape", + )); + } + } + + // Fingerprints include trigger identities/enabled state, but not table + // values. Preserve trigger OIDs while writing only these controlled stamps. + let authority = catalog(connection, core_oid, 0, q_oid).await?; + let full = if q_oid == 0 { + None + } else { + Some(catalog(connection, core_oid, q_oid, 0).await?) + }; + connection.execute_unprepared(&format!("ALTER TABLE {c}.mst2_qualified_family_policy DISABLE TRIGGER mst2_route_family_policy_immutable")).await?; + connection.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres,format!( + "UPDATE {c}.mst2_qualified_family_policy SET implementation_fingerprint=$1,expected_shape=$2,authority_catalog=$3 WHERE singleton=1"), + [current.clone().into(),expected_shape.clone().into(),authority.clone().into()])).await?; + connection.execute_unprepared(&format!("ALTER TABLE {c}.mst2_qualified_family_policy ENABLE TRIGGER mst2_route_family_policy_immutable")).await?; + if let (Some(family), Some(full)) = (&family, &full) { + let q = identifier(&family.schema); + connection.execute_unprepared(&format!( + "ALTER TABLE {q}.mst2_metadata_family_identity DISABLE TRIGGER mst2_00_family_barrier; + ALTER TABLE {q}.mst2_metadata_family_identity DISABLE TRIGGER mst2_metadata_identity_immutable; + ALTER TABLE {c}.mst2_metadata_namespace DISABLE TRIGGER mst2_00_route_statement_barrier; + ALTER TABLE {c}.mst2_metadata_namespace DISABLE TRIGGER mst2_route_immutable")).await?; + connection.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres, + format!("UPDATE {q}.mst2_metadata_family_identity SET implementation_fingerprint=$1 WHERE singleton=1"),[current.clone().into()])).await?; + connection.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres,format!( + "UPDATE {c}.mst2_metadata_namespace SET implementation_fingerprint=$1,catalog_fingerprint=$2 WHERE namespace_uuid=$3::uuid"), + [current.clone().into(),full.clone().into(),family.namespace.clone().into()])).await?; + connection.execute_unprepared(&format!( + "ALTER TABLE {q}.mst2_metadata_family_identity ENABLE TRIGGER mst2_00_family_barrier; + ALTER TABLE {q}.mst2_metadata_family_identity ENABLE TRIGGER mst2_metadata_identity_immutable; + ALTER TABLE {c}.mst2_metadata_namespace ENABLE TRIGGER mst2_00_route_statement_barrier; + ALTER TABLE {c}.mst2_metadata_namespace ENABLE TRIGGER mst2_route_immutable")).await?; + } + connection + .execute_unprepared(&format!("SET LOCAL search_path={c},pg_catalog,pg_temp")) + .await?; + if catalog(connection, core_oid, 0, q_oid).await? != authority { + return Err(rejected( + "reader migration final core authority catalog changed", + )); + } + if let (Some(family), Some(full)) = (&family, &full) + && (catalog(connection, core_oid, q_oid, 0).await? != *full + || shape(connection, family.oid, &family.namespace, &family.storage).await? + != expected_shape) + { + return Err(rejected("reader migration final Q fingerprint changed")); + } + Ok(()) + } + + async fn down(&self, _manager: &SchemaManager) -> Result<(), DbErr> { + // Preserve committed issuance fences and the complete reader identity. + Ok(()) + } + + fn use_transaction(&self) -> Option { + Some(true) + } +} diff --git a/src/jupiter/migration/m20261008_000400_reader_retention_upgrade.sql b/src/jupiter/migration/m20261008_000400_reader_retention_upgrade.sql new file mode 100644 index 00000000..cb7b6876 --- /dev/null +++ b/src/jupiter/migration/m20261008_000400_reader_retention_upgrade.sql @@ -0,0 +1,46 @@ +SET LOCAL search_path=$Q_SCHEMA$,pg_catalog,pg_temp; +ALTER TABLE mst2_metadata_reader_operation DISABLE TRIGGER mst2_00_family_barrier; +ALTER TABLE mst2_metadata_reader_operation DISABLE TRIGGER mst2_metadata_reader_guard; +ALTER TABLE mst2_metadata_reader_operation DISABLE TRIGGER mst2_metadata_reader_complete; +ALTER TABLE mst2_metadata_root_anchor DISABLE TRIGGER mst2_00_family_barrier; +ALTER TABLE mst2_metadata_root_anchor DISABLE TRIGGER mst2_metadata_root_anchor_guard; + +ALTER TABLE mst2_metadata_reader_operation + ADD COLUMN reader_issuance bigint NOT NULL DEFAULT 0 CHECK(reader_issuance>=0), + ADD COLUMN terminal_xid bigint; +UPDATE mst2_metadata_reader_operation SET terminal_xid=0 WHERE state IN ('FINISHED','EXPIRED'); +ALTER TABLE mst2_metadata_reader_operation ALTER COLUMN reader_issuance DROP DEFAULT; +ALTER TABLE mst2_metadata_reader_operation + ADD CONSTRAINT mst2_metadata_reader_identity UNIQUE(operation_id,reader_issuance), + ADD CONSTRAINT mst2_metadata_reader_terminal_state CHECK((state='ACTIVE')=(terminal_xid IS NULL)); +CREATE TABLE mst2_metadata_reader_issuance ( + singleton smallint PRIMARY KEY CHECK(singleton=1),high_water bigint NOT NULL CHECK(high_water>=0) +); +INSERT INTO mst2_metadata_reader_issuance VALUES(1,0); +CREATE INDEX mst2_metadata_reader_terminal ON mst2_metadata_reader_operation(terminal_xid,operation_id) WHERE state IN ('FINISHED','EXPIRED'); +ALTER TABLE mst2_metadata_root_anchor ADD COLUMN reader_issuance bigint; +UPDATE mst2_metadata_root_anchor SET reader_issuance=0 WHERE reader_operation_id IS NOT NULL; +ALTER TABLE mst2_metadata_root_anchor + DROP CONSTRAINT mst2_metadata_root_anchor_reader_operation_id_fkey, + ADD CONSTRAINT mst2_metadata_root_anchor_reader_identity FOREIGN KEY(reader_operation_id,reader_issuance) + REFERENCES mst2_metadata_reader_operation(operation_id,reader_issuance) MATCH FULL, + ADD CONSTRAINT mst2_metadata_root_anchor_reader_kind CHECK((anchor_kind IN ('REQUEST','READER'))=(reader_operation_id IS NOT NULL)), + ADD CONSTRAINT mst2_metadata_root_anchor_reader_fields CHECK((reader_operation_id IS NULL)=(reader_issuance IS NULL)); +CREATE INDEX mst2_metadata_root_anchor_reader ON mst2_metadata_root_anchor(reader_operation_id,reader_issuance) WHERE reader_operation_id IS NOT NULL; +DROP FUNCTION mst2_metadata_begin_reader(text,text,text); +DROP FUNCTION mst2_metadata_finish_reader(uuid); +DROP FUNCTION mst2_metadata_read_source_entries(uuid,uuid,bytea,bigint,bytea,jsonb); + +$READER_FUNCTIONS_SQL$ + +CREATE TRIGGER mst2_00_family_barrier BEFORE INSERT OR UPDATE OR DELETE ON mst2_metadata_reader_issuance + FOR EACH STATEMENT EXECUTE FUNCTION mst2_metadata_dml_barrier(); +CREATE TRIGGER mst2_metadata_truncate_guard BEFORE TRUNCATE ON mst2_metadata_reader_issuance + FOR EACH STATEMENT EXECUTE FUNCTION mst2_metadata_immutable(); +CREATE TRIGGER mst2_metadata_reader_issuance_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_metadata_reader_issuance + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_reader_issuance_guard(); +ALTER TABLE mst2_metadata_reader_operation ENABLE TRIGGER mst2_00_family_barrier; +ALTER TABLE mst2_metadata_reader_operation ENABLE TRIGGER mst2_metadata_reader_guard; +ALTER TABLE mst2_metadata_reader_operation ENABLE TRIGGER mst2_metadata_reader_complete; +ALTER TABLE mst2_metadata_root_anchor ENABLE TRIGGER mst2_00_family_barrier; +ALTER TABLE mst2_metadata_root_anchor ENABLE TRIGGER mst2_metadata_root_anchor_guard; diff --git a/src/jupiter/migration/mod.rs b/src/jupiter/migration/mod.rs index 018716bc..d44427fc 100644 --- a/src/jupiter/migration/mod.rs +++ b/src/jupiter/migration/mod.rs @@ -154,10 +154,31 @@ mod m20261007_000600_add_mst2_storage_routes; mod m20261008_000100_add_mst2_chunk_maps; mod m20261008_000200_add_mst2_rooted_qualified_family; mod m20261008_000300_add_mst2_chunk_map_retention; +mod m20261008_000400_add_mst2_reader_retention; mod runner; pub use m20260905_000100_add_push_queue::ensure_queue_control_seed; pub use runner::apply_migrations; +#[cfg(test)] +pub(crate) async fn test_upgrade_reader_retention( + connection: &sea_orm::DatabaseConnection, +) -> Result<(), sea_orm::DbErr> { + use sea_orm::TransactionTrait; + use sea_orm_migration::{MigrationTrait, SchemaManager}; + + let txn = connection.begin().await?; + let result = m20261008_000400_add_mst2_reader_retention::Migration + .up(&SchemaManager::new(&txn)) + .await; + match result { + Ok(()) => txn.commit().await, + Err(error) => { + txn.rollback().await?; + Err(error) + } + } +} + /// Primary key `BIGINT` (not DB auto-increment); the application assigns `id` (e.g. `idgenerator::IdInstance::next_id`). fn pk_bigint(name: T) -> ColumnDef { big_integer(name).primary_key().take() @@ -296,6 +317,7 @@ impl MigratorTrait for Migrator { Box::new(m20261008_000100_add_mst2_chunk_maps::Migration), Box::new(m20261008_000200_add_mst2_rooted_qualified_family::Migration), Box::new(m20261008_000300_add_mst2_chunk_map_retention::Migration), + Box::new(m20261008_000400_add_mst2_reader_retention::Migration), ] } } @@ -1206,7 +1228,7 @@ mod tests { async fn import_repo_alias_rows_canonicalized() { let names = migration_names(); assert_eq!( - &names[names.len() - 17..names.len() - 8], + &names[names.len() - 18..names.len() - 9], &[ "m20260923_000200_canonicalize_import_repo_paths".to_string(), "m20260925_000100_media_paging".to_string(), @@ -1221,38 +1243,42 @@ mod tests { "native retention, metadata installation and view tables follow media paging" ); assert_eq!( - &names[names.len() - 8], + &names[names.len() - 9], "m20261007_000200_add_mst2_metadata_generations" ); assert_eq!( - &names[names.len() - 7], + &names[names.len() - 8], "m20261007_000300_add_mst2_metadata_lifetime_history" ); assert_eq!( - &names[names.len() - 6], + &names[names.len() - 7], "m20261007_000400_add_mst2_qualified_metadata_gc" ); assert_eq!( - &names[names.len() - 5], + &names[names.len() - 6], "m20261007_000500_add_mst2_install_capability" ); assert_eq!( - &names[names.len() - 4], + &names[names.len() - 5], "m20261007_000600_add_mst2_storage_routes" ); assert_eq!( - &names[names.len() - 3], + &names[names.len() - 4], "m20261008_000100_add_mst2_chunk_maps" ); assert_eq!( - &names[names.len() - 2], + &names[names.len() - 3], "m20261008_000200_add_mst2_rooted_qualified_family" ); assert_eq!( - names.last().unwrap(), + &names[names.len() - 2], "m20261008_000300_add_mst2_chunk_map_retention" ); + assert_eq!( + names.last().unwrap(), + "m20261008_000400_add_mst2_reader_retention" + ); let db = alias_db().await; insert_repo(&db, 1, "/third-party//a").await; insert_repo(&db, 2, "/third-party/b/").await; diff --git a/src/jupiter/storage/qualified_metadata_anchors.sql b/src/jupiter/storage/qualified_metadata_anchors.sql index 08b3d7ea..e7db38e1 100644 --- a/src/jupiter/storage/qualified_metadata_anchors.sql +++ b/src/jupiter/storage/qualified_metadata_anchors.sql @@ -46,8 +46,16 @@ CREATE TABLE mst2_metadata_reader_operation ( snapshot_id text NOT NULL,session_incarnation uuid NOT NULL,root_page bytea NOT NULL,root_generation bigint NOT NULL, lease_epoch bigint NOT NULL CHECK(lease_epoch>0),hard_deadline_unix bigint NOT NULL, state text NOT NULL CHECK(state IN ('ACTIVE','FINISHED','EXPIRED')), + reader_issuance bigint NOT NULL CHECK(reader_issuance>=0),terminal_xid bigint, + CONSTRAINT mst2_metadata_reader_identity UNIQUE(operation_id,reader_issuance), + CONSTRAINT mst2_metadata_reader_terminal_state CHECK((state='ACTIVE')=(terminal_xid IS NULL)), FOREIGN KEY(snapshot_id,session_incarnation) REFERENCES mst2_qualified_session_incarnation(snapshot_id,session_incarnation) ); +CREATE TABLE mst2_metadata_reader_issuance ( + singleton smallint PRIMARY KEY CHECK(singleton=1),high_water bigint NOT NULL CHECK(high_water>=0) +); +INSERT INTO mst2_metadata_reader_issuance VALUES(1,0); +CREATE INDEX mst2_metadata_reader_terminal ON mst2_metadata_reader_operation(terminal_xid,operation_id) WHERE state IN ('FINISHED','EXPIRED'); CREATE INDEX mst2_metadata_reader_active_lease ON mst2_metadata_reader_operation(lease_id,state,operation_id); CREATE INDEX mst2_metadata_reader_active_deadline ON mst2_metadata_reader_operation(hard_deadline_unix,operation_id) WHERE state='ACTIVE'; CREATE TABLE mst2_metadata_root_anchor ( @@ -56,13 +64,17 @@ CREATE TABLE mst2_metadata_root_anchor ( root_page bytea NOT NULL,root_generation bigint NOT NULL,root_certificate_digest bytea NOT NULL, prepare_id text REFERENCES mst2_metadata_prepare(prepare_id), snapshot_id text,session_incarnation uuid,lease_id text REFERENCES mst2_qualified_lease_binding(lease_id), - reader_operation_id uuid REFERENCES mst2_metadata_reader_operation(operation_id), + reader_operation_id uuid,reader_issuance bigint, + CONSTRAINT mst2_metadata_root_anchor_reader_identity FOREIGN KEY(reader_operation_id,reader_issuance) REFERENCES mst2_metadata_reader_operation(operation_id,reader_issuance) MATCH FULL, + CONSTRAINT mst2_metadata_root_anchor_reader_kind CHECK((anchor_kind IN ('REQUEST','READER'))=(reader_operation_id IS NOT NULL)), + CONSTRAINT mst2_metadata_root_anchor_reader_fields CHECK((reader_operation_id IS NULL)=(reader_issuance IS NULL)), UNIQUE(anchor_kind,owner_key,root_page,root_generation), FOREIGN KEY(root_page,root_generation) REFERENCES mst2_metadata_graph_node(page_id,generation), FOREIGN KEY(root_page,root_generation,root_certificate_digest) REFERENCES mst2_metadata_page_certificate(page_id,generation,certificate_digest), FOREIGN KEY(snapshot_id,session_incarnation) REFERENCES mst2_qualified_session_incarnation(snapshot_id,session_incarnation) ); +CREATE INDEX mst2_metadata_root_anchor_reader ON mst2_metadata_root_anchor(reader_operation_id,reader_issuance) WHERE reader_operation_id IS NOT NULL; CREATE INDEX mst2_metadata_root_anchor_page ON mst2_metadata_root_anchor(root_page,root_generation,anchor_kind,owner_key); CREATE INDEX mst2_metadata_root_anchor_prepare ON mst2_metadata_root_anchor(prepare_id,anchor_kind,anchor_id); CREATE INDEX mst2_metadata_root_anchor_lease ON mst2_metadata_root_anchor(lease_id,anchor_kind,anchor_id); @@ -327,7 +339,7 @@ BEGIN RAISE EXCEPTION 'lease root still has its active lease or readers'; END IF; ELSE - IF NOT EXISTS(SELECT 1 FROM mst2_metadata_reader_operation r WHERE r.operation_id=OLD.reader_operation_id + IF NOT EXISTS(SELECT 1 FROM mst2_metadata_reader_operation r WHERE r.operation_id=OLD.reader_operation_id AND r.reader_issuance=OLD.reader_issuance AND (r.state='FINISHED' OR r.state='EXPIRED' AND r.hard_deadline_unix<=floor(extract(epoch FROM clock_timestamp()))::bigint)) THEN RAISE EXCEPTION 'reader root still has an active operation'; END IF; @@ -376,7 +388,7 @@ BEGIN ELSE IF NEW.owner_key IS DISTINCT FROM NEW.reader_operation_id::text OR NEW.prepare_id IS NOT NULL OR NOT EXISTS(SELECT 1 FROM mst2_metadata_reader_operation r JOIN mst2_qualified_lease_binding l USING(lease_id) - WHERE r.operation_id=NEW.reader_operation_id AND r.lease_id=NEW.lease_id AND r.snapshot_id=NEW.snapshot_id + WHERE r.operation_id=NEW.reader_operation_id AND r.reader_issuance=NEW.reader_issuance AND r.lease_id=NEW.lease_id AND r.snapshot_id=NEW.snapshot_id AND r.session_incarnation=NEW.session_incarnation AND r.root_page=NEW.root_page AND r.root_generation=NEW.root_generation AND r.state='ACTIVE' AND l.state='ACTIVE' AND l.lease_epoch=r.lease_epoch AND r.hard_deadline_unix<=l.expires_at_unix AND r.hard_deadline_unix>floor(extract(epoch FROM clock_timestamp()))::bigint) THEN diff --git a/src/jupiter/storage/qualified_metadata_family.rs b/src/jupiter/storage/qualified_metadata_family.rs index 024d325b..c44c8a94 100644 --- a/src/jupiter/storage/qualified_metadata_family.rs +++ b/src/jupiter/storage/qualified_metadata_family.rs @@ -32,11 +32,15 @@ const SOURCE_REVISION_SQL: &str = include_str!("qualified_source_revision.sql"); const ROOTED_SQL: &str = include_str!("qualified_metadata_rooted.sql"); const SERVING_SQL: &str = include_str!("qualified_metadata_serving.sql"); const GC_SQL: &str = include_str!("qualified_metadata_gc.sql"); +const READER_LIFECYCLE_SQL: &str = include_str!("qualified_metadata_reader_lifecycle.sql"); #[path = "qualified_metadata_rooted.rs"] mod rooted; pub(crate) use rooted::{RootedLookupStatus, RootedQualifiedMetadataRepository}; #[cfg(test)] +#[path = "qualified_metadata_reader_previous_fixture.rs"] +pub(crate) mod reader_previous_fixture; +#[cfg(test)] pub(crate) use rooted::{ RootedPrepareIntent, with_rooted_reader_barriers, with_rooted_source_fact_barriers, with_rooted_source_temporary_shadow, @@ -139,6 +143,10 @@ pub(crate) fn implementation_fingerprint() -> Vec { ("indexed-source-read-revision-1", SOURCE_READ_SQL.as_bytes()), ("captured-source-revision-1", SOURCE_REVISION_SQL.as_bytes()), ("rooted-collector-revision-1", GC_SQL.as_bytes()), + ( + "bounded-reader-lifecycle-revision-1", + READER_LIFECYCLE_SQL.as_bytes(), + ), ("typed-certificate-revision-1", CERTIFICATES_SQL.as_bytes()), ( "source-attestation-and-anchor-revision-1", @@ -189,6 +197,7 @@ pub(crate) fn render_family( .replace("$ROOTED_SQL$", ROOTED_SQL) .replace("$SERVING_SQL$", SERVING_SQL) .replace("$GC_SQL$", GC_SQL) + .replace("$READER_LIFECYCLE_SQL$", READER_LIFECYCLE_SQL) .replace("$CORE_SCHEMA$", &identifier(core_schema)) .replace("$CORE_LITERAL$", &literal(core_schema)) .replace("$Q_SCHEMA$", &identifier(q_schema)) diff --git a/src/jupiter/storage/qualified_metadata_family.sql b/src/jupiter/storage/qualified_metadata_family.sql index 853834e5..fae04446 100644 --- a/src/jupiter/storage/qualified_metadata_family.sql +++ b/src/jupiter/storage/qualified_metadata_family.sql @@ -473,7 +473,7 @@ DO $$ DECLARE t text; BEGIN 'mst2_metadata_prepare_page','mst2_metadata_graph_node','mst2_metadata_graph_edge','mst2_metadata_graph_root', 'mst2_metadata_gc_op','mst2_qualified_session_incarnation','mst2_qualified_lease_binding', 'mst2_metadata_page_certificate','mst2_metadata_verified_ref','mst2_metadata_source_root_attestation', - 'mst2_metadata_prepare_reuse_root','mst2_metadata_reuse_index','mst2_metadata_reader_operation','mst2_metadata_root_anchor', + 'mst2_metadata_prepare_reuse_root','mst2_metadata_reuse_index','mst2_metadata_reader_operation','mst2_metadata_reader_issuance','mst2_metadata_root_anchor', 'mst2_metadata_source_entry_reference','mst2_metadata_scope_source_reference'] LOOP EXECUTE format('CREATE TRIGGER mst2_00_family_barrier BEFORE INSERT OR UPDATE OR DELETE ON %I FOR EACH STATEMENT EXECUTE FUNCTION mst2_metadata_dml_barrier()',t); EXECUTE format('CREATE TRIGGER mst2_metadata_truncate_guard BEFORE TRUNCATE ON %I FOR EACH STATEMENT EXECUTE FUNCTION mst2_metadata_immutable()',t); diff --git a/src/jupiter/storage/qualified_metadata_gc.sql b/src/jupiter/storage/qualified_metadata_gc.sql index 13b41998..61276556 100644 --- a/src/jupiter/storage/qualified_metadata_gc.sql +++ b/src/jupiter/storage/qualified_metadata_gc.sql @@ -299,8 +299,8 @@ DO $$ DECLARE name text; BEGIN END LOOP; END $$; --- Append-only identity and replay evidence remains bounded independently of --- resident payloads. The admission check reads actual stored rows, never hints. +-- Permanent identity evidence and bounded reader owners have independent +-- resident quotas. Admission reads actual stored rows, never hints. CREATE FUNCTION mst2_metadata_history_capacity_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ DECLARE actual bigint; bytes bigint; row_limit integer; size_expression text; @@ -529,23 +529,23 @@ BEGIN examined:=0; readers_expired:=0; leases_expired:=0; prepares_aborted:=0; handovers_retired:=0; orphans_retired:=0; now_unix:=floor(extract(epoch FROM clock_timestamp()))::bigint; FOR item IN SELECT * FROM ( - (SELECT 'READER'::text AS kind,operation_id::text AS owner,lease_id,hard_deadline_unix AS deadline + (SELECT 'READER'::text AS kind,operation_id::text AS owner,lease_id,hard_deadline_unix AS deadline,reader_issuance FROM mst2_metadata_reader_operation WHERE state='ACTIVE' AND hard_deadline_unix<=now_unix ORDER BY hard_deadline_unix,operation_id LIMIT bound) UNION ALL - (SELECT 'LEASE'::text,lease_id,lease_id,expires_at_unix FROM mst2_qualified_lease_binding + (SELECT 'LEASE'::text,lease_id,lease_id,expires_at_unix,NULL::bigint FROM mst2_qualified_lease_binding WHERE state='ACTIVE' AND expires_at_unix<=now_unix ORDER BY expires_at_unix,lease_id LIMIT bound) UNION ALL - (SELECT 'PREPARE'::text,prepare_id,NULL::text,floor(extract(epoch FROM orphan_expires_at))::bigint + (SELECT 'PREPARE'::text,prepare_id,NULL::text,floor(extract(epoch FROM orphan_expires_at))::bigint,NULL::bigint FROM mst2_metadata_prepare WHERE (state='PREPARING' OR state='COMMITTED' AND coverage_retired_at IS NULL) AND orphan_expires_at<=clock_timestamp() ORDER BY orphan_expires_at,prepare_id LIMIT bound) ) expired ORDER BY deadline,kind,owner LIMIT bound LOOP examined:=examined+1; IF item.kind='READER' THEN - UPDATE mst2_metadata_reader_operation SET state='EXPIRED' WHERE operation_id=item.owner::uuid AND state='ACTIVE'; + UPDATE mst2_metadata_reader_operation SET state='EXPIRED' WHERE operation_id=item.owner::uuid AND reader_issuance=item.reader_issuance AND state='ACTIVE'; IF NOT FOUND THEN RAISE EXCEPTION 'qualified expired reader changed behind its mutation barrier'; END IF; readers_expired:=readers_expired+1; - DELETE FROM mst2_metadata_root_anchor WHERE reader_operation_id=item.owner::uuid AND anchor_kind IN ('REQUEST','READER'); + DELETE FROM mst2_metadata_root_anchor WHERE reader_operation_id=item.owner::uuid AND reader_issuance=item.reader_issuance AND anchor_kind IN ('REQUEST','READER'); PERFORM mst2_metadata_cleanup_lease(item.lease_id); ELSIF item.kind='LEASE' THEN UPDATE mst2_qualified_lease_binding SET state='EXPIRED',lease_epoch=lease_epoch+1 WHERE lease_id=item.owner AND state='ACTIVE'; diff --git a/src/jupiter/storage/qualified_metadata_reader.rs b/src/jupiter/storage/qualified_metadata_reader.rs index 6e28360a..85b22ff4 100644 --- a/src/jupiter/storage/qualified_metadata_reader.rs +++ b/src/jupiter/storage/qualified_metadata_reader.rs @@ -119,9 +119,15 @@ pub(crate) struct RootedLookupBatch { pub results: Vec, pub proof_pages: Vec<([u8; 32], Vec)>, } +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +struct ReaderIdentity { + operation_id: uuid::Uuid, + issuance: i64, +} + struct Reader<'a> { txn: &'a DatabaseTransaction, - operation: uuid::Uuid, + operation: ReaderIdentity, bindings: HashMap<[u8; 32], Binding>, cache: HashMap<[u8; 32], StoredPage>, work: PersistedMetadataReadWork, @@ -284,7 +290,7 @@ impl RootedQualifiedMetadataRepository { async fn admit_reader( &self, pinned: &SnapshotContext, - ) -> Result<(uuid::Uuid, Binding, uuid::Uuid), SnapshotError> { + ) -> Result<(ReaderIdentity, Binding, uuid::Uuid), SnapshotError> { let txn = self.transaction().await?; let admitted = async { let row = self @@ -312,7 +318,7 @@ impl RootedQualifiedMetadataRepository { } let reader = txn .query_one_raw(sql( - "SELECT operation_id::text,root_generation,certificate_digest + "SELECT operation_id::text,reader_issuance,root_generation,certificate_digest FROM mst2_metadata_begin_reader($1,$2,$3)", [ pinned.built.snapshot_id.clone().into(), @@ -324,12 +330,15 @@ impl RootedQualifiedMetadataRepository { .map_err(database_error)? .ok_or_else(|| unavailable("qualified reader admission returned no owned root"))?; Ok(( - uuid::Uuid::parse_str( - &reader - .try_get::("", "operation_id") - .map_err(internal)?, - ) - .map_err(internal)?, + ReaderIdentity { + operation_id: uuid::Uuid::parse_str( + &reader + .try_get::("", "operation_id") + .map_err(internal)?, + ) + .map_err(internal)?, + issuance: reader.try_get("", "reader_issuance").map_err(internal)?, + }, Binding { generation: reader.try_get("", "root_generation").map_err(internal)?, certificate: digest_column(&reader, "certificate_digest")?, @@ -349,7 +358,7 @@ impl RootedQualifiedMetadataRepository { async fn finish_reader( &self, - operation: uuid::Uuid, + operation: ReaderIdentity, result: Result, ) -> Result { let cleanup = self.transaction().await; @@ -357,8 +366,11 @@ impl RootedQualifiedMetadataRepository { Ok(txn) => { let finished = txn .execute_raw(sql( - "SELECT mst2_metadata_finish_reader($1::uuid)", - [operation.to_string().into()], + "SELECT mst2_metadata_finish_reader($1::uuid,$2::bigint)", + [ + operation.operation_id.to_string().into(), + operation.issuance.into(), + ], )) .await .map_err(internal) @@ -381,7 +393,7 @@ impl RootedQualifiedMetadataRepository { &self, pinned: &SnapshotContext, requests: &[MetadataRouteRequest<'_>], - operation: uuid::Uuid, + operation: ReaderIdentity, binding: Binding, source: uuid::Uuid, ) -> Result { @@ -424,7 +436,7 @@ impl RootedQualifiedMetadataRepository { impl Reader<'_> { fn new( txn: &DatabaseTransaction, - operation: uuid::Uuid, + operation: ReaderIdentity, root: [u8; 32], binding: Binding, source: uuid::Uuid, @@ -456,8 +468,8 @@ impl Reader<'_> { .iter() .map(|entry| hex::encode(&entry.name)) .collect(); - let rows = self.txn.query_all_raw(sql("SELECT * FROM mst2_metadata_read_source_entries($1::uuid,$2::uuid,$3,$4,$5,$6::jsonb)", - [self.operation.to_string().into(), source.to_string().into(), directory.to_vec().into(), + let rows = self.txn.query_all_raw(sql("SELECT * FROM mst2_metadata_read_source_entries($1::uuid,$2::bigint,$3::uuid,$4,$5,$6,$7::jsonb)", + [self.operation.operation_id.to_string().into(), self.operation.issuance.into(), source.to_string().into(), directory.to_vec().into(), binding.generation.into(), binding.certificate.to_vec().into(),json!(names).into()])).await.map_err(database_error)?; #[cfg(test)] if entries.iter().any(|entry| !entry.is_dir()) @@ -712,7 +724,8 @@ impl Reader<'_> { id.to_vec().into(), binding.generation.into(), binding.certificate.to_vec().into(), - self.operation.to_string().into(), + self.operation.operation_id.to_string().into(), + self.operation.issuance.into(), ], )) .await @@ -947,9 +960,9 @@ const PAGE_SQL:&str="SELECT body.payload,body.byte_size,proof.canonical_proof->' AND octet_length(body.payload)=proof.byte_size AND NOT EXISTS(SELECT 1 FROM mst2_metadata_gc_op gc WHERE gc.page_id=proof.page_id AND gc.generation=proof.generation) AND EXISTS(SELECT 1 FROM mst2_metadata_reader_operation reader JOIN mst2_metadata_root_anchor anchor - ON anchor.reader_operation_id=reader.operation_id AND anchor.anchor_kind='READER' + ON anchor.reader_operation_id=reader.operation_id AND anchor.reader_issuance=reader.reader_issuance AND anchor.anchor_kind='READER' AND anchor.root_page=reader.root_page AND anchor.root_generation=reader.root_generation - WHERE reader.operation_id=$4::uuid AND reader.state='ACTIVE' + WHERE reader.operation_id=$4::uuid AND reader.reader_issuance=$5::bigint AND reader.state='ACTIVE' AND reader.hard_deadline_unix>floor(extract(epoch FROM clock_timestamp()))::bigint) AND (SELECT count(*) FROM mst2_metadata_verified_ref ref WHERE ref.parent_page=proof.page_id AND ref.parent_generation=proof.generation)=jsonb_array_length(proof.canonical_proof->'references') diff --git a/src/jupiter/storage/qualified_metadata_reader_lifecycle.sql b/src/jupiter/storage/qualified_metadata_reader_lifecycle.sql new file mode 100644 index 00000000..058c9a84 --- /dev/null +++ b/src/jupiter/storage/qualified_metadata_reader_lifecycle.sql @@ -0,0 +1,110 @@ +CREATE FUNCTION mst2_metadata_next_reader_issuance(issued bigint) RETURNS bigint LANGUAGE plpgsql IMMUTABLE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + IF issued IS NULL OR issued<0 OR issued=9223372036854775807 THEN + RAISE EXCEPTION 'qualified reader issuance is exhausted or invalid'; END IF; + RETURN issued+1; +END $$; + +CREATE FUNCTION mst2_metadata_reader_issuance_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + IF TG_OP<>'UPDATE' THEN RAISE EXCEPTION 'qualified reader issuance cannot be reset or removed'; END IF; + IF NEW.singleton IS DISTINCT FROM OLD.singleton OR NEW.high_water IS DISTINCT FROM mst2_metadata_next_reader_issuance(OLD.high_water) THEN + RAISE EXCEPTION 'qualified reader issuance must advance exactly once'; END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_metadata_reader_issuance_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_metadata_reader_issuance + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_reader_issuance_guard(); + +CREATE FUNCTION mst2_metadata_reader_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE l mst2_qualified_lease_binding%ROWTYPE; issued bigint; + now_unix bigint:=floor(extract(epoch FROM clock_timestamp()))::bigint; +BEGIN + IF TG_OP='DELETE' THEN + IF OLD.state NOT IN ('FINISHED','EXPIRED') OR OLD.terminal_xid IS NULL OR OLD.terminal_xid=txid_current() + OR OLD.state='EXPIRED' AND OLD.hard_deadline_unix>now_unix + OR EXISTS(SELECT 1 FROM mst2_metadata_root_anchor WHERE reader_operation_id=OLD.operation_id) THEN + RAISE EXCEPTION 'qualified reader still has its active owner, deferred completion, or owned roots'; END IF; + RETURN OLD; + END IF; + IF TG_OP='UPDATE' THEN + IF (to_jsonb(NEW)-'state') IS DISTINCT FROM (to_jsonb(OLD)-'state') + OR OLD.state<>'ACTIVE' AND NEW.state<>OLD.state OR NEW.state NOT IN ('ACTIVE','FINISHED','EXPIRED') + OR NEW.state='EXPIRED' AND OLD.hard_deadline_unix>now_unix THEN + RAISE EXCEPTION 'qualified reader identity cannot change or be prematurely expired'; END IF; + IF OLD.state<>'ACTIVE' THEN RETURN NULL; END IF; + IF NEW.state<>'ACTIVE' THEN NEW.terminal_xid:=txid_current(); END IF; + RETURN NEW; + END IF; + SELECT * INTO l FROM mst2_qualified_lease_binding WHERE lease_id=NEW.lease_id AND state='ACTIVE' AND expires_at_unix>now_unix; + IF NOT FOUND OR NEW.state<>'ACTIVE' OR NEW.terminal_xid IS NOT NULL OR NEW.reader_issuance<=0 + OR substr(NEW.operation_id::text,15,1)<>'4' OR substr(NEW.operation_id::text,20,1) NOT IN ('8','9','a','b') + OR ROW(NEW.snapshot_id,NEW.session_incarnation,NEW.root_page,NEW.root_generation,NEW.lease_epoch) IS DISTINCT FROM + ROW(l.snapshot_id,l.session_incarnation,l.metadata_root,l.root_generation,l.lease_epoch) + OR NEW.hard_deadline_unix<=now_unix OR NEW.hard_deadline_unix>least(l.expires_at_unix,now_unix+60) + OR NOT mst2_metadata_incarnation_proof(l.snapshot_id,l.session_incarnation,true) THEN + RAISE EXCEPTION 'qualified reader lacks its exact active lease and bounded deadline'; END IF; + SELECT high_water INTO STRICT issued FROM mst2_metadata_reader_issuance WHERE singleton=1; + IF NEW.reader_issuance<>mst2_metadata_next_reader_issuance(issued) THEN RAISE EXCEPTION 'qualified reader cannot replay an issued identity'; END IF; + UPDATE mst2_metadata_reader_issuance SET high_water=NEW.reader_issuance WHERE singleton=1; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_metadata_reader_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_metadata_reader_operation + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_reader_guard(); + +CREATE FUNCTION mst2_metadata_prune_readers(maximum integer) RETURNS bigint LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE removed bigint; now_unix bigint:=floor(extract(epoch FROM clock_timestamp()))::bigint; +BEGIN + PERFORM $CORE_SCHEMA$.mst2_route_enter($CORE_LITERAL$); + IF maximum IS NULL OR maximum NOT BETWEEN 0 AND 64 THEN RAISE EXCEPTION 'qualified reader prune budget must be 0..=64'; END IF; + WITH candidates AS ( + SELECT r.operation_id,r.reader_issuance FROM mst2_metadata_reader_operation r + WHERE r.state IN ('FINISHED','EXPIRED') AND r.terminal_xid (usize, usize) { + let start = source.find(&format!("CREATE FUNCTION {name}(")).unwrap(); + let f = &source[start..]; + let body = f.find("AS $").unwrap() + 3; + let tag_end = f[body + 1..].find('$').unwrap() + body + 2; + let tag = &f[body..tag_end]; + let end = f[tag_end..].find(tag).unwrap() + tag_end + tag.len() + 1; + assert_eq!(f.as_bytes()[end - 1], b';'); + (start, start + end) +} + +fn render_previous( + core: &str, + core_oid: i64, + q: &str, + q_oid: i64, + namespace: &str, + storage: &str, +) -> String { + let capture = CAPTURE.replace("\r\n", "\n"); + assert_eq!( + hex::encode(Sha256::digest(capture.as_bytes())), + "599a5a6749cb41afd8f7b05287a482ab6ba921d4ffe6142159b0bbf0f00e33f4" + ); + let mut sql = render_family(core, core_oid, q, q_oid, namespace, storage).replace("\r\n", "\n"); + let a = sql + .find("CREATE TABLE mst2_metadata_reader_operation (") + .unwrap(); + let z = sql[a..] + .find("CREATE FUNCTION mst2_metadata_session_covers_prepare(") + .unwrap() + + a; + let old_a = capture.find("-- READER DDL BEGIN\n").unwrap() + "-- READER DDL BEGIN\n".len(); + let old_z = capture.find("-- READER DDL END").unwrap(); + sql.replace_range(a..z, &capture[old_a..old_z]); + for name in [ + "mst2_metadata_dml_barrier", + "mst2_metadata_gc_enabled", + "mst2_metadata_root_anchor_guard", + "mst2_metadata_serving_complete", + "mst2_metadata_reader_guard", + "mst2_metadata_begin_reader", + "mst2_metadata_finish_reader", + "mst2_metadata_read_source_entries", + "mst2_metadata_cleanup_expired", + "mst2_metadata_gc_owner_cleanup", + ] { + let (a, z) = function(&sql, name); + let (old_a, old_z) = function(&capture, name); + sql.replace_range(a..z, &capture[old_a..old_z]); + } + for name in [ + "mst2_metadata_next_reader_issuance", + "mst2_metadata_reader_issuance_guard", + "mst2_metadata_prune_readers", + ] { + let (a, z) = function(&sql, name); + sql.replace_range(a..z, ""); + } + sql=sql.replace("CREATE TRIGGER mst2_metadata_reader_issuance_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_metadata_reader_issuance\n FOR EACH ROW EXECUTE FUNCTION mst2_metadata_reader_issuance_guard();","") + .replace("'mst2_metadata_reader_issuance',","") + .replace(&hex::encode(implementation_fingerprint()),PREVIOUS) + .replace("$CORE_SCHEMA$",&identifier(core)).replace("$CORE_LITERAL$",&literal(core)) + .replace("$Q_SCHEMA$",&identifier(q)).replace("$Q_LITERAL$",&literal(q)) + .replace("$CORE_OID$",&core_oid.to_string()).replace("$Q_OID$",&q_oid.to_string()) + .replace("$NAMESPACE_UUID$",namespace).replace("$STORAGE_UUID$",storage).replace("$IMPLEMENTATION_SHA$",PREVIOUS); + assert!(!sql.contains("reader_issuance")); + assert!(!sql.contains("terminal_xid")); + sql +} + +struct Table { + name: String, + rows: Value, + dependencies: BTreeSet, +} + +async fn fingerprint(txn: &DatabaseTransaction, core_oid: i64, q_oid: i64, exempt: i64) -> Vec { + catalog_with_exemption(txn, core_oid, q_oid, exempt) + .await + .unwrap() +} + +/// Recreate Q relations from captured old DDL within an owned test schema. +/// Preserve its namespace OID, scope and real source/certificate/session rows. +pub(crate) async fn restore_previous(core: &DatabaseConnection, q_schema: &str, keep_q: bool) { + let txn = core.begin().await.unwrap(); + let (core_schema, core_oid) = captured_core(&txn).await.unwrap(); + assert!(core_schema.starts_with("mega2_test_")); + assert!(q_schema.starts_with("mst2q_")); + let c = identifier(&core_schema); + let q = identifier(q_schema); + txn.execute_unprepared(&format!( + "SELECT {c}.mst2_route_enter({})", + literal(&core_schema) + )) + .await + .unwrap(); + let registry=txn.query_one_raw(Statement::from_sql_and_values(DbBackend::Postgres, + format!("SELECT metadata_schema_oid::bigint,namespace_uuid::text,metadata_storage_uuid FROM {c}.mst2_metadata_namespace WHERE metadata_schema=$1 AND graph_domain='qualified-v1'"), + [q_schema.into()])).await.unwrap().unwrap(); + let q_oid: i64 = registry.try_get("", "metadata_schema_oid").unwrap(); + let namespace: String = registry.try_get("", "namespace_uuid").unwrap(); + let storage: String = registry.try_get("", "metadata_storage_uuid").unwrap(); + let table_rows=txn.query_all_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "SELECT relname FROM pg_catalog.pg_class WHERE relnamespace=$1::bigint::oid AND relkind='r' ORDER BY relname",[q_oid.into()])).await.unwrap(); + let mut tables = Vec::new(); + let mut all_names = Vec::new(); + for row in table_rows { + let name: String = row.try_get("", "relname").unwrap(); + all_names.push(name.clone()); + if [ + "mst2_metadata_storage_scope", + "mst2_metadata_family_identity", + "mst2_metadata_reader_issuance", + ] + .contains(&name.as_str()) + { + continue; + } + let rows:Value=txn.query_one_raw(Statement::from_string(DbBackend::Postgres,format!( + "SELECT coalesce(jsonb_agg(to_jsonb(data)),'[]'::jsonb) AS rows FROM {q}.{} data",identifier(&name)))) + .await.unwrap().unwrap().try_get("","rows").unwrap(); + if !keep_q { + assert_eq!( + rows, + json!([]), + "Q-less fixture must have no history to discard" + ); + } + let dependencies=txn.query_all_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "SELECT target.relname FROM pg_catalog.pg_constraint fk JOIN pg_catalog.pg_class source ON source.oid=fk.conrelid + JOIN pg_catalog.pg_class target ON target.oid=fk.confrelid WHERE source.relnamespace=$1::bigint::oid + AND target.relnamespace=source.relnamespace AND source.relname=$2 AND target.relname<>source.relname AND fk.contype='f'", + [q_oid.into(),name.clone().into()])).await.unwrap().into_iter() + .map(|row|row.try_get("","relname").unwrap()).collect(); + tables.push(Table { + name, + rows, + dependencies, + }); + } + txn.execute_unprepared(&format!( + "DROP TABLE {} CASCADE", + all_names + .iter() + .map(|name| format!("{q}.{}", identifier(name))) + .collect::>() + .join(",") + )) + .await + .unwrap(); + let funcs=txn.query_all_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "SELECT proname,pg_catalog.pg_get_function_identity_arguments(oid) AS args FROM pg_catalog.pg_proc WHERE pronamespace=$1::bigint::oid",[q_oid.into()])).await.unwrap(); + for row in funcs { + let name: String = row.try_get("", "proname").unwrap(); + let args: String = row.try_get("", "args").unwrap(); + txn.execute_unprepared(&format!( + "DROP FUNCTION IF EXISTS {q}.{}({args}) CASCADE", + identifier(&name) + )) + .await + .unwrap(); + } + txn.execute_unprepared(&render_previous( + &core_schema, + core_oid, + q_schema, + q_oid, + &namespace, + &storage, + )) + .await + .unwrap(); + let old_tables=txn.query_all_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "SELECT relname FROM pg_catalog.pg_class WHERE relnamespace=$1::bigint::oid AND relkind='r'",[q_oid.into()])).await.unwrap(); + let old_names: Vec = old_tables + .into_iter() + .map(|row| row.try_get("", "relname").unwrap()) + .collect(); + for name in &old_names { + txn.execute_unprepared(&format!( + "ALTER TABLE {q}.{} DISABLE TRIGGER USER", + identifier(name) + )) + .await + .unwrap(); + } + let mut restored = BTreeSet::from([ + "mst2_metadata_storage_scope".to_owned(), + "mst2_metadata_family_identity".to_owned(), + ]); + while !tables.is_empty() { + let index = tables + .iter() + .position(|table| table.dependencies.is_subset(&restored)) + .expect("captured Q foreign keys are acyclic"); + let table = tables.remove(index); + txn.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres,format!( + "INSERT INTO {q}.{t} SELECT * FROM jsonb_populate_recordset(NULL::{q}.{t},$1::jsonb)",t=identifier(&table.name)),[table.rows.into()])).await.unwrap(); + restored.insert(table.name); + } + for name in &old_names { + txn.execute_unprepared(&format!( + "ALTER TABLE {q}.{} ENABLE TRIGGER USER", + identifier(name) + )) + .await + .unwrap(); + } + let capture = CAPTURE.replace("\r\n", "\n"); + let (a, z) = function(&capture, "mst2_route_family_registration_guard"); + let registration = capture[a..z] + .replacen("CREATE FUNCTION", "CREATE OR REPLACE FUNCTION", 1) + .replace("$CORE_SCHEMA$", &c) + .replace("$IMPLEMENTATION_SHA$", PREVIOUS); + txn.execute_unprepared(®istration).await.unwrap(); + let old_shape: Vec = txn + .query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + format!( + "SELECT {c}.mst2_route_family_shape($1::bigint::oid,$2::uuid,$3) AS fingerprint" + ), + [q_oid.into(), namespace.clone().into(), storage.into()], + )) + .await + .unwrap() + .unwrap() + .try_get("", "fingerprint") + .unwrap(); + txn.execute_unprepared(&format!("SET LOCAL search_path={c},pg_catalog,pg_temp")) + .await + .unwrap(); + if !keep_q { + txn.execute_unprepared(&format!( + "ALTER TABLE {c}.mst2_metadata_namespace DISABLE TRIGGER mst2_route_immutable" + )) + .await + .unwrap(); + txn.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + format!("DELETE FROM {c}.mst2_metadata_namespace WHERE namespace_uuid=$1::uuid"), + [namespace.clone().into()], + )) + .await + .unwrap(); + txn.execute_unprepared(&format!("ALTER TABLE {c}.mst2_metadata_namespace ENABLE TRIGGER mst2_route_immutable; DROP SCHEMA {q} CASCADE")).await.unwrap(); + } + let authority = fingerprint(&txn, core_oid, 0, if keep_q { q_oid } else { 0 }).await; + let full = if keep_q { + Some(fingerprint(&txn, core_oid, q_oid, 0).await) + } else { + None + }; + txn.execute_unprepared(&format!("ALTER TABLE {c}.mst2_qualified_family_policy DISABLE TRIGGER mst2_route_family_policy_immutable")).await.unwrap(); + txn.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres,format!( + "UPDATE {c}.mst2_qualified_family_policy SET implementation_fingerprint=$1,expected_shape=$2,authority_catalog=$3 WHERE singleton=1"), + [hex::decode(PREVIOUS).unwrap().into(),old_shape.into(),authority.clone().into()])).await.unwrap(); + txn.execute_unprepared(&format!("ALTER TABLE {c}.mst2_qualified_family_policy ENABLE TRIGGER mst2_route_family_policy_immutable")).await.unwrap(); + if let Some(full) = &full { + txn.execute_unprepared(&format!( + "ALTER TABLE {c}.mst2_metadata_namespace DISABLE TRIGGER mst2_route_immutable" + )) + .await + .unwrap(); + txn.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres,format!( + "UPDATE {c}.mst2_metadata_namespace SET implementation_fingerprint=$1,catalog_fingerprint=$2 WHERE namespace_uuid=$3::uuid"), + [hex::decode(PREVIOUS).unwrap().into(),full.clone().into(),namespace.into()])).await.unwrap(); + txn.execute_unprepared(&format!( + "ALTER TABLE {c}.mst2_metadata_namespace ENABLE TRIGGER mst2_route_immutable" + )) + .await + .unwrap(); + } + assert_eq!( + fingerprint(&txn, core_oid, 0, if keep_q { q_oid } else { 0 }).await, + authority + ); + if let Some(full) = full { + assert_eq!(fingerprint(&txn, core_oid, q_oid, 0).await, full); + } + txn.commit().await.unwrap(); +} diff --git a/src/jupiter/storage/qualified_metadata_reader_previous_fixture.sql b/src/jupiter/storage/qualified_metadata_reader_previous_fixture.sql new file mode 100644 index 00000000..3b087462 --- /dev/null +++ b/src/jupiter/storage/qualified_metadata_reader_previous_fixture.sql @@ -0,0 +1,425 @@ +-- Captured exactly from c1ff280281440891e732ecfaed4732ee59c41a90; test-only. +-- READER DDL BEGIN +CREATE TABLE mst2_metadata_reader_operation ( + operation_id uuid PRIMARY KEY,lease_id text NOT NULL REFERENCES mst2_qualified_lease_binding(lease_id), + snapshot_id text NOT NULL,session_incarnation uuid NOT NULL,root_page bytea NOT NULL,root_generation bigint NOT NULL, + lease_epoch bigint NOT NULL CHECK(lease_epoch>0),hard_deadline_unix bigint NOT NULL, + state text NOT NULL CHECK(state IN ('ACTIVE','FINISHED','EXPIRED')), + FOREIGN KEY(snapshot_id,session_incarnation) REFERENCES mst2_qualified_session_incarnation(snapshot_id,session_incarnation) +); +CREATE INDEX mst2_metadata_reader_active_lease ON mst2_metadata_reader_operation(lease_id,state,operation_id); +CREATE INDEX mst2_metadata_reader_active_deadline ON mst2_metadata_reader_operation(hard_deadline_unix,operation_id) WHERE state='ACTIVE'; +CREATE TABLE mst2_metadata_root_anchor ( + anchor_id uuid PRIMARY KEY,anchor_kind text NOT NULL CHECK(anchor_kind IN ('PREPARE','REUSE','SESSION','LEASE','REQUEST','READER')), + owner_key text NOT NULL CHECK(octet_length(owner_key) BETWEEN 1 AND 512), + root_page bytea NOT NULL,root_generation bigint NOT NULL,root_certificate_digest bytea NOT NULL, + prepare_id text REFERENCES mst2_metadata_prepare(prepare_id), + snapshot_id text,session_incarnation uuid,lease_id text REFERENCES mst2_qualified_lease_binding(lease_id), + reader_operation_id uuid REFERENCES mst2_metadata_reader_operation(operation_id), + UNIQUE(anchor_kind,owner_key,root_page,root_generation), + FOREIGN KEY(root_page,root_generation) REFERENCES mst2_metadata_graph_node(page_id,generation), + FOREIGN KEY(root_page,root_generation,root_certificate_digest) + REFERENCES mst2_metadata_page_certificate(page_id,generation,certificate_digest), + FOREIGN KEY(snapshot_id,session_incarnation) REFERENCES mst2_qualified_session_incarnation(snapshot_id,session_incarnation) +); +CREATE INDEX mst2_metadata_root_anchor_page ON mst2_metadata_root_anchor(root_page,root_generation,anchor_kind,owner_key); +CREATE INDEX mst2_metadata_root_anchor_prepare ON mst2_metadata_root_anchor(prepare_id,anchor_kind,anchor_id); +CREATE INDEX mst2_metadata_root_anchor_lease ON mst2_metadata_root_anchor(lease_id,anchor_kind,anchor_id); + +-- READER DDL END + +CREATE FUNCTION mst2_metadata_dml_barrier() RETURNS trigger LANGUAGE plpgsql VOLATILE AS $$ +BEGIN + IF pg_catalog.current_schema() IS DISTINCT FROM $Q_LITERAL$ OR TG_TABLE_SCHEMA IS DISTINCT FROM $Q_LITERAL$ + OR NOT EXISTS(SELECT 1 FROM pg_catalog.pg_class WHERE oid=TG_RELID AND relnamespace='$Q_OID$'::oid) + OR NOT EXISTS(SELECT 1 FROM pg_catalog.pg_namespace WHERE oid='$Q_OID$'::oid AND nspname=$Q_LITERAL$) + OR pg_catalog.pg_is_in_recovery() OR pg_catalog.current_setting('transaction_isolation')<>'read committed' THEN + RAISE EXCEPTION 'qualified mutation requires its captured primary family and READ COMMITTED'; + END IF; + PERFORM $CORE_SCHEMA$.mst2_route_enter($CORE_LITERAL$); + IF NOT EXISTS(SELECT 1 FROM $CORE_SCHEMA$.mst2_metadata_namespace n + WHERE n.namespace_uuid='$NAMESPACE_UUID$'::uuid AND n.metadata_schema=$Q_LITERAL$ + AND n.metadata_schema_oid='$Q_OID$'::oid AND n.core_schema_oid=$CORE_OID$ + AND n.metadata_storage_uuid='$STORAGE_UUID$' AND n.admission_state='ROOTED_Q_ADMITTED' + AND n.collector_state='ENABLED' AND n.implementation_fingerprint=pg_catalog.decode('$IMPLEMENTATION_SHA$','hex') + AND n.catalog_fingerprint=$CORE_SCHEMA$.mst2_route_family_catalog($CORE_OID$,'$Q_OID$'::oid)) THEN + RAISE EXCEPTION 'qualified physical family catalog fingerprint is unavailable'; + END IF; + RETURN NULL; +END $$; + +CREATE FUNCTION mst2_metadata_gc_enabled() RETURNS boolean LANGUAGE sql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ + SELECT NOT pg_is_in_recovery() AND current_setting('transaction_isolation')='read committed' + AND EXISTS(SELECT 1 FROM $CORE_SCHEMA$.mst2_metadata_namespace n WHERE n.namespace_uuid='$NAMESPACE_UUID$'::uuid + AND n.graph_domain='qualified-v1' AND n.metadata_schema=$Q_LITERAL$ AND n.metadata_schema_oid='$Q_OID$'::oid + AND n.core_schema_oid=$CORE_OID$ AND n.metadata_storage_uuid='$STORAGE_UUID$' + AND n.admission_state='ROOTED_Q_ADMITTED' AND n.collector_state='ENABLED' + AND n.implementation_fingerprint=decode('$IMPLEMENTATION_SHA$','hex') + AND n.catalog_fingerprint=$CORE_SCHEMA$.mst2_route_family_catalog($CORE_OID$,'$Q_OID$'::oid)) +$$; + +CREATE FUNCTION mst2_metadata_root_anchor_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + IF TG_OP='UPDATE' THEN RAISE EXCEPTION 'rooted anchors cannot retarget immutable owner or root identities'; END IF; + IF TG_OP='DELETE' THEN + IF OLD.anchor_kind IN ('PREPARE','REUSE') THEN + IF NOT EXISTS(SELECT 1 FROM mst2_metadata_prepare WHERE prepare_id=OLD.prepare_id AND state IN ('PREPARING','COMMITTED','ABORTED')) THEN + RAISE EXCEPTION 'temporary anchor owner history is missing'; + END IF; + ELSIF OLD.anchor_kind='SESSION' THEN + IF NOT EXISTS(SELECT 1 FROM mst2_qualified_session_incarnation s + WHERE s.snapshot_id=OLD.snapshot_id AND s.session_incarnation=OLD.session_incarnation AND s.state='RETIRED') + OR EXISTS(SELECT 1 FROM mst2_qualified_lease_binding l WHERE l.snapshot_id=OLD.snapshot_id + AND l.session_incarnation=OLD.session_incarnation AND l.state='ACTIVE') THEN + RAISE EXCEPTION 'session root still has its active incarnation or leases'; + END IF; + ELSIF OLD.anchor_kind='LEASE' THEN + IF NOT EXISTS(SELECT 1 FROM mst2_qualified_lease_binding WHERE lease_id=OLD.lease_id AND state IN ('RELEASED','EXPIRED')) + OR EXISTS(SELECT 1 FROM mst2_metadata_reader_operation WHERE lease_id=OLD.lease_id AND state='ACTIVE') THEN + RAISE EXCEPTION 'lease root still has its active lease or readers'; + END IF; + ELSE + IF NOT EXISTS(SELECT 1 FROM mst2_metadata_reader_operation r WHERE r.operation_id=OLD.reader_operation_id + AND (r.state='FINISHED' OR r.state='EXPIRED' AND r.hard_deadline_unix<=floor(extract(epoch FROM clock_timestamp()))::bigint)) THEN + RAISE EXCEPTION 'reader root still has an active operation'; + END IF; + END IF; + RETURN OLD; + END IF; + IF NOT EXISTS(SELECT 1 FROM mst2_metadata_graph_node node + JOIN mst2_metadata_current cur USING(page_id,generation) + JOIN mst2_metadata_lifetime life USING(page_id,generation) + JOIN mst2_metadata_page_certificate proof USING(page_id,generation) + WHERE node.page_id=NEW.root_page AND node.generation=NEW.root_generation AND node.state='LIVE' + AND node.certificate_digest=NEW.root_certificate_digest AND proof.certificate_digest=NEW.root_certificate_digest + AND life.state IN ('RESERVED','LIVE') AND life.graph_domain='qualified-v1' + AND NOT EXISTS(SELECT 1 FROM mst2_metadata_gc_op gc WHERE gc.page_id=node.page_id AND gc.generation=node.generation)) THEN + RAISE EXCEPTION 'rooted anchor does not protect its exact canonical current graph'; + END IF; + IF NEW.anchor_kind IN ('PREPARE','REUSE') THEN + IF NEW.owner_key IS DISTINCT FROM NEW.prepare_id OR NEW.snapshot_id IS NOT NULL OR NEW.session_incarnation IS NOT NULL + OR NEW.lease_id IS NOT NULL OR NEW.reader_operation_id IS NOT NULL + OR NOT EXISTS(SELECT 1 FROM mst2_metadata_prepare WHERE prepare_id=NEW.prepare_id + AND state IN ('PREPARING','COMMITTED') AND coverage_retired_at IS NULL) + OR NEW.anchor_kind='PREPARE' AND NOT (EXISTS(SELECT 1 FROM mst2_metadata_prepare_page + WHERE prepare_id=NEW.prepare_id AND page_id=NEW.root_page AND generation=NEW.root_generation) + OR EXISTS(SELECT 1 FROM mst2_metadata_prepare q JOIN mst2_metadata_prepare_reuse_root r USING(prepare_id) + WHERE q.prepare_id=NEW.prepare_id AND q.plan_kind='ROOTED' AND q.metadata_root=NEW.root_page + AND r.root_page=NEW.root_page AND r.root_generation=NEW.root_generation)) + OR NEW.anchor_kind='REUSE' AND NOT EXISTS(SELECT 1 FROM mst2_metadata_prepare_reuse_root + WHERE prepare_id=NEW.prepare_id AND root_page=NEW.root_page AND root_generation=NEW.root_generation) THEN + RAISE EXCEPTION 'temporary anchor differs from its immutable delta or reused-root owner'; + END IF; + ELSIF NEW.anchor_kind='SESSION' THEN + IF NEW.owner_key IS DISTINCT FROM NEW.snapshot_id||':'||NEW.session_incarnation::text OR NEW.prepare_id IS NOT NULL + OR NEW.lease_id IS NOT NULL OR NEW.reader_operation_id IS NOT NULL + OR NOT EXISTS(SELECT 1 FROM mst2_qualified_session_incarnation s WHERE s.snapshot_id=NEW.snapshot_id + AND s.session_incarnation=NEW.session_incarnation AND s.metadata_root=NEW.root_page AND s.root_generation=NEW.root_generation AND s.state='READY') THEN + RAISE EXCEPTION 'session anchor differs from its exact ready incarnation'; + END IF; + ELSIF NEW.anchor_kind='LEASE' THEN + IF NEW.owner_key IS DISTINCT FROM NEW.lease_id OR NEW.prepare_id IS NOT NULL OR NEW.reader_operation_id IS NOT NULL + OR NOT EXISTS(SELECT 1 FROM mst2_qualified_lease_binding l WHERE l.lease_id=NEW.lease_id + AND l.snapshot_id=NEW.snapshot_id AND l.session_incarnation=NEW.session_incarnation + AND l.metadata_root=NEW.root_page AND l.root_generation=NEW.root_generation AND l.state='ACTIVE' + AND l.expires_at_unix>floor(extract(epoch FROM clock_timestamp()))::bigint) THEN + RAISE EXCEPTION 'lease anchor differs from its exact active lease'; + END IF; + ELSE + IF NEW.owner_key IS DISTINCT FROM NEW.reader_operation_id::text OR NEW.prepare_id IS NOT NULL + OR NOT EXISTS(SELECT 1 FROM mst2_metadata_reader_operation r JOIN mst2_qualified_lease_binding l USING(lease_id) + WHERE r.operation_id=NEW.reader_operation_id AND r.lease_id=NEW.lease_id AND r.snapshot_id=NEW.snapshot_id + AND r.session_incarnation=NEW.session_incarnation AND r.root_page=NEW.root_page AND r.root_generation=NEW.root_generation + AND r.state='ACTIVE' AND l.state='ACTIVE' AND l.lease_epoch=r.lease_epoch + AND r.hard_deadline_unix<=l.expires_at_unix AND r.hard_deadline_unix>floor(extract(epoch FROM clock_timestamp()))::bigint) THEN + RAISE EXCEPTION 'reader anchor differs from its exact active lease operation'; + END IF; + END IF; + RETURN NEW; +END $$; + +CREATE FUNCTION mst2_metadata_serving_complete() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE s mst2_qualified_session_incarnation%ROWTYPE; l mst2_qualified_lease_binding%ROWTYPE; + r mst2_metadata_reader_operation%ROWTYPE; +BEGIN + IF TG_TABLE_NAME='mst2_qualified_session_incarnation' THEN + SELECT * INTO STRICT s FROM mst2_qualified_session_incarnation + WHERE snapshot_id=NEW.snapshot_id AND session_incarnation=NEW.session_incarnation; + IF NOT mst2_metadata_incarnation_proof(s.snapshot_id,s.session_incarnation,s.state='READY') + OR s.state='READY' AND NOT EXISTS(SELECT 1 FROM mst2_qualified_lease_binding lease + WHERE lease.snapshot_id=s.snapshot_id AND lease.session_incarnation=s.session_incarnation AND lease.state='ACTIVE') + OR s.state='RETIRED' AND EXISTS(SELECT 1 FROM mst2_metadata_root_anchor a + WHERE a.anchor_kind='SESSION' AND a.snapshot_id=s.snapshot_id AND a.session_incarnation=s.session_incarnation) THEN + RAISE EXCEPTION 'qualified session cannot commit without its exact final serving roots'; END IF; + ELSIF TG_TABLE_NAME='mst2_qualified_lease_binding' THEN + SELECT * INTO STRICT l FROM mst2_qualified_lease_binding WHERE lease_id=NEW.lease_id; + IF NOT EXISTS(SELECT 1 FROM $CORE_SCHEMA$.mst2_lease_storage_route route + WHERE route.lease_id=l.lease_id AND route.namespace_uuid=l.namespace_uuid AND route.snapshot_id=l.snapshot_id + AND route.session_incarnation=l.session_incarnation AND route.prepare_id=l.prepare_id AND route.metadata_root=l.metadata_root + AND route.authorization_epoch=l.authorization_epoch AND route.publication_sequence=l.publication_sequence + AND route.writer_epoch=l.writer_epoch AND route.certificate_receipt_id=l.certificate_receipt_id) + OR l.state='ACTIVE' AND (l.expires_at_unix<=floor(extract(epoch FROM clock_timestamp()))::bigint + OR NOT mst2_metadata_incarnation_proof(l.snapshot_id,l.session_incarnation,true) + OR NOT EXISTS(SELECT 1 FROM mst2_metadata_root_anchor a WHERE a.anchor_kind='LEASE' AND a.owner_key=l.lease_id + AND a.lease_id=l.lease_id AND a.root_page=l.metadata_root AND a.root_generation=l.root_generation)) + OR l.state<>'ACTIVE' AND NOT EXISTS(SELECT 1 FROM mst2_metadata_reader_operation operation + WHERE operation.lease_id=l.lease_id AND operation.state='ACTIVE') + AND EXISTS(SELECT 1 FROM mst2_metadata_root_anchor a WHERE a.anchor_kind='LEASE' AND a.lease_id=l.lease_id) THEN + RAISE EXCEPTION 'qualified lease cannot commit without its exact final route and owned protection'; END IF; + ELSE + SELECT * INTO STRICT r FROM mst2_metadata_reader_operation WHERE operation_id=NEW.operation_id; + IF r.state='ACTIVE' AND (r.hard_deadline_unix<=floor(extract(epoch FROM clock_timestamp()))::bigint + OR (SELECT count(*) FROM mst2_metadata_root_anchor a WHERE a.reader_operation_id=r.operation_id + AND a.anchor_kind IN ('REQUEST','READER') AND a.owner_key=r.operation_id::text + AND a.lease_id=r.lease_id AND a.snapshot_id=r.snapshot_id AND a.session_incarnation=r.session_incarnation + AND a.root_page=r.root_page AND a.root_generation=r.root_generation)<>2) + OR r.state<>'ACTIVE' AND EXISTS(SELECT 1 FROM mst2_metadata_root_anchor a WHERE a.reader_operation_id=r.operation_id) THEN + RAISE EXCEPTION 'qualified reader cannot commit without both exact owned roots or definitive cleanup'; END IF; + END IF; + RETURN NULL; +END $$; + +CREATE FUNCTION mst2_metadata_reader_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE l mst2_qualified_lease_binding%ROWTYPE; now_unix bigint:=floor(extract(epoch FROM clock_timestamp()))::bigint; +BEGIN + IF TG_OP='DELETE' THEN RAISE EXCEPTION 'qualified reader operation history is immutable'; END IF; + IF TG_OP='UPDATE' THEN + IF (to_jsonb(NEW)-'state') IS DISTINCT FROM (to_jsonb(OLD)-'state') + OR OLD.state<>'ACTIVE' AND NEW.state<>OLD.state OR NEW.state NOT IN ('ACTIVE','FINISHED','EXPIRED') + OR NEW.state='EXPIRED' AND OLD.hard_deadline_unix>now_unix THEN + RAISE EXCEPTION 'qualified reader identity cannot change or be prematurely expired'; END IF; + RETURN NEW; + END IF; + SELECT * INTO l FROM mst2_qualified_lease_binding WHERE lease_id=NEW.lease_id AND state='ACTIVE' AND expires_at_unix>now_unix; + IF NOT FOUND OR NEW.state<>'ACTIVE' OR substr(NEW.operation_id::text,15,1)<>'4' + OR substr(NEW.operation_id::text,20,1) NOT IN ('8','9','a','b') + OR ROW(NEW.snapshot_id,NEW.session_incarnation,NEW.root_page,NEW.root_generation,NEW.lease_epoch) IS DISTINCT FROM + ROW(l.snapshot_id,l.session_incarnation,l.metadata_root,l.root_generation,l.lease_epoch) + OR NEW.hard_deadline_unix<=now_unix OR NEW.hard_deadline_unix>least(l.expires_at_unix,now_unix+60) + OR NOT mst2_metadata_incarnation_proof(l.snapshot_id,l.session_incarnation,true) THEN + RAISE EXCEPTION 'qualified reader lacks its exact active lease and bounded deadline'; END IF; + RETURN NEW; +END $$; + +CREATE FUNCTION mst2_metadata_begin_reader(sid text,lid text,instance text) +RETURNS TABLE(operation_id uuid,root_generation bigint,certificate_digest bytea) LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE l mst2_qualified_lease_binding%ROWTYPE; s record; op uuid:=gen_random_uuid(); deadline bigint; kind text; +BEGIN + PERFORM $CORE_SCHEMA$.mst2_route_enter($CORE_LITERAL$); + PERFORM mst2_metadata_cleanup_expired(64); + SELECT * INTO s FROM mst2_metadata_session_row(sid,lid,instance); + IF NOT FOUND THEN RETURN; END IF; + SELECT * INTO STRICT l FROM mst2_qualified_lease_binding WHERE lease_id=lid; + deadline:=least(l.expires_at_unix,floor(extract(epoch FROM clock_timestamp()))::bigint+60); + INSERT INTO mst2_metadata_reader_operation(operation_id,lease_id,snapshot_id,session_incarnation,root_page,root_generation, + lease_epoch,hard_deadline_unix,state) VALUES(op,lid,sid,l.session_incarnation,l.metadata_root,l.root_generation,l.lease_epoch,deadline,'ACTIVE'); + FOREACH kind IN ARRAY ARRAY['REQUEST','READER'] LOOP + INSERT INTO mst2_metadata_root_anchor(anchor_id,anchor_kind,owner_key,root_page,root_generation,root_certificate_digest, + snapshot_id,session_incarnation,lease_id,reader_operation_id) VALUES(gen_random_uuid(),kind,op::text, + l.metadata_root,l.root_generation,s.certificate_digest,sid,l.session_incarnation,lid,op); + END LOOP; + RETURN QUERY SELECT op,l.root_generation,s.certificate_digest::bytea; +END $$; + +CREATE FUNCTION mst2_metadata_finish_reader(op uuid) RETURNS void LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE r mst2_metadata_reader_operation%ROWTYPE; +BEGIN + PERFORM $CORE_SCHEMA$.mst2_route_enter($CORE_LITERAL$); + SELECT * INTO r FROM mst2_metadata_reader_operation WHERE operation_id=op; + IF NOT FOUND THEN RETURN; END IF; + IF r.state='ACTIVE' THEN UPDATE mst2_metadata_reader_operation SET state='FINISHED' WHERE operation_id=op; END IF; + DELETE FROM mst2_metadata_root_anchor WHERE reader_operation_id=op AND anchor_kind IN ('REQUEST','READER'); + PERFORM mst2_metadata_cleanup_lease(r.lease_id); +END $$; + +CREATE FUNCTION mst2_metadata_read_source_entries(op uuid,source_id uuid,p bytea,g bigint,c bytea,names jsonb) +RETURNS TABLE(name bytea,git_oid text,kind smallint,byte_size bigint,content_digest bytea,child_root bytea, + child_generation bigint,child_certificate_digest bytea,child_attestation_id uuid,fact_state text) +LANGUAGE plpgsql VOLATILE SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE a record; profile jsonb; +BEGIN + IF jsonb_typeof(names)<>'array' OR jsonb_array_length(names)>256 + OR EXISTS(SELECT 1 FROM jsonb_array_elements_text(names) wanted + WHERE wanted !~ '^([0-9a-f]{2}){1,255}$') THEN + RAISE EXCEPTION 'qualified source-name read exceeds its bounded exact request'; END IF; + SELECT source.namespace_uuid,source.source_profile,source.tagged_tree_oid,source.source_body_digest,source.source_revision, + source.root_page,source.root_generation,source.root_certificate_digest + INTO a FROM mst2_metadata_source_root_attestation source + JOIN mst2_metadata_prepare origin ON origin.prepare_id=source.origin_prepare_id + WHERE source.attestation_id=source_id AND origin.state='COMMITTED' + AND source.root_page=p AND source.root_generation=g AND source.root_certificate_digest=c + AND source.namespace_uuid='$NAMESPACE_UUID$'::uuid; + IF NOT FOUND THEN RAISE EXCEPTION 'qualified selected directory has no definitive source attestation'; END IF; + SELECT mst2_metadata_native_profile(session.prepare_id) INTO profile + FROM mst2_metadata_reader_operation reader JOIN mst2_qualified_session_incarnation session + ON session.snapshot_id=reader.snapshot_id AND session.session_incarnation=reader.session_incarnation + WHERE reader.operation_id=op AND reader.state='ACTIVE' + AND reader.hard_deadline_unix>floor(extract(epoch FROM clock_timestamp()))::bigint + AND EXISTS(SELECT 1 FROM mst2_metadata_root_anchor anchor WHERE anchor.reader_operation_id=reader.operation_id + AND anchor.anchor_kind='READER' AND anchor.root_page=reader.root_page AND anchor.root_generation=reader.root_generation); + IF NOT FOUND OR profile IS DISTINCT FROM a.source_profile + OR NOT mst2_metadata_root_live(a.root_page,a.root_generation,a.root_certificate_digest) THEN + RAISE EXCEPTION 'qualified source read lost its exact reader profile and current directory'; END IF; + IF NOT $CORE_SCHEMA$.mst2_route_source_tree_matches(split_part(a.tagged_tree_oid,':',2),a.source_revision,a.source_body_digest) THEN + RAISE EXCEPTION 'qualified selected directory source body changed'; END IF; + RETURN QUERY SELECT reference.name,reference.git_oid,reference.kind,reference.byte_size,reference.content_digest, + reference.child_root,reference.child_generation,reference.child_certificate_digest,child.attestation_id, + CASE WHEN reference.kind=4 THEN CASE WHEN child.attestation_id IS NULL THEN 'SOURCE_UNAVAILABLE' ELSE 'READY' END + WHEN fact.git_oid IS NULL THEN 'MISSING' + WHEN fact.state<>'VERIFIED' OR fact.verification_version NOT IN (1,2) + OR fact.size NOT BETWEEN 0 AND 8796093022208 OR octet_length(fact.raw_sha256)<>32 THEN 'INVALID' + WHEN fact.verification_version=1 THEN 'MISSING' + WHEN fact.size IS DISTINCT FROM reference.byte_size OR reference.kind=3 AND fact.size NOT BETWEEN 1 AND 4095 + OR CASE WHEN octet_length(fact.raw_sha256)=32 THEN fact.raw_sha256 ELSE NULL END + IS DISTINCT FROM reference.content_digest THEN 'INVALID' + ELSE 'READY' END + FROM (SELECT DISTINCT decode(value,'hex') AS name FROM jsonb_array_elements_text(names)) wanted + JOIN mst2_metadata_source_entry_reference reference ON reference.attestation_id=source_id AND reference.name=wanted.name + LEFT JOIN LATERAL ( + SELECT verified.git_oid,verified.state,verified.verification_version,verified.size,verified.raw_sha256 + FROM $CORE_SCHEMA$.mst2_verified_object verified WHERE reference.kind<>4 AND verified.storage_domain='git' + AND verified.object_kind='blob' AND verified.git_oid=split_part(reference.git_oid,':',2) + FOR SHARE OF verified NOWAIT + ) fact ON true + LEFT JOIN LATERAL ( + SELECT candidate.attestation_id FROM mst2_metadata_source_root_attestation candidate + JOIN mst2_metadata_prepare origin ON origin.prepare_id=candidate.origin_prepare_id + JOIN $CORE_SCHEMA$.mega_tree source_tree ON source_tree.tree_id=split_part(candidate.tagged_tree_oid,':',2) + WHERE reference.kind=4 AND candidate.namespace_uuid=a.namespace_uuid AND origin.state='COMMITTED' + AND candidate.tagged_tree_oid=reference.git_oid AND candidate.source_profile=a.source_profile + AND candidate.root_page=reference.child_root AND candidate.root_generation=reference.child_generation + AND candidate.root_certificate_digest=reference.child_certificate_digest + AND $CORE_SCHEMA$.mst2_route_source_tree_matches(split_part(candidate.tagged_tree_oid,':',2),candidate.source_revision,candidate.source_body_digest) + AND mst2_metadata_root_live(candidate.root_page,candidate.root_generation,candidate.root_certificate_digest) + ORDER BY candidate.attestation_id LIMIT 1 + ) child ON true ORDER BY reference.name; +END $$; + +CREATE FUNCTION mst2_metadata_cleanup_expired(maximum integer DEFAULT 64) RETURNS void LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE item record; now_unix bigint; bound integer:=least(64,greatest(0,maximum)); +BEGIN + PERFORM $CORE_SCHEMA$.mst2_route_enter($CORE_LITERAL$); + now_unix:=floor(extract(epoch FROM clock_timestamp()))::bigint; + FOR item IN SELECT * FROM ( + (SELECT 'READER'::text AS kind,operation_id::text AS owner,lease_id,hard_deadline_unix AS deadline + FROM mst2_metadata_reader_operation WHERE state='ACTIVE' AND hard_deadline_unix<=now_unix + ORDER BY hard_deadline_unix,operation_id LIMIT bound) + UNION ALL + (SELECT 'LEASE'::text,lease_id,lease_id,expires_at_unix + FROM mst2_qualified_lease_binding WHERE state='ACTIVE' AND expires_at_unix<=now_unix + ORDER BY expires_at_unix,lease_id LIMIT bound) + ) expired ORDER BY deadline,kind,owner LIMIT bound LOOP + IF item.kind='READER' THEN + UPDATE mst2_metadata_reader_operation SET state='EXPIRED' WHERE operation_id=item.owner::uuid AND state='ACTIVE'; + DELETE FROM mst2_metadata_root_anchor WHERE reader_operation_id=item.owner::uuid AND anchor_kind IN ('REQUEST','READER'); + ELSE + UPDATE mst2_qualified_lease_binding SET state='EXPIRED',lease_epoch=lease_epoch+1 WHERE lease_id=item.owner AND state='ACTIVE'; + END IF; + PERFORM mst2_metadata_cleanup_lease(item.lease_id); + END LOOP; +END $$; + +CREATE FUNCTION mst2_metadata_gc_owner_cleanup(maximum integer) +RETURNS TABLE(examined bigint,readers_expired bigint,leases_expired bigint,prepares_aborted bigint, + handovers_retired bigint,orphans_retired bigint) +LANGUAGE plpgsql VOLATILE SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE bound integer:=maximum; item record; q mst2_metadata_prepare%ROWTYPE; now_unix bigint; +BEGIN + PERFORM $CORE_SCHEMA$.mst2_route_enter($CORE_LITERAL$); + IF maximum IS NULL OR maximum NOT BETWEEN 0 AND 64 THEN RAISE EXCEPTION 'qualified owner cleanup budget must be 0..=64'; END IF; + examined:=0; readers_expired:=0; leases_expired:=0; prepares_aborted:=0; handovers_retired:=0; orphans_retired:=0; + now_unix:=floor(extract(epoch FROM clock_timestamp()))::bigint; + FOR item IN SELECT * FROM ( + (SELECT 'READER'::text AS kind,operation_id::text AS owner,lease_id,hard_deadline_unix AS deadline + FROM mst2_metadata_reader_operation WHERE state='ACTIVE' AND hard_deadline_unix<=now_unix + ORDER BY hard_deadline_unix,operation_id LIMIT bound) + UNION ALL + (SELECT 'LEASE'::text,lease_id,lease_id,expires_at_unix FROM mst2_qualified_lease_binding + WHERE state='ACTIVE' AND expires_at_unix<=now_unix ORDER BY expires_at_unix,lease_id LIMIT bound) + UNION ALL + (SELECT 'PREPARE'::text,prepare_id,NULL::text,floor(extract(epoch FROM orphan_expires_at))::bigint + FROM mst2_metadata_prepare WHERE (state='PREPARING' OR state='COMMITTED' AND coverage_retired_at IS NULL) + AND orphan_expires_at<=clock_timestamp() ORDER BY orphan_expires_at,prepare_id LIMIT bound) + ) expired ORDER BY deadline,kind,owner LIMIT bound LOOP + examined:=examined+1; + IF item.kind='READER' THEN + UPDATE mst2_metadata_reader_operation SET state='EXPIRED' WHERE operation_id=item.owner::uuid AND state='ACTIVE'; + IF NOT FOUND THEN RAISE EXCEPTION 'qualified expired reader changed behind its mutation barrier'; END IF; + readers_expired:=readers_expired+1; + DELETE FROM mst2_metadata_root_anchor WHERE reader_operation_id=item.owner::uuid AND anchor_kind IN ('REQUEST','READER'); + PERFORM mst2_metadata_cleanup_lease(item.lease_id); + ELSIF item.kind='LEASE' THEN + UPDATE mst2_qualified_lease_binding SET state='EXPIRED',lease_epoch=lease_epoch+1 WHERE lease_id=item.owner AND state='ACTIVE'; + IF NOT FOUND THEN RAISE EXCEPTION 'qualified expired lease changed behind its mutation barrier'; END IF; + leases_expired:=leases_expired+1; PERFORM mst2_metadata_cleanup_lease(item.lease_id); + ELSE + SELECT * INTO STRICT q FROM mst2_metadata_prepare WHERE prepare_id=item.owner; + IF q.state='PREPARING' THEN + DELETE FROM mst2_metadata_graph_root WHERE prepare_id=q.prepare_id; + UPDATE mst2_metadata_prepare SET state='ABORTED',aborted_at=clock_timestamp() WHERE prepare_id=q.prepare_id; + DELETE FROM mst2_metadata_root_anchor WHERE prepare_id=q.prepare_id AND anchor_kind IN ('PREPARE','REUSE'); + prepares_aborted:=prepares_aborted+1; + ELSIF mst2_metadata_session_covers_prepare(q.prepare_id) THEN + UPDATE mst2_metadata_prepare SET coverage_retired_at=clock_timestamp() WHERE prepare_id=q.prepare_id; + DELETE FROM mst2_metadata_root_anchor WHERE prepare_id=q.prepare_id AND anchor_kind IN ('PREPARE','REUSE'); + handovers_retired:=handovers_retired+1; + ELSIF mst2_metadata_orphan_prepare_eligible(q.prepare_id) THEN + UPDATE mst2_metadata_prepare SET coverage_retired_at=clock_timestamp() WHERE prepare_id=q.prepare_id; + DELETE FROM mst2_metadata_root_anchor WHERE prepare_id=q.prepare_id AND anchor_kind IN ('PREPARE','REUSE'); + orphans_retired:=orphans_retired+1; + END IF; + END IF; + END LOOP; + RETURN NEXT; +END $$; + +CREATE FUNCTION mst2_route_family_registration_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$CORE_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE g mst2_metadata_namespace%ROWTYPE; stamp record; +BEGIN + SELECT * INTO STRICT g FROM mst2_metadata_namespace WHERE singleton=1; + IF (SELECT count(*) FROM mst2_metadata_namespace)<>1 OR NEW.singleton IS NOT NULL + OR substr(NEW.namespace_uuid::text,15,1)<>'4' OR substr(NEW.namespace_uuid::text,20,1) NOT IN ('8','9','a','b') + OR NEW.graph_domain<>'qualified-v1' OR NEW.family_identity<>'v3-rooted-qualified-1' + OR NEW.admission_state<>'ROOTED_Q_ADMITTED' OR NEW.collector_state<>'ENABLED' + OR ROW(NEW.core_schema,NEW.core_schema_oid,NEW.database_name,NEW.database_oid,NEW.storage_uuid, + NEW.server_address,NEW.server_port,NEW.mono_lock_key2) IS DISTINCT FROM + ROW(g.core_schema,g.core_schema_oid,g.database_name,g.database_oid,g.storage_uuid, + g.server_address,g.server_port,g.mono_lock_key2) + OR NEW.metadata_schema IS DISTINCT FROM 'mst2q_'||replace(NEW.namespace_uuid::text,'-','') + OR NEW.metadata_storage_uuid IS NOT DISTINCT FROM g.storage_uuid + OR NEW.metadata_storage_uuid IS DISTINCT FROM (NEW.metadata_storage_uuid::uuid)::text + OR substr(NEW.metadata_storage_uuid,15,1)<>'4' OR substr(NEW.metadata_storage_uuid,20,1) NOT IN ('8','9','a','b') + OR NOT EXISTS(SELECT 1 FROM pg_catalog.pg_namespace n + WHERE n.oid=NEW.metadata_schema_oid AND n.nspname=NEW.metadata_schema) + OR NEW.implementation_fingerprint IS DISTINCT FROM decode('$IMPLEMENTATION_SHA$','hex') + OR NOT mst2_route_lock_held(1297043024,g.mono_lock_key2) + OR NOT mst2_route_lock_held(1296718001,pg_catalog.hashtext(g.core_schema)) + OR NOT mst2_route_lock_held(1296717362,pg_catalog.hashtext(NEW.metadata_schema)) + OR NOT mst2_route_lock_held(1296717362,pg_catalog.hashtext(g.metadata_schema)) THEN + RAISE EXCEPTION 'qualified namespace registration has no exact admitted rooted family scope and lock set'; + END IF; + EXECUTE pg_catalog.format('SELECT * FROM %I.mst2_metadata_family_identity WHERE singleton=1',NEW.metadata_schema) + INTO STRICT stamp; + IF ROW(stamp.namespace_uuid,stamp.storage_uuid,stamp.core_schema_oid,stamp.metadata_schema_oid, + stamp.family_identity,stamp.implementation_fingerprint) IS DISTINCT FROM + ROW(NEW.namespace_uuid,NEW.metadata_storage_uuid,NEW.core_schema_oid,NEW.metadata_schema_oid, + NEW.family_identity,NEW.implementation_fingerprint) + OR NEW.catalog_fingerprint IS DISTINCT FROM mst2_route_family_catalog(NEW.core_schema_oid,NEW.metadata_schema_oid) THEN + RAISE EXCEPTION 'qualified namespace registration fingerprint disagrees with its physical family'; + END IF; + IF NOT EXISTS(SELECT 1 FROM mst2_qualified_family_policy p WHERE p.singleton=1 + AND p.implementation_fingerprint=NEW.implementation_fingerprint + AND p.authority_catalog=mst2_route_family_catalog(NEW.core_schema_oid,0::oid,NEW.metadata_schema_oid) + AND p.expected_shape=mst2_route_family_shape(NEW.metadata_schema_oid,NEW.namespace_uuid,NEW.metadata_storage_uuid)) THEN + RAISE EXCEPTION 'qualified namespace does not have the trusted complete physical family shape'; + END IF; + RETURN NEW; +END $$; diff --git a/src/jupiter/storage/qualified_metadata_serving.sql b/src/jupiter/storage/qualified_metadata_serving.sql index fa3d729e..56ae3c8f 100644 --- a/src/jupiter/storage/qualified_metadata_serving.sql +++ b/src/jupiter/storage/qualified_metadata_serving.sql @@ -263,30 +263,7 @@ END $$; CREATE TRIGGER mst2_metadata_lease_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_qualified_lease_binding FOR EACH ROW EXECUTE FUNCTION mst2_metadata_lease_guard(); -CREATE FUNCTION mst2_metadata_reader_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE -SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ -DECLARE l mst2_qualified_lease_binding%ROWTYPE; now_unix bigint:=floor(extract(epoch FROM clock_timestamp()))::bigint; -BEGIN - IF TG_OP='DELETE' THEN RAISE EXCEPTION 'qualified reader operation history is immutable'; END IF; - IF TG_OP='UPDATE' THEN - IF (to_jsonb(NEW)-'state') IS DISTINCT FROM (to_jsonb(OLD)-'state') - OR OLD.state<>'ACTIVE' AND NEW.state<>OLD.state OR NEW.state NOT IN ('ACTIVE','FINISHED','EXPIRED') - OR NEW.state='EXPIRED' AND OLD.hard_deadline_unix>now_unix THEN - RAISE EXCEPTION 'qualified reader identity cannot change or be prematurely expired'; END IF; - RETURN NEW; - END IF; - SELECT * INTO l FROM mst2_qualified_lease_binding WHERE lease_id=NEW.lease_id AND state='ACTIVE' AND expires_at_unix>now_unix; - IF NOT FOUND OR NEW.state<>'ACTIVE' OR substr(NEW.operation_id::text,15,1)<>'4' - OR substr(NEW.operation_id::text,20,1) NOT IN ('8','9','a','b') - OR ROW(NEW.snapshot_id,NEW.session_incarnation,NEW.root_page,NEW.root_generation,NEW.lease_epoch) IS DISTINCT FROM - ROW(l.snapshot_id,l.session_incarnation,l.metadata_root,l.root_generation,l.lease_epoch) - OR NEW.hard_deadline_unix<=now_unix OR NEW.hard_deadline_unix>least(l.expires_at_unix,now_unix+60) - OR NOT mst2_metadata_incarnation_proof(l.snapshot_id,l.session_incarnation,true) THEN - RAISE EXCEPTION 'qualified reader lacks its exact active lease and bounded deadline'; END IF; - RETURN NEW; -END $$; -CREATE TRIGGER mst2_metadata_reader_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_metadata_reader_operation - FOR EACH ROW EXECUTE FUNCTION mst2_metadata_reader_guard(); +$READER_LIFECYCLE_SQL$ CREATE FUNCTION mst2_metadata_serving_complete() RETURNS trigger LANGUAGE plpgsql VOLATILE SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ @@ -318,9 +295,9 @@ BEGIN AND EXISTS(SELECT 1 FROM mst2_metadata_root_anchor a WHERE a.anchor_kind='LEASE' AND a.lease_id=l.lease_id) THEN RAISE EXCEPTION 'qualified lease cannot commit without its exact final route and owned protection'; END IF; ELSE - SELECT * INTO STRICT r FROM mst2_metadata_reader_operation WHERE operation_id=NEW.operation_id; + SELECT * INTO STRICT r FROM mst2_metadata_reader_operation WHERE operation_id=NEW.operation_id AND reader_issuance=NEW.reader_issuance; IF r.state='ACTIVE' AND (r.hard_deadline_unix<=floor(extract(epoch FROM clock_timestamp()))::bigint - OR (SELECT count(*) FROM mst2_metadata_root_anchor a WHERE a.reader_operation_id=r.operation_id + OR (SELECT count(*) FROM mst2_metadata_root_anchor a WHERE a.reader_operation_id=r.operation_id AND a.reader_issuance=r.reader_issuance AND a.anchor_kind IN ('REQUEST','READER') AND a.owner_key=r.operation_id::text AND a.lease_id=r.lease_id AND a.snapshot_id=r.snapshot_id AND a.session_incarnation=r.session_incarnation AND a.root_page=r.root_page AND a.root_generation=r.root_generation)<>2) @@ -488,17 +465,17 @@ BEGIN PERFORM $CORE_SCHEMA$.mst2_route_enter($CORE_LITERAL$); now_unix:=floor(extract(epoch FROM clock_timestamp()))::bigint; FOR item IN SELECT * FROM ( - (SELECT 'READER'::text AS kind,operation_id::text AS owner,lease_id,hard_deadline_unix AS deadline + (SELECT 'READER'::text AS kind,operation_id::text AS owner,lease_id,hard_deadline_unix AS deadline,reader_issuance FROM mst2_metadata_reader_operation WHERE state='ACTIVE' AND hard_deadline_unix<=now_unix ORDER BY hard_deadline_unix,operation_id LIMIT bound) UNION ALL - (SELECT 'LEASE'::text,lease_id,lease_id,expires_at_unix + (SELECT 'LEASE'::text,lease_id,lease_id,expires_at_unix,NULL::bigint FROM mst2_qualified_lease_binding WHERE state='ACTIVE' AND expires_at_unix<=now_unix ORDER BY expires_at_unix,lease_id LIMIT bound) ) expired ORDER BY deadline,kind,owner LIMIT bound LOOP IF item.kind='READER' THEN - UPDATE mst2_metadata_reader_operation SET state='EXPIRED' WHERE operation_id=item.owner::uuid AND state='ACTIVE'; - DELETE FROM mst2_metadata_root_anchor WHERE reader_operation_id=item.owner::uuid AND anchor_kind IN ('REQUEST','READER'); + UPDATE mst2_metadata_reader_operation SET state='EXPIRED' WHERE operation_id=item.owner::uuid AND reader_issuance=item.reader_issuance AND state='ACTIVE'; + DELETE FROM mst2_metadata_root_anchor WHERE reader_operation_id=item.owner::uuid AND reader_issuance=item.reader_issuance AND anchor_kind IN ('REQUEST','READER'); ELSE UPDATE mst2_qualified_lease_binding SET state='EXPIRED',lease_epoch=lease_epoch+1 WHERE lease_id=item.owner AND state='ACTIVE'; END IF; @@ -543,39 +520,6 @@ BEGIN RETURN changed; END $$; -CREATE FUNCTION mst2_metadata_begin_reader(sid text,lid text,instance text) -RETURNS TABLE(operation_id uuid,root_generation bigint,certificate_digest bytea) LANGUAGE plpgsql VOLATILE -SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ -DECLARE l mst2_qualified_lease_binding%ROWTYPE; s record; op uuid:=gen_random_uuid(); deadline bigint; kind text; -BEGIN - PERFORM $CORE_SCHEMA$.mst2_route_enter($CORE_LITERAL$); - PERFORM mst2_metadata_cleanup_expired(64); - SELECT * INTO s FROM mst2_metadata_session_row(sid,lid,instance); - IF NOT FOUND THEN RETURN; END IF; - SELECT * INTO STRICT l FROM mst2_qualified_lease_binding WHERE lease_id=lid; - deadline:=least(l.expires_at_unix,floor(extract(epoch FROM clock_timestamp()))::bigint+60); - INSERT INTO mst2_metadata_reader_operation(operation_id,lease_id,snapshot_id,session_incarnation,root_page,root_generation, - lease_epoch,hard_deadline_unix,state) VALUES(op,lid,sid,l.session_incarnation,l.metadata_root,l.root_generation,l.lease_epoch,deadline,'ACTIVE'); - FOREACH kind IN ARRAY ARRAY['REQUEST','READER'] LOOP - INSERT INTO mst2_metadata_root_anchor(anchor_id,anchor_kind,owner_key,root_page,root_generation,root_certificate_digest, - snapshot_id,session_incarnation,lease_id,reader_operation_id) VALUES(gen_random_uuid(),kind,op::text, - l.metadata_root,l.root_generation,s.certificate_digest,sid,l.session_incarnation,lid,op); - END LOOP; - RETURN QUERY SELECT op,l.root_generation,s.certificate_digest::bytea; -END $$; - -CREATE FUNCTION mst2_metadata_finish_reader(op uuid) RETURNS void LANGUAGE plpgsql VOLATILE -SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ -DECLARE r mst2_metadata_reader_operation%ROWTYPE; -BEGIN - PERFORM $CORE_SCHEMA$.mst2_route_enter($CORE_LITERAL$); - SELECT * INTO r FROM mst2_metadata_reader_operation WHERE operation_id=op; - IF NOT FOUND THEN RETURN; END IF; - IF r.state='ACTIVE' THEN UPDATE mst2_metadata_reader_operation SET state='FINISHED' WHERE operation_id=op; END IF; - DELETE FROM mst2_metadata_root_anchor WHERE reader_operation_id=op AND anchor_kind IN ('REQUEST','READER'); - PERFORM mst2_metadata_cleanup_lease(r.lease_id); -END $$; - DROP TRIGGER mst2_01_family_closed ON mst2_qualified_session_incarnation; DROP TRIGGER mst2_01_family_closed ON mst2_qualified_lease_binding; DROP TRIGGER mst2_01_family_closed ON mst2_metadata_reader_operation; diff --git a/src/jupiter/storage/qualified_metadata_source_read.sql b/src/jupiter/storage/qualified_metadata_source_read.sql index 232a0e6d..f2442e1b 100644 --- a/src/jupiter/storage/qualified_metadata_source_read.sql +++ b/src/jupiter/storage/qualified_metadata_source_read.sql @@ -68,7 +68,7 @@ END $$; CREATE CONSTRAINT TRIGGER mst2_metadata_source_entries_complete AFTER INSERT ON mst2_metadata_source_root_attestation DEFERRABLE INITIALLY DEFERRED FOR EACH ROW EXECUTE FUNCTION mst2_metadata_source_entries_complete(); -CREATE FUNCTION mst2_metadata_read_source_entries(op uuid,source_id uuid,p bytea,g bigint,c bytea,names jsonb) +CREATE FUNCTION mst2_metadata_read_source_entries(op uuid,issuance bigint,source_id uuid,p bytea,g bigint,c bytea,names jsonb) RETURNS TABLE(name bytea,git_oid text,kind smallint,byte_size bigint,content_digest bytea,child_root bytea, child_generation bigint,child_certificate_digest bytea,child_attestation_id uuid,fact_state text) LANGUAGE plpgsql VOLATILE SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ @@ -89,9 +89,9 @@ BEGIN SELECT mst2_metadata_native_profile(session.prepare_id) INTO profile FROM mst2_metadata_reader_operation reader JOIN mst2_qualified_session_incarnation session ON session.snapshot_id=reader.snapshot_id AND session.session_incarnation=reader.session_incarnation - WHERE reader.operation_id=op AND reader.state='ACTIVE' + WHERE reader.operation_id=op AND reader.reader_issuance=issuance AND reader.state='ACTIVE' AND reader.hard_deadline_unix>floor(extract(epoch FROM clock_timestamp()))::bigint - AND EXISTS(SELECT 1 FROM mst2_metadata_root_anchor anchor WHERE anchor.reader_operation_id=reader.operation_id + AND EXISTS(SELECT 1 FROM mst2_metadata_root_anchor anchor WHERE anchor.reader_operation_id=reader.operation_id AND anchor.reader_issuance=reader.reader_issuance AND anchor.anchor_kind='READER' AND anchor.root_page=reader.root_page AND anchor.root_generation=reader.root_generation); IF NOT FOUND OR profile IS DISTINCT FROM a.source_profile OR NOT mst2_metadata_root_live(a.root_page,a.root_generation,a.root_certificate_digest) THEN From 21b7f896fb5f1add4a58e4fd63e8f47af997d2b1 Mon Sep 17 00:00:00 2001 From: Xiaoyang Han Date: Thu, 8 Oct 2026 11:50:37 +0800 Subject: [PATCH 47/50] fix(v3): repair native decoder and authenticate existing family upgrades (#101) Native rooted plan admission failed because PL/pgSQL parent and child locals conflicted with unqualified graph-query columns. Qualify both columns and repair the migration shape sampler to deparse under pg_catalog,pg_temp with parameterized caller-path restoration. Share trusted forward migration logic between 400 and new 500: authenticate exact 6d/900e/current implementation, captured historical templates, authority/full physical catalogs, scope and canonical namespace identity; replace the decoder on old 6d and the three changed functions on 900e while preserving its relations, function OIDs, reader high-water and retained ownership/history. Admit the exact known Q-visible legacy deparse shape only for the authenticated 900e policy with no actual Q. Handle the legitimate 900e state with a missing 400 ledger without replaying reader DDL. Move the old reader capture into migration verification and reuse the forward-only template renderer in owned fixtures. Add six PostgreSQL regression sources for active/terminal readers, nonzero issuance, all anchors/OIDs, old SID lookup, Q-less canonical/known legacy policies, missing ledger, tampering and idempotence. Add test-only underlying install error output; the production 503 remains unchanged. Owned nightly formatting, candidate whitespace, historical-source equivalence and patch checks pass. Native build/tests/Clippy, PostgreSQL execution and performance measurements remain pending; this is a source-only submission. --- .../snapshot_native_runtime_upgrade_tests.rs | 276 ++++++++ ...snapshot_reader_retention_upgrade_tests.rs | 3 + ...261008_000400_add_mst2_reader_retention.rs | 380 +---------- ...20261008_000500_fix_mst2_native_runtime.rs | 21 + ...0261008_000500_previous_reader_family.sql} | 2 +- ...0261008_000500_previous_rooted_decoder.sql | 110 ++++ src/jupiter/migration/mod.rs | 35 +- .../qualified_native_runtime_upgrade.rs | 597 ++++++++++++++++++ .../storage/native_snapshot_session.rs | 2 + .../qualified_metadata_family_tests.rs | 2 +- ...lified_metadata_reader_previous_fixture.rs | 259 ++++++-- .../storage/qualified_metadata_rooted.sql | 4 +- 12 files changed, 1254 insertions(+), 437 deletions(-) create mode 100644 src/api/router/snapshot_native_runtime_upgrade_tests.rs create mode 100644 src/jupiter/migration/m20261008_000500_fix_mst2_native_runtime.rs rename src/jupiter/{storage/qualified_metadata_reader_previous_fixture.sql => migration/m20261008_000500_previous_reader_family.sql} (99%) create mode 100644 src/jupiter/migration/m20261008_000500_previous_rooted_decoder.sql create mode 100644 src/jupiter/migration/qualified_native_runtime_upgrade.rs diff --git a/src/api/router/snapshot_native_runtime_upgrade_tests.rs b/src/api/router/snapshot_native_runtime_upgrade_tests.rs new file mode 100644 index 00000000..79a76175 --- /dev/null +++ b/src/api/router/snapshot_native_runtime_upgrade_tests.rs @@ -0,0 +1,276 @@ +use super::*; +use crate::jupiter::{ + migration::{apply_migrations, test_upgrade_native_runtime}, + storage::qualified_metadata_family::reader_previous_fixture::restore_retention, +}; + +async fn retained_owners(fixture: &Fixture) -> Value { + let txn = transaction(fixture).await; + let value = txn.query_one_raw(Statement::from_string(DbBackend::Postgres, + "SELECT jsonb_build_object( + 'readers',(SELECT jsonb_agg(to_jsonb(x) ORDER BY operation_id,reader_issuance) FROM mst2_metadata_reader_operation x), + 'anchors',(SELECT jsonb_agg(to_jsonb(x) ORDER BY anchor_id) FROM mst2_metadata_root_anchor x), + 'issuance',(SELECT to_jsonb(x) FROM mst2_metadata_reader_issuance x), + 'functions',(SELECT jsonb_agg(jsonb_build_array(p.proname,p.oid::bigint) ORDER BY p.proname,p.oid) + FROM pg_catalog.pg_proc p JOIN pg_catalog.pg_namespace n ON n.oid=p.pronamespace + WHERE n.nspname=current_schema() OR (n.nspname=(SELECT c.nspname FROM mst2_metadata_family_identity i + JOIN pg_catalog.pg_namespace c ON c.oid=i.core_schema_oid LIMIT 1) + AND p.proname='mst2_route_family_registration_guard'))) AS owners")) + .await.unwrap().unwrap().try_get("", "owners").unwrap(); + txn.commit().await.unwrap(); + value +} + +async fn policy(core: &sea_orm::DatabaseConnection) -> Value { + core.query_one_raw(Statement::from_string( + DbBackend::Postgres, + "SELECT to_jsonb(p) FROM mst2_qualified_family_policy p WHERE singleton=1", + )) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap() +} + +async fn seed_live_and_terminal(fixture: &Fixture) { + let txn = transaction(fixture).await; + admission(&txn, fixture).await; + txn.commit().await.unwrap(); + let txn = transaction(fixture).await; + let terminal = admission(&txn, fixture).await; + finish(&txn, &terminal).await; + txn.commit().await.unwrap(); + assert_eq!( + q_count( + fixture, + "SELECT count(*) FROM {q}.mst2_metadata_reader_operation WHERE state='ACTIVE'" + ) + .await, + 1 + ); + assert_eq!( + q_count( + fixture, + "SELECT count(*) FROM {q}.mst2_metadata_reader_operation WHERE state='FINISHED'" + ) + .await, + 1 + ); + assert!( + q_count( + fixture, + "SELECT high_water FROM {q}.mst2_metadata_reader_issuance" + ) + .await + >= 2 + ); +} + +async fn assert_old_sid_works(fixture: &Fixture) { + let pinned = fixture + .state + .storage + .snapshot_context(&fixture.snapshot, &fixture.lease) + .await + .unwrap(); + let repository = RootedQualifiedMetadataRepository::open( + fixture.state.storage.mono_storage().get_connection(), + &fixture.state.storage.config().database, + ) + .await + .unwrap(); + let context = repository + .context(&fixture.snapshot, &fixture.lease, &pinned.built.instance_id) + .await + .unwrap(); + assert_eq!(context.built.descriptor, pinned.built.descriptor); + assert_eq!(context.commit_oid, pinned.commit_oid); + let lookup = repository + .lookup_metadata(&context, &["/file".to_owned()]) + .await + .unwrap(); + assert!(matches!( + lookup.results.as_slice(), + [RootedLookupStatus::File { .. }] + )); + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn captured_cc90_upgrade_preserves_nonzero_reader_issuance_all_owners_oids_and_old_sid() { + let fixture = Fixture::new_with_pg_config(true).await; + seed_live_and_terminal(&fixture).await; + let schema = q_schema(&fixture).await; + let mono = fixture.state.storage.mono_storage(); + let core = mono.get_connection(); + let rows = permanent_rows(&fixture).await; + let owners = retained_owners(&fixture).await; + restore_retention(core, &schema, true, false).await; + assert_eq!(retained_owners(&fixture).await, owners); + test_upgrade_native_runtime(core).await.unwrap(); + assert_eq!(permanent_rows(&fixture).await, rows); + assert_eq!(retained_owners(&fixture).await, owners); + let stamp = policy(core).await; + test_upgrade_native_runtime(core).await.unwrap(); + assert_eq!(policy(core).await, stamp); + assert_eq!(retained_owners(&fixture).await, owners); + assert_eq!(permanent_rows(&fixture).await, rows); + assert_old_sid_works(&fixture).await; + assert_eq!(permanent_rows(&fixture).await, rows); + provision_or_verify_rooted_qualified_family(core) + .await + .unwrap(); +} + +#[tokio::test] +async fn captured_cc90_interrupted_before_reader_migration_resumes_without_reader_ddl() { + let fixture = Fixture::new_with_pg_config(true).await; + seed_live_and_terminal(&fixture).await; + let schema = q_schema(&fixture).await; + let mono = fixture.state.storage.mono_storage(); + let core = mono.get_connection(); + restore_retention(core, &schema, true, false).await; + let rows = permanent_rows(&fixture).await; + let owners = retained_owners(&fixture).await; + core.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "DELETE FROM seaql_migrations WHERE version IN ($1,$2)", + [ + "m20261008_000400_add_mst2_reader_retention".into(), + "m20261008_000500_fix_mst2_native_runtime".into(), + ], + )) + .await + .unwrap(); + apply_migrations(core, false).await.unwrap(); + assert_eq!(permanent_rows(&fixture).await, rows); + assert_eq!(retained_owners(&fixture).await, owners); + let count: i64 = core.query_one_raw(Statement::from_string(DbBackend::Postgres, + "SELECT count(*) FROM seaql_migrations WHERE version IN + ('m20261008_000400_add_mst2_reader_retention','m20261008_000500_fix_mst2_native_runtime')")) + .await.unwrap().unwrap().try_get_by_index(0).unwrap(); + assert_eq!(count, 2); + assert_old_sid_works(&fixture).await; +} + +async fn qless_upgrade(legacy: bool) { + let temp = tempfile::tempdir().unwrap(); + let (config, _schema) = test_db_config(temp.path()).await; + let core = crate::jupiter::storage::init::database_connection(&config) + .await + .unwrap(); + let old_schema: String = core + .query_one_raw(Statement::from_string( + DbBackend::Postgres, + "SELECT metadata_schema FROM mst2_metadata_namespace WHERE graph_domain='qualified-v1'", + )) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap(); + restore_retention(&core, &old_schema, false, legacy).await; + test_upgrade_native_runtime(&core).await.unwrap(); + let stamp = policy(&core).await; + test_upgrade_native_runtime(&core).await.unwrap(); + assert_eq!(policy(&core).await, stamp); + provision_or_verify_rooted_qualified_family(&core) + .await + .unwrap(); + provision_or_verify_rooted_qualified_family(&core) + .await + .unwrap(); + let row = core + .query_one_raw(Statement::from_string( + DbBackend::Postgres, + "SELECT metadata_schema FROM mst2_metadata_namespace WHERE graph_domain='qualified-v1'", + )) + .await + .unwrap() + .unwrap(); + let schema: String = row.try_get_by_index(0).unwrap(); + assert_ne!(schema, old_schema); + let water: i64 = core + .query_one_raw(Statement::from_string( + DbBackend::Postgres, + format!( + "SELECT high_water FROM \"{}\".mst2_metadata_reader_issuance", + schema.replace('"', "\"\"") + ), + )) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap(); + assert_eq!(water, 0); +} + +#[tokio::test] +async fn captured_cc90_qless_canonical_policy_upgrades_before_new_provisioning() { + qless_upgrade(false).await; +} + +#[tokio::test] +async fn captured_cc90_qless_exact_legacy_deparse_policy_upgrades_before_new_provisioning() { + qless_upgrade(true).await; +} + +#[tokio::test] +async fn captured_cc90_tampered_decoder_rejects_upgrade_without_rewriting_owners_or_policy() { + let fixture = Fixture::new_with_pg_config(true).await; + seed_live_and_terminal(&fixture).await; + let schema = q_schema(&fixture).await; + let mono = fixture.state.storage.mono_storage(); + let core = mono.get_connection(); + restore_retention(core, &schema, true, false).await; + let before = policy(core).await; + let owners = retained_owners(&fixture).await; + let rows = permanent_rows(&fixture).await; + core.execute_unprepared(&format!( + "CREATE OR REPLACE FUNCTION \"{}\".mst2_metadata_decode_rooted_plan(b bytea) RETURNS jsonb + LANGUAGE plpgsql IMMUTABLE STRICT AS $$ BEGIN RAISE EXCEPTION 'tampered decoder'; END $$", + schema.replace('"', "\"\""))).await.unwrap(); + assert!(test_upgrade_native_runtime(core).await.is_err()); + assert_eq!(policy(core).await, before); + assert_eq!(retained_owners(&fixture).await, owners); + assert_eq!(permanent_rows(&fixture).await, rows); +} + +#[tokio::test] +async fn captured_cc90_qless_arbitrary_policy_shape_is_never_resigned() { + for legacy in [false, true] { + let temp = tempfile::tempdir().unwrap(); + let (config, _schema) = test_db_config(temp.path()).await; + let core = crate::jupiter::storage::init::database_connection(&config) + .await + .unwrap(); + let schema: String = core.query_one_raw(Statement::from_string(DbBackend::Postgres, + "SELECT metadata_schema FROM mst2_metadata_namespace WHERE graph_domain='qualified-v1'")) + .await.unwrap().unwrap().try_get_by_index(0).unwrap(); + restore_retention(&core, &schema, false, legacy).await; + core.execute_unprepared("ALTER TABLE mst2_qualified_family_policy DISABLE TRIGGER mst2_route_family_policy_immutable; + UPDATE mst2_qualified_family_policy SET expected_shape=decode(repeat('a5',32),'hex') WHERE singleton=1; + ALTER TABLE mst2_qualified_family_policy ENABLE TRIGGER mst2_route_family_policy_immutable").await.unwrap(); + let before = policy(&core).await; + let schemas: Value = core.query_one_raw(Statement::from_string(DbBackend::Postgres, + "SELECT coalesce(jsonb_agg(n.nspname ORDER BY n.nspname),'[]'::jsonb) FROM pg_catalog.pg_namespace n + WHERE EXISTS(SELECT 1 FROM pg_catalog.pg_proc p WHERE p.pronamespace=n.oid + AND p.proname='mst2_metadata_dml_barrier' + AND strpos(p.prosrc,chr(34)||current_schema()||chr(34))>0)")) + .await.unwrap().unwrap().try_get_by_index(0).unwrap(); + assert!(test_upgrade_native_runtime(&core).await.is_err()); + assert_eq!(policy(&core).await, before); + let after: Value = core.query_one_raw(Statement::from_string(DbBackend::Postgres, + "SELECT coalesce(jsonb_agg(n.nspname ORDER BY n.nspname),'[]'::jsonb) FROM pg_catalog.pg_namespace n + WHERE EXISTS(SELECT 1 FROM pg_catalog.pg_proc p WHERE p.pronamespace=n.oid + AND p.proname='mst2_metadata_dml_barrier' + AND strpos(p.prosrc,chr(34)||current_schema()||chr(34))>0)")) + .await.unwrap().unwrap().try_get_by_index(0).unwrap(); + assert_eq!( + after, schemas, + "rejected migration must roll back every temporary template" + ); + } +} diff --git a/src/api/router/snapshot_reader_retention_upgrade_tests.rs b/src/api/router/snapshot_reader_retention_upgrade_tests.rs index 1f798a8e..54a6eaef 100644 --- a/src/api/router/snapshot_reader_retention_upgrade_tests.rs +++ b/src/api/router/snapshot_reader_retention_upgrade_tests.rs @@ -7,6 +7,9 @@ use crate::jupiter::{ }, }; +#[path = "snapshot_native_runtime_upgrade_tests.rs"] +mod native_runtime; + async fn permanent_rows(fixture: &Fixture) -> Value { let txn = transaction(fixture).await; let rows=txn.query_one_raw(Statement::from_string(DbBackend::Postgres, diff --git a/src/jupiter/migration/m20261008_000400_add_mst2_reader_retention.rs b/src/jupiter/migration/m20261008_000400_add_mst2_reader_retention.rs index 5be3fd26..674c3a90 100644 --- a/src/jupiter/migration/m20261008_000400_add_mst2_reader_retention.rs +++ b/src/jupiter/migration/m20261008_000400_add_mst2_reader_retention.rs @@ -1,393 +1,17 @@ -//! Upgrade only the trusted Q reader lifecycle, preserving fixed source history. +//! Upgrade the trusted v3 reader lifecycle and its native plan decoder. -use sea_orm::{ConnectionTrait, DbBackend, Statement, Value}; use sea_orm_migration::prelude::*; -use crate::jupiter::storage::qualified_metadata_family::{ - implementation_fingerprint, render_family, -}; - -const PREVIOUS_IMPLEMENTATION: &str = - "6d7095dc052e60bf6de21a987b72e23fdfbfe819355f0e847d2a7f3c1cbd3f27"; - #[derive(DeriveMigrationName)] pub struct Migration; -fn rejected(message: &str) -> DbErr { - DbErr::Custom(message.into()) -} - -fn identifier(value: &str) -> String { - format!("\"{}\"", value.replace('"', "\"\"")) -} - -fn literal(value: &str) -> String { - format!("'{}'", value.replace('\'', "''")) -} - -fn trusted_function(source: &str, name: &str) -> Result { - let prefix = format!("CREATE FUNCTION {name}("); - let start = source - .find(&prefix) - .ok_or_else(|| rejected("reader migration trusted function is missing"))?; - let function = &source[start..]; - let body = function - .find("AS $") - .ok_or_else(|| rejected("reader migration function body is missing"))? - + 3; - let tag_end = function[body + 1..] - .find('$') - .ok_or_else(|| rejected("reader migration function delimiter is missing"))? - + body - + 2; - let tag = &function[body..tag_end]; - let end = function[tag_end..] - .find(tag) - .ok_or_else(|| rejected("reader migration function terminator is missing"))? - + tag_end - + tag.len(); - if function.as_bytes().get(end) != Some(&b';') { - return Err(rejected("reader migration function terminator is invalid")); - } - Ok(function[..=end].replacen("CREATE FUNCTION", "CREATE OR REPLACE FUNCTION", 1)) -} - -async fn digest( - connection: &C, - sql: String, - values: Vec, -) -> Result, DbErr> { - connection - .query_one_raw(Statement::from_sql_and_values( - DbBackend::Postgres, - sql, - values, - )) - .await? - .ok_or_else(|| rejected("reader migration physical fingerprint is missing"))? - .try_get("", "fingerprint") -} - -async fn catalog( - connection: &C, - core_oid: i64, - q_oid: i64, - exempt_oid: i64, -) -> Result, DbErr> { - let source = include_str!("../storage/qualified_family_catalog.sql") - .replace("$CORE_OID$", "$1::bigint::oid") - .replace("$Q_OID$", "$2::bigint::oid") - .replace("$EXEMPT_Q_OID$", "$3::bigint::oid"); - digest( - connection, - source, - vec![core_oid.into(), q_oid.into(), exempt_oid.into()], - ) - .await -} - -async fn shape( - connection: &C, - q_oid: i64, - namespace: &str, - storage: &str, -) -> Result, DbErr> { - let source = include_str!("../storage/qualified_family_shape.sql") - .replace("q_oid", "$1::bigint::oid") - .replace("n_uuid", "$2::uuid") - .replace("s_uuid", "$3::text"); - digest( - connection, - source, - vec![q_oid.into(), namespace.into(), storage.into()], - ) - .await -} - -struct Family { - schema: String, - oid: i64, - namespace: String, - storage: String, - catalog: Vec, -} - #[async_trait::async_trait] impl MigrationTrait for Migration { async fn up(&self, manager: &SchemaManager) -> Result<(), DbErr> { - if manager.get_database_backend() != DbBackend::Postgres { - return Err(rejected( - "qualified reader retention requires primary PostgreSQL", - )); - } - let connection = manager.get_connection(); - let captured = connection.query_one_raw(Statement::from_string(DbBackend::Postgres, - "SELECT n.nspname,n.oid::bigint,registry.mono_lock_key2 FROM pg_catalog.pg_namespace n - JOIN mst2_metadata_namespace registry ON registry.singleton=1 AND registry.core_schema=n.nspname - AND registry.core_schema_oid=n.oid WHERE n.nspname=current_schema()")) - .await?.ok_or_else(|| rejected("reader migration requires its captured core schema"))?; - let core: String = captured.try_get("", "nspname")?; - let core_oid: i64 = captured.try_get("", "oid")?; - let mono: i32 = captured.try_get("", "mono_lock_key2")?; - let c = identifier(&core); - connection - .execute_unprepared("SELECT pg_catalog.set_config('lock_timeout','5000ms',true)") - .await?; - connection - .execute_raw(Statement::from_sql_and_values( - DbBackend::Postgres, - "SELECT pg_catalog.pg_advisory_xact_lock(1297043024,$1)", - [mono.into()], - )) - .await?; - connection - .execute_raw(Statement::from_sql_and_values( - DbBackend::Postgres, - "SELECT pg_catalog.pg_advisory_xact_lock(1296718001,pg_catalog.hashtext($1))", - [core.clone().into()], - )) - .await?; - let namespaces = connection.query_all_raw(Statement::from_string(DbBackend::Postgres, - format!("SELECT metadata_schema FROM {c}.mst2_metadata_namespace ORDER BY namespace_uuid"))).await?; - if !(1..=2).contains(&namespaces.len()) { - return Err(rejected( - "reader migration namespace registry is not bounded", - )); - } - for namespace in namespaces { - let schema: String = namespace.try_get("", "metadata_schema")?; - connection - .execute_raw(Statement::from_sql_and_values( - DbBackend::Postgres, - "SELECT pg_catalog.pg_advisory_xact_lock(1296717362,pg_catalog.hashtext($1))", - [schema.into()], - )) - .await?; - } - connection.execute_unprepared(&format!( - "LOCK TABLE {c}.mst2_metadata_namespace,{c}.mst2_qualified_family_policy IN ACCESS EXCLUSIVE MODE" - )).await?; - let policy=connection.query_one_raw(Statement::from_string(DbBackend::Postgres, - format!("SELECT implementation_fingerprint,expected_shape,authority_catalog FROM {c}.mst2_qualified_family_policy WHERE singleton=1"))) - .await?.ok_or_else(|| rejected("reader migration trusted policy is missing"))?; - let implementation: Vec = policy.try_get("", "implementation_fingerprint")?; - let current = implementation_fingerprint(); - let previous = - hex::decode(PREVIOUS_IMPLEMENTATION).map_err(|error| rejected(&error.to_string()))?; - if implementation != previous && implementation != current { - return Err(rejected( - "reader migration refuses an unsupported prior implementation", - )); - } - let rows=connection.query_all_raw(Statement::from_string(DbBackend::Postgres,format!( - "SELECT namespace_uuid::text,metadata_schema,metadata_schema_oid::bigint,metadata_storage_uuid, - implementation_fingerprint,catalog_fingerprint FROM {c}.mst2_metadata_namespace WHERE graph_domain='qualified-v1'"))).await?; - let family = match rows.as_slice() { - [] => None, - [row] => Some(Family { - schema: row.try_get("", "metadata_schema")?, - oid: row.try_get("", "metadata_schema_oid")?, - namespace: row.try_get("", "namespace_uuid")?, - storage: row.try_get("", "metadata_storage_uuid")?, - catalog: row.try_get("", "catalog_fingerprint")?, - }), - _ => return Err(rejected("reader migration requires at most one Q family")), - }; - let q_oid = family.as_ref().map_or(0, |family| family.oid); - if policy.try_get::>("", "authority_catalog")? - != catalog(connection, core_oid, 0, q_oid).await? - { - return Err(rejected( - "reader migration prior core authority catalog changed", - )); - } - let scope = connection - .query_one_raw(Statement::from_sql_and_values( - DbBackend::Postgres, - format!("SELECT {c}.mst2_route_scope_valid($1) AS valid"), - [core.clone().into()], - )) - .await? - .ok_or_else(|| rejected("reader migration core scope is missing"))?; - if !scope.try_get::("", "valid")? { - return Err(rejected("reader migration prior core scope changed")); - } - if let Some(family) = &family { - let q = identifier(&family.schema); - let identity=connection.query_one_raw(Statement::from_sql_and_values(DbBackend::Postgres,format!( - "SELECT EXISTS(SELECT 1 FROM {c}.mst2_metadata_namespace n - JOIN {c}.mst2_metadata_namespace g ON g.singleton=1 - JOIN {q}.mst2_metadata_family_identity i ON i.singleton=1 - JOIN {q}.mst2_metadata_storage_scope s ON s.singleton=1 - JOIN pg_catalog.pg_namespace physical ON physical.oid=n.metadata_schema_oid AND physical.nspname=n.metadata_schema - WHERE n.namespace_uuid=$1::uuid AND n.metadata_schema='mst2q_'||replace(n.namespace_uuid::text,'-','') - AND n.singleton IS NULL AND n.family_identity='v3-rooted-qualified-1' AND n.graph_domain='qualified-v1' - AND n.admission_state='ROOTED_Q_ADMITTED' AND n.collector_state='ENABLED' - AND ROW(n.core_schema,n.core_schema_oid,n.database_name,n.database_oid,n.storage_uuid,n.server_address,n.server_port,n.mono_lock_key2) - IS NOT DISTINCT FROM ROW(g.core_schema,g.core_schema_oid,g.database_name,g.database_oid,g.storage_uuid,g.server_address,g.server_port,g.mono_lock_key2) - AND n.metadata_storage_uuid<>g.storage_uuid AND n.implementation_fingerprint=$2 - AND ROW(i.namespace_uuid,i.storage_uuid,i.core_schema_oid,i.metadata_schema_oid,i.family_identity,i.implementation_fingerprint) - IS NOT DISTINCT FROM ROW(n.namespace_uuid,n.metadata_storage_uuid,n.core_schema_oid,n.metadata_schema_oid,n.family_identity,n.implementation_fingerprint) - AND s.storage_uuid=i.storage_uuid) AS valid"),[family.namespace.clone().into(),implementation.clone().into()])).await? - .ok_or_else(|| rejected("reader migration prior Q identity is missing"))?; - if !identity.try_get::("", "valid")? - || catalog(connection, core_oid, family.oid, 0).await? != family.catalog - || shape(connection, family.oid, &family.namespace, &family.storage).await? - != policy.try_get::>("", "expected_shape")? - { - return Err(rejected( - "reader migration prior Q physical stamp or shape changed", - )); - } - } - if implementation == current { - return Ok(()); - } - - let registration = include_str!("m20261008_000200_rooted_qualified_family.sql") - .replace("$CORE_SCHEMA$", &c) - .replace("$CORE_LITERAL$", &literal(&core)) - .replace("$IMPLEMENTATION_SHA$", &hex::encode(¤t)); - connection - .execute_unprepared(&trusted_function( - ®istration, - "mst2_route_family_registration_guard", - )?) - .await?; - - let template_namespace = uuid::Uuid::new_v4().to_string(); - let template_schema = format!("mst2q_{}", template_namespace.replace('-', "")); - let template_storage = uuid::Uuid::new_v4().to_string(); - let t = identifier(&template_schema); - connection - .execute_unprepared(&format!("CREATE SCHEMA {t}")) - .await?; - let template_oid: i64 = connection - .query_one_raw(Statement::from_sql_and_values( - DbBackend::Postgres, - "SELECT oid::bigint FROM pg_catalog.pg_namespace WHERE nspname=$1", - [template_schema.clone().into()], - )) - .await? - .ok_or_else(|| rejected("reader migration trusted template schema is missing"))? - .try_get("", "oid")?; - connection - .execute_unprepared(&render_family( - &core, - core_oid, - &template_schema, - template_oid, - &template_namespace, - &template_storage, - )) - .await?; - let expected_shape = shape( - connection, - template_oid, - &template_namespace, - &template_storage, - ) - .await?; - connection - .execute_unprepared(&format!( - "SET LOCAL search_path={c},pg_catalog,pg_temp; DROP SCHEMA {t} CASCADE" - )) - .await?; - - if let Some(family) = &family { - let q = identifier(&family.schema); - connection.execute_unprepared(&format!("LOCK TABLE {q}.mst2_metadata_reader_operation,{q}.mst2_metadata_root_anchor IN ACCESS EXCLUSIVE MODE")).await?; - let rendered = render_family( - &core, - core_oid, - &family.schema, - family.oid, - &family.namespace, - &family.storage, - ); - let mut functions = String::new(); - for name in [ - "mst2_metadata_dml_barrier", - "mst2_metadata_gc_enabled", - "mst2_metadata_root_anchor_guard", - "mst2_metadata_serving_complete", - "mst2_metadata_next_reader_issuance", - "mst2_metadata_reader_issuance_guard", - "mst2_metadata_reader_guard", - "mst2_metadata_prune_readers", - "mst2_metadata_begin_reader", - "mst2_metadata_finish_reader", - "mst2_metadata_read_source_entries", - "mst2_metadata_cleanup_expired", - "mst2_metadata_gc_owner_cleanup", - ] { - functions.push_str(&trusted_function(&rendered, name)?); - functions.push('\n'); - } - let upgrade = include_str!("m20261008_000400_reader_retention_upgrade.sql") - .replace("$Q_SCHEMA$", &q) - .replace("$READER_FUNCTIONS_SQL$", &functions); - connection.execute_unprepared(&upgrade).await?; - if shape(connection, family.oid, &family.namespace, &family.storage).await? - != expected_shape - { - return Err(rejected( - "reader migration upgraded Q differs from the trusted complete shape", - )); - } - } - - // Fingerprints include trigger identities/enabled state, but not table - // values. Preserve trigger OIDs while writing only these controlled stamps. - let authority = catalog(connection, core_oid, 0, q_oid).await?; - let full = if q_oid == 0 { - None - } else { - Some(catalog(connection, core_oid, q_oid, 0).await?) - }; - connection.execute_unprepared(&format!("ALTER TABLE {c}.mst2_qualified_family_policy DISABLE TRIGGER mst2_route_family_policy_immutable")).await?; - connection.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres,format!( - "UPDATE {c}.mst2_qualified_family_policy SET implementation_fingerprint=$1,expected_shape=$2,authority_catalog=$3 WHERE singleton=1"), - [current.clone().into(),expected_shape.clone().into(),authority.clone().into()])).await?; - connection.execute_unprepared(&format!("ALTER TABLE {c}.mst2_qualified_family_policy ENABLE TRIGGER mst2_route_family_policy_immutable")).await?; - if let (Some(family), Some(full)) = (&family, &full) { - let q = identifier(&family.schema); - connection.execute_unprepared(&format!( - "ALTER TABLE {q}.mst2_metadata_family_identity DISABLE TRIGGER mst2_00_family_barrier; - ALTER TABLE {q}.mst2_metadata_family_identity DISABLE TRIGGER mst2_metadata_identity_immutable; - ALTER TABLE {c}.mst2_metadata_namespace DISABLE TRIGGER mst2_00_route_statement_barrier; - ALTER TABLE {c}.mst2_metadata_namespace DISABLE TRIGGER mst2_route_immutable")).await?; - connection.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres, - format!("UPDATE {q}.mst2_metadata_family_identity SET implementation_fingerprint=$1 WHERE singleton=1"),[current.clone().into()])).await?; - connection.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres,format!( - "UPDATE {c}.mst2_metadata_namespace SET implementation_fingerprint=$1,catalog_fingerprint=$2 WHERE namespace_uuid=$3::uuid"), - [current.clone().into(),full.clone().into(),family.namespace.clone().into()])).await?; - connection.execute_unprepared(&format!( - "ALTER TABLE {q}.mst2_metadata_family_identity ENABLE TRIGGER mst2_00_family_barrier; - ALTER TABLE {q}.mst2_metadata_family_identity ENABLE TRIGGER mst2_metadata_identity_immutable; - ALTER TABLE {c}.mst2_metadata_namespace ENABLE TRIGGER mst2_00_route_statement_barrier; - ALTER TABLE {c}.mst2_metadata_namespace ENABLE TRIGGER mst2_route_immutable")).await?; - } - connection - .execute_unprepared(&format!("SET LOCAL search_path={c},pg_catalog,pg_temp")) - .await?; - if catalog(connection, core_oid, 0, q_oid).await? != authority { - return Err(rejected( - "reader migration final core authority catalog changed", - )); - } - if let (Some(family), Some(full)) = (&family, &full) - && (catalog(connection, core_oid, q_oid, 0).await? != *full - || shape(connection, family.oid, &family.namespace, &family.storage).await? - != expected_shape) - { - return Err(rejected("reader migration final Q fingerprint changed")); - } - Ok(()) + super::qualified_native_runtime_upgrade::upgrade(manager, true).await } async fn down(&self, _manager: &SchemaManager) -> Result<(), DbErr> { - // Preserve committed issuance fences and the complete reader identity. Ok(()) } diff --git a/src/jupiter/migration/m20261008_000500_fix_mst2_native_runtime.rs b/src/jupiter/migration/m20261008_000500_fix_mst2_native_runtime.rs new file mode 100644 index 00000000..aed57614 --- /dev/null +++ b/src/jupiter/migration/m20261008_000500_fix_mst2_native_runtime.rs @@ -0,0 +1,21 @@ +//! Repair the exact previously admitted v3 decoder without rewriting history. + +use sea_orm_migration::prelude::*; + +#[derive(DeriveMigrationName)] +pub struct Migration; + +#[async_trait::async_trait] +impl MigrationTrait for Migration { + async fn up(&self, manager: &SchemaManager) -> Result<(), DbErr> { + super::qualified_native_runtime_upgrade::upgrade(manager, false).await + } + + async fn down(&self, _manager: &SchemaManager) -> Result<(), DbErr> { + Ok(()) + } + + fn use_transaction(&self) -> Option { + Some(true) + } +} diff --git a/src/jupiter/storage/qualified_metadata_reader_previous_fixture.sql b/src/jupiter/migration/m20261008_000500_previous_reader_family.sql similarity index 99% rename from src/jupiter/storage/qualified_metadata_reader_previous_fixture.sql rename to src/jupiter/migration/m20261008_000500_previous_reader_family.sql index 3b087462..c76bfded 100644 --- a/src/jupiter/storage/qualified_metadata_reader_previous_fixture.sql +++ b/src/jupiter/migration/m20261008_000500_previous_reader_family.sql @@ -1,4 +1,4 @@ --- Captured exactly from c1ff280281440891e732ecfaed4732ee59c41a90; test-only. +-- Captured exactly from c1ff280281440891e732ecfaed4732ee59c41a90; migration verification only. -- READER DDL BEGIN CREATE TABLE mst2_metadata_reader_operation ( operation_id uuid PRIMARY KEY,lease_id text NOT NULL REFERENCES mst2_qualified_lease_binding(lease_id), diff --git a/src/jupiter/migration/m20261008_000500_previous_rooted_decoder.sql b/src/jupiter/migration/m20261008_000500_previous_rooted_decoder.sql new file mode 100644 index 00000000..94ca2612 --- /dev/null +++ b/src/jupiter/migration/m20261008_000500_previous_rooted_decoder.sql @@ -0,0 +1,110 @@ +CREATE FUNCTION mst2_metadata_decode_rooted_plan(b bytea) RETURNS jsonb +LANGUAGE plpgsql IMMUTABLE STRICT SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE domain bytea:=convert_to('mega.mst2.rooted-install.v1','UTF8')||decode('00','hex'); + cursor_pos integer; part jsonb; source_domain text; tree_oid text; scope text; hash_kind text; identity jsonb; + root bytea; count numeric; i integer; page bytea; parent bytea; child bytea; previous bytea; previous_tree text; + size numeric; generation numeric; attestation bytea; attestation_digest bytea; certificate_digest bytea; + delta_rows jsonb[]:=ARRAY[]::jsonb[]; edge_rows jsonb[]:=ARRAY[]::jsonb[]; + reuse_rows jsonb[]:=ARRAY[]::jsonb[]; source_rows jsonb[]:=ARRAY[]::jsonb[]; total_bytes bigint:=0; + delta_index jsonb; node_index jsonb; adjacency jsonb; +BEGIN + IF octet_length(b)>2097152 OR substring(b FROM 1 FOR octet_length(domain))<>domain THEN + RAISE EXCEPTION 'rooted preparation domain or byte budget is invalid'; + END IF; + cursor_pos:=octet_length(domain); + IF mst2_metadata_read_be(b,cursor_pos,2)<>1 THEN RAISE EXCEPTION 'rooted preparation version is unsupported'; END IF; + cursor_pos:=cursor_pos+2; + part:=mst2_metadata_rooted_string(b,cursor_pos,64); cursor_pos:=(part->>'end')::integer; source_domain:=part->>'text'; + part:=mst2_metadata_rooted_string(b,cursor_pos,128); cursor_pos:=(part->>'end')::integer; tree_oid:=part->>'text'; + part:=mst2_metadata_rooted_string(b,cursor_pos,4096); cursor_pos:=(part->>'end')::integer; scope:=part->>'text'; + IF source_domain<>'native-git' OR tree_oid !~ '^(sha1:[0-9a-f]{40}|sha256:[0-9a-f]{64}|blake3:[0-9a-f]{64})$' + OR scope NOT LIKE '/%' OR scope<>'/' AND (scope LIKE '%/' OR scope LIKE '%//%' OR EXISTS( + SELECT 1 FROM unnest(string_to_array(substring(scope FROM 2),'/')) name WHERE name IN ('','.','..') + OR octet_length(name)>255) OR cardinality(string_to_array(substring(scope FROM 2),'/'))>256) THEN + RAISE EXCEPTION 'rooted source identity or scope is invalid'; END IF; + hash_kind:=split_part(tree_oid,':',1); + identity:=jsonb_build_object('source_domain',source_domain,'tagged_root_tree_oid',tree_oid,'scope',scope, + 'schema_version',mst2_metadata_read_be(b,cursor_pos,2), + 'metadata_codec',mst2_metadata_read_be(b,cursor_pos+2,2), + 'materialization_policy',mst2_metadata_read_be(b,cursor_pos+4,2), + 'fs_semantics',mst2_metadata_read_be(b,cursor_pos+6,2), + 'access_projection',mst2_metadata_read_be(b,cursor_pos+8,2), + 'verification_revision',mst2_metadata_read_be(b,cursor_pos+10,4), + 'projection_revision',mst2_metadata_read_be(b,cursor_pos+14,2)); + cursor_pos:=cursor_pos+16; + IF (identity->>'schema_version')::integer<>2 OR (identity->>'metadata_codec')::integer<>1 + OR (identity->>'materialization_policy')::integer<>1 OR (identity->>'fs_semantics')::integer<>1 + OR (identity->>'access_projection')::integer<>0 OR (identity->>'verification_revision')::integer<>2 + OR (identity->>'projection_revision')::integer<>1 THEN RAISE EXCEPTION 'rooted source profile is not current native'; END IF; + IF cursor_pos>octet_length(b)-32 THEN RAISE EXCEPTION 'rooted metadata root is truncated'; END IF; + root:=substring(b FROM cursor_pos+1 FOR 32); cursor_pos:=cursor_pos+32; + count:=mst2_metadata_read_be(b,cursor_pos,4); cursor_pos:=cursor_pos+4; + IF count>4096 THEN RAISE EXCEPTION 'rooted delta exceeds its node budget'; END IF; + IF count>0 THEN FOR i IN 1..count::integer LOOP + IF cursor_pos>octet_length(b)-40 THEN RAISE EXCEPTION 'rooted delta member is truncated'; END IF; + page:=substring(b FROM cursor_pos+1 FOR 32); size:=mst2_metadata_read_be(b,cursor_pos+32,8); cursor_pos:=cursor_pos+40; + IF previous IS NOT NULL AND previous>=page OR size NOT BETWEEN 20 AND 16384 THEN + RAISE EXCEPTION 'rooted delta is not exactly ordered or has invalid size'; END IF; + previous:=page; total_bytes:=total_bytes+size::bigint; + IF total_bytes>67108864 THEN RAISE EXCEPTION 'rooted delta exceeds its metadata byte budget'; END IF; + delta_rows:=array_append(delta_rows,jsonb_build_object('page',encode(page,'hex'),'size',size)); + END LOOP; END IF; + previous:=NULL; count:=mst2_metadata_read_be(b,cursor_pos,4); cursor_pos:=cursor_pos+4; + IF count>16384 THEN RAISE EXCEPTION 'rooted delta exceeds its edge budget'; END IF; + IF count>0 THEN FOR i IN 1..count::integer LOOP + IF cursor_pos>octet_length(b)-64 THEN RAISE EXCEPTION 'rooted edge is truncated'; END IF; + parent:=substring(b FROM cursor_pos+1 FOR 32); child:=substring(b FROM cursor_pos+33 FOR 32); cursor_pos:=cursor_pos+64; + IF previous IS NOT NULL AND previous>=parent||child OR parent=child THEN RAISE EXCEPTION 'rooted edges are not unique and ordered'; END IF; + previous:=parent||child; + edge_rows:=array_append(edge_rows,jsonb_build_object('parent',encode(parent,'hex'),'child',encode(child,'hex'))); + END LOOP; END IF; + previous:=NULL; count:=mst2_metadata_read_be(b,cursor_pos,4); cursor_pos:=cursor_pos+4; + IF count>4096 OR coalesce(array_length(delta_rows,1),0)+count>4096 THEN RAISE EXCEPTION 'rooted delta and boundaries exceed node budget'; END IF; + IF count>0 THEN FOR i IN 1..count::integer LOOP + IF cursor_pos>octet_length(b)-120 THEN RAISE EXCEPTION 'rooted reuse boundary is truncated'; END IF; + page:=substring(b FROM cursor_pos+1 FOR 32); generation:=mst2_metadata_read_be(b,cursor_pos+32,8); + attestation:=substring(b FROM cursor_pos+41 FOR 16); attestation_digest:=substring(b FROM cursor_pos+57 FOR 32); + certificate_digest:=substring(b FROM cursor_pos+89 FOR 32); cursor_pos:=cursor_pos+120; + IF previous IS NOT NULL AND previous>=page OR generation NOT BETWEEN 1 AND 9223372036854775807 THEN + RAISE EXCEPTION 'rooted reuse boundaries are not exact positive ordered lifetimes'; END IF; + previous:=page; + reuse_rows:=array_append(reuse_rows,jsonb_build_object('page',encode(page,'hex'),'generation',generation, + 'attestation_id',encode(attestation,'hex')::uuid,'attestation_digest',encode(attestation_digest,'hex'), + 'certificate_digest',encode(certificate_digest,'hex'))); + END LOOP; END IF; + count:=mst2_metadata_read_be(b,cursor_pos,4); cursor_pos:=cursor_pos+4; + IF count NOT BETWEEN 1 AND 4096 THEN RAISE EXCEPTION 'rooted source-root budget is invalid'; END IF; + FOR i IN 1..count::integer LOOP + part:=mst2_metadata_rooted_string(b,cursor_pos,128); cursor_pos:=(part->>'end')::integer; tree_oid:=part->>'text'; + IF cursor_pos>octet_length(b)-32 THEN RAISE EXCEPTION 'rooted source root is truncated'; END IF; + page:=substring(b FROM cursor_pos+1 FOR 32); cursor_pos:=cursor_pos+32; + IF tree_oid !~ '^(sha1:[0-9a-f]{40}|sha256:[0-9a-f]{64}|blake3:[0-9a-f]{64})$' + OR split_part(tree_oid,':',1)<>hash_kind OR previous_tree IS NOT NULL AND convert_to(previous_tree,'UTF8')>=convert_to(tree_oid,'UTF8') THEN + RAISE EXCEPTION 'rooted source roots are not unique ordered same-profile identities'; END IF; + previous_tree:=tree_oid; source_rows:=array_append(source_rows,jsonb_build_object('tree_oid',tree_oid,'page',encode(page,'hex'))); + END LOOP; + IF cursor_pos<>octet_length(b) THEN RAISE EXCEPTION 'rooted preparation has trailing bytes'; END IF; + SELECT coalesce(jsonb_object_agg(value->>'page',true),'{}'::jsonb) INTO delta_index FROM unnest(delta_rows) d(value); + SELECT coalesce(jsonb_object_agg(value->>'page',true),'{}'::jsonb) INTO node_index FROM ( + SELECT d.value FROM unnest(delta_rows) d(value) UNION ALL SELECT r.value FROM unnest(reuse_rows) r(value)) nodes; + IF coalesce(array_length(delta_rows,1),0)+coalesce(array_length(reuse_rows,1),0)=0 + OR EXISTS(SELECT 1 FROM unnest(reuse_rows) r(value) WHERE delta_index ? (r.value->>'page')) + OR NOT node_index ? encode(root,'hex') + OR EXISTS(SELECT 1 FROM unnest(edge_rows) e(value) WHERE NOT delta_index ? (e.value->>'parent') + OR NOT node_index ? (e.value->>'child')) + OR EXISTS(SELECT 1 FROM unnest(source_rows) s(value) WHERE NOT node_index ? (s.value->>'page')) + OR NOT EXISTS(SELECT 1 FROM unnest(source_rows) s(value) WHERE value->>'page'=encode(root,'hex')) THEN + RAISE EXCEPTION 'rooted plan has overlap or an unbound graph/source endpoint'; + END IF; + SELECT coalesce(jsonb_object_agg(parent,children),'{}'::jsonb) INTO adjacency FROM ( + SELECT value->>'parent' AS parent,jsonb_agg(value->>'child') AS children FROM unnest(edge_rows) e(value) + GROUP BY value->>'parent') grouped; + IF EXISTS(WITH RECURSIVE reached(page) AS (SELECT encode(root,'hex') UNION + SELECT child FROM reached r CROSS JOIN LATERAL jsonb_array_elements_text(adjacency->r.page) children(child)) + SELECT 1 FROM (SELECT d.value FROM unnest(delta_rows) d(value) UNION ALL SELECT r.value FROM unnest(reuse_rows) r(value)) nodes + WHERE NOT EXISTS(SELECT 1 FROM reached r WHERE r.page=nodes.value->>'page')) THEN + RAISE EXCEPTION 'rooted plan includes members outside its bounded delta and boundary closure'; + END IF; + RETURN identity||jsonb_build_object('root',encode(root,'hex'),'delta',to_jsonb(delta_rows),'edges',to_jsonb(edge_rows), + 'reused',to_jsonb(reuse_rows),'source_roots',to_jsonb(source_rows),'total_delta_bytes',total_bytes); +END $$; \ No newline at end of file diff --git a/src/jupiter/migration/mod.rs b/src/jupiter/migration/mod.rs index d44427fc..09eaa04d 100644 --- a/src/jupiter/migration/mod.rs +++ b/src/jupiter/migration/mod.rs @@ -155,6 +155,8 @@ mod m20261008_000100_add_mst2_chunk_maps; mod m20261008_000200_add_mst2_rooted_qualified_family; mod m20261008_000300_add_mst2_chunk_map_retention; mod m20261008_000400_add_mst2_reader_retention; +mod m20261008_000500_fix_mst2_native_runtime; +pub(crate) mod qualified_native_runtime_upgrade; mod runner; pub use m20260905_000100_add_push_queue::ensure_queue_control_seed; pub use runner::apply_migrations; @@ -179,6 +181,26 @@ pub(crate) async fn test_upgrade_reader_retention( } } +#[cfg(test)] +pub(crate) async fn test_upgrade_native_runtime( + connection: &sea_orm::DatabaseConnection, +) -> Result<(), sea_orm::DbErr> { + use sea_orm::TransactionTrait; + use sea_orm_migration::{MigrationTrait, SchemaManager}; + + let txn = connection.begin().await?; + let result = m20261008_000500_fix_mst2_native_runtime::Migration + .up(&SchemaManager::new(&txn)) + .await; + match result { + Ok(()) => txn.commit().await, + Err(error) => { + txn.rollback().await?; + Err(error) + } + } +} + /// Primary key `BIGINT` (not DB auto-increment); the application assigns `id` (e.g. `idgenerator::IdInstance::next_id`). fn pk_bigint(name: T) -> ColumnDef { big_integer(name).primary_key().take() @@ -318,6 +340,7 @@ impl MigratorTrait for Migrator { Box::new(m20261008_000200_add_mst2_rooted_qualified_family::Migration), Box::new(m20261008_000300_add_mst2_chunk_map_retention::Migration), Box::new(m20261008_000400_add_mst2_reader_retention::Migration), + Box::new(m20261008_000500_fix_mst2_native_runtime::Migration), ] } } @@ -1263,22 +1286,26 @@ mod tests { "m20261007_000600_add_mst2_storage_routes" ); assert_eq!( - &names[names.len() - 4], + &names[names.len() - 5], "m20261008_000100_add_mst2_chunk_maps" ); assert_eq!( - &names[names.len() - 3], + &names[names.len() - 4], "m20261008_000200_add_mst2_rooted_qualified_family" ); assert_eq!( - &names[names.len() - 2], + &names[names.len() - 3], "m20261008_000300_add_mst2_chunk_map_retention" ); assert_eq!( - names.last().unwrap(), + &names[names.len() - 2], "m20261008_000400_add_mst2_reader_retention" ); + assert_eq!( + names.last().unwrap(), + "m20261008_000500_fix_mst2_native_runtime" + ); let db = alias_db().await; insert_repo(&db, 1, "/third-party//a").await; insert_repo(&db, 2, "/third-party/b/").await; diff --git a/src/jupiter/migration/qualified_native_runtime_upgrade.rs b/src/jupiter/migration/qualified_native_runtime_upgrade.rs new file mode 100644 index 00000000..0a7d4b94 --- /dev/null +++ b/src/jupiter/migration/qualified_native_runtime_upgrade.rs @@ -0,0 +1,597 @@ +//! Trusted forward repair of the physical v3 Q family, preserving history. + +use sea_orm::{ConnectionTrait, DbBackend, Statement, Value}; +use sea_orm_migration::prelude::*; +use sha2::{Digest, Sha256}; + +use crate::jupiter::storage::qualified_metadata_family::{ + implementation_fingerprint, render_family, +}; + +pub(crate) const PREVIOUS_IMPLEMENTATION: &str = + "6d7095dc052e60bf6de21a987b72e23fdfbfe819355f0e847d2a7f3c1cbd3f27"; + +pub(crate) const RETENTION_IMPLEMENTATION: &str = + "900e7a340e7365c9f31995e070edbfe2076b870ec09fe89e4dc43e56f26b91ab"; +const PREVIOUS_DECODER: &str = include_str!("m20261008_000500_previous_rooted_decoder.sql"); +const PREVIOUS_READERS: &str = include_str!("m20261008_000500_previous_reader_family.sql"); + +fn rejected(message: &str) -> DbErr { + DbErr::Custom(message.into()) +} + +fn identifier(value: &str) -> String { + format!("\"{}\"", value.replace('"', "\"\"")) +} + +fn literal(value: &str) -> String { + format!("'{}'", value.replace('\'', "''")) +} + +fn trusted_function(source: &str, name: &str) -> Result { + let prefix = format!("CREATE FUNCTION {name}("); + let start = source + .find(&prefix) + .ok_or_else(|| rejected("reader migration trusted function is missing"))?; + let function = &source[start..]; + let body = function + .find("AS $") + .ok_or_else(|| rejected("reader migration function body is missing"))? + + 3; + let tag_end = function[body + 1..] + .find('$') + .ok_or_else(|| rejected("reader migration function delimiter is missing"))? + + body + + 2; + let tag = &function[body..tag_end]; + let end = function[tag_end..] + .find(tag) + .ok_or_else(|| rejected("reader migration function terminator is missing"))? + + tag_end + + tag.len(); + if function.as_bytes().get(end) != Some(&b';') { + return Err(rejected("reader migration function terminator is invalid")); + } + Ok(function[..=end].replacen("CREATE FUNCTION", "CREATE OR REPLACE FUNCTION", 1)) +} + +async fn digest( + connection: &C, + sql: String, + values: Vec, +) -> Result, DbErr> { + connection + .query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + sql, + values, + )) + .await? + .ok_or_else(|| rejected("reader migration physical fingerprint is missing"))? + .try_get("", "fingerprint") +} + +async fn catalog( + connection: &C, + core_oid: i64, + q_oid: i64, + exempt_oid: i64, +) -> Result, DbErr> { + let source = include_str!("../storage/qualified_family_catalog.sql") + .replace("$CORE_OID$", "$1::bigint::oid") + .replace("$Q_OID$", "$2::bigint::oid") + .replace("$EXEMPT_Q_OID$", "$3::bigint::oid"); + digest( + connection, + source, + vec![core_oid.into(), q_oid.into(), exempt_oid.into()], + ) + .await +} + +async fn shape( + connection: &C, + q_oid: i64, + namespace: &str, + storage: &str, +) -> Result, DbErr> { + let search_path: String = connection + .query_one_raw(Statement::from_string( + DbBackend::Postgres, + "SELECT pg_catalog.current_setting('search_path') AS search_path", + )) + .await? + .ok_or_else(|| rejected("reader migration caller search path is missing"))? + .try_get("", "search_path")?; + // Catalog deparsing must use the same path as the trusted shape helper. + connection + .execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT pg_catalog.set_config('search_path',$1,true)", + ["pg_catalog,pg_temp".into()], + )) + .await?; + let source = include_str!("../storage/qualified_family_shape.sql") + .replace("q_oid", "$1::bigint::oid") + .replace("n_uuid", "$2::uuid") + .replace("s_uuid", "$3::text"); + let fingerprint = digest( + connection, + source, + vec![q_oid.into(), namespace.into(), storage.into()], + ) + .await; + let restored = connection + .execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT pg_catalog.set_config('search_path',$1,true)", + [search_path.into()], + )) + .await; + let fingerprint = fingerprint?; + restored?; + Ok(fingerprint) +} + +fn replace_function(source: &mut String, name: &str, captured: &str) -> Result<(), DbErr> { + let plain = |sql: &str| -> Result { + Ok(trusted_function(sql, name)?.replacen( + "CREATE OR REPLACE FUNCTION", + "CREATE FUNCTION", + 1, + )) + }; + let current = plain(source)?; + let previous = plain(captured)?; + *source = source.replacen(¤t, &previous, 1); + Ok(()) +} + +/// Historical source is used only to authenticate forward migration templates. +pub(crate) fn render_prior_family( + core: &str, + core_oid: i64, + q: &str, + q_oid: i64, + namespace: &str, + storage: &str, + implementation: &str, +) -> Result { + if ![PREVIOUS_IMPLEMENTATION, RETENTION_IMPLEMENTATION].contains(&implementation) { + return Err(rejected("native runtime template version is unsupported")); + } + let decoder = PREVIOUS_DECODER.replace("\r\n", "\n"); + let readers = PREVIOUS_READERS.replace("\r\n", "\n"); + if hex::encode(Sha256::digest(decoder.as_bytes())) + != "f14d024cfb821f9d6b7348ab9b69f48c09862ff7dd3fbb2b360b253f949e35a9" + || hex::encode(Sha256::digest(readers.as_bytes())) + != "ec9b422440da111e5e047e1a4c85a28a33053cf0e320a56ea944ebefeaa8ea37" + { + return Err(rejected("native runtime historical source capture changed")); + } + let mut source = + render_family(core, core_oid, q, q_oid, namespace, storage).replace("\r\n", "\n"); + replace_function(&mut source, "mst2_metadata_decode_rooted_plan", &decoder)?; + if implementation == PREVIOUS_IMPLEMENTATION { + let start = source + .find("CREATE TABLE mst2_metadata_reader_operation (") + .ok_or_else(|| rejected("native runtime reader template is missing"))?; + let end = source[start..] + .find("CREATE FUNCTION mst2_metadata_session_covers_prepare(") + .ok_or_else(|| rejected("native runtime reader template boundary is missing"))? + + start; + let captured_start = readers + .find("-- READER DDL BEGIN\n") + .ok_or_else(|| rejected("native runtime captured reader DDL is missing"))? + + "-- READER DDL BEGIN\n".len(); + let captured_end = readers + .find("-- READER DDL END") + .ok_or_else(|| rejected("native runtime captured reader DDL boundary is missing"))?; + source.replace_range(start..end, &readers[captured_start..captured_end]); + for name in [ + "mst2_metadata_dml_barrier", + "mst2_metadata_gc_enabled", + "mst2_metadata_root_anchor_guard", + "mst2_metadata_serving_complete", + "mst2_metadata_reader_guard", + "mst2_metadata_begin_reader", + "mst2_metadata_finish_reader", + "mst2_metadata_read_source_entries", + "mst2_metadata_cleanup_expired", + "mst2_metadata_gc_owner_cleanup", + ] { + replace_function(&mut source, name, &readers)?; + } + for name in [ + "mst2_metadata_next_reader_issuance", + "mst2_metadata_reader_issuance_guard", + "mst2_metadata_prune_readers", + ] { + let function = trusted_function(&source, name)?.replacen( + "CREATE OR REPLACE FUNCTION", + "CREATE FUNCTION", + 1, + ); + source = source.replacen(&function, "", 1); + } + source = source.replace( + "CREATE TRIGGER mst2_metadata_reader_issuance_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_metadata_reader_issuance\n FOR EACH ROW EXECUTE FUNCTION mst2_metadata_reader_issuance_guard();", "", + ).replace("'mst2_metadata_reader_issuance',", ""); + } + Ok(source + .replace(&hex::encode(implementation_fingerprint()), implementation) + .replace("$CORE_SCHEMA$", &identifier(core)) + .replace("$CORE_LITERAL$", &literal(core)) + .replace("$Q_SCHEMA$", &identifier(q)) + .replace("$Q_LITERAL$", &literal(q)) + .replace("$CORE_OID$", &core_oid.to_string()) + .replace("$Q_OID$", &q_oid.to_string()) + .replace("$NAMESPACE_UUID$", namespace) + .replace("$STORAGE_UUID$", storage) + .replace("$IMPLEMENTATION_SHA$", implementation)) +} + +struct TrustedShape { + canonical: Vec, + legacy: Option>, +} + +async fn template_shapes( + connection: &C, + core: &str, + core_oid: i64, + implementation: &[u8], +) -> Result { + let namespace = uuid::Uuid::new_v4().to_string(); + let schema = format!("mst2q_{}", namespace.replace('-', "")); + let storage = uuid::Uuid::new_v4().to_string(); + let q = identifier(&schema); + let c = identifier(core); + connection + .execute_unprepared(&format!("CREATE SCHEMA {q}")) + .await?; + let oid: i64 = connection + .query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT oid::bigint FROM pg_catalog.pg_namespace WHERE nspname=$1", + [schema.clone().into()], + )) + .await? + .ok_or_else(|| rejected("native runtime trusted template schema is missing"))? + .try_get("", "oid")?; + let version = hex::encode(implementation); + let source = if implementation == implementation_fingerprint().as_slice() { + render_family(core, core_oid, &schema, oid, &namespace, &storage) + } else { + render_prior_family(core, core_oid, &schema, oid, &namespace, &storage, &version)? + }; + connection.execute_unprepared(&source).await?; + let canonical = shape(connection, oid, &namespace, &storage).await?; + // Reproduce only the exact historical Q-visible deparsing defect. The + // exception is admitted below solely for the known 900e policy without Q. + let legacy = if version == RETENTION_IMPLEMENTATION { + let source = include_str!("../storage/qualified_family_shape.sql") + .replace("q_oid", "$1::bigint::oid") + .replace("n_uuid", "$2::uuid") + .replace("s_uuid", "$3::text"); + Some( + digest( + connection, + source, + vec![oid.into(), namespace.into(), storage.into()], + ) + .await?, + ) + } else { + None + }; + connection + .execute_unprepared(&format!( + "SET LOCAL search_path={c},pg_catalog,pg_temp; DROP SCHEMA {q} CASCADE" + )) + .await?; + Ok(TrustedShape { canonical, legacy }) +} + +struct Family { + schema: String, + oid: i64, + namespace: String, + storage: String, + catalog: Vec, +} + +pub(super) async fn upgrade( + manager: &SchemaManager<'_>, + allow_previous_readers: bool, +) -> Result<(), DbErr> { + if manager.get_database_backend() != DbBackend::Postgres { + return Err(rejected( + "qualified reader retention requires primary PostgreSQL", + )); + } + let connection = manager.get_connection(); + let captured = connection.query_one_raw(Statement::from_string(DbBackend::Postgres, + "SELECT n.nspname,n.oid::bigint,registry.mono_lock_key2 FROM pg_catalog.pg_namespace n + JOIN mst2_metadata_namespace registry ON registry.singleton=1 AND registry.core_schema=n.nspname + AND registry.core_schema_oid=n.oid WHERE n.nspname=current_schema()")) + .await?.ok_or_else(|| rejected("reader migration requires its captured core schema"))?; + let core: String = captured.try_get("", "nspname")?; + let core_oid: i64 = captured.try_get("", "oid")?; + let mono: i32 = captured.try_get("", "mono_lock_key2")?; + let c = identifier(&core); + connection + .execute_unprepared("SELECT pg_catalog.set_config('lock_timeout','5000ms',true)") + .await?; + connection + .execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT pg_catalog.pg_advisory_xact_lock(1297043024,$1)", + [mono.into()], + )) + .await?; + connection + .execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT pg_catalog.pg_advisory_xact_lock(1296718001,pg_catalog.hashtext($1))", + [core.clone().into()], + )) + .await?; + let namespaces = connection + .query_all_raw(Statement::from_string( + DbBackend::Postgres, + format!( + "SELECT metadata_schema FROM {c}.mst2_metadata_namespace ORDER BY namespace_uuid" + ), + )) + .await?; + if !(1..=2).contains(&namespaces.len()) { + return Err(rejected( + "reader migration namespace registry is not bounded", + )); + } + let namespace_count = namespaces.len(); + for namespace in namespaces { + let schema: String = namespace.try_get("", "metadata_schema")?; + connection + .execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT pg_catalog.pg_advisory_xact_lock(1296717362,pg_catalog.hashtext($1))", + [schema.into()], + )) + .await?; + } + connection.execute_unprepared(&format!( + "LOCK TABLE {c}.mst2_metadata_namespace,{c}.mst2_qualified_family_policy IN ACCESS EXCLUSIVE MODE" + )).await?; + let policy=connection.query_one_raw(Statement::from_string(DbBackend::Postgres, + format!("SELECT implementation_fingerprint,expected_shape,authority_catalog FROM {c}.mst2_qualified_family_policy WHERE singleton=1"))) + .await?.ok_or_else(|| rejected("reader migration trusted policy is missing"))?; + let implementation: Vec = policy.try_get("", "implementation_fingerprint")?; + let current = implementation_fingerprint(); + let previous = + hex::decode(PREVIOUS_IMPLEMENTATION).map_err(|error| rejected(&error.to_string()))?; + let retention = + hex::decode(RETENTION_IMPLEMENTATION).map_err(|error| rejected(&error.to_string()))?; + if implementation != current + && implementation != retention + && !(allow_previous_readers && implementation == previous) + { + return Err(rejected( + "reader migration refuses an unsupported prior implementation", + )); + } + let rows=connection.query_all_raw(Statement::from_string(DbBackend::Postgres,format!( + "SELECT namespace_uuid::text,metadata_schema,metadata_schema_oid::bigint,metadata_storage_uuid, + implementation_fingerprint,catalog_fingerprint FROM {c}.mst2_metadata_namespace WHERE graph_domain='qualified-v1'"))).await?; + let family = match rows.as_slice() { + [] => None, + [row] => Some(Family { + schema: row.try_get("", "metadata_schema")?, + oid: row.try_get("", "metadata_schema_oid")?, + namespace: row.try_get("", "namespace_uuid")?, + storage: row.try_get("", "metadata_storage_uuid")?, + catalog: row.try_get("", "catalog_fingerprint")?, + }), + _ => return Err(rejected("reader migration requires at most one Q family")), + }; + let q_oid = family.as_ref().map_or(0, |family| family.oid); + if namespace_count != if family.is_some() { 2 } else { 1 } { + return Err(rejected( + "native runtime namespace registry identity is not exact", + )); + } + if policy.try_get::>("", "authority_catalog")? + != catalog(connection, core_oid, 0, q_oid).await? + { + return Err(rejected( + "reader migration prior core authority catalog changed", + )); + } + let scope = connection + .query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + format!("SELECT {c}.mst2_route_scope_valid($1) AS valid"), + [core.clone().into()], + )) + .await? + .ok_or_else(|| rejected("reader migration core scope is missing"))?; + if !scope.try_get::("", "valid")? { + return Err(rejected("reader migration prior core scope changed")); + } + if let Some(family) = &family { + for identity in [&family.namespace, &family.storage] { + let parsed = uuid::Uuid::parse_str(identity) + .map_err(|_| rejected("native runtime prior Q UUID is invalid"))?; + if parsed.get_version_num() != 4 + || parsed.get_variant() != uuid::Variant::RFC4122 + || parsed.to_string() != *identity + { + return Err(rejected("native runtime prior Q UUID is not canonical v4")); + } + } + let q = identifier(&family.schema); + let identity=connection.query_one_raw(Statement::from_sql_and_values(DbBackend::Postgres,format!( + "SELECT EXISTS(SELECT 1 FROM {c}.mst2_metadata_namespace n + JOIN {c}.mst2_metadata_namespace g ON g.singleton=1 + JOIN {q}.mst2_metadata_family_identity i ON i.singleton=1 + JOIN {q}.mst2_metadata_storage_scope s ON s.singleton=1 + JOIN pg_catalog.pg_namespace physical ON physical.oid=n.metadata_schema_oid AND physical.nspname=n.metadata_schema + WHERE n.namespace_uuid=$1::uuid AND n.metadata_schema='mst2q_'||replace(n.namespace_uuid::text,'-','') + AND n.singleton IS NULL AND n.family_identity='v3-rooted-qualified-1' AND n.graph_domain='qualified-v1' + AND n.admission_state='ROOTED_Q_ADMITTED' AND n.collector_state='ENABLED' + AND ROW(n.core_schema,n.core_schema_oid,n.database_name,n.database_oid,n.storage_uuid,n.server_address,n.server_port,n.mono_lock_key2) + IS NOT DISTINCT FROM ROW(g.core_schema,g.core_schema_oid,g.database_name,g.database_oid,g.storage_uuid,g.server_address,g.server_port,g.mono_lock_key2) + AND n.metadata_storage_uuid<>g.storage_uuid AND n.implementation_fingerprint=$2 + AND ROW(i.namespace_uuid,i.storage_uuid,i.core_schema_oid,i.metadata_schema_oid,i.family_identity,i.implementation_fingerprint) + IS NOT DISTINCT FROM ROW(n.namespace_uuid,n.metadata_storage_uuid,n.core_schema_oid,n.metadata_schema_oid,n.family_identity,n.implementation_fingerprint) + AND s.storage_uuid=i.storage_uuid) AS valid"),[family.namespace.clone().into(),implementation.clone().into()])).await? + .ok_or_else(|| rejected("reader migration prior Q identity is missing"))?; + if !identity.try_get::("", "valid")? + || catalog(connection, core_oid, family.oid, 0).await? != family.catalog + || shape(connection, family.oid, &family.namespace, &family.storage).await? + != policy.try_get::>("", "expected_shape")? + { + return Err(rejected( + "reader migration prior Q physical stamp or shape changed", + )); + } + } + let prior = template_shapes(connection, &core, core_oid, &implementation).await?; + let recorded_shape: Vec = policy.try_get("", "expected_shape")?; + if recorded_shape != prior.canonical + && !(implementation == retention + && family.is_none() + && prior.legacy.as_ref() == Some(&recorded_shape)) + { + return Err(rejected( + "native runtime migration prior policy shape is not trusted", + )); + } + if implementation == current { + return Ok(()); + } + + let registration = include_str!("m20261008_000200_rooted_qualified_family.sql") + .replace("$CORE_SCHEMA$", &c) + .replace("$CORE_LITERAL$", &literal(&core)) + .replace("$IMPLEMENTATION_SHA$", &hex::encode(¤t)); + connection + .execute_unprepared(&trusted_function( + ®istration, + "mst2_route_family_registration_guard", + )?) + .await?; + + let expected_shape = template_shapes(connection, &core, core_oid, ¤t) + .await? + .canonical; + + if let Some(family) = &family { + let q = identifier(&family.schema); + connection.execute_unprepared(&format!("LOCK TABLE {q}.mst2_metadata_reader_operation,{q}.mst2_metadata_root_anchor IN ACCESS EXCLUSIVE MODE")).await?; + let rendered = render_family( + &core, + core_oid, + &family.schema, + family.oid, + &family.namespace, + &family.storage, + ); + let mut functions = String::new(); + let replacements: &[&str] = if implementation == previous { + &[ + "mst2_metadata_decode_rooted_plan", + "mst2_metadata_dml_barrier", + "mst2_metadata_gc_enabled", + "mst2_metadata_root_anchor_guard", + "mst2_metadata_serving_complete", + "mst2_metadata_next_reader_issuance", + "mst2_metadata_reader_issuance_guard", + "mst2_metadata_reader_guard", + "mst2_metadata_prune_readers", + "mst2_metadata_begin_reader", + "mst2_metadata_finish_reader", + "mst2_metadata_read_source_entries", + "mst2_metadata_cleanup_expired", + "mst2_metadata_gc_owner_cleanup", + ] + } else { + &[ + "mst2_metadata_decode_rooted_plan", + "mst2_metadata_dml_barrier", + "mst2_metadata_gc_enabled", + ] + }; + for name in replacements { + functions.push_str(&trusted_function(&rendered, name)?); + functions.push('\n'); + } + if implementation == previous { + let upgrade = include_str!("m20261008_000400_reader_retention_upgrade.sql") + .replace("$Q_SCHEMA$", &q) + .replace("$READER_FUNCTIONS_SQL$", &functions); + connection.execute_unprepared(&upgrade).await?; + } else { + connection + .execute_unprepared(&format!("SET LOCAL search_path={q},pg_catalog,pg_temp")) + .await?; + connection.execute_unprepared(&functions).await?; + } + if shape(connection, family.oid, &family.namespace, &family.storage).await? + != expected_shape + { + return Err(rejected( + "reader migration upgraded Q differs from the trusted complete shape", + )); + } + } + + // Fingerprints include trigger identities/enabled state, but not table + // values. Preserve trigger OIDs while writing only these controlled stamps. + let authority = catalog(connection, core_oid, 0, q_oid).await?; + let full = if q_oid == 0 { + None + } else { + Some(catalog(connection, core_oid, q_oid, 0).await?) + }; + connection.execute_unprepared(&format!("ALTER TABLE {c}.mst2_qualified_family_policy DISABLE TRIGGER mst2_route_family_policy_immutable")).await?; + connection.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres,format!( + "UPDATE {c}.mst2_qualified_family_policy SET implementation_fingerprint=$1,expected_shape=$2,authority_catalog=$3 WHERE singleton=1"), + [current.clone().into(),expected_shape.clone().into(),authority.clone().into()])).await?; + connection.execute_unprepared(&format!("ALTER TABLE {c}.mst2_qualified_family_policy ENABLE TRIGGER mst2_route_family_policy_immutable")).await?; + if let (Some(family), Some(full)) = (&family, &full) { + let q = identifier(&family.schema); + connection.execute_unprepared(&format!( + "ALTER TABLE {q}.mst2_metadata_family_identity DISABLE TRIGGER mst2_00_family_barrier; + ALTER TABLE {q}.mst2_metadata_family_identity DISABLE TRIGGER mst2_metadata_identity_immutable; + ALTER TABLE {c}.mst2_metadata_namespace DISABLE TRIGGER mst2_00_route_statement_barrier; + ALTER TABLE {c}.mst2_metadata_namespace DISABLE TRIGGER mst2_route_immutable")).await?; + connection.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres, + format!("UPDATE {q}.mst2_metadata_family_identity SET implementation_fingerprint=$1 WHERE singleton=1"),[current.clone().into()])).await?; + connection.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres,format!( + "UPDATE {c}.mst2_metadata_namespace SET implementation_fingerprint=$1,catalog_fingerprint=$2 WHERE namespace_uuid=$3::uuid"), + [current.clone().into(),full.clone().into(),family.namespace.clone().into()])).await?; + connection.execute_unprepared(&format!( + "ALTER TABLE {q}.mst2_metadata_family_identity ENABLE TRIGGER mst2_00_family_barrier; + ALTER TABLE {q}.mst2_metadata_family_identity ENABLE TRIGGER mst2_metadata_identity_immutable; + ALTER TABLE {c}.mst2_metadata_namespace ENABLE TRIGGER mst2_00_route_statement_barrier; + ALTER TABLE {c}.mst2_metadata_namespace ENABLE TRIGGER mst2_route_immutable")).await?; + } + connection + .execute_unprepared(&format!("SET LOCAL search_path={c},pg_catalog,pg_temp")) + .await?; + if catalog(connection, core_oid, 0, q_oid).await? != authority { + return Err(rejected( + "reader migration final core authority catalog changed", + )); + } + if let (Some(family), Some(full)) = (&family, &full) + && (catalog(connection, core_oid, q_oid, 0).await? != *full + || shape(connection, family.oid, &family.namespace, &family.storage).await? + != expected_shape) + { + return Err(rejected("reader migration final Q fingerprint changed")); + } + Ok(()) +} diff --git a/src/jupiter/storage/native_snapshot_session.rs b/src/jupiter/storage/native_snapshot_session.rs index da15a716..a13180bd 100644 --- a/src/jupiter/storage/native_snapshot_session.rs +++ b/src/jupiter/storage/native_snapshot_session.rs @@ -671,6 +671,8 @@ pub(crate) fn install_error(error: MetadataInstallError) -> SnapshotError { match error { MetadataInstallError::Rejected(error) if error.code == SnapshotErrorCode::Internal => { tracing::error!(%error,"native metadata installation unavailable"); + #[cfg(test)] + eprintln!("MST2 test native metadata installation rejected: {error}"); SnapshotError::new( SnapshotErrorCode::TemporaryUnavailable, "native metadata installation unavailable", diff --git a/src/jupiter/storage/qualified_metadata_family_tests.rs b/src/jupiter/storage/qualified_metadata_family_tests.rs index 54106a11..176c465d 100644 --- a/src/jupiter/storage/qualified_metadata_family_tests.rs +++ b/src/jupiter/storage/qualified_metadata_family_tests.rs @@ -136,7 +136,7 @@ async fn physical_q_shadow_writer_is_distinct_and_restart_reuses_its_exact_catal count(&q, "SELECT count(*) FROM mst2_qualified_lease_binding").await, 0 ); - assert_eq!(count(&q,"SELECT count(*) FROM pg_catalog.pg_class WHERE relnamespace=(SELECT oid FROM pg_catalog.pg_namespace WHERE nspname=current_schema()) AND relkind='r'").await,22); + assert_eq!(count(&q,"SELECT count(*) FROM pg_catalog.pg_class WHERE relnamespace=(SELECT oid FROM pg_catalog.pg_namespace WHERE nspname=current_schema()) AND relkind='r'").await,23); assert_eq!(count(&q,"SELECT count(*) FROM pg_catalog.pg_class WHERE relnamespace=(SELECT oid FROM pg_catalog.pg_namespace WHERE nspname=current_schema()) AND relname IN ('mst2_retention_node','mst2_retention_edge','mst2_snapshot_context','mst2_snapshot_lease','mega_refs','git_repo','seaql_migrations')").await,0); let q_scope = q .query_one_raw(Statement::from_string( diff --git a/src/jupiter/storage/qualified_metadata_reader_previous_fixture.rs b/src/jupiter/storage/qualified_metadata_reader_previous_fixture.rs index 84ce5232..ba120126 100644 --- a/src/jupiter/storage/qualified_metadata_reader_previous_fixture.rs +++ b/src/jupiter/storage/qualified_metadata_reader_previous_fixture.rs @@ -8,7 +8,7 @@ use serde_json::{Value, json}; use super::*; const PREVIOUS: &str = "6d7095dc052e60bf6de21a987b72e23fdfbfe819355f0e847d2a7f3c1cbd3f27"; -const CAPTURE: &str = include_str!("qualified_metadata_reader_previous_fixture.sql"); +const CAPTURE: &str = include_str!("../migration/m20261008_000500_previous_reader_family.sql"); fn function(source: &str, name: &str) -> (usize, usize) { let start = source.find(&format!("CREATE FUNCTION {name}(")).unwrap(); @@ -29,56 +29,10 @@ fn render_previous( namespace: &str, storage: &str, ) -> String { - let capture = CAPTURE.replace("\r\n", "\n"); - assert_eq!( - hex::encode(Sha256::digest(capture.as_bytes())), - "599a5a6749cb41afd8f7b05287a482ab6ba921d4ffe6142159b0bbf0f00e33f4" - ); - let mut sql = render_family(core, core_oid, q, q_oid, namespace, storage).replace("\r\n", "\n"); - let a = sql - .find("CREATE TABLE mst2_metadata_reader_operation (") - .unwrap(); - let z = sql[a..] - .find("CREATE FUNCTION mst2_metadata_session_covers_prepare(") - .unwrap() - + a; - let old_a = capture.find("-- READER DDL BEGIN\n").unwrap() + "-- READER DDL BEGIN\n".len(); - let old_z = capture.find("-- READER DDL END").unwrap(); - sql.replace_range(a..z, &capture[old_a..old_z]); - for name in [ - "mst2_metadata_dml_barrier", - "mst2_metadata_gc_enabled", - "mst2_metadata_root_anchor_guard", - "mst2_metadata_serving_complete", - "mst2_metadata_reader_guard", - "mst2_metadata_begin_reader", - "mst2_metadata_finish_reader", - "mst2_metadata_read_source_entries", - "mst2_metadata_cleanup_expired", - "mst2_metadata_gc_owner_cleanup", - ] { - let (a, z) = function(&sql, name); - let (old_a, old_z) = function(&capture, name); - sql.replace_range(a..z, &capture[old_a..old_z]); - } - for name in [ - "mst2_metadata_next_reader_issuance", - "mst2_metadata_reader_issuance_guard", - "mst2_metadata_prune_readers", - ] { - let (a, z) = function(&sql, name); - sql.replace_range(a..z, ""); - } - sql=sql.replace("CREATE TRIGGER mst2_metadata_reader_issuance_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_metadata_reader_issuance\n FOR EACH ROW EXECUTE FUNCTION mst2_metadata_reader_issuance_guard();","") - .replace("'mst2_metadata_reader_issuance',","") - .replace(&hex::encode(implementation_fingerprint()),PREVIOUS) - .replace("$CORE_SCHEMA$",&identifier(core)).replace("$CORE_LITERAL$",&literal(core)) - .replace("$Q_SCHEMA$",&identifier(q)).replace("$Q_LITERAL$",&literal(q)) - .replace("$CORE_OID$",&core_oid.to_string()).replace("$Q_OID$",&q_oid.to_string()) - .replace("$NAMESPACE_UUID$",namespace).replace("$STORAGE_UUID$",storage).replace("$IMPLEMENTATION_SHA$",PREVIOUS); - assert!(!sql.contains("reader_issuance")); - assert!(!sql.contains("terminal_xid")); - sql + crate::jupiter::migration::qualified_native_runtime_upgrade::render_prior_family( + core, core_oid, q, q_oid, namespace, storage, PREVIOUS, + ) + .unwrap() } struct Table { @@ -220,6 +174,9 @@ pub(crate) async fn restore_previous(core: &DatabaseConnection, q_schema: &str, .await .unwrap(); } + txn.execute_unprepared(&format!("SET LOCAL search_path={c},pg_catalog,pg_temp")) + .await + .unwrap(); let capture = CAPTURE.replace("\r\n", "\n"); let (a, z) = function(&capture, "mst2_route_family_registration_guard"); let registration = capture[a..z] @@ -293,3 +250,203 @@ pub(crate) async fn restore_previous(core: &DatabaseConnection, q_schema: &str, } txn.commit().await.unwrap(); } + +/// Restore the exact cc90 functions and stamps in an explicitly owned schema. +/// Existing reader generations, rows, relations and function OIDs are retained. +pub(crate) async fn restore_retention( + core: &DatabaseConnection, + q_schema: &str, + keep_q: bool, + legacy_shape: bool, +) { + use crate::jupiter::migration::qualified_native_runtime_upgrade::{ + RETENTION_IMPLEMENTATION, render_prior_family, + }; + + assert!(!legacy_shape || !keep_q); + let txn = core.begin().await.unwrap(); + let (core_schema, core_oid) = captured_core(&txn).await.unwrap(); + assert!(core_schema.starts_with("mega2_test_")); + assert!(q_schema.starts_with("mst2q_")); + let c = identifier(&core_schema); + let q = identifier(q_schema); + txn.execute_unprepared(&format!( + "SELECT {c}.mst2_route_enter({})", + literal(&core_schema) + )) + .await + .unwrap(); + let registry = txn.query_one_raw(Statement::from_sql_and_values(DbBackend::Postgres, + format!("SELECT metadata_schema_oid::bigint,namespace_uuid::text,metadata_storage_uuid + FROM {c}.mst2_metadata_namespace WHERE metadata_schema=$1 AND graph_domain='qualified-v1'"), + [q_schema.into()])).await.unwrap().unwrap(); + let oid: i64 = registry.try_get("", "metadata_schema_oid").unwrap(); + let namespace: String = registry.try_get("", "namespace_uuid").unwrap(); + let storage: String = registry.try_get("", "metadata_storage_uuid").unwrap(); + let rendered = render_prior_family( + &core_schema, + core_oid, + q_schema, + oid, + &namespace, + &storage, + RETENTION_IMPLEMENTATION, + ) + .unwrap(); + txn.execute_unprepared(&format!("SET LOCAL search_path={q},pg_catalog,pg_temp")) + .await + .unwrap(); + for name in [ + "mst2_metadata_decode_rooted_plan", + "mst2_metadata_dml_barrier", + "mst2_metadata_gc_enabled", + ] { + let (a, z) = function(&rendered, name); + txn.execute_unprepared(&rendered[a..z].replacen( + "CREATE FUNCTION", + "CREATE OR REPLACE FUNCTION", + 1, + )) + .await + .unwrap(); + } + txn.execute_unprepared(&format!("SET LOCAL search_path={c},pg_catalog,pg_temp")) + .await + .unwrap(); + let capture = CAPTURE.replace("\r\n", "\n"); + let (a, z) = function(&capture, "mst2_route_family_registration_guard"); + txn.execute_unprepared( + &capture[a..z] + .replacen("CREATE FUNCTION", "CREATE OR REPLACE FUNCTION", 1) + .replace("$CORE_SCHEMA$", &c) + .replace("$IMPLEMENTATION_SHA$", RETENTION_IMPLEMENTATION), + ) + .await + .unwrap(); + let old_shape: Vec = if legacy_shape { + txn.execute_unprepared(&format!("SET LOCAL search_path={q},pg_catalog,pg_temp")) + .await + .unwrap(); + txn.query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + include_str!("qualified_family_shape.sql") + .replace("q_oid", "$1::bigint::oid") + .replace("n_uuid", "$2::uuid") + .replace("s_uuid", "$3::text"), + [oid.into(), namespace.clone().into(), storage.clone().into()], + )) + .await + .unwrap() + .unwrap() + .try_get("", "fingerprint") + .unwrap() + } else { + txn.query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + format!( + "SELECT {c}.mst2_route_family_shape($1::bigint::oid,$2::uuid,$3) AS fingerprint" + ), + [oid.into(), namespace.clone().into(), storage.into()], + )) + .await + .unwrap() + .unwrap() + .try_get("", "fingerprint") + .unwrap() + }; + txn.execute_unprepared(&format!("SET LOCAL search_path={c},pg_catalog,pg_temp")) + .await + .unwrap(); + if !keep_q { + let tables = txn.query_all_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "SELECT relname FROM pg_catalog.pg_class WHERE relnamespace=$1::bigint::oid AND relkind='r'", + [oid.into()])).await.unwrap(); + for row in tables { + let name: String = row.try_get("", "relname").unwrap(); + let count: i64 = txn + .query_one_raw(Statement::from_string( + DbBackend::Postgres, + format!("SELECT count(*) FROM {q}.{}", identifier(&name)), + )) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap(); + assert_eq!( + count, + if [ + "mst2_metadata_storage_scope", + "mst2_metadata_family_identity", + "mst2_metadata_reader_issuance" + ] + .contains(&name.as_str()) + { + 1 + } else { + 0 + }, + "Q-less cc90 fixture must have no history to discard: {name}" + ); + } + let water: i64 = txn + .query_one_raw(Statement::from_string( + DbBackend::Postgres, + format!("SELECT high_water FROM {q}.mst2_metadata_reader_issuance"), + )) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap(); + assert_eq!(water, 0); + txn.execute_unprepared(&format!( + "ALTER TABLE {c}.mst2_metadata_namespace DISABLE TRIGGER mst2_route_immutable" + )) + .await + .unwrap(); + txn.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + format!("DELETE FROM {c}.mst2_metadata_namespace WHERE namespace_uuid=$1::uuid"), + [namespace.clone().into()], + )) + .await + .unwrap(); + txn.execute_unprepared(&format!("ALTER TABLE {c}.mst2_metadata_namespace ENABLE TRIGGER mst2_route_immutable; DROP SCHEMA {q} CASCADE")).await.unwrap(); + } + let authority = fingerprint(&txn, core_oid, 0, if keep_q { oid } else { 0 }).await; + let full = if keep_q { + Some(fingerprint(&txn, core_oid, oid, 0).await) + } else { + None + }; + txn.execute_unprepared(&format!("ALTER TABLE {c}.mst2_qualified_family_policy DISABLE TRIGGER mst2_route_family_policy_immutable")).await.unwrap(); + txn.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres, + format!("UPDATE {c}.mst2_qualified_family_policy SET implementation_fingerprint=$1,expected_shape=$2,authority_catalog=$3 WHERE singleton=1"), + [hex::decode(RETENTION_IMPLEMENTATION).unwrap().into(), old_shape.into(), authority.clone().into()])).await.unwrap(); + txn.execute_unprepared(&format!("ALTER TABLE {c}.mst2_qualified_family_policy ENABLE TRIGGER mst2_route_family_policy_immutable")).await.unwrap(); + if let Some(full) = &full { + txn.execute_unprepared(&format!( + "ALTER TABLE {q}.mst2_metadata_family_identity DISABLE TRIGGER mst2_00_family_barrier; + ALTER TABLE {q}.mst2_metadata_family_identity DISABLE TRIGGER mst2_metadata_identity_immutable; + ALTER TABLE {c}.mst2_metadata_namespace DISABLE TRIGGER mst2_00_route_statement_barrier; + ALTER TABLE {c}.mst2_metadata_namespace DISABLE TRIGGER mst2_route_immutable")).await.unwrap(); + txn.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres, + format!("UPDATE {q}.mst2_metadata_family_identity SET implementation_fingerprint=$1 WHERE singleton=1"), + [hex::decode(RETENTION_IMPLEMENTATION).unwrap().into()])).await.unwrap(); + txn.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres, + format!("UPDATE {c}.mst2_metadata_namespace SET implementation_fingerprint=$1,catalog_fingerprint=$2 WHERE namespace_uuid=$3::uuid"), + [hex::decode(RETENTION_IMPLEMENTATION).unwrap().into(), full.clone().into(), namespace.into()])).await.unwrap(); + txn.execute_unprepared(&format!( + "ALTER TABLE {q}.mst2_metadata_family_identity ENABLE TRIGGER mst2_00_family_barrier; + ALTER TABLE {q}.mst2_metadata_family_identity ENABLE TRIGGER mst2_metadata_identity_immutable; + ALTER TABLE {c}.mst2_metadata_namespace ENABLE TRIGGER mst2_00_route_statement_barrier; + ALTER TABLE {c}.mst2_metadata_namespace ENABLE TRIGGER mst2_route_immutable")).await.unwrap(); + assert_eq!(fingerprint(&txn, core_oid, oid, 0).await, *full); + } + assert_eq!( + fingerprint(&txn, core_oid, 0, if keep_q { oid } else { 0 }).await, + authority + ); + txn.commit().await.unwrap(); +} diff --git a/src/jupiter/storage/qualified_metadata_rooted.sql b/src/jupiter/storage/qualified_metadata_rooted.sql index e95655db..fdea989c 100644 --- a/src/jupiter/storage/qualified_metadata_rooted.sql +++ b/src/jupiter/storage/qualified_metadata_rooted.sql @@ -115,11 +115,11 @@ BEGIN OR NOT EXISTS(SELECT 1 FROM unnest(source_rows) s(value) WHERE value->>'page'=encode(root,'hex')) THEN RAISE EXCEPTION 'rooted plan has overlap or an unbound graph/source endpoint'; END IF; - SELECT coalesce(jsonb_object_agg(parent,children),'{}'::jsonb) INTO adjacency FROM ( + SELECT coalesce(jsonb_object_agg(grouped.parent,grouped.children),'{}'::jsonb) INTO adjacency FROM ( SELECT value->>'parent' AS parent,jsonb_agg(value->>'child') AS children FROM unnest(edge_rows) e(value) GROUP BY value->>'parent') grouped; IF EXISTS(WITH RECURSIVE reached(page) AS (SELECT encode(root,'hex') UNION - SELECT child FROM reached r CROSS JOIN LATERAL jsonb_array_elements_text(adjacency->r.page) children(child)) + SELECT children.child FROM reached r CROSS JOIN LATERAL jsonb_array_elements_text(adjacency->r.page) children(child)) SELECT 1 FROM (SELECT d.value FROM unnest(delta_rows) d(value) UNION ALL SELECT r.value FROM unnest(reuse_rows) r(value)) nodes WHERE NOT EXISTS(SELECT 1 FROM reached r WHERE r.page=nodes.value->>'page')) THEN RAISE EXCEPTION 'rooted plan includes members outside its bounded delta and boundary closure'; From b56c06b25565efb55309e30853cef479bd8cf406 Mon Sep 17 00:00:00 2001 From: Xiaoyang Han Date: Thu, 8 Oct 2026 12:53:53 +0800 Subject: [PATCH 48/50] perf(v3): cache compiled qualified implementation fingerprint (#102) Cache the immutable compiled qualified-family implementation fingerprint in a process-local OnceLock<[u8; 32]>, preserving the existing Vec return API. Keep all 15 source components, domain separator, length prefixes and digest bytes unchanged; every database catalog, shape, policy, identity and physical-stamp check still runs. Owned nightly formatting, offline source equivalence and candidate/patch checks pass. Native typecheck, Clippy, PostgreSQL tests and performance measurements remain pending. --- .../storage/qualified_metadata_family.rs | 104 ++++++++++-------- 1 file changed, 56 insertions(+), 48 deletions(-) diff --git a/src/jupiter/storage/qualified_metadata_family.rs b/src/jupiter/storage/qualified_metadata_family.rs index c44c8a94..13ffb440 100644 --- a/src/jupiter/storage/qualified_metadata_family.rs +++ b/src/jupiter/storage/qualified_metadata_family.rs @@ -1,5 +1,7 @@ //! Captured physical Q provisioning and the sealed rooted production repository. +use std::sync::OnceLock; + use mst2_codec::metapage::{HEADER_LEN, PAGE_MAX_BYTES}; use sea_orm::{ ConnectionTrait, DatabaseConnection, DatabaseTransaction, DbBackend, IsolationLevel, Statement, @@ -132,54 +134,60 @@ fn literal(value: &str) -> String { format!("'{}'", value.replace('\'', "''")) } pub(crate) fn implementation_fingerprint() -> Vec { - let mut hash = Sha256::new(); - hash.update(b"mega.mst2.rooted-qualified-implementation.v1\0"); - // Length-prefix every source component so source selection and proof - // revisions cannot change while retaining an accepted physical stamp. - for (name, bytes) in [ - ("family", FAMILY.as_bytes()), - ("family-ddl", FAMILY_SQL.as_bytes()), - ("canonical-proof-revision-1", CANONICAL_SQL.as_bytes()), - ("indexed-source-read-revision-1", SOURCE_READ_SQL.as_bytes()), - ("captured-source-revision-1", SOURCE_REVISION_SQL.as_bytes()), - ("rooted-collector-revision-1", GC_SQL.as_bytes()), - ( - "bounded-reader-lifecycle-revision-1", - READER_LIFECYCLE_SQL.as_bytes(), - ), - ("typed-certificate-revision-1", CERTIFICATES_SQL.as_bytes()), - ( - "source-attestation-and-anchor-revision-1", - ANCHORS_SQL.as_bytes(), - ), - ("rooted-plan-and-bindings-revision-1", ROOTED_SQL.as_bytes()), - ( - "rooted-session-and-reader-revision-1", - SERVING_SQL.as_bytes(), - ), - ( - "rooted-physical-route-revision-1", - include_bytes!("../migration/m20261008_000200_rooted_routes.sql").as_slice(), - ), - ( - "authority-catalog-selector", - include_bytes!("qualified_family_catalog.sql").as_slice(), - ), - ( - "normalized-family-shape", - include_bytes!("qualified_family_shape.sql").as_slice(), - ), - ( - "initial-core-registration", - include_bytes!("../migration/m20261008_000200_rooted_qualified_family.sql").as_slice(), - ), - ] { - hash.update((name.len() as u64).to_le_bytes()); - hash.update(name.as_bytes()); - hash.update((bytes.len() as u64).to_le_bytes()); - hash.update(bytes); - } - hash.finalize().to_vec() + static FINGERPRINT: OnceLock<[u8; 32]> = OnceLock::new(); + FINGERPRINT + .get_or_init(|| { + let mut hash = Sha256::new(); + hash.update(b"mega.mst2.rooted-qualified-implementation.v1\0"); + // Length-prefix every source component so source selection and proof + // revisions cannot change while retaining an accepted physical stamp. + for (name, bytes) in [ + ("family", FAMILY.as_bytes()), + ("family-ddl", FAMILY_SQL.as_bytes()), + ("canonical-proof-revision-1", CANONICAL_SQL.as_bytes()), + ("indexed-source-read-revision-1", SOURCE_READ_SQL.as_bytes()), + ("captured-source-revision-1", SOURCE_REVISION_SQL.as_bytes()), + ("rooted-collector-revision-1", GC_SQL.as_bytes()), + ( + "bounded-reader-lifecycle-revision-1", + READER_LIFECYCLE_SQL.as_bytes(), + ), + ("typed-certificate-revision-1", CERTIFICATES_SQL.as_bytes()), + ( + "source-attestation-and-anchor-revision-1", + ANCHORS_SQL.as_bytes(), + ), + ("rooted-plan-and-bindings-revision-1", ROOTED_SQL.as_bytes()), + ( + "rooted-session-and-reader-revision-1", + SERVING_SQL.as_bytes(), + ), + ( + "rooted-physical-route-revision-1", + include_bytes!("../migration/m20261008_000200_rooted_routes.sql").as_slice(), + ), + ( + "authority-catalog-selector", + include_bytes!("qualified_family_catalog.sql").as_slice(), + ), + ( + "normalized-family-shape", + include_bytes!("qualified_family_shape.sql").as_slice(), + ), + ( + "initial-core-registration", + include_bytes!("../migration/m20261008_000200_rooted_qualified_family.sql") + .as_slice(), + ), + ] { + hash.update((name.len() as u64).to_le_bytes()); + hash.update(name.as_bytes()); + hash.update((bytes.len() as u64).to_le_bytes()); + hash.update(bytes); + } + hash.finalize().into() + }) + .to_vec() } pub(crate) fn render_family( core_schema: &str, From e502ccfd91481b4071488f0fe2974f1f24178f6f Mon Sep 17 00:00:00 2001 From: Xiaoyang Han Date: Thu, 8 Oct 2026 13:12:27 +0800 Subject: [PATCH 49/50] fix(v3): encode canonical descriptors with the codec wire order (#103) Rooted PostgreSQL resolve failed its exact descriptor handoff because SQL encoded descriptor u16 fields in big endian while mst2-codec 0.3.1 writes little endian. Encode version, flags and UTF-8 byte length with the locked codec order. Add independent transactional migration600; freeze400/500 at authenticated87b so ledger replay cannot downgrade the repaired family. Authenticate unchanged core/Q catalogs, complete physical shape, scope and exact identities before every no-op or forward repair. Preserve existing OIDs, issuance, anchors, readers, source roots and canonical sessions. Add four real PostgreSQL regression sources for bytewise root/ASCII/256-byte/UTF-8 descriptors, authenticated87b upgrades and missing ledgers, Q-less provisioning, idempotence and descriptor tampering; verify compiled-family400/500 replay without a downgrade in the same strong fixture. Correct the complete20-name migration-order assertion while preserving alias DB behavior; historical900e replay includes600. Add a focused24-test native gate with exact inventory presence checks before the retained full repository suite. Formatting, historical-source equivalence, exact function replacement sets, candidate whitespace and bidirectional patch checks pass. Native compile, PostgreSQL regression execution and performance measurements for this candidate remain pending. --- .github/workflows/repository-gates.yml | 38 ++++ .../snapshot_descriptor_upgrade_tests.rs | 181 ++++++++++++++++++ .../router/snapshot_descriptor_wire_tests.rs | 118 ++++++++++++ .../snapshot_native_runtime_upgrade_tests.rs | 11 +- .../router/snapshot_rooted_metadata_tests.rs | 3 + ...0261008_000600_fix_mst2_descriptor_wire.rs | 21 ++ .../m20261008_000600_previous_descriptor.sql | 27 +++ src/jupiter/migration/mod.rs | 106 +++++----- .../qualified_native_runtime_upgrade.rs | 91 +++++++-- ...lified_metadata_reader_previous_fixture.rs | 50 ++++- .../storage/qualified_metadata_serving.sql | 6 +- 11 files changed, 560 insertions(+), 92 deletions(-) create mode 100644 src/api/router/snapshot_descriptor_upgrade_tests.rs create mode 100644 src/api/router/snapshot_descriptor_wire_tests.rs create mode 100644 src/jupiter/migration/m20261008_000600_fix_mst2_descriptor_wire.rs create mode 100644 src/jupiter/migration/m20261008_000600_previous_descriptor.sql diff --git a/.github/workflows/repository-gates.yml b/.github/workflows/repository-gates.yml index 313fc5a5..0e26b36f 100644 --- a/.github/workflows/repository-gates.yml +++ b/.github/workflows/repository-gates.yml @@ -66,6 +66,44 @@ jobs: docker compose -p mega2-it -f docker/docker-compose.test.yml --profile git \ up -d --build --wait postgres redis rustfs rustfs-init git-cli + - name: Verify necessary v3 publication and performance regressions + shell: bash + run: | + set -euo pipefail + source .env.test + export no_proxy="$NO_PROXY" + test_inventory="$(cargo test --lib -- --list)" + tests=( + api::router::snapshot_router::content::tests::rooted_metadata::descriptor_wire::rooted_postgres_descriptor_matches_codec_bytes_for_root_ascii_long_and_utf8_scopes + jupiter::migration::tests::import_repo_alias_rows_canonicalized + api::router::snapshot_router::content::tests::rooted_metadata::reader_retention::committed_metadata_requests_retire_owners_without_retiring_source_history + api::router::snapshot_router::content::tests::rooted_metadata::reader_retention::terminal_owner_survives_its_deferred_completion_and_prunes_in_a_later_transaction + api::router::snapshot_router::content::tests::rooted_metadata::reader_retention::reader_pruning_respects_the_64_owner_budget_with_a_committed_backlog + api::router::snapshot_router::content::tests::rooted_metadata::reader_retention::stale_actual_reader_cannot_read_or_finish_a_reissued_uuid + api::router::snapshot_router::content::tests::rooted_metadata::reader_retention::rolled_back_reader_issuance_is_unpublished_and_extreme_issuance_fails_closed + api::router::snapshot_router::content::tests::rooted_metadata::reader_retention::upgrade::current_reader_retention_migration_verifies_fresh_family_without_rewriting_history + api::router::snapshot_router::content::tests::rooted_metadata::reader_retention::upgrade::captured_previous_q_upgrade_preserves_old_sid_source_proofs_and_active_legacy_reader + api::router::snapshot_router::content::tests::rooted_metadata::reader_retention::upgrade::captured_previous_core_without_q_upgrades_before_new_provisioning + api::router::snapshot_router::content::tests::rooted_metadata::reader_retention::upgrade::captured_previous_tampered_q_rejects_upgrade_without_partial_issuance_schema + api::router::snapshot_router::content::tests::rooted_metadata::reader_retention::upgrade::native_runtime::captured_cc90_upgrade_preserves_nonzero_reader_issuance_all_owners_oids_and_old_sid + api::router::snapshot_router::content::tests::rooted_metadata::reader_retention::upgrade::native_runtime::captured_cc90_interrupted_before_reader_migration_resumes_without_reader_ddl + api::router::snapshot_router::content::tests::rooted_metadata::reader_retention::upgrade::native_runtime::captured_cc90_qless_canonical_policy_upgrades_before_new_provisioning + api::router::snapshot_router::content::tests::rooted_metadata::reader_retention::upgrade::native_runtime::captured_cc90_qless_exact_legacy_deparse_policy_upgrades_before_new_provisioning + api::router::snapshot_router::content::tests::rooted_metadata::reader_retention::upgrade::native_runtime::captured_cc90_tampered_decoder_rejects_upgrade_without_rewriting_owners_or_policy + api::router::snapshot_router::content::tests::rooted_metadata::reader_retention::upgrade::native_runtime::captured_cc90_qless_arbitrary_policy_shape_is_never_resigned + api::router::snapshot_router::content::tests::mst2_rooted_streamed_fact_persists_one_source_pass_and_reuses_warm_alias_facts + jupiter::storage::qualified_metadata_family::tests::canonical_tests::actual_rooted_cold_and_zero_delta_reuse_keep_one_canonical_graph + jupiter::storage::qualified_metadata_family::tests::canonical_tests::changed_ancestor_installs_only_delta_and_accepts_independently_attested_source_aliases + api::router::snapshot_router::content::tests::rooted_metadata::rooted_wide_directory_windows_and_lookup_survive_rebuild_with_valid_proofs + api::router::snapshot_router::content::tests::rooted_metadata::reader_retention::upgrade::native_runtime::descriptor_wire::captured_87b_descriptor_upgrade_preserves_owners_oids_and_replays_missing_ledgers + api::router::snapshot_router::content::tests::rooted_metadata::reader_retention::upgrade::native_runtime::descriptor_wire::captured_87b_qless_descriptor_upgrade_replays_missing_ledgers_before_provisioning + api::router::snapshot_router::content::tests::rooted_metadata::reader_retention::upgrade::native_runtime::descriptor_wire::captured_87b_tampered_descriptor_rejects_without_resigning_history + ) + for test_name in "${tests[@]}"; do + grep -Fqx "$test_name: test" <<< "$test_inventory" + done + cargo test --lib -- --exact --nocapture "${tests[@]}" + - name: Run repository tests with test environment shell: bash run: | diff --git a/src/api/router/snapshot_descriptor_upgrade_tests.rs b/src/api/router/snapshot_descriptor_upgrade_tests.rs new file mode 100644 index 00000000..9cb37b3b --- /dev/null +++ b/src/api/router/snapshot_descriptor_upgrade_tests.rs @@ -0,0 +1,181 @@ +use super::*; +use crate::jupiter::storage::qualified_metadata_family::reader_previous_fixture::restore_native_runtime; + +async fn remove_upgrade_ledgers(core: &sea_orm::DatabaseConnection) { + core.execute_unprepared( + "DELETE FROM seaql_migrations WHERE version IN ( + 'm20261008_000400_add_mst2_reader_retention', + 'm20261008_000500_fix_mst2_native_runtime', + 'm20261008_000600_fix_mst2_descriptor_wire')", + ) + .await + .unwrap(); +} + +async fn assert_upgrade_ledgers(core: &sea_orm::DatabaseConnection) { + let count: i64 = core + .query_one_raw(Statement::from_string( + DbBackend::Postgres, + "SELECT count(*) FROM seaql_migrations WHERE version IN ( + 'm20261008_000400_add_mst2_reader_retention', + 'm20261008_000500_fix_mst2_native_runtime', + 'm20261008_000600_fix_mst2_descriptor_wire')", + )) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap(); + assert_eq!(count, 3); +} + +async fn descriptor_matches_stored(fixture: &Fixture) -> i64 { + q_count( + fixture, + "SELECT count(*) FROM {q}.mst2_qualified_session_incarnation session + WHERE session.state='READY' AND session.canonical_descriptor= + {q}.mst2_metadata_descriptor(session.prepare_id,session.instance_id, + session.commit_oid,session.root_tree_oid)", + ) + .await +} + +#[tokio::test] +async fn captured_87b_descriptor_upgrade_preserves_owners_oids_and_replays_missing_ledgers() { + let fixture = Fixture::new_with_pg_config(true).await; + seed_live_and_terminal(&fixture).await; + let schema = q_schema(&fixture).await; + let mono = fixture.state.storage.mono_storage(); + let core = mono.get_connection(); + for missing_ledgers in [false, true] { + let rows = permanent_rows(&fixture).await; + let owners = retained_owners(&fixture).await; + restore_native_runtime(core, &schema, true).await; + assert_eq!(descriptor_matches_stored(&fixture).await, 0); + assert_eq!(permanent_rows(&fixture).await, rows); + assert_eq!(retained_owners(&fixture).await, owners); + if missing_ledgers { + remove_upgrade_ledgers(core).await; + apply_migrations(core, false).await.unwrap(); + assert_upgrade_ledgers(core).await; + } else { + test_upgrade_native_runtime(core).await.unwrap(); + } + assert_eq!(descriptor_matches_stored(&fixture).await, 1); + assert_eq!(permanent_rows(&fixture).await, rows); + assert_eq!(retained_owners(&fixture).await, owners); + let stamp = policy(core).await; + if missing_ledgers { + apply_migrations(core, false).await.unwrap(); + } else { + test_upgrade_native_runtime(core).await.unwrap(); + } + assert_eq!(policy(core).await, stamp); + assert_eq!(permanent_rows(&fixture).await, rows); + assert_eq!(retained_owners(&fixture).await, owners); + assert_old_sid_works(&fixture).await; + } + let rows = permanent_rows(&fixture).await; + let owners = retained_owners(&fixture).await; + let stamp = policy(core).await; + core.execute_unprepared( + "DELETE FROM seaql_migrations WHERE version IN ( + 'm20261008_000400_add_mst2_reader_retention', + 'm20261008_000500_fix_mst2_native_runtime')", + ) + .await + .unwrap(); + apply_migrations(core, false).await.unwrap(); + assert_upgrade_ledgers(core).await; + assert_eq!(descriptor_matches_stored(&fixture).await, 1); + assert_eq!(policy(core).await, stamp); + assert_eq!(permanent_rows(&fixture).await, rows); + assert_eq!(retained_owners(&fixture).await, owners); + assert_old_sid_works(&fixture).await; +} + +#[tokio::test] +async fn captured_87b_qless_descriptor_upgrade_replays_missing_ledgers_before_provisioning() { + let temp = tempfile::tempdir().unwrap(); + let (config, _schema) = test_db_config(temp.path()).await; + let core = crate::jupiter::storage::init::database_connection(&config) + .await + .unwrap(); + let schema: String = core + .query_one_raw(Statement::from_string( + DbBackend::Postgres, + "SELECT metadata_schema FROM mst2_metadata_namespace WHERE graph_domain='qualified-v1'", + )) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap(); + restore_native_runtime(&core, &schema, false).await; + remove_upgrade_ledgers(&core).await; + apply_migrations(&core, false).await.unwrap(); + assert_upgrade_ledgers(&core).await; + let stamp = policy(&core).await; + apply_migrations(&core, false).await.unwrap(); + test_upgrade_native_runtime(&core).await.unwrap(); + assert_eq!(policy(&core).await, stamp); + let namespace = provision_or_verify_rooted_qualified_family(&core) + .await + .unwrap(); + let current_schema: String = core + .query_one_raw(Statement::from_string( + DbBackend::Postgres, + "SELECT metadata_schema FROM mst2_metadata_namespace WHERE graph_domain='qualified-v1'", + )) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap(); + assert_ne!(current_schema, schema); + assert_eq!( + provision_or_verify_rooted_qualified_family(&core) + .await + .unwrap(), + namespace + ); + let high_water: i64 = core + .query_one_raw(Statement::from_string( + DbBackend::Postgres, + format!( + "SELECT high_water FROM \"{}\".mst2_metadata_reader_issuance", + current_schema.replace('"', "\"\"") + ), + )) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap(); + assert_eq!(high_water, 0); +} + +#[tokio::test] +async fn captured_87b_tampered_descriptor_rejects_without_resigning_history() { + let fixture = Fixture::new_with_pg_config(true).await; + seed_live_and_terminal(&fixture).await; + let schema = q_schema(&fixture).await; + let mono = fixture.state.storage.mono_storage(); + let core = mono.get_connection(); + restore_native_runtime(core, &schema, true).await; + let rows = permanent_rows(&fixture).await; + let owners = retained_owners(&fixture).await; + let before = policy(core).await; + core.execute_unprepared(&format!( + "CREATE OR REPLACE FUNCTION \"{}\".mst2_metadata_descriptor( + pid text,instance text,commit_id text,tree_id text) RETURNS bytea + LANGUAGE sql AS 'SELECT decode(''0000'',''hex'')'", + schema.replace('"', "\"\"") + )) + .await + .unwrap(); + assert!(test_upgrade_native_runtime(core).await.is_err()); + assert_eq!(policy(core).await, before); + assert_eq!(permanent_rows(&fixture).await, rows); + assert_eq!(retained_owners(&fixture).await, owners); +} diff --git a/src/api/router/snapshot_descriptor_wire_tests.rs b/src/api/router/snapshot_descriptor_wire_tests.rs new file mode 100644 index 00000000..f5d5536e --- /dev/null +++ b/src/api/router/snapshot_descriptor_wire_tests.rs @@ -0,0 +1,118 @@ +use mst2_codec::descriptor::ServingDescriptor; + +use super::*; + +async fn assert_postgres_descriptor_matches_codec(fixture: &Fixture, scope: &str) { + let resolved = success_json( + fixture + .app + .clone() + .oneshot( + Request::builder() + .method("POST") + .uri("/api/v2/snapshots/resolve") + .header("authorization", format!("Bearer {TOKEN}")) + .header("content-type", "application/json") + .body(Body::from( + json!({"target":{"kind":"latest"},"scope":scope}).to_string(), + )) + .unwrap(), + ) + .await + .unwrap(), + ) + .await; + let snapshot = resolved["descriptor"]["snapshot_id"].as_str().unwrap(); + let lease = resolved["lease_id"].as_str().unwrap(); + assert_eq!( + fixture + .state + .storage + .snapshot_metadata_family(lease, true) + .await + .unwrap(), + Some(SnapshotMetadataFamily::Rooted) + ); + let context = fixture + .state + .storage + .snapshot_context(snapshot, lease) + .await + .unwrap(); + let descriptor = &context.built.descriptor; + assert_eq!(descriptor.scope, scope); + let codec_bytes = descriptor.encode().unwrap(); + let schema = q_schema(fixture).await; + let quoted = format!("\"{}\"", schema.replace('"', "\"\"")); + let row = fixture + .state + .storage + .mono_storage() + .get_connection() + .query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + format!( + "SELECT {q}.mst2_metadata_descriptor(session.prepare_id,session.instance_id, + session.commit_oid,session.root_tree_oid) AS descriptor, + session.canonical_descriptor,session.metadata_root,session.snapshot_id + FROM {q}.mst2_qualified_session_incarnation session + JOIN {q}.mst2_qualified_lease_binding lease USING(snapshot_id,session_incarnation) + WHERE session.snapshot_id=$1 AND lease.lease_id=$2 AND session.state='READY' + AND lease.state='ACTIVE'", + q = quoted, + ), + [snapshot.into(), lease.into()], + )) + .await + .unwrap() + .unwrap(); + let database_bytes: Vec = row.try_get("", "descriptor").unwrap(); + assert_eq!(database_bytes, codec_bytes, "scope {scope:?}"); + assert_eq!( + row.try_get::>("", "canonical_descriptor").unwrap(), + codec_bytes + ); + assert_eq!( + ServingDescriptor::decode(&database_bytes).unwrap(), + *descriptor + ); + assert_eq!( + &database_bytes[56..58], + &u16::try_from(scope.len()).unwrap().to_le_bytes() + ); + assert_eq!(&database_bytes[58..58 + scope.len()], scope.as_bytes()); + assert_eq!( + row.try_get::>("", "metadata_root").unwrap(), + descriptor.metadata_root.to_vec() + ); + assert_eq!( + resolved["descriptor"]["metadata_root"], + format!("sha256:{}", hex::encode(descriptor.metadata_root)) + ); + let codec_snapshot = format!("sha256:{}", hex::encode(descriptor.snapshot_id().unwrap())); + assert_eq!(snapshot, codec_snapshot); + assert_eq!(context.built.snapshot_id, codec_snapshot); + assert_eq!( + row.try_get::("", "snapshot_id").unwrap(), + codec_snapshot + ); +} + +#[tokio::test] +async fn rooted_postgres_descriptor_matches_codec_bytes_for_root_ascii_long_and_utf8_scopes() { + let component = "a".repeat(247); + let fixture = Fixture::new_with_pg_config_directories_and_objects( + true, + 0, + &[(format!("{component}/é/file"), b"descriptor wire".to_vec())], + ) + .await; + let long_scope = format!("/project/{component}"); + let utf8_scope = format!("{long_scope}/é"); + assert_eq!(long_scope.len(), 256); + assert_eq!(utf8_scope.len(), 259); + assert_ne!(utf8_scope.len(), utf8_scope.chars().count()); + for scope in ["/", "/project", long_scope.as_str(), utf8_scope.as_str()] { + assert_postgres_descriptor_matches_codec(&fixture, scope).await; + } +} diff --git a/src/api/router/snapshot_native_runtime_upgrade_tests.rs b/src/api/router/snapshot_native_runtime_upgrade_tests.rs index 79a76175..2fdee7d6 100644 --- a/src/api/router/snapshot_native_runtime_upgrade_tests.rs +++ b/src/api/router/snapshot_native_runtime_upgrade_tests.rs @@ -4,6 +4,9 @@ use crate::jupiter::{ storage::qualified_metadata_family::reader_previous_fixture::restore_retention, }; +#[path = "snapshot_descriptor_upgrade_tests.rs"] +mod descriptor_wire; + async fn retained_owners(fixture: &Fixture) -> Value { let txn = transaction(fixture).await; let value = txn.query_one_raw(Statement::from_string(DbBackend::Postgres, @@ -135,10 +138,11 @@ async fn captured_cc90_interrupted_before_reader_migration_resumes_without_reade let owners = retained_owners(&fixture).await; core.execute_raw(Statement::from_sql_and_values( DbBackend::Postgres, - "DELETE FROM seaql_migrations WHERE version IN ($1,$2)", + "DELETE FROM seaql_migrations WHERE version IN ($1,$2,$3)", [ "m20261008_000400_add_mst2_reader_retention".into(), "m20261008_000500_fix_mst2_native_runtime".into(), + "m20261008_000600_fix_mst2_descriptor_wire".into(), ], )) .await @@ -148,9 +152,10 @@ async fn captured_cc90_interrupted_before_reader_migration_resumes_without_reade assert_eq!(retained_owners(&fixture).await, owners); let count: i64 = core.query_one_raw(Statement::from_string(DbBackend::Postgres, "SELECT count(*) FROM seaql_migrations WHERE version IN - ('m20261008_000400_add_mst2_reader_retention','m20261008_000500_fix_mst2_native_runtime')")) + ('m20261008_000400_add_mst2_reader_retention','m20261008_000500_fix_mst2_native_runtime', + 'm20261008_000600_fix_mst2_descriptor_wire')")) .await.unwrap().unwrap().try_get_by_index(0).unwrap(); - assert_eq!(count, 2); + assert_eq!(count, 3); assert_old_sid_works(&fixture).await; } diff --git a/src/api/router/snapshot_rooted_metadata_tests.rs b/src/api/router/snapshot_rooted_metadata_tests.rs index 46316703..33179e3d 100644 --- a/src/api/router/snapshot_rooted_metadata_tests.rs +++ b/src/api/router/snapshot_rooted_metadata_tests.rs @@ -10,6 +10,9 @@ use crate::jupiter::storage::qualified_metadata_family::{ #[path = "snapshot_reader_retention_tests.rs"] mod reader_retention; +#[path = "snapshot_descriptor_wire_tests.rs"] +mod descriptor_wire; + async fn q_schema(fixture: &Fixture) -> String { fixture .state diff --git a/src/jupiter/migration/m20261008_000600_fix_mst2_descriptor_wire.rs b/src/jupiter/migration/m20261008_000600_fix_mst2_descriptor_wire.rs new file mode 100644 index 00000000..7efcf97f --- /dev/null +++ b/src/jupiter/migration/m20261008_000600_fix_mst2_descriptor_wire.rs @@ -0,0 +1,21 @@ +//! Repair canonical descriptor integer encoding without rewriting ownership. + +use sea_orm_migration::prelude::*; + +#[derive(DeriveMigrationName)] +pub struct Migration; + +#[async_trait::async_trait] +impl MigrationTrait for Migration { + async fn up(&self, manager: &SchemaManager) -> Result<(), DbErr> { + super::qualified_native_runtime_upgrade::upgrade_descriptor(manager).await + } + + async fn down(&self, _manager: &SchemaManager) -> Result<(), DbErr> { + Ok(()) + } + + fn use_transaction(&self) -> Option { + Some(true) + } +} diff --git a/src/jupiter/migration/m20261008_000600_previous_descriptor.sql b/src/jupiter/migration/m20261008_000600_previous_descriptor.sql new file mode 100644 index 00000000..dab92001 --- /dev/null +++ b/src/jupiter/migration/m20261008_000600_previous_descriptor.sql @@ -0,0 +1,27 @@ +CREATE FUNCTION mst2_metadata_descriptor(pid text,instance text,commit_id text,tree_id text) +RETURNS bytea LANGUAGE plpgsql VOLATILE STRICT SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE p record; scope_bytes bytea; view_digest bytea; instance_bytes bytea; +BEGIN + SELECT tagged_root_tree_oid,scope,metadata_root INTO p FROM mst2_metadata_prepare WHERE prepare_id=pid AND plan_kind='ROOTED' + AND state='COMMITTED' AND graph_domain='qualified-v1' AND mst2_metadata_scope_matches(primary_scope); + IF NOT FOUND OR instance IS DISTINCT FROM (instance::uuid)::text + OR split_part(p.tagged_root_tree_oid,':',2) IS DISTINCT FROM tree_id + OR NOT EXISTS(SELECT 1 FROM $CORE_SCHEMA$.mega_commit c WHERE c.commit_id=$3 AND c.tree=$4) + OR octet_length(commit_id)<>octet_length(tree_id) + OR commit_id !~ '^([0-9a-f]{40}|[0-9a-f]{64})$' THEN + RAISE EXCEPTION 'qualified descriptor has no exact canonical instance and fixed source'; + END IF; + PERFORM mst2_metadata_native_profile(pid); + scope_bytes:=convert_to(p.scope,'UTF8'); + IF p.scope !~ '^/' OR octet_length(scope_bytes)>4096 OR p.scope<>'/' AND + (p.scope ~ '/$|//|/(\.|\.\.)(/|$)' OR cardinality(string_to_array(substring(p.scope FROM 2),'/'))>256) + OR EXISTS(SELECT 1 FROM unnest(string_to_array(substring(p.scope FROM 2),'/')) component + WHERE octet_length(component)>255) THEN + RAISE EXCEPTION 'qualified descriptor scope is not canonical'; + END IF; + instance_bytes:=decode(replace(instance,'-',''),'hex'); + view_digest:=sha256(convert_to('mega.mst2.namespaceview','UTF8')||decode('00','hex')||convert_to(commit_id,'UTF8')); + RETURN convert_to('MSD2','UTF8')||decode('00020001','hex')||instance_bytes||view_digest + ||decode(lpad(to_hex(octet_length(scope_bytes)),4,'0'),'hex')||scope_bytes + ||decode('0001000100000000','hex')||p.metadata_root; +END $$; diff --git a/src/jupiter/migration/mod.rs b/src/jupiter/migration/mod.rs index 09eaa04d..e3a737a5 100644 --- a/src/jupiter/migration/mod.rs +++ b/src/jupiter/migration/mod.rs @@ -156,6 +156,7 @@ mod m20261008_000200_add_mst2_rooted_qualified_family; mod m20261008_000300_add_mst2_chunk_map_retention; mod m20261008_000400_add_mst2_reader_retention; mod m20261008_000500_fix_mst2_native_runtime; +mod m20261008_000600_fix_mst2_descriptor_wire; pub(crate) mod qualified_native_runtime_upgrade; mod runner; pub use m20260905_000100_add_push_queue::ensure_queue_control_seed; @@ -169,9 +170,15 @@ pub(crate) async fn test_upgrade_reader_retention( use sea_orm_migration::{MigrationTrait, SchemaManager}; let txn = connection.begin().await?; - let result = m20261008_000400_add_mst2_reader_retention::Migration - .up(&SchemaManager::new(&txn)) - .await; + let result = async { + m20261008_000400_add_mst2_reader_retention::Migration + .up(&SchemaManager::new(&txn)) + .await?; + m20261008_000600_fix_mst2_descriptor_wire::Migration + .up(&SchemaManager::new(&txn)) + .await + } + .await; match result { Ok(()) => txn.commit().await, Err(error) => { @@ -189,9 +196,15 @@ pub(crate) async fn test_upgrade_native_runtime( use sea_orm_migration::{MigrationTrait, SchemaManager}; let txn = connection.begin().await?; - let result = m20261008_000500_fix_mst2_native_runtime::Migration - .up(&SchemaManager::new(&txn)) - .await; + let result = async { + m20261008_000500_fix_mst2_native_runtime::Migration + .up(&SchemaManager::new(&txn)) + .await?; + m20261008_000600_fix_mst2_descriptor_wire::Migration + .up(&SchemaManager::new(&txn)) + .await + } + .await; match result { Ok(()) => txn.commit().await, Err(error) => { @@ -341,6 +354,7 @@ impl MigratorTrait for Migrator { Box::new(m20261008_000300_add_mst2_chunk_map_retention::Migration), Box::new(m20261008_000400_add_mst2_reader_retention::Migration), Box::new(m20261008_000500_fix_mst2_native_runtime::Migration), + Box::new(m20261008_000600_fix_mst2_descriptor_wire::Migration), ] } } @@ -1250,61 +1264,33 @@ mod tests { #[tokio::test] async fn import_repo_alias_rows_canonicalized() { let names = migration_names(); + let expected = [ + "m20260923_000200_canonicalize_import_repo_paths", + "m20260925_000100_media_paging", + "m20261005_000100_add_mst2_publication_request_digest", + "m20261005_000200_add_mst2_native_head", + "m20261005_000100_add_mst2_retention_durability", + "m20261005_000200_harden_mst2_retention_graph", + "m20261005_000300_add_mst2_metadata_install", + VIEW_MIGRATION_NAME, + "m20261007_000100_add_mst2_snapshot_sessions", + "m20261007_000200_add_mst2_metadata_generations", + "m20261007_000300_add_mst2_metadata_lifetime_history", + "m20261007_000400_add_mst2_qualified_metadata_gc", + "m20261007_000500_add_mst2_install_capability", + "m20261007_000600_add_mst2_storage_routes", + "m20261008_000100_add_mst2_chunk_maps", + "m20261008_000200_add_mst2_rooted_qualified_family", + "m20261008_000300_add_mst2_chunk_map_retention", + "m20261008_000400_add_mst2_reader_retention", + "m20261008_000500_fix_mst2_native_runtime", + "m20261008_000600_fix_mst2_descriptor_wire", + ] + .map(str::to_owned); assert_eq!( - &names[names.len() - 18..names.len() - 9], - &[ - "m20260923_000200_canonicalize_import_repo_paths".to_string(), - "m20260925_000100_media_paging".to_string(), - "m20261005_000100_add_mst2_publication_request_digest".to_string(), - "m20261005_000200_add_mst2_native_head".to_string(), - "m20261005_000100_add_mst2_retention_durability".to_string(), - "m20261005_000200_harden_mst2_retention_graph".to_string(), - "m20261005_000300_add_mst2_metadata_install".to_string(), - VIEW_MIGRATION_NAME.to_string(), - "m20261007_000100_add_mst2_snapshot_sessions".to_string(), - ], - "native retention, metadata installation and view tables follow media paging" - ); - assert_eq!( - &names[names.len() - 9], - "m20261007_000200_add_mst2_metadata_generations" - ); - assert_eq!( - &names[names.len() - 8], - "m20261007_000300_add_mst2_metadata_lifetime_history" - ); - assert_eq!( - &names[names.len() - 7], - "m20261007_000400_add_mst2_qualified_metadata_gc" - ); - assert_eq!( - &names[names.len() - 6], - "m20261007_000500_add_mst2_install_capability" - ); - assert_eq!( - &names[names.len() - 5], - "m20261007_000600_add_mst2_storage_routes" - ); - assert_eq!( - &names[names.len() - 5], - "m20261008_000100_add_mst2_chunk_maps" - ); - assert_eq!( - &names[names.len() - 4], - "m20261008_000200_add_mst2_rooted_qualified_family" - ); - - assert_eq!( - &names[names.len() - 3], - "m20261008_000300_add_mst2_chunk_map_retention" - ); - assert_eq!( - &names[names.len() - 2], - "m20261008_000400_add_mst2_reader_retention" - ); - assert_eq!( - names.last().unwrap(), - "m20261008_000500_fix_mst2_native_runtime" + &names[names.len() - expected.len()..], + expected.as_slice(), + "the complete native migration suffix follows media paging in registered order" ); let db = alias_db().await; insert_repo(&db, 1, "/third-party//a").await; diff --git a/src/jupiter/migration/qualified_native_runtime_upgrade.rs b/src/jupiter/migration/qualified_native_runtime_upgrade.rs index 0a7d4b94..85cd256b 100644 --- a/src/jupiter/migration/qualified_native_runtime_upgrade.rs +++ b/src/jupiter/migration/qualified_native_runtime_upgrade.rs @@ -13,8 +13,11 @@ pub(crate) const PREVIOUS_IMPLEMENTATION: &str = pub(crate) const RETENTION_IMPLEMENTATION: &str = "900e7a340e7365c9f31995e070edbfe2076b870ec09fe89e4dc43e56f26b91ab"; +pub(crate) const NATIVE_RUNTIME_IMPLEMENTATION: &str = + "87b104e6b5593b1d6e5904a4e4be94e6ba0814dbb2c25dd8b38a0ea950d42307"; const PREVIOUS_DECODER: &str = include_str!("m20261008_000500_previous_rooted_decoder.sql"); const PREVIOUS_READERS: &str = include_str!("m20261008_000500_previous_reader_family.sql"); +const PREVIOUS_DESCRIPTOR: &str = include_str!("m20261008_000600_previous_descriptor.sql"); fn rejected(message: &str) -> DbErr { DbErr::Custom(message.into()) @@ -157,21 +160,33 @@ pub(crate) fn render_prior_family( storage: &str, implementation: &str, ) -> Result { - if ![PREVIOUS_IMPLEMENTATION, RETENTION_IMPLEMENTATION].contains(&implementation) { + if ![ + PREVIOUS_IMPLEMENTATION, + RETENTION_IMPLEMENTATION, + NATIVE_RUNTIME_IMPLEMENTATION, + ] + .contains(&implementation) + { return Err(rejected("native runtime template version is unsupported")); } let decoder = PREVIOUS_DECODER.replace("\r\n", "\n"); let readers = PREVIOUS_READERS.replace("\r\n", "\n"); + let descriptor = PREVIOUS_DESCRIPTOR.replace("\r\n", "\n"); if hex::encode(Sha256::digest(decoder.as_bytes())) != "f14d024cfb821f9d6b7348ab9b69f48c09862ff7dd3fbb2b360b253f949e35a9" || hex::encode(Sha256::digest(readers.as_bytes())) != "ec9b422440da111e5e047e1a4c85a28a33053cf0e320a56ea944ebefeaa8ea37" + || hex::encode(Sha256::digest(descriptor.as_bytes())) + != "c833666fcb1b3c9fd0bf186e5ad732f5e8b9b9dbf4360ce8d93a0d91d17d5c1a" { return Err(rejected("native runtime historical source capture changed")); } let mut source = render_family(core, core_oid, q, q_oid, namespace, storage).replace("\r\n", "\n"); - replace_function(&mut source, "mst2_metadata_decode_rooted_plan", &decoder)?; + replace_function(&mut source, "mst2_metadata_descriptor", &descriptor)?; + if implementation != NATIVE_RUNTIME_IMPLEMENTATION { + replace_function(&mut source, "mst2_metadata_decode_rooted_plan", &decoder)?; + } if implementation == PREVIOUS_IMPLEMENTATION { let start = source .find("CREATE TABLE mst2_metadata_reader_operation (") @@ -304,6 +319,21 @@ struct Family { pub(super) async fn upgrade( manager: &SchemaManager<'_>, allow_previous_readers: bool, +) -> Result<(), DbErr> { + upgrade_to(manager, allow_previous_readers, false).await +} + +pub(super) async fn upgrade_descriptor(manager: &SchemaManager<'_>) -> Result<(), DbErr> { + // Finish an interrupted reader/runtime upgrade before the independent + // descriptor repair, without replaying reader DDL on an admitted family. + upgrade(manager, true).await?; + upgrade_to(manager, false, true).await +} + +async fn upgrade_to( + manager: &SchemaManager<'_>, + allow_previous_readers: bool, + repair_descriptor: bool, ) -> Result<(), DbErr> { if manager.get_database_backend() != DbBackend::Postgres { return Err(rejected( @@ -368,15 +398,24 @@ pub(super) async fn upgrade( format!("SELECT implementation_fingerprint,expected_shape,authority_catalog FROM {c}.mst2_qualified_family_policy WHERE singleton=1"))) .await?.ok_or_else(|| rejected("reader migration trusted policy is missing"))?; let implementation: Vec = policy.try_get("", "implementation_fingerprint")?; - let current = implementation_fingerprint(); + let compiled = implementation_fingerprint(); + let runtime = + hex::decode(NATIVE_RUNTIME_IMPLEMENTATION).map_err(|error| rejected(&error.to_string()))?; + let current = if repair_descriptor { + compiled.clone() + } else { + runtime.clone() + }; let previous = hex::decode(PREVIOUS_IMPLEMENTATION).map_err(|error| rejected(&error.to_string()))?; let retention = hex::decode(RETENTION_IMPLEMENTATION).map_err(|error| rejected(&error.to_string()))?; - if implementation != current - && implementation != retention - && !(allow_previous_readers && implementation == previous) - { + let supported = implementation == compiled + || implementation == runtime + || !repair_descriptor + && (implementation == retention + || allow_previous_readers && implementation == previous); + if !supported { return Err(rejected( "reader migration refuses an unsupported prior implementation", )); @@ -468,7 +507,7 @@ pub(super) async fn upgrade( "native runtime migration prior policy shape is not trusted", )); } - if implementation == current { + if implementation == compiled || implementation == current { return Ok(()); } @@ -490,16 +529,34 @@ pub(super) async fn upgrade( if let Some(family) = &family { let q = identifier(&family.schema); connection.execute_unprepared(&format!("LOCK TABLE {q}.mst2_metadata_reader_operation,{q}.mst2_metadata_root_anchor IN ACCESS EXCLUSIVE MODE")).await?; - let rendered = render_family( - &core, - core_oid, - &family.schema, - family.oid, - &family.namespace, - &family.storage, - ); + let rendered = if repair_descriptor { + render_family( + &core, + core_oid, + &family.schema, + family.oid, + &family.namespace, + &family.storage, + ) + } else { + render_prior_family( + &core, + core_oid, + &family.schema, + family.oid, + &family.namespace, + &family.storage, + NATIVE_RUNTIME_IMPLEMENTATION, + )? + }; let mut functions = String::new(); - let replacements: &[&str] = if implementation == previous { + let replacements: &[&str] = if repair_descriptor { + &[ + "mst2_metadata_descriptor", + "mst2_metadata_dml_barrier", + "mst2_metadata_gc_enabled", + ] + } else if implementation == previous { &[ "mst2_metadata_decode_rooted_plan", "mst2_metadata_dml_barrier", diff --git a/src/jupiter/storage/qualified_metadata_reader_previous_fixture.rs b/src/jupiter/storage/qualified_metadata_reader_previous_fixture.rs index ba120126..a7975c05 100644 --- a/src/jupiter/storage/qualified_metadata_reader_previous_fixture.rs +++ b/src/jupiter/storage/qualified_metadata_reader_previous_fixture.rs @@ -259,9 +259,40 @@ pub(crate) async fn restore_retention( keep_q: bool, legacy_shape: bool, ) { - use crate::jupiter::migration::qualified_native_runtime_upgrade::{ - RETENTION_IMPLEMENTATION, render_prior_family, - }; + restore_reader_family( + core, + q_schema, + keep_q, + legacy_shape, + crate::jupiter::migration::qualified_native_runtime_upgrade::RETENTION_IMPLEMENTATION, + ) + .await; +} + +/// Restore the exact 87b runtime family without rebuilding any owned history. +pub(crate) async fn restore_native_runtime( + core: &DatabaseConnection, + q_schema: &str, + keep_q: bool, +) { + restore_reader_family( + core, + q_schema, + keep_q, + false, + crate::jupiter::migration::qualified_native_runtime_upgrade::NATIVE_RUNTIME_IMPLEMENTATION, + ) + .await; +} + +async fn restore_reader_family( + core: &DatabaseConnection, + q_schema: &str, + keep_q: bool, + legacy_shape: bool, + implementation: &str, +) { + use crate::jupiter::migration::qualified_native_runtime_upgrade::render_prior_family; assert!(!legacy_shape || !keep_q); let txn = core.begin().await.unwrap(); @@ -290,7 +321,7 @@ pub(crate) async fn restore_retention( oid, &namespace, &storage, - RETENTION_IMPLEMENTATION, + implementation, ) .unwrap(); txn.execute_unprepared(&format!("SET LOCAL search_path={q},pg_catalog,pg_temp")) @@ -298,6 +329,7 @@ pub(crate) async fn restore_retention( .unwrap(); for name in [ "mst2_metadata_decode_rooted_plan", + "mst2_metadata_descriptor", "mst2_metadata_dml_barrier", "mst2_metadata_gc_enabled", ] { @@ -319,7 +351,7 @@ pub(crate) async fn restore_retention( &capture[a..z] .replacen("CREATE FUNCTION", "CREATE OR REPLACE FUNCTION", 1) .replace("$CORE_SCHEMA$", &c) - .replace("$IMPLEMENTATION_SHA$", RETENTION_IMPLEMENTATION), + .replace("$IMPLEMENTATION_SHA$", implementation), ) .await .unwrap(); @@ -386,7 +418,7 @@ pub(crate) async fn restore_retention( } else { 0 }, - "Q-less cc90 fixture must have no history to discard: {name}" + "Q-less captured runtime fixture must have no history to discard: {name}" ); } let water: i64 = txn @@ -423,7 +455,7 @@ pub(crate) async fn restore_retention( txn.execute_unprepared(&format!("ALTER TABLE {c}.mst2_qualified_family_policy DISABLE TRIGGER mst2_route_family_policy_immutable")).await.unwrap(); txn.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres, format!("UPDATE {c}.mst2_qualified_family_policy SET implementation_fingerprint=$1,expected_shape=$2,authority_catalog=$3 WHERE singleton=1"), - [hex::decode(RETENTION_IMPLEMENTATION).unwrap().into(), old_shape.into(), authority.clone().into()])).await.unwrap(); + [hex::decode(implementation).unwrap().into(), old_shape.into(), authority.clone().into()])).await.unwrap(); txn.execute_unprepared(&format!("ALTER TABLE {c}.mst2_qualified_family_policy ENABLE TRIGGER mst2_route_family_policy_immutable")).await.unwrap(); if let Some(full) = &full { txn.execute_unprepared(&format!( @@ -433,10 +465,10 @@ pub(crate) async fn restore_retention( ALTER TABLE {c}.mst2_metadata_namespace DISABLE TRIGGER mst2_route_immutable")).await.unwrap(); txn.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres, format!("UPDATE {q}.mst2_metadata_family_identity SET implementation_fingerprint=$1 WHERE singleton=1"), - [hex::decode(RETENTION_IMPLEMENTATION).unwrap().into()])).await.unwrap(); + [hex::decode(implementation).unwrap().into()])).await.unwrap(); txn.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres, format!("UPDATE {c}.mst2_metadata_namespace SET implementation_fingerprint=$1,catalog_fingerprint=$2 WHERE namespace_uuid=$3::uuid"), - [hex::decode(RETENTION_IMPLEMENTATION).unwrap().into(), full.clone().into(), namespace.into()])).await.unwrap(); + [hex::decode(implementation).unwrap().into(), full.clone().into(), namespace.into()])).await.unwrap(); txn.execute_unprepared(&format!( "ALTER TABLE {q}.mst2_metadata_family_identity ENABLE TRIGGER mst2_00_family_barrier; ALTER TABLE {q}.mst2_metadata_family_identity ENABLE TRIGGER mst2_metadata_identity_immutable; diff --git a/src/jupiter/storage/qualified_metadata_serving.sql b/src/jupiter/storage/qualified_metadata_serving.sql index 56ae3c8f..18ad4663 100644 --- a/src/jupiter/storage/qualified_metadata_serving.sql +++ b/src/jupiter/storage/qualified_metadata_serving.sql @@ -48,9 +48,9 @@ BEGIN END IF; instance_bytes:=decode(replace(instance,'-',''),'hex'); view_digest:=sha256(convert_to('mega.mst2.namespaceview','UTF8')||decode('00','hex')||convert_to(commit_id,'UTF8')); - RETURN convert_to('MSD2','UTF8')||decode('00020001','hex')||instance_bytes||view_digest - ||decode(lpad(to_hex(octet_length(scope_bytes)),4,'0'),'hex')||scope_bytes - ||decode('0001000100000000','hex')||p.metadata_root; + RETURN convert_to('MSD2','UTF8')||decode('02000100','hex')||instance_bytes||view_digest + ||mst2_metadata_write_le(octet_length(scope_bytes),2)||scope_bytes + ||decode('0100010000000000','hex')||p.metadata_root; END $$; CREATE FUNCTION mst2_metadata_root_live(p bytea,g bigint,c bytea) RETURNS boolean From 5ff2db2888c6f9fb775ba2e1bfb091c057681e1a Mon Sep 17 00:00:00 2001 From: Xiaoyang Han Date: Thu, 8 Oct 2026 13:45:06 +0800 Subject: [PATCH 50/50] test(v3): complete reader owners before committing a prune backlog (#104) The native 64-owner backlog regression failed at its first bulk ACTIVE-owner commit, before any explicit prune assertion, because the shared deferred ownership guard rejected the final state. Source inspection supports the 60-second reader deadline as the most likely explanation; the runtime log does not identify which guard predicate failed. Complete each of the 65 admitted readers immediately in the same transaction, preserving the terminal transaction fence and deferred validation. After its first successful commit, assert 65 FINISHED owners, no ACTIVE owners or REQUEST/READER anchors, and an issuance high-water advance of 65. Preserve the same-transaction prune=0 and later prune=64 then1, with retained-row counts1 then0 and all existing tests. Production SQL, deadlines, ownership guards, implementation fingerprints and the focused24-test workflow remain unchanged. Nightly formatting, scoped source checks, candidate whitespace and bidirectional patch checks pass; native execution of this new fixture remains pending. --- .../router/snapshot_reader_retention_tests.rs | 41 +++++++++++++++---- 1 file changed, 34 insertions(+), 7 deletions(-) diff --git a/src/api/router/snapshot_reader_retention_tests.rs b/src/api/router/snapshot_reader_retention_tests.rs index 546ece7a..52890bd7 100644 --- a/src/api/router/snapshot_reader_retention_tests.rs +++ b/src/api/router/snapshot_reader_retention_tests.rs @@ -189,18 +189,45 @@ async fn terminal_owner_survives_its_deferred_completion_and_prunes_in_a_later_t #[tokio::test] async fn reader_pruning_respects_the_64_owner_budget_with_a_committed_backlog() { let fixture = Fixture::new_with_pg_config(true).await; + let initial = q_count( + &fixture, + "SELECT high_water FROM {q}.mst2_metadata_reader_issuance", + ) + .await; let txn = transaction(&fixture).await; - let mut tickets = Vec::new(); + // Complete each owner before admitting the next. Their terminal transaction + // fence retains the entire backlog until its deferred checks have committed. for _ in 0..65 { - tickets.push(admission(&txn, &fixture).await); - } - txn.commit().await.unwrap(); - let txn = transaction(&fixture).await; - for ticket in &tickets { - finish(&txn, ticket).await; + let ticket = admission(&txn, &fixture).await; + finish(&txn, &ticket).await; } assert_eq!(prune(&txn, 64).await, 0); txn.commit().await.unwrap(); + assert_eq!( + q_count( + &fixture, + "SELECT count(*) FROM {q}.mst2_metadata_reader_operation WHERE state='FINISHED'" + ) + .await, + 65 + ); + assert_eq!( + q_count( + &fixture, + "SELECT count(*) FROM {q}.mst2_metadata_reader_operation WHERE state='ACTIVE'" + ) + .await, + 0 + ); + assert_eq!(q_count(&fixture,"SELECT count(*) FROM {q}.mst2_metadata_root_anchor WHERE anchor_kind IN ('REQUEST','READER')").await,0); + assert_eq!( + q_count( + &fixture, + "SELECT high_water FROM {q}.mst2_metadata_reader_issuance" + ) + .await, + initial + 65 + ); let txn = transaction(&fixture).await; assert_eq!(prune(&txn, 64).await, 64); txn.commit().await.unwrap();