diff --git a/.github/workflows/repository-gates.yml b/.github/workflows/repository-gates.yml index b656c665..0e26b36f 100644 --- a/.github/workflows/repository-gates.yml +++ b/.github/workflows/repository-gates.yml @@ -66,13 +66,51 @@ jobs: docker compose -p mega2-it -f docker/docker-compose.test.yml --profile git \ up -d --build --wait postgres redis rustfs rustfs-init git-cli + - name: Verify necessary v3 publication and performance regressions + shell: bash + run: | + set -euo pipefail + source .env.test + export no_proxy="$NO_PROXY" + test_inventory="$(cargo test --lib -- --list)" + tests=( + api::router::snapshot_router::content::tests::rooted_metadata::descriptor_wire::rooted_postgres_descriptor_matches_codec_bytes_for_root_ascii_long_and_utf8_scopes + jupiter::migration::tests::import_repo_alias_rows_canonicalized + api::router::snapshot_router::content::tests::rooted_metadata::reader_retention::committed_metadata_requests_retire_owners_without_retiring_source_history + api::router::snapshot_router::content::tests::rooted_metadata::reader_retention::terminal_owner_survives_its_deferred_completion_and_prunes_in_a_later_transaction + api::router::snapshot_router::content::tests::rooted_metadata::reader_retention::reader_pruning_respects_the_64_owner_budget_with_a_committed_backlog + api::router::snapshot_router::content::tests::rooted_metadata::reader_retention::stale_actual_reader_cannot_read_or_finish_a_reissued_uuid + api::router::snapshot_router::content::tests::rooted_metadata::reader_retention::rolled_back_reader_issuance_is_unpublished_and_extreme_issuance_fails_closed + api::router::snapshot_router::content::tests::rooted_metadata::reader_retention::upgrade::current_reader_retention_migration_verifies_fresh_family_without_rewriting_history + api::router::snapshot_router::content::tests::rooted_metadata::reader_retention::upgrade::captured_previous_q_upgrade_preserves_old_sid_source_proofs_and_active_legacy_reader + api::router::snapshot_router::content::tests::rooted_metadata::reader_retention::upgrade::captured_previous_core_without_q_upgrades_before_new_provisioning + api::router::snapshot_router::content::tests::rooted_metadata::reader_retention::upgrade::captured_previous_tampered_q_rejects_upgrade_without_partial_issuance_schema + api::router::snapshot_router::content::tests::rooted_metadata::reader_retention::upgrade::native_runtime::captured_cc90_upgrade_preserves_nonzero_reader_issuance_all_owners_oids_and_old_sid + api::router::snapshot_router::content::tests::rooted_metadata::reader_retention::upgrade::native_runtime::captured_cc90_interrupted_before_reader_migration_resumes_without_reader_ddl + api::router::snapshot_router::content::tests::rooted_metadata::reader_retention::upgrade::native_runtime::captured_cc90_qless_canonical_policy_upgrades_before_new_provisioning + api::router::snapshot_router::content::tests::rooted_metadata::reader_retention::upgrade::native_runtime::captured_cc90_qless_exact_legacy_deparse_policy_upgrades_before_new_provisioning + api::router::snapshot_router::content::tests::rooted_metadata::reader_retention::upgrade::native_runtime::captured_cc90_tampered_decoder_rejects_upgrade_without_rewriting_owners_or_policy + api::router::snapshot_router::content::tests::rooted_metadata::reader_retention::upgrade::native_runtime::captured_cc90_qless_arbitrary_policy_shape_is_never_resigned + api::router::snapshot_router::content::tests::mst2_rooted_streamed_fact_persists_one_source_pass_and_reuses_warm_alias_facts + jupiter::storage::qualified_metadata_family::tests::canonical_tests::actual_rooted_cold_and_zero_delta_reuse_keep_one_canonical_graph + jupiter::storage::qualified_metadata_family::tests::canonical_tests::changed_ancestor_installs_only_delta_and_accepts_independently_attested_source_aliases + api::router::snapshot_router::content::tests::rooted_metadata::rooted_wide_directory_windows_and_lookup_survive_rebuild_with_valid_proofs + api::router::snapshot_router::content::tests::rooted_metadata::reader_retention::upgrade::native_runtime::descriptor_wire::captured_87b_descriptor_upgrade_preserves_owners_oids_and_replays_missing_ledgers + api::router::snapshot_router::content::tests::rooted_metadata::reader_retention::upgrade::native_runtime::descriptor_wire::captured_87b_qless_descriptor_upgrade_replays_missing_ledgers_before_provisioning + api::router::snapshot_router::content::tests::rooted_metadata::reader_retention::upgrade::native_runtime::descriptor_wire::captured_87b_tampered_descriptor_rejects_without_resigning_history + ) + for test_name in "${tests[@]}"; do + grep -Fqx "$test_name: test" <<< "$test_inventory" + done + cargo test --lib -- --exact --nocapture "${tests[@]}" + - name: Run repository tests with test environment shell: bash run: | set -euo pipefail source .env.test export no_proxy="$NO_PROXY" - cargo test --all + cargo test --all -- --nocapture - name: Stop isolated test dependencies if: always() diff --git a/src/api/router/snapshot_chunk_map_retention_tests.rs b/src/api/router/snapshot_chunk_map_retention_tests.rs new file mode 100644 index 00000000..6f326b57 --- /dev/null +++ b/src/api/router/snapshot_chunk_map_retention_tests.rs @@ -0,0 +1,1422 @@ +use sea_orm::{DatabaseConnection, DbBackend, Statement}; +use tokio::{sync::Notify, time::timeout}; + +use super::*; +use crate::{ + ceres::snapshot::{ + chunks::{ChunkMapSource, VerifiedSourceChunkMap}, + error::SnapshotErrorCode, + }, + jupiter::storage::{ + native_chunk_map::PostgresChunkMapRepository, object_storage::mock_object_storage, + }, +}; + +fn statement(sql: &str, values: [sea_orm::Value; N]) -> Statement { + Statement::from_sql_and_values(DbBackend::Postgres, sql, values) +} + +fn router(state: &MonoApiServiceState) -> Router { + Router::new().nest("/api/v2", routers(state.clone()).with_state(state.clone())) +} + +fn budgeted_chunks_app( + fixture: &Fixture, + response_budget: &Arc, + scratch_budget: &Arc, +) -> Router { + use axum::{ + extract::{Path as AxumPath, State}, + routing::post, + }; + + use crate::api::router::snapshot_router::{ + content, request::Mst2Bytes, snapshot_auth_middleware, + }; + + let response_budget = response_budget.clone(); + let scratch_budget = scratch_budget.clone(); + Router::new() + .route( + "/api/v2/snapshots/{snapshot_id}/chunks", + post( + move |state: State, + path: AxumPath, + body: Mst2Bytes| { + content::chunks_with_budgets( + state, + path, + body, + content::ChunksBudgets { + response: response_budget.clone(), + scratch: scratch_budget.clone(), + }, + ) + }, + ), + ) + .route_layer(axum::middleware::from_fn_with_state( + fixture.state.clone(), + snapshot_auth_middleware, + )) + .with_state(fixture.state.clone()) +} + +async fn count(db: &DatabaseConnection, table: &str) -> i64 { + db.query_one_raw(statement( + &format!("SELECT count(*) AS count FROM {table}"), + [], + )) + .await + .unwrap() + .unwrap() + .try_get("", "count") + .unwrap() +} + +async fn wait_count(db: &DatabaseConnection, table: &str, expected: i64) { + timeout(Duration::from_secs(10), async { + loop { + if count(db, table).await == expected { + return; + } + tokio::time::sleep(Duration::from_millis(5)).await; + } + }) + .await + .expect("actual durable owner release did not complete"); +} + +async fn age_unowned_candidates(db: &DatabaseConnection) { + db.execute_unprepared("UPDATE mst2_chunk_receipt_generation SET last_progress=pg_catalog.clock_timestamp()-interval '2 hours' WHERE state='LIVE'; UPDATE mst2_chunk_map_lifetime SET last_used=pg_catalog.clock_timestamp()-interval '2 hours' WHERE state='LIVE'").await.unwrap(); +} + +async fn oid_for(fixture: &Fixture, path: &str) -> String { + let handler = MonoApiService::from(&fixture.state); + let main = fixture + .state + .storage + .mono_storage() + .get_main_ref("/project") + .await + .unwrap() + .unwrap(); + let tree = handler.get_tree_by_hash(&main.ref_tree_hash).await.unwrap(); + match resolve_abs_metadata(&handler, &tree, path).await.unwrap() { + MetadataWalkOutcome::FoundFile { oid, .. } => oid, + other => panic!("retention fixture did not resolve: {other:?}"), + } +} + +#[tokio::test] +async fn shared_map_survives_other_source_retirement_while_actual_reader_is_live() { + let fixture = + Fixture::new_in_metadata_family(false, 0, &[("other".to_string(), vec![19; 4096])], true) + .await; + let original = fixture.map("/file").await; + let other = oid_for(&fixture, "/other").await; + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + // This G fixture isolates independent source receipts without changing a + // Q-certified file tuple. Two exact facts consume the same actual raw + // bytes independently; map sharing never admits the second source. + db.execute_raw(statement( + "UPDATE mst2_verified_object SET size=$1,raw_sha256=$2 WHERE git_oid=$3", + [ + (fixture.raw.len() as i64).into(), + fixture.digest.to_vec().into(), + other.clone().into(), + ], + )) + .await + .unwrap(); + *fixture.counts.object_size_override.lock().unwrap() = Some(fixture.raw.len() as i64); + *fixture.counts.object_fault.lock().unwrap() = Some(bounded_objects::StreamFault { + oid: other.clone(), + kind: bounded_objects::FaultKind::Parts( + fixture + .raw + .chunks(CHUNK_SIZE as usize) + .map(Bytes::copy_from_slice) + .collect(), + ), + }); + fixture.counts.reset(); + assert_eq!(fixture.map("/other").await["map"], original["map"]); + fixture.counts.assert(1, fixture.raw.len()); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 1); + wait_count(db, "mst2_chunk_reader", 0).await; + assert_eq!(count(db, "mst2_chunk_map_source").await, 2); + assert_eq!(count(db, "mst2_chunk_map").await, 1); + assert_eq!(count(db, "mst2_chunk_map_lifetime").await, 1); + let fact = fixture + .state + .storage + .mono_storage() + .get_verified_blobs(vec![other.clone()]) + .await + .unwrap() + .remove(&other) + .unwrap(); + let source = ChunkMapSource::from_fact(fact, &other).unwrap(); + let repository = fixture.state.storage.chunk_maps().await.unwrap(); + let objects = &fixture.state.storage.git_service.obj_storage; + let reader = repository.read(&source, objects).await.unwrap().unwrap(); + age_unowned_candidates(db).await; + repository.maintain(objects, 64).await.unwrap(); + assert_eq!(count(db, "mst2_chunk_map_source").await, 1); + assert_eq!(count(db, "mst2_chunk_map").await, 1); + assert_eq!(count(db, "mst2_chunk_map_leaf").await, 1); + assert_eq!(count(db, "mst2_chunk_map_node").await, 1); + assert_eq!(fixture.counts.receipt_deletes.load(Ordering::SeqCst), 1); + repository + .selected_page(&reader, 0) + .await + .unwrap() + .verify_chunk(&reader.map, 0, &fixture.raw[..CHUNK_SIZE as usize]) + .unwrap(); + drop(reader); + wait_count(db, "mst2_chunk_reader", 0).await; + age_unowned_candidates(db).await; + repository.maintain(objects, 64).await.unwrap(); + assert_eq!(count(db, "mst2_chunk_map_source").await, 0); + assert_eq!(count(db, "mst2_chunk_map").await, 0); + assert_eq!(fixture.counts.receipt_deletes.load(Ordering::SeqCst), 2); +} + +#[tokio::test] +async fn pending_receipt_is_replayed_before_another_aged_live_victim() { + let fixture = Fixture::new_with_pg_config_directories_and_objects( + false, + 0, + &[("other".to_string(), vec![19; 4096])], + ) + .await; + fixture.map("/file").await; + let other = fixture.map("/other").await; + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + wait_count(db, "mst2_chunk_reader", 0).await; + age_unowned_candidates(db).await; + db.execute_raw(statement( + "UPDATE mst2_chunk_receipt_generation g SET state='DELETING' WHERE EXISTS(SELECT 1 FROM mst2_chunk_map_source s WHERE s.receipt_key=g.receipt_key AND s.git_oid=$1)", + [fixture.oid.clone().into()], + )) + .await + .unwrap(); + let repository = fixture.state.storage.chunk_maps().await.unwrap(); + repository + .maintain(&fixture.state.storage.git_service.obj_storage, 1) + .await + .unwrap(); + assert_eq!(count(db, "mst2_chunk_map_source").await, 1); + assert_eq!(count(db, "mst2_chunk_map").await, 1); + assert_eq!(fixture.counts.receipt_deletes.load(Ordering::SeqCst), 1); + fixture.counts.reset(); + assert_eq!(fixture.map("/other").await["map"], other["map"]); + fixture.counts.assert(0, 0); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 0); +} + +#[tokio::test] +async fn actual_http_replays_a_retiring_source_row_before_rebuilding_its_next_generation() { + let fixture = Fixture::new().await; + let original = fixture.map("/file").await; + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + wait_count(db, "mst2_chunk_reader", 0).await; + let old_key: String = db + .query_one_raw(statement( + "SELECT receipt_key FROM mst2_chunk_map_source", + [], + )) + .await + .unwrap() + .unwrap() + .try_get("", "receipt_key") + .unwrap(); + db.execute_unprepared( + "UPDATE mst2_chunk_receipt_generation SET state='DELETING' WHERE state='LIVE'", + ) + .await + .unwrap(); + // The source row is deliberately still present; inline replay cannot + // depend on the source-index delete having already finished. + assert_eq!(count(db, "mst2_chunk_map_source").await, 1); + fixture.counts.reset(); + assert_eq!(fixture.map("/file").await["map"], original["map"]); + fixture.counts.assert(1, fixture.raw.len()); + assert_eq!(fixture.counts.receipt_deletes.load(Ordering::SeqCst), 1); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 1); + let new_key: String = db + .query_one_raw(statement( + "SELECT receipt_key FROM mst2_chunk_map_source", + [], + )) + .await + .unwrap() + .unwrap() + .try_get("", "receipt_key") + .unwrap(); + assert_ne!(old_key, new_key); + assert_eq!(count(db, "mst2_chunk_map").await, 1); + fixture.counts.reset(); + fixture.map("/alias").await; + fixture.counts.assert(0, 0); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 0); +} + +#[tokio::test] +async fn actual_warm_body_owner_cancels_never_returning_open_and_next_without_backend_release() { + use crate::ceres::snapshot::content_budget::{MemoryBudget, RANGE_WORK_BYTES}; + + for raw_route in [true, false] { + for stage in 0..3 { + let held_open = stage == 0; + let complete_without_eof = stage == 2; + let fixture = Fixture::new().await; + let map = fixture.map("/file").await; + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + wait_count(db, "mst2_chunk_reader", 0).await; + fixture.counts.reset(); + let entered = Arc::new(Notify::new()); + let release = Arc::new(Notify::new()); + let drops = Arc::new(AtomicUsize::new(0)); + let tail_polls = Arc::new(AtomicUsize::new(0)); + if held_open { + let holds = Some((entered.clone(), release.clone(), drops.clone())); + if raw_route { + *fixture.counts.whole_open_holds.lock().unwrap() = holds; + } else { + *fixture.counts.range_open_holds.lock().unwrap() = holds; + } + } else { + *fixture.counts.object_fault.lock().unwrap() = Some(bounded_objects::StreamFault { + oid: fixture.oid.clone(), + kind: bounded_objects::FaultKind::HeldFragment { + prefix: if complete_without_eof { + Bytes::copy_from_slice(if raw_route { + &fixture.raw + } else { + &fixture.raw[..CHUNK_SIZE as usize] + }) + } else { + Bytes::new() + }, + fragment: None, + entered: entered.clone(), + release: release.clone(), + drops: drops.clone(), + tail_polls: tail_polls.clone(), + }, + }); + } + let response_bytes = if raw_route { + CHUNK_SIZE as usize + } else { + CHUNK_SIZE as usize + 2048 + }; + let response_budget = MemoryBudget::new(response_bytes); + let scratch_budget = MemoryBudget::new(RANGE_WORK_BYTES); + let app = if raw_route { + super::raw_blob::budgeted_app(&fixture, &response_budget, &scratch_budget) + } else { + budgeted_chunks_app(&fixture, &response_budget, &scratch_budget) + }; + let request = if raw_route { + fixture.request("GET", "blob?path=/file", Body::empty()) + } else { + fixture.request( + "POST", + "chunks", + Body::from( + fixture + .chunk_body("/file", map["map"]["map_id"].as_str().unwrap(), "0") + .to_string(), + ), + ) + }; + let expected_raw = fixture.raw.clone(); + let mut task = tokio::spawn(async move { + let response = app.oneshot(request).await.unwrap(); + if raw_route && !held_open { + assert_eq!(response.status(), 200); + let mut body = response.into_body().into_data_stream(); + if complete_without_eof { + let first = body.next().await.unwrap().unwrap(); + assert_eq!(first.as_ref(), &expected_raw[..CHUNK_SIZE as usize]); + drop(first); + } + assert!(body.next().await.unwrap().is_err()); + assert!(body.next().await.is_none()); + } else { + error(response, 410, "LEASE_EXPIRED", false).await; + } + }); + timeout(Duration::from_secs(10), entered.notified()) + .await + .unwrap(); + assert_eq!(count(db, "mst2_chunk_reader").await, 1); + assert_eq!(scratch_budget.used(), RANGE_WORK_BYTES); + assert_eq!( + response_budget.used(), + if raw_route && complete_without_eof { + fixture.raw.len() - CHUNK_SIZE as usize + } else { + response_bytes + } + ); + db.execute_unprepared("UPDATE mst2_chunk_reader SET deadline=pg_catalog.clock_timestamp()+interval '20 milliseconds'").await.unwrap(); + // Never release the backend and never cancel the caller. The + // request must observe actual expiry during its pending await. + let finished = timeout(Duration::from_secs(12), &mut task).await; + if finished.is_err() { + task.abort(); + let _ = task.await; + panic!("actual body owner did not cancel its permanently held backend"); + } + finished.unwrap().unwrap(); + assert_eq!(drops.load(Ordering::SeqCst), 1); + assert_eq!(tail_polls.load(Ordering::SeqCst), 0); + let consumed = if complete_without_eof { + if raw_route { + fixture.raw.len() + } else { + CHUNK_SIZE as usize + } + } else { + 0 + }; + assert_eq!(fixture.counts.bytes.load(Ordering::SeqCst), consumed); + assert_eq!(scratch_budget.used(), 0); + assert_eq!(response_budget.used(), 0); + if raw_route { + fixture.counts.assert(1, consumed); + } else { + assert_eq!(fixture.counts.whole.load(Ordering::SeqCst), 0); + assert_eq!(fixture.counts.range.load(Ordering::SeqCst), 1); + } + wait_count(db, "mst2_chunk_reader", 0).await; + assert_eq!(count(db, "mst2_chunk_map_source").await, 1); + drop(release); + } + } +} + +#[tokio::test] +async fn actual_cold_builder_uses_remaining_database_deadline_for_open_and_next_and_refunds_credit() +{ + use crate::ceres::snapshot::content_budget::MemoryBudget; + + for stage in 0..3 { + let held_open = stage == 0; + let complete_without_eof = stage == 2; + let fixture = Fixture::new().await; + let repository = fixture.state.storage.chunk_maps().await.unwrap(); + let source = ChunkMapSource::from_fact(fixture.fact().await, &fixture.oid).unwrap(); + let admission = repository + .admit_install(&source, &fixture.state.storage.git_service.obj_storage) + .await + .unwrap(); + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + db.execute_unprepared("UPDATE mst2_chunk_receipt_generation SET deadline=pg_catalog.clock_timestamp()+interval '3 seconds' WHERE state='RESERVED'").await.unwrap(); + admission.test_check_next_owner_operation().await; + fixture.counts.reset(); + let entered = Arc::new(Notify::new()); + let release = Arc::new(Notify::new()); + let drops = Arc::new(AtomicUsize::new(0)); + let tail_polls = Arc::new(AtomicUsize::new(0)); + if held_open { + *fixture.counts.whole_open_holds.lock().unwrap() = + Some((entered.clone(), release.clone(), drops.clone())); + } else { + *fixture.counts.object_fault.lock().unwrap() = Some(bounded_objects::StreamFault { + oid: fixture.oid.clone(), + kind: bounded_objects::FaultKind::HeldFragment { + prefix: if complete_without_eof { + Bytes::copy_from_slice(&fixture.raw) + } else { + Bytes::new() + }, + fragment: None, + entered: entered.clone(), + release: release.clone(), + drops: drops.clone(), + tail_polls: tail_polls.clone(), + }, + }); + } + let budget = MemoryBudget::new(512 * 1024 * 1024); + let task_budget = budget.clone(); + let handler = MonoApiService::from(&fixture.state); + let mut task = tokio::spawn(async move { + VerifiedSourceChunkMap::verify(&handler, source, &task_budget, &admission).await + }); + timeout(Duration::from_secs(2), entered.notified()) + .await + .unwrap(); + assert!(budget.used() > 0); + let result = timeout(Duration::from_secs(5), &mut task).await; + if result.is_err() { + task.abort(); + let _ = task.await; + panic!("cold source ignored its actual remaining database deadline"); + } + let result = result.unwrap().unwrap(); + let error = match result { + Ok(_) => panic!("incomplete cold source produced trust"), + Err(error) => error, + }; + assert_eq!(error.code, SnapshotErrorCode::LeaseExpired); + assert_eq!(budget.used(), 0); + assert_eq!(drops.load(Ordering::SeqCst), 1); + assert_eq!(tail_polls.load(Ordering::SeqCst), 0); + fixture.counts.assert( + 1, + if complete_without_eof { + fixture.raw.len() + } else { + 0 + }, + ); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 0); + assert_eq!(count(db, "mst2_chunk_map").await, 0); + assert_eq!(count(db, "mst2_chunk_map_source").await, 0); + drop(release); + } +} + +#[tokio::test] +async fn actual_receipt_create_and_read_waits_cancel_without_backend_release_or_source_publication() +{ + use crate::ceres::snapshot::content_budget::MemoryBudget; + + for creating in [true, false] { + let fixture = Fixture::new().await; + let repository = fixture.state.storage.chunk_maps().await.unwrap(); + let source = ChunkMapSource::from_fact(fixture.fact().await, &fixture.oid).unwrap(); + let objects = fixture.state.storage.git_service.obj_storage.clone(); + let admission = repository.admit_install(&source, &objects).await.unwrap(); + let budget = MemoryBudget::new(512 * 1024 * 1024); + let handler = MonoApiService::from(&fixture.state); + let verified = VerifiedSourceChunkMap::verify(&handler, source, &budget, &admission) + .await + .unwrap(); + assert!(budget.used() > 0); + let entered = Arc::new(Notify::new()); + let release = Arc::new(Notify::new()); + let drops = if creating { + *fixture.counts.receipt_write_holds.lock().unwrap() = + Some((entered.clone(), release.clone())); + fixture.counts.receipt_write_wait_drops.clone() + } else { + let drops = Arc::new(AtomicUsize::new(0)); + *fixture.counts.receipt_open_holds.lock().unwrap() = + Some((entered.clone(), release.clone(), drops.clone())); + drops + }; + let task_storage = fixture.state.storage.clone(); + let mut task = tokio::spawn(async move { + let repository = task_storage.chunk_maps().await.unwrap(); + repository.install(verified, &objects, &admission).await + }); + timeout(Duration::from_secs(10), entered.notified()) + .await + .unwrap(); + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 1); + if creating { + db.execute_unprepared("UPDATE mst2_chunk_receipt_generation SET deadline=pg_catalog.clock_timestamp()+interval '20 milliseconds' WHERE state='CREATING'").await.unwrap(); + } + // Atomic create uses the actual owner expiry; receipt input also keeps + // its stricter five-second cap. Neither wait needs backend release. + let finished = timeout( + Duration::from_secs(if creating { 12 } else { 7 }), + &mut task, + ) + .await; + if finished.is_err() { + task.abort(); + let _ = task.await; + panic!("receipt wait retained its install workspace indefinitely"); + } + let error = finished.unwrap().unwrap().unwrap_err(); + assert_eq!( + error.code, + if creating { + SnapshotErrorCode::LeaseExpired + } else { + SnapshotErrorCode::TemporaryUnavailable + } + ); + assert_eq!(drops.load(Ordering::SeqCst), 1); + assert_eq!(budget.used(), 0); + fixture.counts.assert(1, fixture.raw.len()); + assert_eq!( + fixture.counts.receipt_reads.load(Ordering::SeqCst), + usize::from(!creating) + ); + assert_eq!(count(db, "mst2_chunk_reader").await, 0); + assert_eq!(count(db, "mst2_chunk_map_source").await, 0); + assert_eq!(count(db, "mst2_chunk_map").await, 0); + *fixture.counts.receipt_open_holds.lock().unwrap() = None; + let repository = fixture.state.storage.chunk_maps().await.unwrap(); + repository + .maintain(&fixture.state.storage.git_service.obj_storage, 64) + .await + .unwrap(); + assert_eq!(fixture.counts.receipt_deletes.load(Ordering::SeqCst), 1); + drop(release); + } +} + +#[tokio::test] +async fn held_raw_and_exact_range_do_not_resume_after_actual_reader_expiry() { + for raw_route in [true, false] { + let fixture = Fixture::new().await; + let map = fixture.map("/file").await; + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + wait_count(db, "mst2_chunk_reader", 0).await; + fixture.counts.reset(); + let entered = Arc::new(Notify::new()); + let release = Arc::new(Notify::new()); + let drops = Arc::new(AtomicUsize::new(0)); + let tail_polls = Arc::new(AtomicUsize::new(0)); + *fixture.counts.object_fault.lock().unwrap() = Some(bounded_objects::StreamFault { + oid: fixture.oid.clone(), + kind: bounded_objects::FaultKind::HeldFragment { + prefix: Bytes::new(), + fragment: Some(Bytes::new()), + entered: entered.clone(), + release: release.clone(), + tail_polls: tail_polls.clone(), + drops: drops.clone(), + }, + }); + let app = fixture.app.clone(); + let request = if raw_route { + fixture.request("GET", "blob?path=/file", Body::empty()) + } else { + fixture.request( + "POST", + "chunks", + Body::from( + fixture + .chunk_body("/file", map["map"]["map_id"].as_str().unwrap(), "0") + .to_string(), + ), + ) + }; + let mut task = tokio::spawn(async move { + let response = app.oneshot(request).await.unwrap(); + if raw_route { + assert_eq!(response.status(), 200); + let mut body = response.into_body().into_data_stream(); + assert!(body.next().await.unwrap().is_err()); + assert!(body.next().await.is_none()); + } else { + error(response, 410, "LEASE_EXPIRED", false).await; + } + }); + timeout(Duration::from_secs(10), entered.notified()) + .await + .unwrap(); + assert_eq!(count(db, "mst2_chunk_reader").await, 1); + db.execute_unprepared("UPDATE mst2_chunk_reader SET deadline=pg_catalog.clock_timestamp()+interval '20 milliseconds'").await.unwrap(); + // Exercise the real request-held reader without reaching into it: + // let the existing ten-second local check interval elapse too. + tokio::time::sleep(Duration::from_secs(11)).await; + release.notify_one(); + let completed = timeout(Duration::from_secs(5), &mut task).await; + if completed.is_err() { + task.abort(); + let _ = task.await; + panic!("expired reader continued into another held backend poll"); + } + completed.unwrap().unwrap(); + assert_eq!(tail_polls.load(Ordering::SeqCst), 0); + assert_eq!(drops.load(Ordering::SeqCst), 1); + assert_eq!(fixture.counts.bytes.load(Ordering::SeqCst), 0); + wait_count(db, "mst2_chunk_reader", 0).await; + assert_eq!(count(db, "mst2_chunk_map_source").await, 1); + } +} + +#[tokio::test] +async fn production_q_bootstrap_after_retention_preserves_catalog_and_reconstructed_warm_routes() { + use crate::jupiter::storage::qualified_metadata_family::{ + SnapshotMetadataFamily, provision_or_verify_rooted_qualified_family, + }; + + let fixture = Fixture::new_with_pg_config(true).await; + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + let namespace = provision_or_verify_rooted_qualified_family(db) + .await + .unwrap(); + assert_eq!( + fixture + .state + .storage + .snapshot_metadata_family(&fixture.lease, true) + .await + .unwrap(), + Some(SnapshotMetadataFamily::Rooted) + ); + let applied: i64 = db + .query_one_raw(statement( + "SELECT count(*) AS count FROM seaql_migrations WHERE version='m20261008_000300_add_mst2_chunk_map_retention'", + [], + )) + .await + .unwrap() + .unwrap() + .try_get("", "count") + .unwrap(); + assert_eq!(applied, 1); + let original = fixture.map("/file").await; + fixture.counts.assert(1, fixture.raw.len()); + wait_count(db, "mst2_chunk_reader", 0).await; + + // This is the production connection path: it reapplies migration checks + // and verifies the complete existing Q authority catalog without a test + // exemption or a refreshed registration fingerprint. + let config = fixture.state.storage.config(); + let connection = crate::jupiter::storage::init::database_connection(&config.database) + .await + .unwrap(); + assert_eq!( + provision_or_verify_rooted_qualified_family(&connection) + .await + .unwrap(), + namespace + ); + let storage = crate::jupiter::storage::Storage::new_with_connection( + config, + Arc::new(connection), + fixture.state.storage.git_service.obj_storage.clone(), + ) + .await + .unwrap(); + let state = MonoApiServiceState { + storage, + ..fixture.state.clone() + }; + assert_eq!( + state + .storage + .snapshot_metadata_family(&fixture.lease, true) + .await + .unwrap(), + Some(SnapshotMetadataFamily::Rooted) + ); + let app = router(&state); + fixture.counts.reset(); + let warm = success_json( + app.clone() + .oneshot(fixture.request("GET", "chunk-map?path=/alias", Body::empty())) + .await + .unwrap(), + ) + .await; + assert_eq!(warm["map"], original["map"]); + fixture.counts.assert(0, 0); + assert_eq!(fixture.counts.receipt_reads.load(Ordering::SeqCst), 1); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 0); + + fixture.counts.reset(); + let response = app + .oneshot( + fixture.request( + "POST", + "chunks", + Body::from( + fixture + .chunk_body("/alias", original["map"]["map_id"].as_str().unwrap(), "0") + .to_string(), + ), + ), + ) + .await + .unwrap(); + assert_eq!(response.status(), 200); + let wire = to_bytes(response.into_body(), 2 * 1024 * 1024) + .await + .unwrap(); + let frames = parse_stream(&wire).unwrap(); + let [Frame::Chunk(chunk), Frame::End(end)] = frames.as_slice() else { + panic!("reconstructed rooted route must emit CHUNK and terminal END"); + }; + assert_eq!(chunk.chunk_bytes, &fixture.raw[..CHUNK_SIZE as usize]); + assert_eq!(chunk.file_content_id, fixture.digest); + assert_eq!(chunk.chunk_index, 0); + assert_eq!(end.request_item_count, 1); + assert_eq!(end.logical_bytes, u64::from(CHUNK_SIZE)); + assert_eq!(fixture.counts.whole.load(Ordering::SeqCst), 0); + assert_eq!(fixture.counts.range.load(Ordering::SeqCst), 1); + assert_eq!( + fixture.counts.bytes.load(Ordering::SeqCst), + CHUNK_SIZE as usize + ); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 0); + drop(wire); + wait_count(db, "mst2_chunk_reader", 0).await; + assert_eq!(count(db, "mst2_chunk_map_source").await, 1); + assert_eq!(count(db, "mst2_chunk_map").await, 1); +} + +#[tokio::test] +async fn one_connection_install_commits_bounded_stages_and_reconstructed_warm_reads() { + let fixture = Fixture::new_with_pg_config(true).await; + let mut config = fixture.state.storage.config().database.clone(); + config.max_connection = 1; + config.min_connection = 1; + let connection = crate::jupiter::storage::init::database_connection(&config) + .await + .unwrap(); + let repository = PostgresChunkMapRepository::new(connection.clone()) + .await + .unwrap(); + assert!( + fixture + .state + .storage + .native_chunk_maps + .set(repository) + .is_ok() + ); + connection.execute_unprepared("CREATE TABLE chunk_stage_audit(relation text NOT NULL, writer_xid bigint NOT NULL); CREATE FUNCTION chunk_stage_audit() RETURNS trigger LANGUAGE plpgsql AS $audit$ BEGIN INSERT INTO chunk_stage_audit VALUES(TG_TABLE_NAME,pg_catalog.txid_current()); RETURN NEW; END $audit$; CREATE TRIGGER chunk_stage_audit AFTER INSERT ON mst2_chunk_map FOR EACH ROW EXECUTE FUNCTION chunk_stage_audit(); CREATE TRIGGER chunk_stage_audit AFTER INSERT ON mst2_chunk_map_leaf FOR EACH ROW EXECUTE FUNCTION chunk_stage_audit(); CREATE TRIGGER chunk_stage_audit AFTER INSERT ON mst2_chunk_map_node FOR EACH ROW EXECUTE FUNCTION chunk_stage_audit(); CREATE TRIGGER chunk_stage_audit AFTER INSERT ON mst2_chunk_map_source FOR EACH ROW EXECUTE FUNCTION chunk_stage_audit()").await.unwrap(); + let map = timeout(Duration::from_secs(10), fixture.map("/file")) + .await + .expect("single-connection staging waited for its own held connection"); + fixture.counts.assert(1, fixture.raw.len()); + let transactions: i64 = connection + .query_one_raw(statement( + "SELECT count(DISTINCT writer_xid) AS count FROM chunk_stage_audit", + [], + )) + .await + .unwrap() + .unwrap() + .try_get("", "count") + .unwrap(); + assert!(transactions >= 4); + for table in [ + "mst2_chunk_map", + "mst2_chunk_map_leaf", + "mst2_chunk_map_node", + "mst2_chunk_map_source", + ] { + assert_eq!(count(&connection, table).await, 1); + } + fixture.counts.reset(); + let reconstructed = PostgresChunkMapRepository::new(connection.clone()) + .await + .unwrap(); + let source = ChunkMapSource::from_fact(fixture.fact().await, &fixture.oid).unwrap(); + let persisted = timeout( + Duration::from_secs(10), + reconstructed.read(&source, &fixture.state.storage.git_service.obj_storage), + ) + .await + .unwrap() + .unwrap() + .unwrap(); + assert_eq!( + format!("sha256:{}", hex_of(&persisted.map_id)), + map["map"]["map_id"] + ); + let page = reconstructed.selected_page(&persisted, 0).await.unwrap(); + page.verify_chunk(&persisted.map, 0, &fixture.raw[..CHUNK_SIZE as usize]) + .unwrap(); + fixture.counts.assert(0, 0); + assert_eq!(fixture.counts.receipt_reads.load(Ordering::SeqCst), 1); +} + +#[tokio::test] +async fn cached_inventory_is_coalesced_and_bound_to_the_actual_backend_arc() { + let fixture = Fixture::new().await; + let repository = fixture.state.storage.chunk_maps().await.unwrap(); + let objects = &fixture.state.storage.git_service.obj_storage; + repository.maintain(objects, 8).await.unwrap(); + let first = fixture + .counts + .receipt_inventory_calls + .load(Ordering::SeqCst); + repository.maintain(objects, 8).await.unwrap(); + assert_eq!( + fixture + .counts + .receipt_inventory_calls + .load(Ordering::SeqCst), + first + ); + let counts = Arc::new(ReadCounts::default()); + counts + .receipt_retention_unsupported + .store(true, Ordering::SeqCst); + let other = MegaObjectStorageWrapper::new(Arc::new(CountingStorage { + inner: objects.clone(), + counts: counts.clone(), + })); + let error = repository.maintain(&other, 8).await.unwrap_err(); + assert_eq!(error.code, SnapshotErrorCode::TemporaryUnavailable); + assert_eq!(counts.receipt_inventory_calls.load(Ordering::SeqCst), 1); + fixture.counts.assert(0, 0); + assert_eq!( + count( + fixture.state.storage.mono_storage().get_connection(), + "mst2_chunk_receipt_generation" + ) + .await, + 0 + ); +} + +#[tokio::test] +async fn completion_during_real_inventory_keeps_backing_credit_after_pending_install_disappears() { + let mut fixture = Fixture::new().await; + let backing = mock_object_storage(); + backing + .inner + .put_stream( + &ObjectKey { + namespace: ObjectNamespace::Git, + key: fixture.oid.clone(), + }, + Box::pin(futures::stream::iter([Ok(Bytes::copy_from_slice( + &fixture.raw, + ))])), + ObjectMeta::default(), + ) + .await + .unwrap(); + fixture.state.storage.git_service = GitService { + obj_storage: MegaObjectStorageWrapper::new(Arc::new(CountingStorage { + inner: backing.clone(), + counts: fixture.counts.clone(), + })), + }; + fixture.app = router(&fixture.state); + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + let create_entered = Arc::new(Notify::new()); + let create_release = Arc::new(Notify::new()); + *fixture.counts.receipt_late_create_holds.lock().unwrap() = + Some((create_entered.clone(), create_release.clone())); + let app = fixture.app.clone(); + let request = fixture.request("GET", "chunk-map?path=/file", Body::empty()); + let leader = tokio::spawn(async move { app.oneshot(request).await.unwrap() }); + timeout(Duration::from_secs(10), create_entered.notified()) + .await + .unwrap(); + *fixture.counts.receipt_late_create_holds.lock().unwrap() = None; + for index in 0..crate::orbit_api::object_storage::MAX_CHUNK_MAP_RECEIPTS - 1 { + backing + .inner + .put_metadata_atomic_create( + &ObjectKey { + namespace: ObjectNamespace::ChunkMapReceipt, + key: format!("{index:064x}"), + }, + Bytes::from_static(&[1; 129]), + ObjectMeta::default(), + ) + .await + .unwrap(); + } + // Synthetic occupants test physical capacity; they confer no source trust. + let actual = &fixture.state.storage.git_service.obj_storage; + let inventory_entered = Arc::new(Notify::new()); + let inventory_release = Arc::new(Notify::new()); + let observer_counts = Arc::new(ReadCounts::default()); + *observer_counts.receipt_inventory_holds.lock().unwrap() = + Some((inventory_entered.clone(), inventory_release.clone())); + let observer = PostgresChunkMapRepository::new(db.clone()).await.unwrap(); + let observed = MegaObjectStorageWrapper::new(Arc::new(CountingStorage { + inner: actual.clone(), + counts: observer_counts, + })); + let capacity = tokio::spawn(async move { observer.test_new_install_capacity(&observed).await }); + timeout(Duration::from_secs(30), inventory_entered.notified()) + .await + .unwrap(); + create_release.notify_one(); + success_json( + timeout(Duration::from_secs(30), leader) + .await + .unwrap() + .unwrap(), + ) + .await; + assert_eq!(count(db, "mst2_chunk_map_source").await, 1); + let pending:i64=db.query_one_raw(statement("SELECT count(*) AS count FROM mst2_chunk_receipt_generation WHERE state IN ('RESERVED','CREATING')",[])).await.unwrap().unwrap().try_get("","count").unwrap(); + assert_eq!(pending, 0); + inventory_release.notify_one(); + let error = timeout(Duration::from_secs(10), capacity) + .await + .unwrap() + .unwrap() + .unwrap_err(); + assert_eq!(error.code, SnapshotErrorCode::LimitExceeded); + fixture.counts.assert(1, fixture.raw.len()); + assert_eq!( + actual + .inner + .chunk_map_receipt_inventory() + .await + .unwrap() + .objects + .len(), + crate::orbit_api::object_storage::MAX_CHUNK_MAP_RECEIPTS + ); +} + +#[tokio::test] +async fn json_chunk_and_raw_last_transport_clones_block_actual_collection() { + use crate::ceres::snapshot::content_budget::{MemoryBudget, RANGE_WORK_BYTES}; + + for route in 0..3 { + let fixture = Fixture::new().await; + let map = fixture.map("/file").await; + let repository = fixture.state.storage.chunk_maps().await.unwrap(); + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + wait_count(db, "mst2_chunk_reader", 0).await; + fixture.counts.reset(); + let response_budget = MemoryBudget::new(CHUNK_SIZE as usize + 2048); + let scratch_budget = MemoryBudget::new(RANGE_WORK_BYTES); + let response = match route { + 0 => { + fixture + .send("GET", "chunk-map?path=/file", Body::empty()) + .await + } + 1 => budgeted_chunks_app(&fixture, &response_budget, &scratch_budget) + .oneshot( + fixture.request( + "POST", + "chunks", + Body::from( + json!({"items":[{ + "path":"/file", "expected_digest":fixture.digest_string(), + "map_id":map["map"]["map_id"], "chunk_index":"0" + }],"encoding":"identity"}) + .to_string(), + ), + ), + ) + .await + .unwrap(), + _ => super::raw_blob::budgeted_app(&fixture, &response_budget, &scratch_budget) + .oneshot(fixture.request("GET", "blob?path=/file", Body::empty())) + .await + .unwrap(), + }; + assert_eq!(response.status(), 200); + let mut body = response.into_body().into_data_stream(); + let mut wire = Vec::new(); + let mut transport = Vec::new(); + while let Some(frame) = body.next().await { + let bytes = frame.unwrap(); + wire.extend_from_slice(&bytes); + transport.push(bytes.clone()); + } + drop(body); + if route == 2 { + assert_eq!(wire, fixture.raw); + } + if route == 1 { + let frames = parse_stream(&wire).unwrap(); + assert!(frames.iter().any(|frame| matches!(frame, Frame::End(_)))); + } + assert!(!transport.is_empty()); + let last = transport.pop().unwrap(); + let last_transport = last.clone(); + drop(last); + drop(transport); + assert_eq!(scratch_budget.used(), 0); + if route == 1 { + assert_eq!(response_budget.used(), CHUNK_SIZE as usize + 2048); + } else if route == 2 { + assert_eq!( + response_budget.used(), + fixture.raw.len() - CHUNK_SIZE as usize + ); + } + assert_eq!(count(db, "mst2_chunk_reader").await, 1); + age_unowned_candidates(db).await; + repository + .maintain(&fixture.state.storage.git_service.obj_storage, 64) + .await + .unwrap(); + assert_eq!(count(db, "mst2_chunk_map_source").await, 1); + assert_eq!(count(db, "mst2_chunk_map").await, 1); + assert_eq!(fixture.counts.receipt_deletes.load(Ordering::SeqCst), 0); + drop(last_transport); + assert_eq!(response_budget.used(), 0); + wait_count(db, "mst2_chunk_reader", 0).await; + repository + .maintain(&fixture.state.storage.git_service.obj_storage, 64) + .await + .unwrap(); + for table in [ + "mst2_chunk_map_source", + "mst2_chunk_map", + "mst2_chunk_map_leaf", + "mst2_chunk_map_node", + "mst2_chunk_map_lifetime", + ] { + assert_eq!(count(db, table).await, 0); + } + assert_eq!(fixture.counts.receipt_deletes.load(Ordering::SeqCst), 1); + assert_eq!(count(db, "mst2_chunk_receipt_generation").await, 1); + assert_eq!(count(db, "mst2_chunk_map_gc").await, 1); + } +} + +#[tokio::test] +async fn expired_reader_arc_cannot_resurrect_or_authenticate_a_page() { + let fixture = Fixture::new().await; + fixture.map("/file").await; + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + wait_count(db, "mst2_chunk_reader", 0).await; + let repository = fixture.state.storage.chunk_maps().await.unwrap(); + let source = ChunkMapSource::from_fact(fixture.fact().await, &fixture.oid).unwrap(); + let map = repository + .read(&source, &fixture.state.storage.git_service.obj_storage) + .await + .unwrap() + .unwrap(); + let old_generation: i64 = db + .query_one_raw(statement( + "SELECT generation FROM mst2_chunk_map_lifetime WHERE map_id=$1", + [map.map_id.to_vec().into()], + )) + .await + .unwrap() + .unwrap() + .try_get("", "generation") + .unwrap(); + db.execute_unprepared("UPDATE mst2_chunk_reader SET deadline=pg_catalog.clock_timestamp()+interval '20 milliseconds'").await.unwrap(); + tokio::time::sleep(Duration::from_millis(50)).await; + map.test_check_next_owner_operation().await; + assert_eq!( + map.ensure_live().await.unwrap_err().code, + SnapshotErrorCode::LeaseExpired + ); + assert_eq!( + map.record_progress().await.unwrap_err().code, + SnapshotErrorCode::LeaseExpired + ); + assert_eq!( + repository.selected_page(&map, 0).await.err().unwrap().code, + SnapshotErrorCode::LeaseExpired + ); + assert!(db.execute_unprepared("UPDATE mst2_chunk_reader SET deadline=pg_catalog.clock_timestamp()+interval '59 seconds'").await.is_err()); + age_unowned_candidates(db).await; + repository + .maintain(&fixture.state.storage.git_service.obj_storage, 64) + .await + .unwrap(); + assert_eq!(count(db, "mst2_chunk_map").await, 0); + assert_eq!( + map.ensure_live().await.unwrap_err().code, + SnapshotErrorCode::LeaseExpired + ); + fixture.counts.reset(); + assert_eq!( + fixture.map("/file").await["map"]["file_size"], + fixture.raw.len().to_string() + ); + fixture.counts.assert(1, fixture.raw.len()); + let new_generation: i64 = db + .query_one_raw(statement( + "SELECT generation FROM mst2_chunk_map_lifetime WHERE map_id=$1", + [map.map_id.to_vec().into()], + )) + .await + .unwrap() + .unwrap() + .try_get("", "generation") + .unwrap(); + assert!(new_generation > old_generation); + assert!( + db.execute_raw(statement( + "UPDATE mst2_chunk_map_lifetime SET state='DELETING' WHERE map_id=$1", + [map.map_id.to_vec().into()] + )) + .await + .is_err() + ); + assert!( + db.execute_raw(statement( + "DELETE FROM mst2_chunk_map_leaf WHERE map_id=$1", + [map.map_id.to_vec().into()] + )) + .await + .is_err() + ); + fixture.counts.reset(); + fixture.map("/alias").await; + fixture.counts.assert(0, 0); + assert_eq!( + map.ensure_live().await.unwrap_err().code, + SnapshotErrorCode::LeaseExpired + ); +} + +#[tokio::test] +async fn late_cancelled_create_is_reserved_and_old_key_replay_preserves_new_generation() { + let fixture = Fixture::new().await; + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + let entered = Arc::new(Notify::new()); + let release = Arc::new(Notify::new()); + *fixture.counts.receipt_late_create_holds.lock().unwrap() = + Some((entered.clone(), release.clone())); + let app = fixture.app.clone(); + let request = fixture.request("GET", "chunk-map?path=/file", Body::empty()); + let cancelled = tokio::spawn(async move { app.oneshot(request).await }); + timeout(Duration::from_secs(10), entered.notified()) + .await + .unwrap(); + let old_key: String = db + .query_one_raw(statement( + "SELECT receipt_key FROM mst2_chunk_receipt_generation", + [], + )) + .await + .unwrap() + .unwrap() + .try_get("", "receipt_key") + .unwrap(); + assert!( + !fixture + .state + .storage + .git_service + .obj_storage + .inner + .exists(&ObjectKey { + namespace: ObjectNamespace::ChunkMapReceipt, + key: old_key.clone() + }) + .await + .unwrap() + ); + cancelled.abort(); + assert!(cancelled.await.err().unwrap().is_cancelled()); + *fixture.counts.receipt_late_create_holds.lock().unwrap() = None; + let current = fixture.map("/file").await; + let repository = fixture.state.storage.chunk_maps().await.unwrap(); + wait_count(db, "mst2_chunk_reader", 0).await; + let history = db + .query_one_raw(statement( + "SELECT state,create_completed FROM mst2_chunk_receipt_generation WHERE receipt_key=$1", + [old_key.clone().into()], + )) + .await + .unwrap() + .unwrap(); + assert_eq!(history.try_get::("", "state").unwrap(), "APPLIED"); + assert!(!history.try_get::("", "create_completed").unwrap()); + let new_key: String = db + .query_one_raw(statement( + "SELECT receipt_key FROM mst2_chunk_map_source", + [], + )) + .await + .unwrap() + .unwrap() + .try_get("", "receipt_key") + .unwrap(); + assert_ne!(old_key, new_key); + release.notify_one(); + timeout(Duration::from_secs(10), async { + loop { + if fixture + .state + .storage + .git_service + .obj_storage + .inner + .exists(&ObjectKey { + namespace: ObjectNamespace::ChunkMapReceipt, + key: old_key.clone(), + }) + .await + .unwrap() + && fixture.counts.receipt_writes.load(Ordering::SeqCst) == 2 + { + break; + } + tokio::task::yield_now().await; + } + }) + .await + .unwrap(); + repository + .maintain(&fixture.state.storage.git_service.obj_storage, 64) + .await + .unwrap(); + assert!( + !fixture + .state + .storage + .git_service + .obj_storage + .inner + .exists(&ObjectKey { + namespace: ObjectNamespace::ChunkMapReceipt, + key: old_key.clone() + }) + .await + .unwrap() + ); + assert!( + fixture + .state + .storage + .git_service + .obj_storage + .inner + .exists(&ObjectKey { + namespace: ObjectNamespace::ChunkMapReceipt, + key: new_key + }) + .await + .unwrap() + ); + fixture.counts.reset(); + assert_eq!(fixture.map("/file").await["map"], current["map"]); + fixture.counts.assert(0, 0); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 0); + let history = db + .query_one_raw(statement( + "SELECT create_completed FROM mst2_chunk_receipt_generation WHERE receipt_key=$1", + [old_key.into()], + )) + .await + .unwrap() + .unwrap(); + assert!(!history.try_get::("", "create_completed").unwrap()); +} + +#[tokio::test] +async fn actual_backing_quota_and_unsupported_capability_reject_before_source_open() { + for unsupported in [true, false] { + let fixture = Fixture::new().await; + let counts = Arc::new(ReadCounts::default()); + counts + .receipt_retention_unsupported + .store(unsupported, Ordering::SeqCst); + let backend = mock_object_storage(); + if !unsupported { + for index in 0..=crate::orbit_api::object_storage::MAX_CHUNK_MAP_RECEIPTS { + backend + .inner + .put_metadata_atomic_create( + &ObjectKey { + namespace: ObjectNamespace::ChunkMapReceipt, + key: format!("{index:064x}"), + }, + Bytes::from_static(&[1; 129]), + ObjectMeta::default(), + ) + .await + .unwrap(); + } + } + let mut state = fixture.state.clone(); + state.storage.git_service = GitService { + obj_storage: MegaObjectStorageWrapper::new(Arc::new(CountingStorage { + inner: backend, + counts: counts.clone(), + })), + }; + let response = router(&state) + .oneshot(fixture.request("GET", "chunk-map?path=/file", Body::empty())) + .await + .unwrap(); + if unsupported { + error(response, 503, "TEMPORARY_UNAVAILABLE", true).await; + } else { + error(response, 413, "LIMIT_EXCEEDED", false).await; + } + counts.assert(0, 0); + assert_eq!(counts.receipt_writes.load(Ordering::SeqCst), 0); + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + assert_eq!(count(db, "mst2_chunk_receipt_generation").await, 0); + assert_eq!(count(db, "mst2_chunk_map").await, 0); + } +} + +#[tokio::test] +async fn empty_source_fragment_after_deadline_does_not_renew_install_or_publish() { + let fixture = Fixture::new().await; + let repository = fixture.state.storage.chunk_maps().await.unwrap(); + let source = ChunkMapSource::from_fact(fixture.fact().await, &fixture.oid).unwrap(); + let admission = Arc::new( + repository + .admit_install(&source, &fixture.state.storage.git_service.obj_storage) + .await + .unwrap(), + ); + let entered = Arc::new(Notify::new()); + let release = Arc::new(Notify::new()); + let drops = Arc::new(AtomicUsize::new(0)); + let tail_polls = Arc::new(AtomicUsize::new(0)); + *fixture.counts.object_fault.lock().unwrap() = Some(bounded_objects::StreamFault { + oid: fixture.oid.clone(), + kind: bounded_objects::FaultKind::HeldFragment { + prefix: Bytes::new(), + fragment: Some(Bytes::new()), + entered: entered.clone(), + release: release.clone(), + tail_polls: tail_polls.clone(), + drops: drops.clone(), + }, + }); + let handler = MonoApiService::from(&fixture.state); + let budget = crate::ceres::snapshot::content_budget::MemoryBudget::new(8 * 1024 * 1024); + let verifier_budget = budget.clone(); + let owner = admission.clone(); + let task = tokio::spawn(async move { + VerifiedSourceChunkMap::verify(&handler, source, &verifier_budget, &owner).await + }); + timeout(Duration::from_secs(10), entered.notified()) + .await + .unwrap(); + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + db.execute_unprepared("UPDATE mst2_chunk_receipt_generation SET deadline=pg_catalog.clock_timestamp()+interval '20 milliseconds' WHERE state='RESERVED'").await.unwrap(); + tokio::time::sleep(Duration::from_millis(50)).await; + admission.test_check_next_owner_operation().await; + release.notify_one(); + let error = timeout(Duration::from_secs(5), task) + .await + .unwrap() + .unwrap() + .err() + .unwrap(); + assert_eq!(error.code, SnapshotErrorCode::LeaseExpired); + assert_eq!(drops.load(Ordering::SeqCst), 1); + assert_eq!(tail_polls.load(Ordering::SeqCst), 0); + assert_eq!(fixture.counts.bytes.load(Ordering::SeqCst), 0); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 0); + assert_eq!(count(db, "mst2_chunk_map_source").await, 0); + assert_eq!(budget.used(), 0); + drop(admission); + repository + .maintain(&fixture.state.storage.git_service.obj_storage, 64) + .await + .unwrap(); + assert_eq!(count(db, "mst2_chunk_map").await, 0); +} diff --git a/src/api/router/snapshot_chunks_bounded_tests.rs b/src/api/router/snapshot_chunks_bounded_tests.rs new file mode 100644 index 00000000..243d326a --- /dev/null +++ b/src/api/router/snapshot_chunks_bounded_tests.rs @@ -0,0 +1,514 @@ +use std::io; + +use tokio::{sync::Notify, time::timeout}; + +use super::*; +use crate::orbit_api::error::IoOrbitError; + +#[derive(Clone)] +pub(super) struct ChunkFault { + pub(super) oid: String, + pub(super) size: u64, + pattern: Bytes, + mode: RangeMode, + requests: Arc>>, +} + +#[derive(Clone)] +enum RangeMode { + Good, + WrongMeta, + WrongDigest, + Truncated, + TooLong, + LateError, + Unsupported, + WrongOffset, + Missing, + Held { + entered: Arc, + release: Arc, + drops: Arc, + }, +} + +struct DropCount(Arc); +impl Drop for DropCount { + fn drop(&mut self) { + self.0.fetch_add(1, Ordering::SeqCst); + } +} + +impl ChunkFault { + pub(super) fn full_stream(self) -> ObjectByteStream { + Box::pin(futures::stream::unfold( + (self.pattern, 0u64, self.size), + |(pattern, offset, size)| async move { + if offset == size { + return None; + } + let len = (size - offset).min(CHUNK_SIZE as u64) as usize; + let bytes = pattern.slice(..len); + Some((Ok(bytes), (pattern, offset + len as u64, size))) + }, + )) + } + + pub(super) fn range_stream( + self, + start: u64, + end: u64, + ) -> OrbitResult> { + self.requests + .lock() + .unwrap() + .push((self.oid.clone(), start, end)); + if matches!(self.mode, RangeMode::Unsupported) { + return Ok(None); + } + if matches!(self.mode, RangeMode::WrongOffset) { + return Err(io::Error::new(io::ErrorKind::InvalidData, "wrong backend range").into()); + } + if matches!(self.mode, RangeMode::Missing) { + return Err(IoOrbitError::object_store_not_found("fixed raw missing")); + } + assert!(start < end && end <= self.size && end - start <= CHUNK_SIZE as u64); + assert_eq!(start % CHUNK_SIZE as u64, 0); + let len = (end - start) as usize; + let raw = self.pattern.slice(..len); + let meta = ObjectMeta { + size: if matches!(self.mode, RangeMode::WrongMeta) { + self.size as i64 + 1 + } else { + self.size as i64 + }, + ..Default::default() + }; + let stream: ObjectByteStream = match self.mode { + RangeMode::WrongDigest => { + Box::pin(futures::stream::iter([Ok(Bytes::from(vec![0; len]))])) + } + RangeMode::Truncated => Box::pin(futures::stream::iter([Ok(raw.slice(..len - 1))])), + RangeMode::TooLong => Box::pin(futures::stream::iter([ + Ok(raw), + Ok(Bytes::from_static(b"!")), + ])), + RangeMode::LateError => Box::pin(futures::stream::iter([ + Ok(raw), + Err(io::Error::other("late range error")), + ])), + RangeMode::Held { + entered, + release, + drops, + } => Box::pin(futures::stream::unfold( + (Some(raw), entered, release, DropCount(drops)), + |(raw, entered, release, owner)| async move { + let raw = raw?; + entered.notify_one(); + release.notified().await; + Some((Ok(raw), (None, entered, release, owner))) + }, + )), + _ => Box::pin(futures::stream::iter([Ok(raw)])), + }; + Ok(Some((stream, meta))) + } +} + +async fn oid_for(fixture: &Fixture, path: &str) -> String { + let handler = MonoApiService::from(&fixture.state); + let main = fixture + .state + .storage + .mono_storage() + .get_main_ref("/project") + .await + .unwrap() + .unwrap(); + let tree = handler.get_tree_by_hash(&main.ref_tree_hash).await.unwrap(); + match resolve_abs_metadata(&handler, &tree, path).await.unwrap() { + MetadataWalkOutcome::FoundFile { oid, .. } => oid, + other => panic!("fixture did not resolve: {other:?}"), + } +} + +async fn set_fact(fixture: &Fixture, oid: &str, size: u64, digest: [u8; 32]) { + let storage = fixture.state.storage.mono_storage(); + let db = storage.get_connection(); + let fact = mst2_verified_object::Entity::find() + .filter(mst2_verified_object::Column::GitOid.eq(oid)) + .one(db) + .await + .unwrap() + .unwrap(); + let mut fact = fact.into_active_model(); + fact.size = Set(size as i64); + fact.raw_sha256 = Set(digest.to_vec()); + fact.update(db).await.unwrap(); +} + +fn request(path: &str, digest: [u8; 32], map_id: &str, index: u64) -> Value { + json!({"items":[{"path":path,"expected_digest":format!("sha256:{}",hex_of(&digest)),"map_id":map_id,"chunk_index":index.to_string()}],"encoding":"identity"}) +} + +async fn assert_chunk(response: Response, body: &Value, raw: &[u8], index: u64) { + assert_eq!(response.status(), 200); + let wire = to_bytes(response.into_body(), 2 * 1024 * 1024) + .await + .unwrap(); + let frames = parse_stream(&wire).unwrap(); + let [Frame::Chunk(chunk), Frame::End(end)] = frames.as_slice() else { + panic!("one CHUNK and END required"); + }; + assert_eq!(chunk.chunk_bytes, raw); + assert_eq!(chunk.chunk_index, index); + assert_eq!(end.request_item_count, 1); + assert_eq!(end.unique_unit_count, 1); + assert_eq!(end.logical_bytes, raw.len() as u64); + assert_eq!( + end.request_body_sha256, + <[u8; 32]>::from(Sha256::digest(body.to_string().as_bytes())) + ); +} + +#[tokio::test] +async fn mst2_large_chunk_uses_current_oid_strict_range_faults_cancel_retry_and_lease() { + let fixture = Fixture::new_with_pg_config_directories_and_objects( + false, + 0, + &[("other".to_string(), vec![19; 4096])], + ) + .await; + let mut pattern = vec![51; CHUNK_SIZE as usize]; + let seed = uuid::Uuid::new_v4(); + pattern[..16].copy_from_slice(seed.as_bytes()); + let pattern = Bytes::from(pattern); + let size = 512 * CHUNK_SIZE as u64 + 7; + let mut hash = Sha256::new(); + for _ in 0..512 { + hash.update(&pattern); + } + hash.update(&pattern[..7]); + let digest: [u8; 32] = hash.finalize().into(); + let other = oid_for(&fixture, "/other").await; + let requests = Arc::new(std::sync::Mutex::new(Vec::new())); + for oid in [&fixture.oid, &other] { + set_fact(&fixture, oid, size, digest).await; + fixture + .counts + .chunk_faults + .lock() + .unwrap() + .push(ChunkFault { + oid: oid.clone(), + size, + pattern: pattern.clone(), + mode: RangeMode::Good, + requests: requests.clone(), + }); + } + fixture.counts.reset(); + let map = fixture.map("/file").await; + assert_eq!(map["map"]["file_size"], size.to_string()); + fixture.counts.assert(1, size as usize); + let map_id = map["map"]["map_id"].as_str().unwrap(); + fixture.counts.reset(); + super::persisted_chunk_maps::assert_three_page_proofs_and_selected_sibling_faults( + &fixture, map_id, digest, &pattern, + ) + .await; + // Each distinct OID earns its own full-stream receipt before map reuse. + assert_eq!(fixture.map("/other").await["map"], map["map"]); + fixture.counts.assert(1, size as usize); + fixture.counts.reset(); + // Equal content map sharing never carries the first source's OID. + let body = request("/other", digest, map_id, 512); + assert_chunk( + fixture + .send("POST", "chunks", Body::from(body.to_string())) + .await, + &body, + &pattern[..7], + 512, + ) + .await; + assert_eq!( + requests.lock().unwrap().as_slice(), + &[(other.clone(), 512 * CHUNK_SIZE as u64, size)] + ); + assert_eq!(fixture.counts.whole.load(Ordering::SeqCst), 0); + assert_eq!(fixture.counts.range.load(Ordering::SeqCst), 1); + assert_eq!(fixture.counts.bytes.load(Ordering::SeqCst), 7); + for (index, encoding) in [(0, "identity"), (511, "zstd")] { + let mut full = request("/other", digest, map_id, index); + full["encoding"] = json!(encoding); + fixture.counts.reset(); + assert_chunk( + fixture + .send("POST", "chunks", Body::from(full.to_string())) + .await, + &full, + &pattern, + index, + ) + .await; + assert_eq!(fixture.counts.whole.load(Ordering::SeqCst), 0); + assert_eq!(fixture.counts.range.load(Ordering::SeqCst), 1); + assert_eq!( + fixture.counts.bytes.load(Ordering::SeqCst), + CHUNK_SIZE as usize + ); + assert_eq!( + requests.lock().unwrap().last().unwrap(), + &( + other.clone(), + index * CHUNK_SIZE as u64, + (index + 1) * CHUNK_SIZE as u64 + ) + ); + } + for (mode, status, code, bytes) in [ + (RangeMode::WrongMeta, 502, "INTEGRITY_ERROR", 0), + (RangeMode::WrongDigest, 502, "INTEGRITY_ERROR", 7), + (RangeMode::Truncated, 502, "INTEGRITY_ERROR", 6), + (RangeMode::TooLong, 502, "INTEGRITY_ERROR", 8), + (RangeMode::LateError, 503, "OBJECT_UNAVAILABLE", 7), + (RangeMode::Unsupported, 400, "RANGE_NOT_SUPPORTED", 0), + (RangeMode::WrongOffset, 502, "INTEGRITY_ERROR", 0), + (RangeMode::Missing, 503, "OBJECT_UNAVAILABLE", 0), + ] { + fixture.counts.chunk_faults.lock().unwrap()[1].mode = mode; + fixture.counts.reset(); + error( + fixture + .send("POST", "chunks", Body::from(body.to_string())) + .await, + status, + code, + false, + ) + .await; + assert_eq!(fixture.counts.whole.load(Ordering::SeqCst), 0); + assert_eq!(fixture.counts.range.load(Ordering::SeqCst), 1); + assert_eq!(fixture.counts.bytes.load(Ordering::SeqCst), bytes); + } + fixture.counts.chunk_faults.lock().unwrap()[1].mode = RangeMode::LateError; + let first = request("/file", digest, map_id, 0); + let batch = json!({"items":[first["items"][0],body["items"][0]],"encoding":"identity"}); + fixture.counts.reset(); + error( + fixture + .send("POST", "chunks", Body::from(batch.to_string())) + .await, + 503, + "OBJECT_UNAVAILABLE", + false, + ) + .await; + assert_eq!(fixture.counts.whole.load(Ordering::SeqCst), 0); + assert_eq!(fixture.counts.range.load(Ordering::SeqCst), 2); + assert_eq!( + fixture.counts.bytes.load(Ordering::SeqCst), + CHUNK_SIZE as usize + 7 + ); + let entered = Arc::new(Notify::new()); + let release = Arc::new(Notify::new()); + let drops = Arc::new(AtomicUsize::new(0)); + fixture.counts.chunk_faults.lock().unwrap()[1].mode = RangeMode::Held { + entered: entered.clone(), + release: release.clone(), + drops: drops.clone(), + }; + let task = tokio::spawn(fixture.app.clone().oneshot(fixture.request( + "POST", + "chunks", + Body::from(body.to_string()), + ))); + timeout(Duration::from_secs(10), entered.notified()) + .await + .unwrap(); + task.abort(); + assert!(task.await.err().unwrap().is_cancelled()); + assert_eq!(drops.load(Ordering::SeqCst), 1); + fixture.counts.chunk_faults.lock().unwrap()[1].mode = RangeMode::Good; + assert_chunk( + fixture + .send("POST", "chunks", Body::from(body.to_string())) + .await, + &body, + &pattern[..7], + 512, + ) + .await; + // Revocation while the range is held is a pre-header 410 JSON response. + fixture.counts.chunk_faults.lock().unwrap()[1].mode = RangeMode::Held { + entered: entered.clone(), + release: release.clone(), + drops: drops.clone(), + }; + let task = tokio::spawn(fixture.app.clone().oneshot(fixture.request( + "POST", + "chunks", + Body::from(body.to_string()), + ))); + timeout(Duration::from_secs(10), entered.notified()) + .await + .unwrap(); + let revoked = fixture + .app + .clone() + .oneshot( + Request::builder() + .method("DELETE") + .uri(format!("/api/v2/snapshots/leases/{}", fixture.lease)) + .header("authorization", format!("Bearer {TOKEN}")) + .body(Body::empty()) + .unwrap(), + ) + .await + .unwrap(); + assert_eq!(revoked.status(), 200); + release.notify_one(); + error( + timeout(Duration::from_secs(10), task) + .await + .unwrap() + .unwrap() + .unwrap(), + 410, + "LEASE_EXPIRED", + false, + ) + .await; + assert_eq!(drops.load(Ordering::SeqCst), 2); +} + +#[tokio::test] +async fn mst2_chunk_batch_live_budget_and_invalid_later_path_reject_before_body_io() { + let fixture = Fixture::new_with_pg_config_directories_and_objects( + false, + 0, + &[ + ("one".to_string(), vec![11; 4096]), + ("two".to_string(), vec![12; 4096]), + ], + ) + .await; + let budget = crate::ceres::snapshot::content_budget::MemoryBudget::new(1024 * 1024); + let repository = crate::jupiter::storage::native_chunk_map::PostgresChunkMapRepository::new( + fixture + .state + .storage + .mono_storage() + .get_connection() + .clone(), + ) + .await + .unwrap() + .with_test_budget(budget.clone()); + assert!( + fixture + .state + .storage + .native_chunk_maps + .set(repository) + .is_ok() + ); + let mut items = Vec::new(); + for (index, path) in ["/file", "/one", "/two"].iter().enumerate() { + let oid = oid_for(&fixture, path).await; + let digest = [index as u8 + 1; 32]; + set_fact(&fixture, &oid, 512 * CHUNK_SIZE as u64, digest).await; + items.push( + request( + path, + digest, + &format!("sha256:{}", hex_of(&[index as u8 + 1; 32])), + 0, + )["items"][0] + .clone(), + ); + } + fixture.counts.reset(); + error( + fixture + .send( + "POST", + "chunks", + Body::from(json!({"items":items,"encoding":"identity"}).to_string()), + ) + .await, + 413, + "LIMIT_EXCEEDED", + false, + ) + .await; + fixture.counts.assert(0, 0); + assert_eq!(budget.used(), 0); + let fixture = Fixture::new().await; + let mut body = fixture.chunk_body("/file", &format!("sha256:{}", hex_of(&[1; 32])), "0"); + let mut invalid = body["items"][0].clone(); + invalid["path"] = json!("/absent"); + body["items"].as_array_mut().unwrap().push(invalid); + error( + fixture + .send("POST", "chunks", Body::from(body.to_string())) + .await, + 404, + "PATH_NOT_FOUND", + false, + ) + .await; + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn mst2_budgeted_chunk_body_preserves_per_frame_revocation_and_no_end() { + let fixture = Fixture::new().await; + let map = fixture.map("/file").await; + let map_id = map["map"]["map_id"].as_str().unwrap(); + let mut body = fixture.chunk_body("/file", map_id, "0"); + body["items"] + .as_array_mut() + .unwrap() + .push(fixture.chunk_body("/file", map_id, "1")["items"][0].clone()); + let response = fixture + .send("POST", "chunks", Body::from(body.to_string())) + .await; + assert_eq!(response.status(), 200); + let mut data = response.into_body().into_data_stream(); + let first = data.next().await.unwrap().unwrap(); + let (frame, consumed) = mst2_codec::treeframe::parse_frame(&first).unwrap(); + assert_eq!(consumed, first.len()); + let Frame::Chunk(chunk) = frame else { + panic!("first DATA must contain exactly one CHUNK frame"); + }; + assert_eq!(format!("sha256:{}", hex_of(&chunk.map_id)), map_id); + assert_eq!(chunk.file_content_id, fixture.digest); + assert_eq!(chunk.chunk_index, 0); + assert_eq!(chunk.chunk_bytes, fixture.raw[..CHUNK_SIZE as usize]); + let received = first.to_vec(); + let revoked = fixture + .app + .clone() + .oneshot( + Request::builder() + .method("DELETE") + .uri(format!("/api/v2/snapshots/leases/{}", fixture.lease)) + .header("authorization", format!("Bearer {TOKEN}")) + .body(Body::empty()) + .unwrap(), + ) + .await + .unwrap(); + assert_eq!(revoked.status(), 200); + assert!(data.next().await.unwrap().is_err()); + assert!(data.next().await.is_none()); + assert!(matches!( + parse_stream(&received), + Err(mst2_codec::CodecError::BadOrdering( + "stream missing END/ERROR frame" + )) + )); +} diff --git a/src/api/router/snapshot_content.rs b/src/api/router/snapshot_content.rs index b52dbba6..961d1b00 100644 --- a/src/api/router/snapshot_content.rs +++ b/src/api/router/snapshot_content.rs @@ -4,98 +4,361 @@ //! WP and `frame_encodings` advertises identity alone. use axum::{ - Json, extract::{Path as AxumPath, Query, State}, http::HeaderMap, - response::{IntoResponse, Response}, + response::Response, }; -use bytes::Bytes; use futures::stream::StreamExt; use serde::Deserialize; use serde_json::json; +use sha2::{Digest, Sha256}; -use super::{abs_view_path, internal, mst2_error_response}; +use super::{ + abs_view_path, guarded_treeframe_response, guarded_treeframe_response_with_budget, internal, + mst2_error_response, request::Mst2Bytes, +}; use crate::ceres::snapshot::{ - chunks::{ChunkProjection, get_or_project}, + chunks::{ChunkMapSource, VerifiedSourceChunkMap, map_build_reservation_bytes}, + content_budget::{BudgetedFrame, MemoryLease, reserve_response}, error::{SnapshotError, SnapshotErrorCode}, - pages::{WalkOutcome, base64_of, hex_of, resolve_abs}, + pages::{MetadataWalkOutcome, base64_of, hex_of, resolve_abs_metadata}, resolver::FsKind, - runtime::runtime, view::validate_scope_relative_path, }; -/// One file resolved at a fixed path with verified content. -struct ResolvedFile { - fs_kind: FsKind, - digest: [u8; 32], - size: u64, - raw: Vec, +pub(super) struct ResolvedFileMetadata { + pub(super) fs_kind: FsKind, + pub(super) oid: String, + pub(super) digest: [u8; 32], + pub(super) size: u64, + fact: crate::callisto::mst2_verified_object::Model, } -pub(super) fn fs_kind_str(k: FsKind) -> &'static str { - match k { - FsKind::Regular => "regular", - FsKind::Executable => "executable", - FsKind::Symlink => "symlink", - FsKind::Directory => "directory", +#[allow(clippy::result_large_err)] +pub(super) async fn fixed_root_tree( + handler: &T, + oid: &str, +) -> Result { + let tree = handler.get_tree_by_hash(oid).await.map_err(internal)?; + if tree.id.to_string() != oid { + return Err(mst2_error_response(SnapshotError::new( + SnapshotErrorCode::IntegrityError, + "fetched fixed root tree identity mismatch", + ))); } + Ok(tree) } -/// Resolve a scope-relative path against the fixed tree and verify the -/// optional `expected_digest`. Absence/directory/intermediate outcomes stay -/// typed errors, never an empty body. #[allow(clippy::result_large_err)] -async fn resolve_file( +async fn resolve_legacy_file_metadata( handler: &T, root_tree: &git_internal::internal::object::tree::Tree, scope: &str, path: &str, expected_digest: Option<&str>, -) -> Result { - let abs_path = abs_view_path(scope, path); - match resolve_abs(handler, root_tree, &abs_path) +) -> Result { + validate_scope_relative_path(path).map_err(mst2_error_response)?; + let (fs_kind, oid) = match resolve_abs_metadata(handler, root_tree, &abs_view_path(scope, path)) .await .map_err(mst2_error_response)? { - WalkOutcome::FoundFile { - fs_kind, - raw, - size, - digest, - .. - } => { - if let Some(expected) = expected_digest - && expected != format!("sha256:{}", hex_of(&digest)) - { - return Err(mst2_error_response(SnapshotError::new( - SnapshotErrorCode::DigestMismatch, - format!("{path}: content does not match expected_digest"), - ))); - } - Ok(ResolvedFile { - fs_kind, - digest, - size, - raw, - }) + MetadataWalkOutcome::FoundFile { fs_kind, oid } => (fs_kind, oid), + MetadataWalkOutcome::FoundDir => { + return Err(mst2_error_response(SnapshotError::new( + SnapshotErrorCode::NotDirectory, + format!("{path} is a directory"), + ))); } - WalkOutcome::FoundDir => Err(mst2_error_response(SnapshotError::new( - SnapshotErrorCode::NotDirectory, - format!("{path} is a directory"), - ))), - WalkOutcome::Absent => Err(mst2_error_response(SnapshotError::new( - SnapshotErrorCode::PathNotFound, - format!("{path} absent in the fixed view"), + MetadataWalkOutcome::Absent => { + return Err(mst2_error_response(SnapshotError::new( + SnapshotErrorCode::PathNotFound, + format!("{path} absent in the fixed view"), + ))); + } + MetadataWalkOutcome::NotDirectory { symlink } => { + return Err(mst2_error_response(SnapshotError::new( + if symlink { + SnapshotErrorCode::SymlinkTraversal + } else { + SnapshotErrorCode::NotDirectory + }, + format!("{path}: intermediate component is not a directory"), + ))); + } + }; + verified_file_metadata(handler, fs_kind, oid, path, expected_digest).await +} + +#[allow(clippy::result_large_err)] +async fn fixed_content_root( + handler: &T, + ctx: &crate::ceres::snapshot::runtime::SnapshotContext, +) -> Result, Response> { + use crate::jupiter::storage::qualified_metadata_family::SnapshotMetadataFamily; + if !handler.get_context().config().mst2.publication_enabled { + return fixed_root_tree(handler, &ctx.root_tree_oid).await.map(Some); + } + match handler + .get_context() + .snapshot_metadata_family(&ctx.lease_id, true) + .await + .map_err(mst2_error_response)? + { + Some(SnapshotMetadataFamily::Rooted) => Ok(None), + Some(SnapshotMetadataFamily::Generic) => { + fixed_root_tree(handler, &ctx.root_tree_oid).await.map(Some) + } + None => Err(mst2_error_response(SnapshotError::new( + SnapshotErrorCode::SnapshotGone, + "fixed content lease has no permanent storage route", ))), - WalkOutcome::NotDirectory { symlink } => Err(mst2_error_response(SnapshotError::new( - if symlink { - SnapshotErrorCode::SymlinkTraversal + } +} + +#[allow(clippy::result_large_err)] +async fn resolve_file_metadata( + handler: &T, + root_tree: Option<&git_internal::internal::object::tree::Tree>, + ctx: &crate::ceres::snapshot::runtime::SnapshotContext, + path: &str, + expected_digest: Option<&str>, +) -> Result { + if let Some(root_tree) = root_tree { + return resolve_legacy_file_metadata( + handler, + root_tree, + &ctx.built.descriptor.scope, + path, + expected_digest, + ) + .await; + } + use crate::jupiter::storage::qualified_metadata_family::RootedLookupStatus; + let storage = handler.get_context(); + let repository = storage + .rooted_qualified_metadata_writer() + .await + .map_err(internal)?; + let (entry, git_oid) = match repository + .fixed_path_metadata(ctx, path) + .await + .map_err(mst2_error_response)? + { + RootedLookupStatus::File { entry, git_oid } => (entry, git_oid), + RootedLookupStatus::Directory(_) => { + return Err(mst2_error_response(SnapshotError::new( + SnapshotErrorCode::NotDirectory, + format!("{path} is a directory"), + ))); + } + RootedLookupStatus::Absent => { + return Err(mst2_error_response(SnapshotError::new( + SnapshotErrorCode::PathNotFound, + format!("{path} absent in the fixed view"), + ))); + } + RootedLookupStatus::NotDirectory { symlink } => { + return Err(mst2_error_response(SnapshotError::new( + if symlink { + SnapshotErrorCode::SymlinkTraversal + } else { + SnapshotErrorCode::NotDirectory + }, + format!("{path}: intermediate component is not a directory"), + ))); + } + }; + let fs_kind = match entry.kind { + mst2_codec::metapage::EntryKind::Regular => FsKind::Regular, + mst2_codec::metapage::EntryKind::Executable => FsKind::Executable, + mst2_codec::metapage::EntryKind::Symlink => FsKind::Symlink, + mst2_codec::metapage::EntryKind::Directory => { + return Err(mst2_error_response(internal( + "fixed source file has a directory kind", + ))); + } + }; + let oid = git_oid + .split_once(':') + .ok_or_else(|| mst2_error_response(internal("fixed source OID has no hash kind")))? + .1 + .to_owned(); + let file = verified_file_metadata(handler, fs_kind, oid, path, expected_digest).await?; + if file.size != entry.size || file.digest != entry.content_id { + return Err(mst2_error_response(SnapshotError::new( + SnapshotErrorCode::IntegrityError, + "fixed content metadata changed from its certified current source occurrence", + ))); + } + Ok(file) +} + +#[allow(clippy::result_large_err)] +pub(super) async fn resolve_snapshot_file_metadata( + state: &crate::api::MonoApiServiceState, + ctx: &crate::ceres::snapshot::runtime::SnapshotContext, + path: &str, + expected_digest: Option<&str>, +) -> Result { + let handler = state + .api_handler(std::path::Path::new("/")) + .await + .map_err(internal)?; + let root_tree = fixed_content_root(handler.as_ref(), ctx).await?; + resolve_file_metadata( + handler.as_ref(), + root_tree.as_ref(), + ctx, + path, + expected_digest, + ) + .await +} + +#[allow(clippy::result_large_err)] +pub(super) async fn verified_file_metadata( + handler: &T, + fs_kind: FsKind, + oid: String, + path: &str, + expected_digest: Option<&str>, +) -> Result { + let storage = handler.get_context(); + let mut verified = storage + .mono_storage() + .get_verified_blobs(vec![oid.clone()]) + .await + .map_err(|error| { + let code = if matches!( + error, + crate::common::errors::MegaError::ObjStorageInconsistent(_) + ) { + SnapshotErrorCode::IntegrityError } else { - SnapshotErrorCode::NotDirectory - }, - format!("{path}: intermediate component is not a directory"), - ))), + SnapshotErrorCode::Internal + }; + tracing::warn!(error = %error, "fixed content metadata lookup failed"); + mst2_error_response(SnapshotError::new( + code, + "fixed content metadata lookup failed", + )) + })?; + let fact = verified.remove(&oid).ok_or_else(|| { + mst2_error_response(SnapshotError::new( + SnapshotErrorCode::MetadataNotReady, + "fixed object has no current verified size and digest fact", + )) + })?; + let digest: [u8; 32] = fact.raw_sha256.as_slice().try_into().map_err(|_| { + mst2_error_response(SnapshotError::new( + SnapshotErrorCode::IntegrityError, + "invalid verified content digest", + )) + })?; + let size = u64::try_from(fact.size).map_err(|_| { + mst2_error_response(SnapshotError::new( + SnapshotErrorCode::IntegrityError, + "invalid verified content size", + )) + })?; + if fs_kind == FsKind::Symlink && !(1..=4095).contains(&size) { + return Err(mst2_error_response(SnapshotError::new( + SnapshotErrorCode::IntegrityError, + "verified symlink size is outside the fixed filesystem profile", + ))); + } + if let Some(expected) = expected_digest + && expected != format!("sha256:{}", hex_of(&digest)) + { + return Err(mst2_error_response(SnapshotError::new( + SnapshotErrorCode::DigestMismatch, + format!("{path}: content does not match expected_digest"), + ))); + } + Ok(ResolvedFileMetadata { + fs_kind, + oid, + digest, + size, + fact, + }) +} + +pub(super) fn fs_kind_str(k: FsKind) -> &'static str { + match k { + FsKind::Regular => "regular", + FsKind::Executable => "executable", + FsKind::Symlink => "symlink", + FsKind::Directory => "directory", + } +} + +#[allow(clippy::result_large_err)] +async fn read_object( + handler: &T, + file: &ResolvedFileMetadata, + path: &str, +) -> Result, Response> { + if file.size > OBJECT_ITEM_MAX { + return Err(mst2_error_response(SnapshotError::new( + SnapshotErrorCode::ScopeInvalid, + "object exceeds the 256KiB item cap", + ))); + } + let expected_size = file.size as usize; + let mut input = handler + .get_raw_blob_stream_by_hash(&file.oid) + .await + .map_err(|error| { + let code = match error { + crate::common::errors::MegaError::ObjStorageNotFound(_) => { + SnapshotErrorCode::ObjectUnavailable + } + crate::common::errors::MegaError::ObjStorageInconsistent(_) => { + SnapshotErrorCode::IntegrityError + } + _ => SnapshotErrorCode::Internal, + }; + tracing::warn!(error = %error, "fixed-view blob fetch failed"); + mst2_error_response(SnapshotError::new( + code, + "fixed-view content could not be read", + )) + })?; + let mut raw = Vec::with_capacity(expected_size); + let mut hash = Sha256::new(); + while let Some(part) = input.next().await { + let bytes = part.map_err(|error| { + tracing::warn!(error = %error, "fixed-view blob stream failed"); + mst2_error_response(SnapshotError::new( + SnapshotErrorCode::Internal, + "fixed-view content could not be read", + )) + })?; + // Reject oversized producer chunks before copying or hashing them. + if bytes.len() > expected_size - raw.len() { + return Err(mst2_error_response(SnapshotError::new( + SnapshotErrorCode::IntegrityError, + "fixed blob length disagrees with its verified size fact", + ))); + } + hash.update(&bytes); + raw.extend_from_slice(&bytes); } + if raw.len() != expected_size { + return Err(mst2_error_response(SnapshotError::new( + SnapshotErrorCode::IntegrityError, + "fixed blob length disagrees with its verified size fact", + ))); + } + let digest: [u8; 32] = hash.finalize().into(); + if digest != file.digest { + return Err(mst2_error_response(SnapshotError::new( + SnapshotErrorCode::DigestMismatch, + format!("{path}: content does not match expected_digest"), + ))); + } + Ok(raw) } #[derive(Deserialize, Debug)] @@ -115,9 +378,7 @@ pub(super) async fn blob_head( headers: HeaderMap, ) -> Result { ensure(&state)?; - let ctx = runtime() - .context(&snapshot_id) - .map_err(mst2_error_response)?; + let ctx = super::request_context(&state, &snapshot_id).map_err(mst2_error_response)?; validate_scope_relative_path(&q.path).map_err(mst2_error_response)?; if headers.contains_key("range") { return Err(mst2_error_response(SnapshotError::new( @@ -129,14 +390,11 @@ pub(super) async fn blob_head( .api_handler(std::path::Path::new("/")) .await .map_err(internal)?; - let root_tree = handler - .get_tree_by_hash(&ctx.root_tree_oid) - .await - .map_err(internal)?; - let f = resolve_file( + let root_tree = fixed_content_root(handler.as_ref(), &ctx).await?; + let f = resolve_file_metadata( handler.as_ref(), - &root_tree, - &ctx.built.descriptor.scope, + root_tree.as_ref(), + &ctx, &q.path, q.expected_digest.as_deref(), ) @@ -145,6 +403,7 @@ pub(super) async fn blob_head( .header("content-length", f.size.to_string()) .header("etag", format!("\"sha256:{}\"", hex_of(&f.digest))) .header("cache-control", "private, no-cache, no-transform") + .header("vary", "Authorization, Accept") .header("x-mega-fs-kind", fs_kind_str(f.fs_kind)) .header("x-mega-content-size", f.size.to_string()) .body(axum::body::Body::empty()) @@ -174,12 +433,10 @@ const OBJECT_TOTAL_MAX: usize = 8 * 1024 * 1024; pub(super) async fn objects( state: State, AxumPath(snapshot_id): AxumPath, - body: Bytes, + Mst2Bytes(body): Mst2Bytes, ) -> Result { ensure(&state)?; - let ctx = runtime() - .context(&snapshot_id) - .map_err(mst2_error_response)?; + let ctx = super::request_context(&state, &snapshot_id).map_err(mst2_error_response)?; let req: ObjectsRequest = super::parse_json_body(&body)?; if req.items.is_empty() || req.items.len() > OBJECT_MAX_ITEMS { return Err(mst2_error_response(SnapshotError::new( @@ -195,30 +452,24 @@ pub(super) async fn objects( .map_err(mst2_error_response)? .unwrap_or(crate::ceres::snapshot::frame_stream::Encoding::Identity); - // Verify every member at its fixed path before any 200 is produced. - // Members resolve concurrently: a batch is up to 128 files, and a - // sequential S3 read per member dominated cold-mount time. + // Admit the whole fixed-path batch before opening any object body. let handler = state .api_handler(std::path::Path::new("/")) .await .map_err(internal)?; - let root_tree = handler - .get_tree_by_hash(&ctx.root_tree_oid) - .await - .map_err(internal)?; - let scope = ctx.built.descriptor.scope.clone(); + let root_tree = fixed_content_root(handler.as_ref(), &ctx).await?; // Copied references: each per-item future borrows the shared walk state // without moving it (the stream is an FnMut over owned items). let handler_ref = handler.as_ref(); - let root_ref = &root_tree; - let scope_ref = &scope; + let root_ref = root_tree.as_ref(); + let ctx_ref = &ctx; let resolved: Vec> = futures::stream::iter(req.items.clone()) .map(move |item| async move { validate_scope_relative_path(&item.path).map_err(mst2_error_response)?; - let f = resolve_file( + let f = resolve_file_metadata( handler_ref, root_ref, - scope_ref, + ctx_ref, &item.path, Some(&item.expected_digest), ) @@ -237,21 +488,55 @@ pub(super) async fn objects( .buffered(16) .collect() .await; - let mut unique: Vec<([u8; 32], Vec)> = Vec::new(); - let mut seen: Vec<[u8; 32]> = Vec::new(); - let mut logical_bytes = 0u64; + let mut sources: Vec<(ObjectItem, ResolvedFileMetadata)> = Vec::new(); + let mut content_sizes: Vec<([u8; 32], u64)> = Vec::new(); + let mut planned_bytes = 0u64; for pair in resolved { - let (_, f) = pair?; - if !seen.contains(&f.digest) { - if logical_bytes as usize + f.raw.len() > OBJECT_TOTAL_MAX { + let (item, f) = pair?; + if let Some((_, size)) = content_sizes.iter().find(|(digest, _)| *digest == f.digest) { + if *size != f.size { + return Err(mst2_error_response(SnapshotError::new( + SnapshotErrorCode::IntegrityError, + "fixed content digest has conflicting verified sizes", + ))); + } + } else { + if planned_bytes + f.size > OBJECT_TOTAL_MAX as u64 { return Err(mst2_error_response(SnapshotError::new( SnapshotErrorCode::ScopeInvalid, "unique object content exceeds the 8MiB batch cap", ))); } - logical_bytes += f.raw.len() as u64; - seen.push(f.digest); - unique.push((f.digest, f.raw)); + planned_bytes += f.size; + content_sizes.push((f.digest, f.size)); + } + if let Some((_, source)) = sources.iter().find(|(_, source)| source.oid == f.oid) { + if source.digest != f.digest || source.size != f.size { + return Err(mst2_error_response(SnapshotError::new( + SnapshotErrorCode::IntegrityError, + "fixed object has conflicting verified facts", + ))); + } + } else { + sources.push((item, f)); + } + } + let loaded = futures::stream::iter(sources) + .map(|(item, file)| async move { + let raw = read_object(handler_ref, &file, &item.path).await?; + Ok::<_, Response>((file.digest, raw)) + }) + .buffered(16); + tokio::pin!(loaded); + let mut unique: Vec<([u8; 32], Vec)> = Vec::new(); + let mut seen: Vec<[u8; 32]> = Vec::new(); + let mut logical_bytes = 0u64; + while let Some(pair) = loaded.next().await { + let (digest, raw) = pair?; + if !seen.contains(&digest) { + logical_bytes += raw.len() as u64; + seen.push(digest); + unique.push((digest, raw)); } } @@ -259,7 +544,7 @@ pub(super) async fn objects( // then exactly one END. Encoding is negotiated per request. use crate::ceres::snapshot::frame_stream::FrameStream; let mut stream = FrameStream::new(1, encoding); - let mut out: Vec = Vec::new(); + let mut out: Vec> = Vec::new(); let mut frame: Vec<([u8; 32], Vec)> = Vec::new(); let mut frame_raw = 0usize; for (cid, data) in unique { @@ -271,7 +556,7 @@ pub(super) async fn objects( let bytes = stream .object(std::mem::take(&mut frame)) .map_err(mst2_error_response)?; - out.extend_from_slice(&bytes); + out.push(bytes); frame_raw = 0; } frame_raw += data.len(); @@ -279,22 +564,17 @@ pub(super) async fn objects( } if !frame.is_empty() { let bytes = stream.object(frame).map_err(mst2_error_response)?; - out.extend_from_slice(&bytes); + out.push(bytes); } - let end = stream.end( req.items.len() as u32, seen.len() as u32, logical_bytes, sha256_of(&body), ); - out.extend_from_slice(&end); + out.push(end); - axum::response::Response::builder() - .header("content-type", "application/octet-stream") - .header("cache-control", "private, no-cache, no-transform") - .body(axum::body::Body::from(Bytes::from(out))) - .map_err(|e| mst2_error_response(internal(format!("body build failed: {e}")))) + guarded_treeframe_response(&state, &ctx, &snapshot_id, &body, out).map_err(mst2_error_response) } #[derive(Deserialize, Debug)] @@ -304,35 +584,92 @@ pub(super) struct ChunkMapQuery { #[serde(default)] pub(super) expected_digest: Option, #[serde(default)] - pub(super) page: Option, + pub(super) map_id: Option, + #[serde(default)] + pub(super) page_index: Option, } /// Project a fixed-path file into its range-readable representation. #[allow(clippy::result_large_err)] async fn project_for( handler: &T, - root_tree: &git_internal::internal::object::tree::Tree, - scope: &str, + root_tree: Option<&git_internal::internal::object::tree::Tree>, + ctx: &crate::ceres::snapshot::runtime::SnapshotContext, path: &str, expected_digest: Option<&str>, -) -> Result, Response> { - let f = resolve_file(handler, root_tree, scope, path, expected_digest).await?; - // The first request for a digest builds the projection from the fixed - // Git object; later requests slice the cached representation. A miss - // rebuilds, never errors with "missing chunk". - let digest = f.digest; - get_or_project(digest, || async move { - let raw = f.raw; - if raw.len() as u64 != f.size { - return Err(SnapshotError::new( - SnapshotErrorCode::Internal, - "resolved blob size disagrees with its length", - )); - } - Ok(raw) - }) +) -> Result, Response> +{ + let f = resolve_file_metadata(handler, root_tree, ctx, path, expected_digest).await?; + project_resolved(handler, &f).await +} + +#[allow(clippy::result_large_err)] +pub(super) async fn project_resolved( + handler: &T, + f: &ResolvedFileMetadata, +) -> Result, Response> +{ + if f.size == 0 { + return Err(mst2_error_response(SnapshotError::new( + SnapshotErrorCode::ScopeInvalid, + "empty files have no chunk map", + ))); + } + let source = ChunkMapSource::from_fact(f.fact.clone(), &f.oid).map_err(mst2_error_response)?; + let storage = handler.get_context(); + let repository = storage.chunk_maps().await.map_err(mst2_error_response)?; + let objects = &storage.git_service.obj_storage; + if let Some(map) = repository + .read(&source, objects) + .await + .map_err(mst2_error_response)? + { + return Ok(map); + } + let flight = crate::ceres::snapshot::chunk_map_gate::InstallFlight::acquire( + repository + .source_identity(&source) + .map_err(mst2_error_response)?, + ) + .map_err(mst2_error_response)?; + let _gate = flight.lock().await.map_err(mst2_error_response)?; + if let Some(map) = repository + .read(&source, objects) + .await + .map_err(mst2_error_response)? + { + return Ok(map); + } + let admission = repository + .admit_install(&source, objects) + .await + .map_err(mst2_error_response)?; + let verified = VerifiedSourceChunkMap::verify( + handler, + source.clone(), + repository.memory_budget(), + &admission, + ) .await - .map_err(mst2_error_response) + .map_err(|error| { + mst2_error_response(if error.code == SnapshotErrorCode::DigestMismatch { + SnapshotError::new( + SnapshotErrorCode::IntegrityError, + "fixed blob digest disagrees with its verified fact", + ) + } else { + error + }) + })?; + repository + .install(verified, objects, &admission) + .await + .map_err(mst2_error_response)?; + repository + .read(&source, objects) + .await + .map_err(mst2_error_response)? + .ok_or_else(|| mst2_error_response(internal("installed chunk map source is missing"))) } #[allow(clippy::result_large_err)] @@ -342,40 +679,50 @@ pub(super) async fn chunk_map( Query(q): Query, ) -> Result { ensure(&state)?; - let ctx = runtime() - .context(&snapshot_id) - .map_err(mst2_error_response)?; + let ctx = super::request_context(&state, &snapshot_id).map_err(mst2_error_response)?; validate_scope_relative_path(&q.path).map_err(mst2_error_response)?; let handler = state .api_handler(std::path::Path::new("/")) .await .map_err(internal)?; - let root_tree = handler - .get_tree_by_hash(&ctx.root_tree_oid) - .await - .map_err(internal)?; - let scope = ctx.built.descriptor.scope.clone(); + let root_tree = fixed_content_root(handler.as_ref(), &ctx).await?; let proj = project_for( handler.as_ref(), - &root_tree, - &scope, + root_tree.as_ref(), + &ctx, &q.path, q.expected_digest.as_deref(), ) .await?; + let wire_bound = q + .path + .len() + .checked_mul(6) + .and_then(|n| n.checked_add(4096)) + .ok_or_else(|| mst2_error_response(internal("chunk map response bound overflow")))?; + let memory = reserve_response( + wire_bound + .checked_mul(2) + .ok_or_else(|| mst2_error_response(internal("chunk map response credit overflow")))?, + ) + .map_err(mst2_error_response)?; let body = json!({ "snapshot_id": snapshot_id, "path": q.path, - "schema_version": 2, - "file_content_id": format!("sha256:{}", hex_of(&proj.map.file_content_id)), - "file_size": proj.map.file_size.to_string(), - "chunk_size": mst2_codec::chunkmap::CHUNK_SIZE, - "chunk_count": proj.map.chunk_count.to_string(), - "page_count": proj.map.page_count.to_string(), - "pages_root": format!("sha256:{}", hex_of(&proj.map.pages_root)), - "map_id": format!("sha256:{}", hex_of(&proj.map_id)), + "map": { + "schema_version": 2, + "file_content_id": format!("sha256:{}", hex_of(&proj.map.file_content_id)), + "file_size": proj.map.file_size.to_string(), + "chunk_size": mst2_codec::chunkmap::CHUNK_SIZE, + "chunk_count": proj.map.chunk_count.to_string(), + "page_count": proj.map.page_count.to_string(), + "pages_root": format!("sha256:{}", hex_of(&proj.map.pages_root)), + "map_id": format!("sha256:{}", hex_of(&proj.map_id)), + }, }); - Ok(Json(body).into_response()) + guarded_map_json_response(&state, &ctx, &body, memory, proj) + .await + .map_err(mst2_error_response) } #[allow(clippy::result_large_err)] @@ -385,34 +732,56 @@ pub(super) async fn chunk_map_pages( Query(q): Query, ) -> Result { ensure(&state)?; - let ctx = runtime() - .context(&snapshot_id) - .map_err(mst2_error_response)?; + let ctx = super::request_context(&state, &snapshot_id).map_err(mst2_error_response)?; validate_scope_relative_path(&q.path).map_err(mst2_error_response)?; - let page_index: u64 = match q.page.as_deref() { - Some(s) => parse_decimal_count(s, "page").map_err(mst2_error_response)?, - None => 0, - }; + let page_index = q + .page_index + .as_deref() + .map(|s| parse_decimal_count(s, "page_index")) + .transpose() + .map_err(mst2_error_response)? + .unwrap_or(0); + if q.page_index.is_some() != q.map_id.is_some() { + return Err(mst2_error_response(SnapshotError::new( + SnapshotErrorCode::ScopeInvalid, + "canonical page requests require map_id and page_index together", + ))); + } let handler = state .api_handler(std::path::Path::new("/")) .await .map_err(internal)?; - let root_tree = handler - .get_tree_by_hash(&ctx.root_tree_oid) - .await - .map_err(internal)?; - let scope = ctx.built.descriptor.scope.clone(); + let root_tree = fixed_content_root(handler.as_ref(), &ctx).await?; let proj = project_for( handler.as_ref(), - &root_tree, - &scope, + root_tree.as_ref(), + &ctx, &q.path, q.expected_digest.as_deref(), ) .await?; - let (leaf, proof) = proj - .leaf_and_proof(page_index) + if let Some(expected_map) = q.map_id.as_deref() { + let actual_map = format!("sha256:{}", hex_of(&proj.map_id)); + if expected_map != actual_map { + return Err(mst2_error_response(SnapshotError::new( + SnapshotErrorCode::DigestMismatch, + "chunk-map page map_id does not bind to the fixed file", + ))); + } + } + let storage = handler.get_context(); + let page = storage + .chunk_maps() + .await + .map_err(mst2_error_response)? + .selected_page(&proj, page_index) + .await .map_err(mst2_error_response)?; + // Reserve both JSON values and the encoded wire buffer before either + // allocation. The authenticated leaf/proof retain their own lease. + let memory = reserve_response(64 * 1024).map_err(mst2_error_response)?; + let leaf = &page.leaf; + let proof = &page.proof; let leaf_bytes = leaf .encode() .map_err(|e| mst2_error_response(internal(format!("chunk leaf encode failed: {e}"))))?; @@ -429,19 +798,137 @@ pub(super) async fn chunk_map_pages( }) }) .collect(); + // Canonical v3 page responses are closed and carry the encoded leaf + // directly. The map descriptor already authenticated page_count and the + // client verifies this page_index against that fixed descriptor. let body = json!({ - "snapshot_id": snapshot_id, - "path": q.path, "map_id": format!("sha256:{}", hex_of(&proj.map_id)), - "page_count": proj.map.page_count.to_string(), - "leaf": { - "page_index": leaf.page_index.to_string(), - "count": leaf.chunk_sha256.len().to_string(), - "data_base64": base64_of(&leaf_bytes), - }, + "page_index": leaf.page_index.to_string(), + "leaf_base64": base64_of(&leaf_bytes), "proof": proof_json, }); - Ok(Json(body).into_response()) + guarded_map_json_response(&state, &ctx, &body, memory, proj) + .await + .map_err(mst2_error_response) +} + +fn map_json_bytes( + value: &serde_json::Value, + memory: MemoryLease, +) -> Result { + let mut bytes = Vec::new(); + bytes + .try_reserve_exact(memory.bytes / 2) + .map_err(|_| internal("chunk map JSON allocation failed"))?; + let limit = memory.bytes / 2; + let mut writer = BoundedMapJsonWriter { + bytes: &mut bytes, + limit, + }; + serde_json::to_writer(&mut writer, value) + .map_err(|_| internal("chunk map JSON encoding exceeds its owned credit"))?; + if bytes.capacity() > memory.bytes { + return Err(internal( + "chunk map JSON allocation exceeds its owned credit", + )); + } + Ok(bytes::Bytes::from_owner(BudgetedFrame { + bytes, + lease: std::sync::Arc::new(memory), + })) +} + +struct BoundedMapJsonWriter<'a> { + bytes: &'a mut Vec, + limit: usize, +} + +impl std::io::Write for BoundedMapJsonWriter<'_> { + fn write(&mut self, input: &[u8]) -> std::io::Result { + if input.len() > self.limit - self.bytes.len() { + return Err(std::io::Error::other( + "chunk map JSON exceeds its wire bound", + )); + } + self.bytes.extend_from_slice(input); + Ok(input.len()) + } + + fn flush(&mut self) -> std::io::Result<()> { + Ok(()) + } +} + +async fn guarded_map_json_response( + state: &crate::api::MonoApiServiceState, + context: &crate::ceres::snapshot::runtime::SnapshotContext, + value: &serde_json::Value, + memory: MemoryLease, + map: std::sync::Arc, +) -> Result { + let bytes = map_json_bytes(value, memory)?; + let headers = super::REQUEST_HEADERS.try_with(Clone::clone).map_err(|_| { + SnapshotError::new( + SnapshotErrorCode::Unauthenticated, + "request authentication context missing", + ) + })?; + super::revalidate_access(state, context, &headers).await?; + map.ensure_live().await?; + let state = state.clone(); + let context = context.clone(); + let stream = futures::stream::once(async move { + super::revalidate_access(&state, &context, &headers).await?; + map.ensure_live().await?; + map.record_progress().await?; + Ok::<_, SnapshotError>(bytes::Bytes::from_owner(RetainedChunkFrame { + bytes, + _maps: std::sync::Arc::new(vec![map]), + })) + }); + Response::builder() + .header("content-type", "application/json") + .header("cache-control", "private, no-cache, no-transform") + .header("vary", "Authorization, Accept") + .body(axum::body::Body::from_stream(stream)) + .map_err(|_| internal("chunk map JSON response build failed")) +} + +struct RetainedChunkFrame { + bytes: bytes::Bytes, + _maps: std::sync::Arc< + Vec>, + >, +} +impl AsRef<[u8]> for RetainedChunkFrame { + fn as_ref(&self) -> &[u8] { + &self.bytes + } +} + +fn retain_chunk_maps( + response: Response, + maps: Vec>, +) -> Response { + let maps = std::sync::Arc::new(maps); + let (parts, body) = response.into_parts(); + let stream = body.into_data_stream().then(move |frame| { + let maps = maps.clone(); + async move { + let bytes = frame.map_err(std::io::Error::other)?; + for map in maps.iter() { + map.ensure_live().await.map_err(std::io::Error::other)?; + if !bytes.is_empty() { + map.record_progress().await.map_err(std::io::Error::other)?; + } + } + Ok::<_, std::io::Error>(bytes::Bytes::from_owner(RetainedChunkFrame { + bytes, + _maps: maps, + })) + } + }); + Response::from_parts(parts, axum::body::Body::from_stream(stream)) } #[derive(Deserialize, Debug)] @@ -464,21 +951,174 @@ struct ChunkItem { const CHUNKS_MAX_ITEMS: usize = 128; const CHUNKS_TOTAL_MAX: u64 = 128 * 1024 * 1024; +pub(super) struct ChunksBudgets { + pub(super) response: std::sync::Arc, + pub(super) scratch: std::sync::Arc, +} + +impl Default for ChunksBudgets { + fn default() -> Self { + Self { + response: crate::ceres::snapshot::content_budget::response_budget().clone(), + scratch: crate::ceres::snapshot::content_budget::range_budget().clone(), + } + } +} + struct Planned { - projection: std::sync::Arc, + projection: std::sync::Arc, + page: std::sync::Arc, + oid: String, index: u64, } +struct ResolvedChunk { + file: ResolvedFileMetadata, + index: u64, + map_id: String, +} + +pub(super) fn content_read_error(error: crate::common::errors::MegaError) -> SnapshotError { + use crate::common::errors::MegaError; + let code = match &error { + MegaError::ObjStorageNotFound(_) => SnapshotErrorCode::ObjectUnavailable, + MegaError::ObjStorageInconsistent(_) => SnapshotErrorCode::IntegrityError, + MegaError::Io(error) if error.kind() == std::io::ErrorKind::InvalidData => { + SnapshotErrorCode::IntegrityError + } + _ => SnapshotErrorCode::Internal, + }; + tracing::warn!(error = %error, "fixed-view content read failed"); + SnapshotError::new(code, "fixed-view content could not be read") +} + +#[allow(clippy::result_large_err)] +async fn read_chunk_range( + handler: &T, + planned: &Planned, +) -> Result, Response> { + let projection = &planned.projection; + projection + .ensure_live() + .await + .map_err(mst2_error_response)?; + let len = projection.map.chunk_len(planned.index).map_err(|error| { + mst2_error_response(internal(format!("invalid admitted chunk: {error}"))) + })?; + let start = planned + .index + .checked_mul(mst2_codec::chunkmap::CHUNK_SIZE as u64) + .ok_or_else(|| mst2_error_response(internal("chunk offset overflow")))?; + let end = start + .checked_add(len) + .ok_or_else(|| mst2_error_response(internal("chunk end overflow")))?; + let (mut input, meta) = projection + .await_backend(handler.get_raw_blob_range_stream_exact(&planned.oid, start, end)) + .await + .map_err(mst2_error_response)? + .map_err(|error| mst2_error_response(content_read_error(error)))? + .ok_or_else(|| { + mst2_error_response(SnapshotError::new( + SnapshotErrorCode::RangeNotSupported, + "fixed source does not support exact raw ranges", + )) + })?; + projection + .ensure_live() + .await + .map_err(mst2_error_response)?; + if u64::try_from(meta.size).ok() != Some(projection.map.file_size) { + return Err(mst2_error_response(SnapshotError::new( + SnapshotErrorCode::IntegrityError, + "range source size disagrees with the fixed verified fact", + ))); + } + let mut raw = Vec::new(); + raw.try_reserve_exact(len as usize).map_err(|_| { + mst2_error_response(SnapshotError::new( + SnapshotErrorCode::TemporaryUnavailable, + "range allocation could not be admitted", + )) + })?; + let mut empty_parts = 0usize; + loop { + projection + .ensure_live() + .await + .map_err(mst2_error_response)?; + let part = projection + .await_backend(input.next()) + .await + .map_err(mst2_error_response)?; + projection + .ensure_live() + .await + .map_err(mst2_error_response)?; + let Some(part) = part else { break }; + let bytes = part.map_err(|error| { + tracing::warn!(error = %error, "fixed-view range stream failed"); + mst2_error_response(SnapshotError::new( + SnapshotErrorCode::ObjectUnavailable, + "fixed-view range stream failed", + )) + })?; + if bytes.len() > len as usize - raw.len() { + return Err(mst2_error_response(SnapshotError::new( + SnapshotErrorCode::IntegrityError, + "range response exceeds its exact requested length", + ))); + } + raw.extend_from_slice(&bytes); + if !bytes.is_empty() { + projection + .record_progress() + .await + .map_err(mst2_error_response)?; + } else { + empty_parts += 1; + if empty_parts == 32 { + tokio::task::yield_now().await; + projection + .ensure_live() + .await + .map_err(mst2_error_response)?; + empty_parts = 0; + } + } + } + planned + .page + .verify_chunk(&projection.map, planned.index, &raw) + .map_err(mst2_error_response)?; + projection + .ensure_live() + .await + .map_err(mst2_error_response)?; + projection + .record_progress() + .await + .map_err(mst2_error_response)?; + Ok(raw) +} + #[allow(clippy::result_large_err)] pub(super) async fn chunks( + state: State, + path: AxumPath, + body: Mst2Bytes, +) -> Result { + chunks_with_budgets(state, path, body, ChunksBudgets::default()).await +} + +#[allow(clippy::result_large_err)] +pub(super) async fn chunks_with_budgets( state: State, AxumPath(snapshot_id): AxumPath, - body: Bytes, + Mst2Bytes(body): Mst2Bytes, + budgets: ChunksBudgets, ) -> Result { ensure(&state)?; - let ctx = runtime() - .context(&snapshot_id) - .map_err(mst2_error_response)?; + let ctx = super::request_context(&state, &snapshot_id).map_err(mst2_error_response)?; let req: ChunksRequest = super::parse_json_body(&body)?; if req.items.is_empty() || req.items.len() > CHUNKS_MAX_ITEMS { return Err(mst2_error_response(SnapshotError::new( @@ -500,39 +1140,31 @@ pub(super) async fn chunks( .api_handler(std::path::Path::new("/")) .await .map_err(internal)?; - let root_tree = handler - .get_tree_by_hash(&ctx.root_tree_oid) - .await - .map_err(internal)?; - let scope = ctx.built.descriptor.scope.clone(); + let root_tree = fixed_content_root(handler.as_ref(), &ctx).await?; let mut planned: Vec = Vec::new(); + let mut resolved: Vec = Vec::new(); let mut units: Vec<(String, u64)> = Vec::new(); + let mut distinct: std::collections::HashMap<[u8; 32], u64> = std::collections::HashMap::new(); let mut logical_bytes = 0u64; for item in &req.items { validate_scope_relative_path(&item.path).map_err(mst2_error_response)?; let index = parse_decimal_count(&item.chunk_index, "chunk_index").map_err(mst2_error_response)?; - let proj = project_for( + let file = resolve_file_metadata( handler.as_ref(), - &root_tree, - &scope, + root_tree.as_ref(), + &ctx, &item.path, Some(&item.expected_digest), ) .await?; - let want_map = format!("sha256:{}", hex_of(&proj.map_id)); - if want_map != item.map_id { - return Err(mst2_error_response(SnapshotError::new( - SnapshotErrorCode::DigestMismatch, - format!("{}: map_id does not bind to this file", item.path), - ))); - } - if index >= proj.map.chunk_count { + let chunk_count = file.size.div_ceil(mst2_codec::chunkmap::CHUNK_SIZE as u64); + if index >= chunk_count { return Err(mst2_error_response(SnapshotError::new( SnapshotErrorCode::ScopeInvalid, format!( "{}: chunk_index {index} >= chunk_count {}", - item.path, proj.map.chunk_count + item.path, chunk_count ), ))); } @@ -544,10 +1176,8 @@ pub(super) async fn chunks( format!("duplicate chunk unit map_id={} index={index}", item.map_id), ))); } - let len = proj - .map - .chunk_len(index) - .map_err(|e| mst2_error_response(internal(e.to_string())))?; + let start = index * mst2_codec::chunkmap::CHUNK_SIZE as u64; + let len = (file.size - start).min(mst2_codec::chunkmap::CHUNK_SIZE as u64); if logical_bytes + len > CHUNKS_TOTAL_MAX { return Err(mst2_error_response(SnapshotError::new( SnapshotErrorCode::ScopeInvalid, @@ -556,30 +1186,103 @@ pub(super) async fn chunks( } logical_bytes += len; units.push(unit); + match distinct.entry(file.digest) { + std::collections::hash_map::Entry::Occupied(entry) => { + if *entry.get() != file.size { + return Err(mst2_error_response(SnapshotError::new( + SnapshotErrorCode::IntegrityError, + "fixed facts disagree for the same content digest", + ))); + } + } + std::collections::hash_map::Entry::Vacant(entry) => { + // Check the profile before any body I/O. Cold construction + // owns its credits and is dropped after durable installation; + // warm requests retain only descriptors and selected pages. + map_build_reservation_bytes(file.size).map_err(mst2_error_response)?; + entry.insert(file.size); + } + } + resolved.push(ResolvedChunk { + file, + index, + map_id: item.map_id.clone(), + }); + } + let response_bytes = usize::try_from(logical_bytes) + .ok() + .and_then(|bytes| bytes.checked_add(req.items.len() * 1024 + 1024)) + .ok_or_else(|| mst2_error_response(internal("chunk response memory overflow")))?; + let response_memory = budgets + .response + .reserve(response_bytes) + .map_err(mst2_error_response)?; + let mut maps = std::collections::HashMap::new(); + let mut pages = std::collections::HashMap::new(); + for item in resolved { + let source = ChunkMapSource::from_fact(item.file.fact.clone(), &item.file.oid) + .map_err(mst2_error_response)?; + let source_key = source.canonical_bytes().map_err(mst2_error_response)?; + let proj = if let Some(map) = maps.get(&source_key) { + std::sync::Arc::clone(map) + } else { + let map = project_resolved(handler.as_ref(), &item.file).await?; + maps.insert(source_key, std::sync::Arc::clone(&map)); + map + }; + let want_map = format!("sha256:{}", hex_of(&proj.map_id)); + if want_map != item.map_id { + return Err(mst2_error_response(SnapshotError::new( + SnapshotErrorCode::DigestMismatch, + "map_id does not bind to the fixed file", + ))); + } + let page_key = ( + proj.source_id(), + item.index / mst2_codec::chunkmap::CHUNKS_PER_PAGE as u64, + ); + let page = if let Some(page) = pages.get(&page_key) { + std::sync::Arc::clone(page) + } else { + let storage = handler.get_context(); + let page = storage + .chunk_maps() + .await + .map_err(mst2_error_response)? + .selected_page(&proj, page_key.1) + .await + .map_err(mst2_error_response)?; + pages.insert(page_key, std::sync::Arc::clone(&page)); + page + }; planned.push(Planned { projection: proj, - index, + page, + oid: item.file.oid, + index: item.index, }); } use crate::ceres::snapshot::frame_stream::FrameStream; let mut stream = FrameStream::new(1, encoding); - let mut out: Vec = Vec::new(); + let mut out: Vec> = Vec::new(); for p in planned.iter() { - // Re-verified slice (digest + length) from the staged projection. - let bytes = p - .projection - .chunk_bytes(p.index) + let _work_memory = budgets + .scratch + .reserve(crate::ceres::snapshot::content_budget::RANGE_WORK_BYTES) .map_err(mst2_error_response)?; + // The source is the current request's fixed OID, never a cached + // handler/backend/credential from a different scope. + let bytes = read_chunk_range(handler.as_ref(), p).await?; let frame = stream .chunk( p.projection.map_id, p.projection.map.file_content_id, p.index, - bytes.to_vec(), + bytes, ) .map_err(mst2_error_response)?; - out.extend_from_slice(&frame); + out.push(frame); } let end = stream.end( req.items.len() as u32, @@ -587,13 +1290,18 @@ pub(super) async fn chunks( logical_bytes, sha256_of(&body), ); - out.extend_from_slice(&end); + out.push(end); - axum::response::Response::builder() - .header("content-type", "application/octet-stream") - .header("cache-control", "private, no-cache, no-transform") - .body(axum::body::Body::from(Bytes::from(out))) - .map_err(|e| mst2_error_response(internal(format!("body build failed: {e}")))) + let response = guarded_treeframe_response_with_budget( + &state, + &ctx, + &snapshot_id, + &body, + out, + Some(response_memory), + ) + .map_err(mst2_error_response)?; + Ok(retain_chunk_maps(response, maps.into_values().collect())) } /// Strict decimal-string parse for unsigned counts (spec 04 §1: no leading @@ -635,3 +1343,7 @@ fn ensure(state: &crate::api::MonoApiServiceState) -> Result<(), Response> { ))) } } + +#[cfg(test)] +#[path = "snapshot_content_tests.rs"] +mod tests; diff --git a/src/api/router/snapshot_content_tests.rs b/src/api/router/snapshot_content_tests.rs new file mode 100644 index 00000000..906752ff --- /dev/null +++ b/src/api/router/snapshot_content_tests.rs @@ -0,0 +1,2088 @@ +use std::{ + sync::{ + Arc, + atomic::{AtomicBool, AtomicUsize, Ordering}, + }, + time::Duration, +}; + +use axum::{ + Router, + body::{Body, to_bytes}, + http::{Method, Request}, + response::Response, +}; +use base64::{Engine, engine::general_purpose::STANDARD}; +use bytes::Bytes; +use futures::StreamExt; +use git_internal::{ + hash::{HashKind, ObjectHash}, + internal::object::{ + commit::Commit, + tree::{Tree, TreeItem, TreeItemMode}, + }, +}; +use mst2_codec::{ + chunkmap::{CHUNK_SIZE, ChunkLeaf, verify_leaf}, + treeframe::{Frame, parse_stream}, +}; +use sea_orm::{ + ActiveModelTrait, ColumnTrait, ConnectionTrait, EntityTrait, IntoActiveModel, QueryFilter, Set, +}; +use serde_json::{Value, json}; +use sha2::{Digest, Sha256}; +use tower::ServiceExt; + +use crate::{ + api::{ + MonoApiServiceState, oauth::api_store::BrowserSessionStore, + router::snapshot_router::routers, + }, + callisto::{mega_refs, mst2_verified_object, sea_orm_active_enums::PushQueueKindEnum}, + ceres::{ + api_service::{ApiHandler, cache::GitObjectCache, mono_api_service::MonoApiService}, + snapshot::{ + error::SnapshotErrorCode, + pages::{MetadataWalkOutcome, hex_of, resolve_abs_metadata}, + resolver::FsKind, + }, + }, + common::utils::MEGA_BRANCH_NAME, + config::{PushPolicy, testing::isolated_config}, + jupiter::{ + service::{ + git_service::GitService, + push_queue_service::{ + EnqueueRequest, ExecuteOutcome, ExecuteRequest, PushExecContext, PushPayload, + push_operation_id, + }, + }, + storage::{ + base_storage::StorageConnector, + mono_storage::MST2_VERIFICATION_VERSION, + object_storage::{MegaObjectStorageWrapper, build_object_storage}, + push_queue_storage::{ClaimOutcome, EnqueueOutcome}, + }, + tests::{ + TestSchemaGuard, test_db_config, test_redis_manager, test_storage_with_config, + with_test_vault, + }, + }, + orbit_api::{ + error::OrbitResult, + log_storage::{LogManifest, LogStorage}, + object_storage::{ + MegaObjectStorage, ObjectByteStream, ObjectKey, ObjectMeta, ObjectNamespace, + }, + }, +}; + +const TOKEN: &str = "mst2-fixed-content-test"; + +#[path = "snapshot_objects_bounded_tests.rs"] +mod bounded_objects; + +#[path = "snapshot_chunks_bounded_tests.rs"] +mod bounded_chunks; + +#[path = "snapshot_persisted_chunk_map_tests.rs"] +mod persisted_chunk_maps; + +#[path = "snapshot_chunk_map_retention_tests.rs"] +mod chunk_map_retention; + +#[path = "snapshot_raw_blob_tests.rs"] +mod raw_blob; + +#[path = "snapshot_session_tests.rs"] +mod durable_sessions; + +#[path = "snapshot_generation_upgrade_tests.rs"] +mod generation_upgrade; + +#[path = "snapshot_generation_history_fixture.rs"] +mod generation_history_fixture; + +#[path = "snapshot_generation_qualified_fixture.rs"] +mod generation_qualified_fixture; + +#[path = "snapshot_install_capability_fixture.rs"] +mod install_capability_fixture; + +type ReceiptWriteHold = (Arc, Arc); +type ReadOpenHold = ( + Arc, + Arc, + Arc, +); + +struct ReadOpenDrop(Arc); + +impl Drop for ReadOpenDrop { + fn drop(&mut self) { + self.0.fetch_add(1, Ordering::SeqCst); + } +} + +async fn await_read_open_hold(hold: Option) { + if let Some((entered, release, drops)) = hold { + let _owner = ReadOpenDrop(drops); + entered.notify_one(); + release.notified().await; + } +} + +struct NoRootedReuse; + +#[async_trait::async_trait] +impl crate::ceres::snapshot::rooted_metadata_projection::RootedReuseLookup for NoRootedReuse { + async fn lookup_reuse( + &self, + _tree_oid: &str, + _identity: &crate::ceres::snapshot::metadata_install::MetadataInstallIdentity, + ) -> Result< + Option, + crate::ceres::snapshot::error::SnapshotError, + > { + Ok(None) + } +} +#[path = "snapshot_rooted_metadata_tests.rs"] +mod rooted_metadata; + +#[derive(Default)] +struct ReadCounts { + whole: AtomicUsize, + range: AtomicUsize, + bytes: AtomicUsize, + object_fault: std::sync::Mutex>, + object_size_override: std::sync::Mutex>, + chunk_faults: std::sync::Mutex>, + receipt_reads: AtomicUsize, + receipt_writes: AtomicUsize, + receipt_write_fail_after_create: AtomicBool, + receipt_read_failure: AtomicBool, + receipt_read_corruption: std::sync::Mutex>, + receipt_read_meta_size: std::sync::atomic::AtomicI64, + receipt_read_late_error: AtomicBool, + receipt_write_holds: std::sync::Mutex>, + receipt_write_wait_drops: Arc, + receipt_late_create_holds: std::sync::Mutex>, + receipt_inventory_holds: std::sync::Mutex>, + receipt_inventory_calls: AtomicUsize, + receipt_retention_unsupported: AtomicBool, + receipt_deletes: AtomicUsize, + whole_open_holds: std::sync::Mutex>, + range_open_holds: std::sync::Mutex>, + receipt_open_holds: std::sync::Mutex>, +} + +impl ReadCounts { + fn reset(&self) { + self.whole.store(0, Ordering::SeqCst); + self.range.store(0, Ordering::SeqCst); + self.bytes.store(0, Ordering::SeqCst); + self.receipt_reads.store(0, Ordering::SeqCst); + self.receipt_writes.store(0, Ordering::SeqCst); + } + + fn assert(&self, whole: usize, bytes: usize) { + assert_eq!(self.whole.load(Ordering::SeqCst), whole); + assert_eq!(self.range.load(Ordering::SeqCst), 0); + assert_eq!(self.bytes.load(Ordering::SeqCst), bytes); + } +} + +struct CountingStorage { + inner: MegaObjectStorageWrapper, + counts: Arc, +} + +#[async_trait::async_trait] +impl MegaObjectStorage for CountingStorage { + fn supports_chunk_map_retention(&self) -> bool { + !self + .counts + .receipt_retention_unsupported + .load(Ordering::SeqCst) + && self.inner.inner.supports_chunk_map_retention() + } + + async fn chunk_map_receipt_inventory( + &self, + ) -> OrbitResult { + self.counts + .receipt_inventory_calls + .fetch_add(1, Ordering::SeqCst); + if self + .counts + .receipt_retention_unsupported + .load(Ordering::SeqCst) + { + return Err(crate::orbit_api::error::IoOrbitError::ChunkMapRetentionUnsupported); + } + let inventory = self.inner.inner.chunk_map_receipt_inventory().await?; + let holds = self.counts.receipt_inventory_holds.lock().unwrap().clone(); + if let Some((entered, release)) = holds { + entered.notify_one(); + release.notified().await; + } + Ok(inventory) + } + + async fn delete_chunk_map_receipt( + &self, + authority: &crate::orbit_api::object_storage::ChunkMapReceiptDeletion, + ) -> OrbitResult { + let deleted = self.inner.inner.delete_chunk_map_receipt(authority).await?; + if deleted { + self.counts.receipt_deletes.fetch_add(1, Ordering::SeqCst); + } + Ok(deleted) + } + + async fn put_metadata_atomic_create( + &self, + key: &ObjectKey, + bytes: Bytes, + meta: ObjectMeta, + ) -> OrbitResult<()> { + let late = if key.namespace == ObjectNamespace::ChunkMapReceipt { + self.counts + .receipt_late_create_holds + .lock() + .unwrap() + .clone() + } else { + None + }; + if let Some((entered, release)) = late { + let inner = self.inner.clone(); + let key = key.clone(); + let counts = self.counts.clone(); + return tokio::spawn(async move { + entered.notify_one(); + release.notified().await; + inner + .inner + .put_metadata_atomic_create(&key, bytes, meta) + .await?; + counts.receipt_writes.fetch_add(1, Ordering::SeqCst); + Ok(()) + }) + .await + .unwrap(); + } + self.inner + .inner + .put_metadata_atomic_create(key, bytes, meta) + .await?; + if key.namespace == ObjectNamespace::ChunkMapReceipt { + self.counts.receipt_writes.fetch_add(1, Ordering::SeqCst); + let holds = self.counts.receipt_write_holds.lock().unwrap().clone(); + if let Some((entered, release)) = holds { + let _hold = ReadOpenDrop(self.counts.receipt_write_wait_drops.clone()); + entered.notify_one(); + release.notified().await; + } + if self + .counts + .receipt_write_fail_after_create + .swap(false, Ordering::SeqCst) + { + return Err(crate::orbit_api::error::IoOrbitError::Other( + "injected post-create receipt failure".into(), + )); + } + } + Ok(()) + } + + async fn put_stream( + &self, + key: &ObjectKey, + data: ObjectByteStream, + meta: ObjectMeta, + ) -> OrbitResult<()> { + self.inner.inner.put_stream(key, data, meta).await + } + + async fn get_stream(&self, key: &ObjectKey) -> OrbitResult<(ObjectByteStream, ObjectMeta)> { + if key.namespace == ObjectNamespace::ChunkMapReceipt { + self.counts.receipt_reads.fetch_add(1, Ordering::SeqCst); + let holds = self.counts.receipt_open_holds.lock().unwrap().clone(); + await_read_open_hold(holds).await; + if self.counts.receipt_read_failure.load(Ordering::SeqCst) { + return Err( + crate::orbit_api::error::IoOrbitError::object_store_not_found( + key.default_sharding(), + ), + ); + } + let bad = self.counts.receipt_read_corruption.lock().unwrap().clone(); + if let Some(bytes) = bad { + let declared = self.counts.receipt_read_meta_size.load(Ordering::SeqCst); + let size = if declared > 0 { + declared + } else { + bytes.len() as i64 + }; + let mut parts = vec![Ok(bytes)]; + if self.counts.receipt_read_late_error.load(Ordering::SeqCst) { + parts.push(Err(std::io::Error::other( + "injected late receipt read error", + ))); + } + return Ok(( + Box::pin(futures::stream::iter(parts)), + ObjectMeta { + size, + ..Default::default() + }, + )); + } + return self.inner.inner.get_stream(key).await; + } + self.counts.whole.fetch_add(1, Ordering::SeqCst); + let holds = self.counts.whole_open_holds.lock().unwrap().clone(); + await_read_open_hold(holds).await; + let chunk_fault = self + .counts + .chunk_faults + .lock() + .unwrap() + .iter() + .find(|fault| fault.oid == key.key) + .cloned(); + if let Some(fault) = chunk_fault { + let counts = self.counts.clone(); + let size = fault.size; + let stream = fault.full_stream().map(move |part| { + if let Ok(bytes) = &part { + counts.bytes.fetch_add(bytes.len(), Ordering::SeqCst); + } + part + }); + return Ok(( + Box::pin(stream), + ObjectMeta { + size: size as i64, + ..Default::default() + }, + )); + } + let (stream, mut meta) = self.inner.inner.get_stream(key).await?; + if let Some(size) = *self.counts.object_size_override.lock().unwrap() { + meta.size = size; + } + let fault = self.counts.object_fault.lock().unwrap().clone(); + let stream = match fault { + Some(fault) if fault.oid == key.key => fault.stream(), + _ => stream, + }; + let counts = self.counts.clone(); + let stream = stream.map(move |part| { + if let Ok(bytes) = &part { + counts.bytes.fetch_add(bytes.len(), Ordering::SeqCst); + } + part + }); + Ok((Box::pin(stream), meta)) + } + + async fn get_range_stream( + &self, + key: &ObjectKey, + start: u64, + end: Option, + ) -> OrbitResult<(ObjectByteStream, ObjectMeta)> { + self.counts.range.fetch_add(1, Ordering::SeqCst); + self.inner.inner.get_range_stream(key, start, end).await + } + + async fn get_range_stream_exact( + &self, + key: &ObjectKey, + start: u64, + end: u64, + ) -> OrbitResult> { + self.counts.range.fetch_add(1, Ordering::SeqCst); + let holds = self.counts.range_open_holds.lock().unwrap().clone(); + await_read_open_hold(holds).await; + let fault = self + .counts + .chunk_faults + .lock() + .unwrap() + .iter() + .find(|fault| fault.oid == key.key) + .cloned(); + if let Some(fault) = fault { + let range = fault.range_stream(start, end)?; + return Ok(range.map(|(stream, meta)| { + let counts = self.counts.clone(); + let stream = stream.map(move |part| { + if let Ok(bytes) = &part { + counts.bytes.fetch_add(bytes.len(), Ordering::SeqCst); + } + part + }); + (Box::pin(stream) as ObjectByteStream, meta) + })); + } + let result = self + .inner + .inner + .get_range_stream_exact(key, start, end) + .await?; + let fault = self.counts.object_fault.lock().unwrap().clone(); + Ok(result.map(|(stream, meta)| { + let stream = match fault { + Some(fault) if fault.oid == key.key => fault.stream(), + _ => stream, + }; + let counts = self.counts.clone(); + let stream = stream.map(move |part| { + if let Ok(bytes) = &part { + counts.bytes.fetch_add(bytes.len(), Ordering::SeqCst); + } + part + }); + (Box::pin(stream) as ObjectByteStream, meta) + })) + } + + async fn exists(&self, key: &ObjectKey) -> OrbitResult { + self.inner.inner.exists(key).await + } + + async fn signed_url( + &self, + key: &ObjectKey, + method: Method, + expires_in: Duration, + ) -> OrbitResult> { + self.inner.inner.signed_url(key, method, expires_in).await + } + + async fn delete(&self, key: &ObjectKey) -> OrbitResult<()> { + self.inner.inner.delete(key).await + } +} + +#[async_trait::async_trait] +impl LogStorage for CountingStorage { + async fn append( + &self, + key: &ObjectKey, + data: ObjectByteStream, + meta: ObjectMeta, + ) -> OrbitResult<()> { + self.inner.inner.append(key, data, meta).await + } + + async fn read_range( + &self, + key: &ObjectKey, + offset: u64, + length: u64, + ) -> OrbitResult { + self.inner.inner.read_range(key, offset, length).await + } + + async fn read_lines_range( + &self, + key: &ObjectKey, + start_line: u64, + end_line: u64, + ) -> OrbitResult { + self.inner + .inner + .read_lines_range(key, start_line, end_line) + .await + } + + async fn append_concurrently( + &self, + key: &ObjectKey, + data: ObjectByteStream, + meta: ObjectMeta, + ) -> OrbitResult<()> { + self.inner.inner.append_concurrently(key, data, meta).await + } + + async fn load_manifest(&self, key: &ObjectKey) -> OrbitResult { + self.inner.inner.load_manifest(key).await + } + + async fn log_exists(&self, key: &ObjectKey) -> OrbitResult { + self.inner.inner.log_exists(key).await + } +} + +struct Fixture { + app: Router, + state: MonoApiServiceState, + generic_history: bool, + snapshot: String, + lease: String, + oid: String, + raw: Vec, + digest: [u8; 32], + counts: Arc, + _temp: tempfile::TempDir, + _schema: Option, +} + +fn tree(items: Vec) -> Tree { + crate::ceres::view::tree_source::build_tree(HashKind::Sha1, items).unwrap() +} + +fn item(mode: TreeItemMode, oid: ObjectHash, name: &str) -> TreeItem { + TreeItem::new(mode, oid, name.to_string()) +} + +fn digest(bytes: &[u8]) -> [u8; 32] { + Sha256::digest(bytes).into() +} + +async fn publish_native_push( + storage: &crate::jupiter::storage::Storage, + path: &str, + old: ObjectHash, + new: &Commit, +) { + let old_id = old.to_string(); + let new_id = new.id.to_string(); + let payload = PushPayload { + commits: vec![new_id.clone()], + fork_base: Some(old_id.clone()), + n: 1, + }; + let outcome = storage + .push_queue_service + .enqueue(EnqueueRequest { + kind: PushQueueKindEnum::Push, + operation_id: push_operation_id(&old_id, &new_id), + path: path.to_string(), + old_id, + new_id, + requester: None, + payload: payload.to_json(), + ref_name: Some(MEGA_BRANCH_NAME.to_string()), + is_delete: false, + }) + .await + .unwrap(); + let EnqueueOutcome::Inserted { id } = outcome else { + panic!("fresh fixture push must insert: {outcome:?}"); + }; + assert_eq!( + storage + .push_queue_service + .storage() + .claim_for_execution(id) + .await + .unwrap(), + ClaimOutcome::Claimed + ); + let context = PushExecContext { + storage: storage.clone(), + git_object_cache: Arc::new(GitObjectCache { + connection: test_redis_manager().await, + prefix: String::new(), + }), + pre_apply_enter_barrier: None, + pre_apply_release_barrier: None, + }; + let outcome = storage + .push_queue_service + .execute_b3( + ExecuteRequest { + id, + ..Default::default() + }, + None, + None, + Some(&context), + ) + .await + .unwrap(); + assert!(matches!( + outcome, + ExecuteOutcome::Done { + root_cas_writes: 1, + .. + } + )); +} + +impl Fixture { + async fn new() -> Self { + Self::new_with_pg_config(false).await + } + + async fn new_with_pg_config(rebuildable: bool) -> Self { + Self::new_with_pg_config_and_directories(rebuildable, 0).await + } + + async fn new_with_pg_config_and_directories(rebuildable: bool, directory_count: usize) -> Self { + Self::new_with_pg_config_directories_and_objects(rebuildable, directory_count, &[]).await + } + + async fn new_with_pg_config_directories_and_objects( + rebuildable: bool, + directory_count: usize, + objects: &[(String, Vec)], + ) -> Self { + Self::new_in_metadata_family(rebuildable, directory_count, objects, false).await + } + + async fn new_generic_history_with_pg_config(rebuildable: bool) -> Self { + Self::new_generic_history_with_pg_config_and_directories(rebuildable, 0).await + } + + async fn new_generic_history_with_pg_config_and_directories( + rebuildable: bool, + directory_count: usize, + ) -> Self { + Self::new_in_metadata_family(rebuildable, directory_count, &[], true).await + } + + async fn new_in_metadata_family( + rebuildable: bool, + directory_count: usize, + objects: &[(String, Vec)], + generic_history: bool, + ) -> Self { + Self::new_in_publication_mode(rebuildable, directory_count, objects, generic_history, true) + .await + } + + async fn new_without_publication() -> Self { + Self::new_in_publication_mode(true, 0, &[], false, false).await + } + + async fn new_in_publication_mode( + rebuildable: bool, + directory_count: usize, + objects: &[(String, Vec)], + generic_history: bool, + publication_enabled: bool, + ) -> Self { + let temp = tempfile::tempdir().unwrap(); + let mut config = isolated_config(temp.path().join("config")); + config.monorepo.push_policy = PushPolicy::Trunk; + config.mst2.enabled = true; + config.mst2.publication_enabled = publication_enabled; + config.mst2.instance_uuid = Some(uuid::Uuid::new_v4().to_string()); + config.mst2.auth_token = Some(TOKEN.to_string()); + let backend = build_object_storage(&config.object_storage).await.unwrap(); + let counts = Arc::new(ReadCounts::default()); + let (mut storage, schema) = if rebuildable || !generic_history { + let (database, schema) = test_db_config(temp.path()).await; + config.database = database; + let assembly = async { + let connection = + crate::jupiter::storage::init::database_connection(&config.database) + .await + .unwrap(); + crate::jupiter::storage::Storage::new_with_connection( + Arc::new(config), + Arc::new(connection), + backend.clone(), + ) + .await + .unwrap() + }; + let storage = if generic_history { + crate::jupiter::storage::init::with_generic_history_bootstrap(assembly).await + } else { + assembly.await + }; + (storage, Some(schema)) + } else { + (test_storage_with_config(temp.path(), config).await, None) + }; + if generic_history { + let q_rows:i64=storage.mono_storage().get_connection().query_one_raw(sea_orm::Statement::from_string( + sea_orm::DbBackend::Postgres,"SELECT count(*) FROM mst2_metadata_namespace WHERE graph_domain='qualified-v1'")) + .await.unwrap().unwrap().try_get_by_index(0).unwrap(); + assert_eq!( + q_rows, 0, + "G history fixtures must not create or erase actual Q ownership" + ); + } + storage.git_service = GitService { + obj_storage: MegaObjectStorageWrapper::new(Arc::new(CountingStorage { + inner: backend, + counts: counts.clone(), + })), + }; + let storage = with_test_vault(storage, temp.path()).await; + // Unique bytes prevent another test's process-wide digest cache from + // turning a required cold read into a hit. These bytes also look like + // a Git object header, and contain NUL and non-UTF-8 content. + let prefix = format!("blob 3\0abc\0{}", uuid::Uuid::new_v4()); + let mut raw = vec![0xff; CHUNK_SIZE as usize + 113]; + raw[..prefix.len()].copy_from_slice(prefix.as_bytes()); + let oid = storage + .git_service + .save_object_from_raw(Bytes::copy_from_slice(&raw)) + .await + .unwrap(); + let blob_oid = ObjectHash::from_hex_for_kind(HashKind::Sha1, &oid).unwrap(); + let link_oid = storage + .git_service + .save_object_from_raw(Bytes::from_static(b"file")) + .await + .unwrap(); + let link_oid = ObjectHash::from_hex_for_kind(HashKind::Sha1, &link_oid).unwrap(); + let empty_oid = storage + .git_service + .save_object_from_raw(Bytes::new()) + .await + .unwrap(); + let empty_oid = ObjectHash::from_hex_for_kind(HashKind::Sha1, &empty_oid).unwrap(); + let empty_dir = tree(vec![]); + let nested = tree(vec![item(TreeItemMode::Blob, blob_oid, "file")]); + let mut project_items = vec![ + item(TreeItemMode::Blob, blob_oid, "alias"), + item(TreeItemMode::Tree, empty_dir.id, "directory"), + item(TreeItemMode::Blob, empty_oid, "empty"), + item(TreeItemMode::BlobExecutable, blob_oid, "executable"), + item(TreeItemMode::Blob, blob_oid, "file"), + item(TreeItemMode::Link, link_oid, "link"), + item(TreeItemMode::Tree, nested.id, "nested"), + ]; + let mut extra_trees = Vec::new(); + for index in 0..directory_count { + // Distinct names make distinct canonical pages even though all + // directories share the same already-verified immutable blob. + let child = tree(vec![item( + TreeItemMode::Blob, + blob_oid, + &format!("file-{index:03}"), + )]); + project_items.push(item( + TreeItemMode::Tree, + child.id, + &format!("wide-{index:03}"), + )); + extra_trees.push(child); + } + for (name, raw) in objects { + let oid = storage + .git_service + .save_object_from_raw(Bytes::copy_from_slice(raw)) + .await + .unwrap(); + let mut oid = ObjectHash::from_hex_for_kind(HashKind::Sha1, &oid).unwrap(); + let mut mode = TreeItemMode::Blob; + let components: Vec<_> = name.split('/').collect(); + for index in (1..components.len()).rev() { + let child = tree(vec![item(mode, oid, components[index])]); + oid = child.id; + mode = TreeItemMode::Tree; + extra_trees.push(child); + } + project_items.push(item(mode, oid, components[0])); + } + let project = tree(project_items); + let old_tip = Commit::from_tree_id_with_kind( + HashKind::Sha1, + project.id, + vec![], + "fixed content initial path tip", + ) + .unwrap(); + let new_tip = Commit::from_tree_id_with_kind( + HashKind::Sha1, + project.id, + vec![old_tip.id], + "fixed content published path tip", + ) + .unwrap(); + let root = tree(vec![ + item(TreeItemMode::Blob, blob_oid, "outside"), + item(TreeItemMode::Tree, project.id, "project"), + ]); + let commit = Commit::from_tree_id_with_kind( + HashKind::Sha1, + root.id, + vec![], + "fixed content HTTP test", + ) + .unwrap(); + let mono = storage.mono_storage(); + let mut trees = vec![empty_dir, nested, project, root.clone()]; + trees.extend(extra_trees); + mono.save_mega_trees(trees, commit.id, None).await.unwrap(); + mono.save_mega_commits(vec![commit.clone(), old_tip.clone(), new_tip.clone()], None) + .await + .unwrap(); + mono.save_refs( + mega_refs::Model::new( + "/", + MEGA_BRANCH_NAME.to_string(), + commit.id.to_string(), + root.id.to_string(), + false, + ), + None, + ) + .await + .unwrap(); + mono.save_refs( + mega_refs::Model::new( + "/project", + MEGA_BRANCH_NAME.to_string(), + old_tip.id.to_string(), + old_tip.tree_id.to_string(), + false, + ), + None, + ) + .await + .unwrap(); + if publication_enabled { + mono.initialize_native_publication( + storage.config().mst2.instance_uuid.as_deref().unwrap(), + ) + .await + .unwrap(); + publish_native_push(&storage, "/project", old_tip.id, &new_tip).await; + let head = mono + .read_native_publication_head( + storage.config().mst2.instance_uuid.as_deref().unwrap(), + ) + .await + .unwrap(); + assert_eq!(head.token.sequence, 1); + assert!(head.token.certificate.is_some()); + assert_eq!(head.root.commit, commit.id.to_string()); + assert_eq!(head.root.tree, root.id.to_string()); + } + let state = MonoApiServiceState { + entity_store: storage.entity_store.clone(), + storage, + session_store: BrowserSessionStore::Anonymous, + git_object_cache: Arc::new(GitObjectCache { + connection: test_redis_manager().await, + prefix: String::new(), + }), + listen_addr: "127.0.0.1:0".to_string(), + }; + let routes = if generic_history { + crate::api::router::snapshot_router::generic_history_routers(state.clone()) + } else { + routers(state.clone()) + }; + let app = Router::new().nest("/api/v2", routes.with_state(state.clone())); + let response = app + .clone() + .oneshot( + Request::builder() + .method("POST") + .uri("/api/v2/snapshots/resolve") + .header("authorization", format!("Bearer {TOKEN}")) + .header("content-type", "application/json") + .body(Body::from( + json!({"target":{"kind":"latest"},"scope":"/project"}).to_string(), + )) + .unwrap(), + ) + .await + .unwrap(); + let resolved = success_json(response).await; + assert!(counts.whole.load(Ordering::SeqCst) > 0); + assert!( + !mono + .get_verified_blobs(vec![oid.clone()]) + .await + .unwrap() + .is_empty() + ); + counts.reset(); + Self { + app, + state, + snapshot: resolved["descriptor"]["snapshot_id"] + .as_str() + .unwrap() + .to_string(), + lease: resolved["lease_id"].as_str().unwrap().to_string(), + oid, + digest: Sha256::digest(&raw).into(), + raw, + counts, + generic_history, + _temp: temp, + _schema: schema, + } + } + + fn request(&self, method: &str, suffix: &str, body: Body) -> Request { + Request::builder() + .method(method) + .uri(format!("/api/v2/snapshots/{}/{suffix}", self.snapshot)) + .header("authorization", format!("Bearer {TOKEN}")) + .header("x-mega-snapshot-lease", &self.lease) + .header("content-type", "application/json") + .body(body) + .unwrap() + } + + async fn send(&self, method: &str, suffix: &str, body: Body) -> Response { + self.app + .clone() + .oneshot(self.request(method, suffix, body)) + .await + .unwrap() + } + + fn digest_string(&self) -> String { + format!("sha256:{}", hex_of(&self.digest)) + } + + async fn map(&self, path: &str) -> Value { + success_json( + self.send("GET", &format!("chunk-map?path={path}"), Body::empty()) + .await, + ) + .await + } + + async fn fact(&self) -> mst2_verified_object::Model { + self.state + .storage + .mono_storage() + .get_verified_blobs(vec![self.oid.clone()]) + .await + .unwrap() + .remove(&self.oid) + .unwrap() + } + + async fn delete_fact(&self) { + mst2_verified_object::Entity::delete_many() + .filter(mst2_verified_object::Column::GitOid.eq(&self.oid)) + .exec(self.state.storage.mono_storage().get_connection()) + .await + .unwrap(); + } + + async fn replace_fact(&self, fact: mst2_verified_object::Model) { + self.delete_fact().await; + mst2_verified_object::Entity::insert(fact.into_active_model()) + .exec(self.state.storage.mono_storage().get_connection()) + .await + .unwrap(); + } + + async fn write_raw(&self, bytes: Vec) { + // Git SinglePut is create-if-absent. Remove only this isolated + // fixture's object so corruption and repair actually change its bytes. + let objects = &self.state.storage.git_service.obj_storage.inner; + let key = ObjectKey { + namespace: ObjectNamespace::Git, + key: self.oid.clone(), + }; + if objects.exists(&key).await.unwrap() { + objects.delete(&key).await.unwrap(); + } + self.state + .storage + .git_service + .save_object_from_model(bytes.clone(), &self.oid) + .await + .unwrap(); + assert_eq!( + self.state + .storage + .git_service + .get_object_as_bytes(&self.oid) + .await + .unwrap(), + bytes + ); + // Fault setup is outside the HTTP read measurement window. + self.counts.reset(); + } + + fn chunk_body(&self, path: &str, map_id: &str, index: &str) -> Value { + json!({"items":[{ + "path":path,"expected_digest":self.digest_string(), + "map_id":map_id,"chunk_index":index + }],"encoding":"identity"}) + } +} + +async fn success_json(response: Response) -> Value { + let status = response.status(); + let bytes = to_bytes(response.into_body(), 2 * 1024 * 1024) + .await + .unwrap(); + assert_eq!(status.as_u16(), 200, "{}", String::from_utf8_lossy(&bytes)); + serde_json::from_slice(&bytes).unwrap() +} + +async fn error(response: Response, status: u16, code: &str, retryable: bool) { + assert_eq!(response.status().as_u16(), status); + assert_eq!(response.headers()["content-type"], "application/json"); + let request_id = response.headers()["x-request-id"] + .to_str() + .unwrap() + .to_string(); + assert!(!request_id.is_empty()); + let bytes = to_bytes(response.into_body(), 16 * 1024).await.unwrap(); + let value: Value = serde_json::from_slice(&bytes).unwrap(); + assert_eq!(value["error"]["code"], code); + assert_eq!(value["error"]["retryable"], retryable); + assert_eq!(value["error"]["request_id"], request_id); +} + +#[tokio::test] +async fn mst2_rooted_streamed_fact_persists_one_source_pass_and_reuses_warm_alias_facts() { + use crate::ceres::snapshot::rooted_metadata_projection::prepare_rooted_native_metadata; + + let fixture = Fixture::new().await; + let handler = MonoApiService::from(&fixture.state); + let mut raw = vec![0xff; crate::orbit_api::object_storage::OBJECT_STREAM_ITEM_BYTES + 113]; + let prefix = format!("blob 3\0abc\0{}", uuid::Uuid::new_v4()); + raw[..prefix.len()].copy_from_slice(prefix.as_bytes()); + let oid = fixture + .state + .storage + .git_service + .save_object_from_raw(Bytes::copy_from_slice(&raw)) + .await + .unwrap(); + let source = ObjectHash::from_hex_for_kind(HashKind::Sha1, &oid).unwrap(); + let root = tree(vec![ + item(TreeItemMode::Blob, source, "alias"), + item(TreeItemMode::Blob, source, "file"), + ]); + let mono = fixture.state.storage.mono_storage(); + assert!( + mono.get_verified_blobs(vec![oid.clone()]) + .await + .unwrap() + .is_empty() + ); + fixture.counts.reset(); + let first = prepare_rooted_native_metadata(&handler, &root, "/", &NoRootedReuse) + .await + .unwrap(); + fixture.counts.assert(1, raw.len()); + assert_eq!(first.work.blob_fetches, 1); + assert_eq!(first.work.raw_bytes_fetched, raw.len() as u64); + assert_eq!(first.work.raw_bytes_hashed, raw.len() as u64); + assert_eq!(first.work.verified_blob_persistence_batches, 1); + let stored = mono.get_verified_blobs(vec![oid.clone()]).await.unwrap(); + assert_eq!(stored[&oid].size, raw.len() as i64); + assert_eq!(stored[&oid].raw_sha256, Sha256::digest(&raw).to_vec()); + fixture.counts.reset(); + let warm = prepare_rooted_native_metadata(&handler, &root, "/", &NoRootedReuse) + .await + .unwrap(); + fixture.counts.assert(0, 0); + assert_eq!(warm.plan.identity, first.plan.identity); + assert_eq!(warm.plan.root, first.plan.root); + assert_eq!(warm.work.blob_fetches, 0); + assert_eq!(warm.work.raw_bytes_fetched, 0); + assert_eq!(warm.work.raw_bytes_hashed, 0); + assert_eq!(warm.work.verified_blob_persistence_batches, 0); + raw.push(b'2'); + let next_oid = fixture + .state + .storage + .git_service + .save_object_from_raw(Bytes::copy_from_slice(&raw)) + .await + .unwrap(); + let next_source = ObjectHash::from_hex_for_kind(HashKind::Sha1, &next_oid).unwrap(); + let next_root = tree(vec![ + item(TreeItemMode::Blob, source, "alias"), + item(TreeItemMode::Blob, next_source, "file"), + ]); + fixture.counts.reset(); + let next = prepare_rooted_native_metadata(&handler, &next_root, "/", &NoRootedReuse) + .await + .unwrap(); + fixture.counts.assert(1, raw.len()); + assert_ne!( + next.plan.identity.tagged_root_tree_oid, + first.plan.identity.tagged_root_tree_oid + ); + assert_ne!(next.plan.root, first.plan.root); + assert_eq!(next.work.blob_fetches, 1); + assert_eq!(next.work.raw_bytes_fetched, raw.len() as u64); + assert_eq!(next.work.raw_bytes_hashed, raw.len() as u64); + assert_eq!(next.work.verified_blob_persistence_batches, 1); + assert_eq!(next.work.verified_blob_facts_loaded, 2); +} + +#[tokio::test] +async fn mst2_rooted_streamed_fact_never_persists_truncated_or_late_failed_sources() { + use crate::ceres::snapshot::rooted_metadata_projection::prepare_rooted_native_metadata; + + let fixture = Fixture::new().await; + let handler = MonoApiService::from(&fixture.state); + for late_error in [false, true] { + let raw = Bytes::from(format!("raw\0{}", uuid::Uuid::new_v4())); + let oid = fixture + .state + .storage + .git_service + .save_object_from_raw(raw.clone()) + .await + .unwrap(); + let source = ObjectHash::from_hex_for_kind(HashKind::Sha1, &oid).unwrap(); + let root = tree(vec![item(TreeItemMode::Blob, source, "file")]); + *fixture.counts.object_fault.lock().unwrap() = Some(bounded_objects::StreamFault { + oid: oid.clone(), + kind: if late_error { + bounded_objects::FaultKind::LateError(raw.clone()) + } else { + bounded_objects::FaultKind::Parts(vec![raw.slice(..raw.len() - 1)]) + }, + }); + fixture.counts.reset(); + let error = prepare_rooted_native_metadata(&handler, &root, "/", &NoRootedReuse) + .await + .unwrap_err(); + assert_eq!( + error.code, + if late_error { + SnapshotErrorCode::ObjectUnavailable + } else { + SnapshotErrorCode::IntegrityError + } + ); + fixture + .counts + .assert(1, raw.len() - usize::from(!late_error)); + assert!( + fixture + .state + .storage + .mono_storage() + .get_verified_blobs(vec![oid]) + .await + .unwrap() + .is_empty() + ); + } +} + +#[tokio::test] +async fn mst2_capabilities_use_actual_backend_without_inventory_and_preserve_metadata_and_objects() +{ + let fixture = Fixture::new().await; + let inventory_before = fixture + .counts + .receipt_inventory_calls + .load(Ordering::SeqCst); + let get_capabilities = || async { + success_json( + fixture + .app + .clone() + .oneshot( + Request::builder() + .uri("/api/v2/snapshots/capabilities") + .body(Body::empty()) + .unwrap(), + ) + .await + .unwrap(), + ) + .await + }; + let supported = get_capabilities().await; + for feature in ["raw_blob", "small_objects", "chunk_reads", "full_hydration"] { + assert_eq!(supported["features"][feature], true); + } + fixture + .counts + .receipt_retention_unsupported + .store(true, Ordering::SeqCst); + let mut unsupported = supported.clone(); + for feature in ["raw_blob", "chunk_reads", "full_hydration"] { + unsupported["features"][feature] = json!(false); + } + for _ in 0..3 { + assert_eq!(get_capabilities().await, unsupported); + } + fixture.counts.assert(0, 0); + assert_eq!( + fixture + .counts + .receipt_inventory_calls + .load(Ordering::SeqCst), + inventory_before + ); + assert_eq!(fixture.counts.receipt_reads.load(Ordering::SeqCst), 0); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 0); + let head = fixture.send("HEAD", "blob?path=/file", Body::empty()).await; + assert_eq!(head.status(), 200); + assert_eq!( + head.headers()["content-length"], + fixture.raw.len().to_string() + ); + assert_eq!( + head.headers()["etag"], + format!("\"{}\"", fixture.digest_string()) + ); + assert!(to_bytes(head.into_body(), 1024).await.unwrap().is_empty()); + success_json(fixture.send("GET", "directory?path=/", Body::empty()).await).await; + fixture.counts.assert(0, 0); + for path in ["blob?path=/file", "chunk-map?path=/file"] { + error( + fixture.send("GET", path, Body::empty()).await, + 503, + "TEMPORARY_UNAVAILABLE", + true, + ) + .await; + } + fixture.counts.assert(0, 0); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 0); + let link_digest = digest(b"file"); + let request = json!({"items":[{"path":"/link", "expected_digest":format!("sha256:{}", hex_of(&link_digest))}], "encoding":"identity"}).to_string(); + let response = fixture + .send("POST", "objects", Body::from(request.clone())) + .await; + assert_eq!(response.status(), 200); + let wire = to_bytes(response.into_body(), 2 * 1024 * 1024) + .await + .unwrap(); + let frames = parse_stream(&wire).unwrap(); + let [Frame::Object(object), Frame::End(end)] = frames.as_slice() else { + panic!("expected OBJECT and terminal END"); + }; + assert_eq!(object.objects, vec![(link_digest, b"file".to_vec())]); + assert_eq!(end.request_item_count, 1); + assert_eq!(end.logical_bytes, 4); + assert_eq!( + end.request_body_sha256, + <[u8; 32]>::from(Sha256::digest(request.as_bytes())) + ); + fixture.counts.assert(1, 4); + fixture + .counts + .receipt_retention_unsupported + .store(false, Ordering::SeqCst); + assert_eq!(get_capabilities().await, supported); + fixture.counts.assert(1, 4); +} + +#[tokio::test] +async fn mst2_fixed_head_uses_verified_facts_without_body_reads_and_preserves_raw_bytes() { + let fixture = Fixture::new().await; + for (path, kind, size, digest) in [ + ( + "/file", + "regular", + fixture.raw.len(), + fixture.digest_string(), + ), + ( + "/alias", + "regular", + fixture.raw.len(), + fixture.digest_string(), + ), + ( + "/executable", + "executable", + fixture.raw.len(), + fixture.digest_string(), + ), + ( + "/empty", + "regular", + 0, + format!("sha256:{}", hex_of(&digest(&[]))), + ), + ( + "/link", + "symlink", + 4, + format!("sha256:{}", hex_of(&digest(b"file"))), + ), + ] { + let response = fixture + .send("HEAD", &format!("blob?path={path}"), Body::empty()) + .await; + assert_eq!(response.status(), 200); + assert_eq!(response.headers()["content-length"], size.to_string()); + assert_eq!(response.headers()["x-mega-content-size"], size.to_string()); + assert_eq!(response.headers()["x-mega-fs-kind"], kind); + assert_eq!(response.headers()["etag"], format!("\"{digest}\"")); + assert_eq!(response.headers()["vary"], "Authorization, Accept"); + assert_eq!( + response.headers()["cache-control"], + "private, no-cache, no-transform" + ); + assert!( + to_bytes(response.into_body(), 1024) + .await + .unwrap() + .is_empty() + ); + fixture.counts.assert(0, 0); + } + let response = fixture.send("GET", "blob?path=/file", Body::empty()).await; + assert_eq!(response.status(), 200); + let bytes = to_bytes(response.into_body(), 2 * 1024 * 1024) + .await + .unwrap(); + assert_eq!(bytes.as_ref(), fixture.raw); + fixture.counts.assert(2, 2 * fixture.raw.len()); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 1); +} + +#[tokio::test] +async fn mst2_publication_disabled_resolve_preserves_head_body_object_map_page_and_chunk_content() { + let fixture = Fixture::new_without_publication().await; + assert!(!fixture.state.storage.config().mst2.publication_enabled); + assert_eq!( + fixture + .state + .storage + .snapshot_metadata_family(&fixture.lease, true) + .await + .unwrap(), + None + ); + let routes: i64 = fixture + .state + .storage + .mono_storage() + .get_connection() + .query_one_raw(sea_orm::Statement::from_string( + sea_orm::DbBackend::Postgres, + "SELECT count(*) FROM mst2_snapshot_storage_route", + )) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap(); + assert_eq!( + routes, 0, + "runtime resolve does not create a permanent SID route" + ); + let head = fixture.send("HEAD", "blob?path=/file", Body::empty()).await; + assert_eq!(head.status(), 200); + assert_eq!( + head.headers()["content-length"], + fixture.raw.len().to_string() + ); + assert_eq!( + head.headers()["etag"], + format!("\"{}\"", fixture.digest_string()) + ); + assert!(to_bytes(head.into_body(), 1024).await.unwrap().is_empty()); + fixture.counts.assert(0, 0); + let body = fixture.send("GET", "blob?path=/file", Body::empty()).await; + assert_eq!(body.status(), 200); + assert_eq!( + to_bytes(body.into_body(), 2 * 1024 * 1024) + .await + .unwrap() + .as_ref(), + fixture.raw + ); + let link_digest = digest(b"file"); + let request=json!({"items":[{"path":"/link","expected_digest":format!("sha256:{}",hex_of(&link_digest))}],"encoding":"identity"}).to_string(); + let objects = fixture + .send("POST", "objects", Body::from(request.clone())) + .await; + assert_eq!(objects.status(), 200); + let wire = to_bytes(objects.into_body(), 2 * 1024 * 1024) + .await + .unwrap(); + let frames = parse_stream(&wire).unwrap(); + let [Frame::Object(object), Frame::End(end)] = frames.as_slice() else { + panic!("expected OBJECT and terminal END"); + }; + assert_eq!(object.objects, vec![(link_digest, b"file".to_vec())]); + assert_eq!(end.request_item_count, 1); + assert_eq!(end.logical_bytes, 4); + assert_eq!( + end.request_body_sha256, + <[u8; 32]>::from(Sha256::digest(request.as_bytes())) + ); + let map = fixture.map("/file").await; + let map_id = map["map"]["map_id"].as_str().unwrap(); + assert_eq!(map["map"]["file_content_id"], fixture.digest_string()); + let page = success_json( + fixture + .send( + "GET", + &format!("chunk-map/pages?path=/file&map_id={map_id}&page_index=0"), + Body::empty(), + ) + .await, + ) + .await; + let leaf = ChunkLeaf::decode( + &STANDARD + .decode(page["leaf_base64"].as_str().unwrap()) + .unwrap(), + ) + .unwrap(); + let hashes: Vec<[u8; 32]> = fixture + .raw + .chunks(CHUNK_SIZE as usize) + .map(|bytes| Sha256::digest(bytes).into()) + .collect(); + assert_eq!(leaf.chunk_sha256, hashes); + let chunks = fixture + .send( + "POST", + "chunks", + Body::from(fixture.chunk_body("/file", map_id, "0").to_string()), + ) + .await; + assert_eq!(chunks.status(), 200); + let wire = to_bytes(chunks.into_body(), 2 * 1024 * 1024).await.unwrap(); + let frames = parse_stream(&wire).unwrap(); + let [Frame::Chunk(chunk), Frame::End(end)] = frames.as_slice() else { + panic!("expected CHUNK and terminal END"); + }; + assert_eq!(chunk.chunk_bytes, &fixture.raw[..CHUNK_SIZE as usize]); + assert_eq!(chunk.file_content_id, fixture.digest); + assert_eq!(chunk.chunk_index, 0); + assert_eq!(end.request_item_count, 1); + assert_eq!(end.logical_bytes, u64::from(CHUNK_SIZE)); +} + +#[tokio::test] +async fn mst2_fixed_warm_map_and_leaf_aliases_skip_body_reads_and_chunks_use_current_ranges() { + let fixture = Fixture::new().await; + let initial = fixture.map("/file").await; + fixture.counts.assert(1, fixture.raw.len()); + assert_eq!(initial["map"]["file_content_id"], fixture.digest_string()); + assert_eq!(initial["map"]["file_size"], fixture.raw.len().to_string()); + assert_eq!(initial["map"]["chunk_count"], "2"); + let map_id = initial["map"]["map_id"].as_str().unwrap(); + fixture.counts.reset(); + for path in ["/file", "/alias", "/executable", "/nested/file"] { + let cached = fixture.map(path).await; + assert_eq!(cached["path"], path); + assert_eq!(cached["map"], initial["map"]); + let leaf_response = fixture + .send( + "GET", + &format!("chunk-map/pages?path={path}&map_id={map_id}&page_index=0"), + Body::empty(), + ) + .await; + let value = success_json(leaf_response).await; + let leaf = ChunkLeaf::decode( + &STANDARD + .decode(value["leaf_base64"].as_str().unwrap()) + .unwrap(), + ) + .unwrap(); + assert_eq!(leaf.page_index, 0); + let hashes: Vec<[u8; 32]> = fixture + .raw + .chunks(CHUNK_SIZE as usize) + .map(|bytes| Sha256::digest(bytes).into()) + .collect(); + assert_eq!(leaf.chunk_sha256, hashes); + assert_eq!(value["proof"], json!([])); + assert_eq!( + initial["map"]["pages_root"], + format!("sha256:{}", hex_of(&leaf.leaf_hash().unwrap())) + ); + verify_leaf( + 1, + 0, + leaf.leaf_hash().unwrap(), + &[], + leaf.leaf_hash().unwrap(), + ) + .unwrap(); + for index in 0..2 { + let body = fixture + .chunk_body(path, map_id, &index.to_string()) + .to_string(); + let response = fixture + .send("POST", "chunks", Body::from(body.clone())) + .await; + assert_eq!(response.status(), 200); + let wire = to_bytes(response.into_body(), 2 * 1024 * 1024) + .await + .unwrap(); + let frames = parse_stream(&wire).unwrap(); + let [Frame::Chunk(chunk), Frame::End(end)] = frames.as_slice() else { + panic!("expected one CHUNK and authenticated END"); + }; + let begin = index as usize * CHUNK_SIZE as usize; + let want = &fixture.raw[begin..fixture.raw.len().min(begin + CHUNK_SIZE as usize)]; + assert_eq!(chunk.chunk_index, index); + assert_eq!(chunk.file_content_id, fixture.digest); + assert_eq!(format!("sha256:{}", hex_of(&chunk.map_id)), map_id); + assert_eq!(chunk.chunk_bytes, want); + assert_eq!(end.request_item_count, 1); + assert_eq!(end.unique_unit_count, 1); + assert_eq!(end.logical_bytes, want.len() as u64); + assert_eq!( + end.request_body_sha256, + <[u8; 32]>::from(Sha256::digest(body.as_bytes())) + ); + } + assert_eq!(fixture.counts.whole.load(Ordering::SeqCst), 0); + assert_eq!(fixture.counts.range.load(Ordering::SeqCst), 2); + assert_eq!( + fixture.counts.bytes.load(Ordering::SeqCst), + fixture.raw.len() + ); + fixture.counts.reset(); + } + fixture + .state + .storage + .git_service + .obj_storage + .inner + .delete(&ObjectKey { + namespace: ObjectNamespace::Git, + key: fixture.oid.clone(), + }) + .await + .unwrap(); + assert_eq!(fixture.map("/alias").await["map"], initial["map"]); + fixture.counts.assert(0, 0); + error( + fixture + .send( + "POST", + "chunks", + Body::from(fixture.chunk_body("/alias", map_id, "0").to_string()), + ) + .await, + 503, + "OBJECT_UNAVAILABLE", + false, + ) + .await; + assert_eq!(fixture.counts.whole.load(Ordering::SeqCst), 0); + assert_eq!(fixture.counts.range.load(Ordering::SeqCst), 1); +} + +#[tokio::test] +async fn mst2_fixed_missing_and_legacy_facts_never_backfill_or_use_warm_cache() { + let fixture = Fixture::new().await; + fixture.map("/file").await; + let original = fixture.fact().await; + fixture.counts.reset(); + fixture.delete_fact().await; + error( + fixture + .send("GET", "chunk-map?path=/file", Body::empty()) + .await, + 503, + "METADATA_NOT_READY", + true, + ) + .await; + let response = fixture.send("HEAD", "blob?path=/file", Body::empty()).await; + assert_eq!(response.status(), 503); + assert!( + to_bytes(response.into_body(), 1024) + .await + .unwrap() + .is_empty() + ); + fixture.counts.assert(0, 0); + for (domain, kind, generation) in [ + ("git", "blob", 1), + ("other", "blob", MST2_VERIFICATION_VERSION), + ("git", "tree", MST2_VERIFICATION_VERSION), + ] { + let mut fact = original.clone(); + fact.storage_domain = domain.to_string(); + fact.object_kind = kind.to_string(); + fact.verification_version = generation; + fixture.replace_fact(fact).await; + error( + fixture + .send("GET", "chunk-map?path=/alias", Body::empty()) + .await, + 503, + "METADATA_NOT_READY", + true, + ) + .await; + fixture.counts.assert(0, 0); + } + fixture.replace_fact(original).await; + fixture.map("/file").await; + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn mst2_fixed_invalid_facts_and_cached_size_conflict_fail_before_body_read() { + let fixture = Fixture::new().await; + fixture.map("/file").await; + let original = fixture.fact().await; + fixture.counts.reset(); + for case in 0..6 { + let mut fact = original.clone(); + match case { + 0 => fact.state = "PENDING".to_string(), + 1 => fact.verification_version = MST2_VERIFICATION_VERSION + 1, + 2 => fact.size = -1, + 3 => fact.size = 8_796_093_022_209, + 4 => { + fact.raw_sha256.pop().unwrap(); + } + 5 => { + fact.size += 1; + } + _ => unreachable!(), + }; + fixture.replace_fact(fact).await; + error( + fixture + .send("GET", "chunk-map?path=/file", Body::empty()) + .await, + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + fixture.counts.assert(0, 0); + } + fixture.replace_fact(original).await; + fixture.map("/file").await; + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn mst2_fixed_verified_fact_db_error_is_not_missing_metadata_or_empty_content() { + let fixture = Fixture::new().await; + let mono = fixture.state.storage.mono_storage(); + mono.get_connection() + .execute_unprepared( + "ALTER TABLE mst2_verified_object RENAME TO mst2_verified_object_unavailable", + ) + .await + .unwrap(); + error( + fixture + .send("GET", "chunk-map?path=/file", Body::empty()) + .await, + 500, + "INTERNAL", + true, + ) + .await; + fixture.counts.assert(0, 0); + mono.get_connection() + .execute_unprepared( + "ALTER TABLE mst2_verified_object_unavailable RENAME TO mst2_verified_object", + ) + .await + .unwrap(); + fixture.map("/file").await; + fixture.counts.assert(1, fixture.raw.len()); +} + +#[tokio::test] +async fn mst2_fixed_symlink_facts_outside_profile_fail_without_body_reads() { + let fixture = Fixture::new().await; + let mono = fixture.state.storage.mono_storage(); + let main = mono.get_main_ref("/").await.unwrap().unwrap(); + let handler = MonoApiService::from(&fixture.state); + let root = handler.get_tree_by_hash(&main.ref_tree_hash).await.unwrap(); + let MetadataWalkOutcome::FoundFile { oid, .. } = + resolve_abs_metadata(&handler, &root, "/project/link") + .await + .unwrap() + else { + panic!("fixture link must be a fixed file"); + }; + let original = mono + .get_verified_blobs(vec![oid.clone()]) + .await + .unwrap() + .remove(&oid) + .unwrap(); + for size in [0, 4096] { + let mut fact = original.clone().into_active_model(); + fact.size = Set(size); + fact.update(mono.get_connection()).await.unwrap(); + error( + fixture + .send("GET", "chunk-map?path=/link", Body::empty()) + .await, + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + fixture.counts.assert(0, 0); + } + let mut restored = original.into_active_model(); + restored.size = Set(4); + restored.update(mono.get_connection()).await.unwrap(); + let response = fixture.send("HEAD", "blob?path=/link", Body::empty()).await; + assert_eq!(response.status(), 200); + assert_eq!(response.headers()["x-mega-content-size"], "4"); + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn mst2_fixed_digest_and_path_semantics_are_checked_before_projection_cache() { + let fixture = Fixture::new().await; + let wrong = format!("sha256:{}", "0".repeat(64)); + error( + fixture + .send( + "GET", + &format!("chunk-map?path=/file&expected_digest={wrong}"), + Body::empty(), + ) + .await, + 409, + "EXPECTED_DIGEST_MISMATCH", + false, + ) + .await; + fixture.counts.assert(0, 0); + fixture.map("/file").await; + fixture.counts.reset(); + for (path, status, code) in [ + ("/absent", 404, "PATH_NOT_FOUND"), + ("/outside", 404, "PATH_NOT_FOUND"), + ("/directory", 409, "NOT_DIRECTORY"), + ("/file/child", 409, "NOT_DIRECTORY"), + ("/link/child", 409, "SYMLINK_TRAVERSAL"), + ("/../outside", 400, "SCOPE_INVALID"), + ] { + error( + fixture + .send( + "GET", + &format!( + "chunk-map?path={path}&expected_digest={}", + fixture.digest_string() + ), + Body::empty(), + ) + .await, + status, + code, + false, + ) + .await; + fixture.counts.assert(0, 0); + } + error( + fixture + .send( + "GET", + &format!("chunk-map?path=/alias&expected_digest={wrong}"), + Body::empty(), + ) + .await, + 409, + "EXPECTED_DIGEST_MISMATCH", + false, + ) + .await; + let mut request = fixture.request("HEAD", "blob?path=/file", Body::empty()); + request + .headers_mut() + .insert("range", "bytes=0-9".parse().unwrap()); + assert_eq!( + fixture.app.clone().oneshot(request).await.unwrap().status(), + 400 + ); + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn mst2_fixed_cold_corrupt_or_missing_bodies_publish_no_projection_and_retry() { + for mode in 0..3 { + let fixture = Fixture::new().await; + match mode { + 0 => { + let mut corrupt = fixture.raw.clone(); + corrupt[0] ^= 1; + fixture.write_raw(corrupt).await; + } + 1 => { + fixture + .write_raw(fixture.raw[..fixture.raw.len() - 1].to_vec()) + .await + } + 2 => fixture + .state + .storage + .git_service + .obj_storage + .inner + .delete(&ObjectKey { + namespace: ObjectNamespace::Git, + key: fixture.oid.clone(), + }) + .await + .unwrap(), + _ => unreachable!(), + } + let (status, code, bytes) = match mode { + 0 => (502, "INTEGRITY_ERROR", fixture.raw.len()), + 1 => (502, "INTEGRITY_ERROR", fixture.raw.len() - 1), + 2 => (503, "OBJECT_UNAVAILABLE", 0), + _ => unreachable!(), + }; + error( + fixture + .send("GET", "chunk-map?path=/file", Body::empty()) + .await, + status, + code, + false, + ) + .await; + fixture.counts.assert(1, bytes); + fixture.write_raw(fixture.raw.clone()).await; + fixture.counts.reset(); + fixture.map("/file").await; + fixture.counts.assert(1, fixture.raw.len()); + fixture.counts.reset(); + fixture.map("/alias").await; + fixture.counts.assert(0, 0); + } +} + +#[tokio::test] +async fn mst2_fixed_warm_map_still_checks_map_indices_batches_and_http_lease() { + let fixture = Fixture::new().await; + let map = fixture.map("/file").await; + let map_id = map["map"]["map_id"].as_str().unwrap(); + let wrong_map = format!("sha256:{}", "1".repeat(64)); + fixture.counts.reset(); + error( + fixture + .send( + "GET", + &format!("chunk-map/pages?path=/file&map_id={wrong_map}&page_index=0"), + Body::empty(), + ) + .await, + 409, + "EXPECTED_DIGEST_MISMATCH", + false, + ) + .await; + error( + fixture + .send( + "GET", + &format!("chunk-map/pages?path=/file&map_id={map_id}&page_index=1"), + Body::empty(), + ) + .await, + 404, + "PATH_NOT_FOUND", + false, + ) + .await; + for (id, index, status, code) in [ + (wrong_map.as_str(), "0", 409, "EXPECTED_DIGEST_MISMATCH"), + (map_id, "2", 400, "SCOPE_INVALID"), + ] { + error( + fixture + .send( + "POST", + "chunks", + Body::from(fixture.chunk_body("/file", id, index).to_string()), + ) + .await, + status, + code, + false, + ) + .await; + } + let mut duplicate = fixture.chunk_body("/file", map_id, "0"); + let entry = duplicate["items"][0].clone(); + duplicate["items"].as_array_mut().unwrap().push(entry); + error( + fixture + .send("POST", "chunks", Body::from(duplicate.to_string())) + .await, + 400, + "SCOPE_INVALID", + false, + ) + .await; + let mut request = fixture.request("GET", "chunk-map?path=/file", Body::empty()); + request.headers_mut().remove("authorization"); + error( + fixture.app.clone().oneshot(request).await.unwrap(), + 401, + "UNAUTHENTICATED", + false, + ) + .await; + let mut request = fixture.request("GET", "chunk-map?path=/file", Body::empty()); + request.headers_mut().remove("x-mega-snapshot-lease"); + error( + fixture.app.clone().oneshot(request).await.unwrap(), + 401, + "UNAUTHENTICATED", + false, + ) + .await; + let response = fixture + .app + .clone() + .oneshot( + Request::builder() + .method("DELETE") + .uri(format!("/api/v2/snapshots/leases/{}", fixture.lease)) + .header("authorization", format!("Bearer {TOKEN}")) + .body(Body::empty()) + .unwrap(), + ) + .await + .unwrap(); + assert_eq!(response.status(), 200); + error( + fixture + .send("GET", "chunk-map?path=/file", Body::empty()) + .await, + 410, + "LEASE_EXPIRED", + false, + ) + .await; + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn mst2_fixed_old_snapshot_reads_its_original_oid_after_real_ref_advances() { + let fixture = Fixture::new().await; + let mono = fixture.state.storage.mono_storage(); + let main = mono.get_main_ref("/").await.unwrap().unwrap(); + let project = mono.get_main_ref("/project").await.unwrap().unwrap(); + let old_commit = + ObjectHash::from_hex_for_kind(HashKind::Sha1, &project.ref_commit_hash).unwrap(); + let blob_oid = ObjectHash::from_hex_for_kind(HashKind::Sha1, &fixture.oid).unwrap(); + let new_tree = tree(vec![item(TreeItemMode::Blob, blob_oid, "new-only")]); + let new_commit = Commit::from_tree_id_with_kind( + HashKind::Sha1, + new_tree.id, + vec![old_commit], + "advanced project without old paths", + ) + .unwrap(); + mono.save_mega_trees(vec![new_tree.clone()], new_commit.id, None) + .await + .unwrap(); + mono.save_mega_commits(vec![new_commit.clone()], None) + .await + .unwrap(); + publish_native_push(&fixture.state.storage, "/project", old_commit, &new_commit).await; + let head = mono + .read_native_publication_head( + fixture + .state + .storage + .config() + .mst2 + .instance_uuid + .as_deref() + .unwrap(), + ) + .await + .unwrap(); + assert_eq!(head.token.sequence, 2); + assert!(head.token.certificate.is_some()); + assert_ne!(head.root.commit, main.ref_commit_hash); + assert_ne!(head.root.tree, main.ref_tree_hash); + let published = mono.get_main_ref("/").await.unwrap().unwrap(); + assert_eq!(head.root.commit, published.ref_commit_hash); + assert_eq!(head.root.tree, published.ref_tree_hash); + let published_project = mono.get_main_ref("/project").await.unwrap().unwrap(); + assert_eq!(published_project.ref_commit_hash, new_commit.id.to_string()); + assert_eq!(published_project.ref_tree_hash, new_tree.id.to_string()); + let response = fixture.send("HEAD", "blob?path=/file", Body::empty()).await; + assert_eq!(response.status(), 200); + assert_eq!( + response.headers()["etag"], + format!("\"{}\"", fixture.digest_string()) + ); + fixture.counts.assert(0, 0); + let map = fixture.map("/file").await; + assert_eq!(map["map"]["file_content_id"], fixture.digest_string()); + fixture.counts.assert(1, fixture.raw.len()); +} + +#[tokio::test] +async fn mst2_fixed_metadata_walker_rejects_real_tree_gitlinks_without_body_reads() { + let fixture = Fixture::new().await; + let handler = MonoApiService::from(&fixture.state); + let commit_oid = ObjectHash::from_hex_for_kind( + HashKind::Sha1, + &fixture + .state + .storage + .mono_storage() + .get_main_ref("/") + .await + .unwrap() + .unwrap() + .ref_commit_hash, + ) + .unwrap(); + let root = tree(vec![item(TreeItemMode::Commit, commit_oid, "gitlink")]); + fixture + .state + .storage + .mono_storage() + .save_mega_trees(vec![root.clone()], commit_oid, None) + .await + .unwrap(); + let error = resolve_abs_metadata(&handler, &root, "/gitlink") + .await + .unwrap_err(); + assert_eq!(error.code, SnapshotErrorCode::UnsupportedEntry); + let file_oid = ObjectHash::from_hex_for_kind(HashKind::Sha1, &fixture.oid).unwrap(); + let file_root = tree(vec![item(TreeItemMode::Blob, file_oid, "file")]); + match resolve_abs_metadata(&handler, &file_root, "/file") + .await + .unwrap() + { + MetadataWalkOutcome::FoundFile { fs_kind, oid } => { + assert_eq!(fs_kind, FsKind::Regular); + assert_eq!(oid, fixture.oid); + } + other => panic!("unexpected fixed outcome: {other:?}"), + } + assert!(matches!( + resolve_abs_metadata(&handler, &file_root, "/missing/deep") + .await + .unwrap(), + MetadataWalkOutcome::Absent + )); + fixture.counts.assert(0, 0); +} + +#[path = "snapshot_storage_route_tests.rs"] +mod storage_routes; + +#[path = "snapshot_storage_route_fixture.rs"] +mod storage_route_fixture; diff --git a/src/api/router/snapshot_descriptor_upgrade_tests.rs b/src/api/router/snapshot_descriptor_upgrade_tests.rs new file mode 100644 index 00000000..9cb37b3b --- /dev/null +++ b/src/api/router/snapshot_descriptor_upgrade_tests.rs @@ -0,0 +1,181 @@ +use super::*; +use crate::jupiter::storage::qualified_metadata_family::reader_previous_fixture::restore_native_runtime; + +async fn remove_upgrade_ledgers(core: &sea_orm::DatabaseConnection) { + core.execute_unprepared( + "DELETE FROM seaql_migrations WHERE version IN ( + 'm20261008_000400_add_mst2_reader_retention', + 'm20261008_000500_fix_mst2_native_runtime', + 'm20261008_000600_fix_mst2_descriptor_wire')", + ) + .await + .unwrap(); +} + +async fn assert_upgrade_ledgers(core: &sea_orm::DatabaseConnection) { + let count: i64 = core + .query_one_raw(Statement::from_string( + DbBackend::Postgres, + "SELECT count(*) FROM seaql_migrations WHERE version IN ( + 'm20261008_000400_add_mst2_reader_retention', + 'm20261008_000500_fix_mst2_native_runtime', + 'm20261008_000600_fix_mst2_descriptor_wire')", + )) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap(); + assert_eq!(count, 3); +} + +async fn descriptor_matches_stored(fixture: &Fixture) -> i64 { + q_count( + fixture, + "SELECT count(*) FROM {q}.mst2_qualified_session_incarnation session + WHERE session.state='READY' AND session.canonical_descriptor= + {q}.mst2_metadata_descriptor(session.prepare_id,session.instance_id, + session.commit_oid,session.root_tree_oid)", + ) + .await +} + +#[tokio::test] +async fn captured_87b_descriptor_upgrade_preserves_owners_oids_and_replays_missing_ledgers() { + let fixture = Fixture::new_with_pg_config(true).await; + seed_live_and_terminal(&fixture).await; + let schema = q_schema(&fixture).await; + let mono = fixture.state.storage.mono_storage(); + let core = mono.get_connection(); + for missing_ledgers in [false, true] { + let rows = permanent_rows(&fixture).await; + let owners = retained_owners(&fixture).await; + restore_native_runtime(core, &schema, true).await; + assert_eq!(descriptor_matches_stored(&fixture).await, 0); + assert_eq!(permanent_rows(&fixture).await, rows); + assert_eq!(retained_owners(&fixture).await, owners); + if missing_ledgers { + remove_upgrade_ledgers(core).await; + apply_migrations(core, false).await.unwrap(); + assert_upgrade_ledgers(core).await; + } else { + test_upgrade_native_runtime(core).await.unwrap(); + } + assert_eq!(descriptor_matches_stored(&fixture).await, 1); + assert_eq!(permanent_rows(&fixture).await, rows); + assert_eq!(retained_owners(&fixture).await, owners); + let stamp = policy(core).await; + if missing_ledgers { + apply_migrations(core, false).await.unwrap(); + } else { + test_upgrade_native_runtime(core).await.unwrap(); + } + assert_eq!(policy(core).await, stamp); + assert_eq!(permanent_rows(&fixture).await, rows); + assert_eq!(retained_owners(&fixture).await, owners); + assert_old_sid_works(&fixture).await; + } + let rows = permanent_rows(&fixture).await; + let owners = retained_owners(&fixture).await; + let stamp = policy(core).await; + core.execute_unprepared( + "DELETE FROM seaql_migrations WHERE version IN ( + 'm20261008_000400_add_mst2_reader_retention', + 'm20261008_000500_fix_mst2_native_runtime')", + ) + .await + .unwrap(); + apply_migrations(core, false).await.unwrap(); + assert_upgrade_ledgers(core).await; + assert_eq!(descriptor_matches_stored(&fixture).await, 1); + assert_eq!(policy(core).await, stamp); + assert_eq!(permanent_rows(&fixture).await, rows); + assert_eq!(retained_owners(&fixture).await, owners); + assert_old_sid_works(&fixture).await; +} + +#[tokio::test] +async fn captured_87b_qless_descriptor_upgrade_replays_missing_ledgers_before_provisioning() { + let temp = tempfile::tempdir().unwrap(); + let (config, _schema) = test_db_config(temp.path()).await; + let core = crate::jupiter::storage::init::database_connection(&config) + .await + .unwrap(); + let schema: String = core + .query_one_raw(Statement::from_string( + DbBackend::Postgres, + "SELECT metadata_schema FROM mst2_metadata_namespace WHERE graph_domain='qualified-v1'", + )) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap(); + restore_native_runtime(&core, &schema, false).await; + remove_upgrade_ledgers(&core).await; + apply_migrations(&core, false).await.unwrap(); + assert_upgrade_ledgers(&core).await; + let stamp = policy(&core).await; + apply_migrations(&core, false).await.unwrap(); + test_upgrade_native_runtime(&core).await.unwrap(); + assert_eq!(policy(&core).await, stamp); + let namespace = provision_or_verify_rooted_qualified_family(&core) + .await + .unwrap(); + let current_schema: String = core + .query_one_raw(Statement::from_string( + DbBackend::Postgres, + "SELECT metadata_schema FROM mst2_metadata_namespace WHERE graph_domain='qualified-v1'", + )) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap(); + assert_ne!(current_schema, schema); + assert_eq!( + provision_or_verify_rooted_qualified_family(&core) + .await + .unwrap(), + namespace + ); + let high_water: i64 = core + .query_one_raw(Statement::from_string( + DbBackend::Postgres, + format!( + "SELECT high_water FROM \"{}\".mst2_metadata_reader_issuance", + current_schema.replace('"', "\"\"") + ), + )) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap(); + assert_eq!(high_water, 0); +} + +#[tokio::test] +async fn captured_87b_tampered_descriptor_rejects_without_resigning_history() { + let fixture = Fixture::new_with_pg_config(true).await; + seed_live_and_terminal(&fixture).await; + let schema = q_schema(&fixture).await; + let mono = fixture.state.storage.mono_storage(); + let core = mono.get_connection(); + restore_native_runtime(core, &schema, true).await; + let rows = permanent_rows(&fixture).await; + let owners = retained_owners(&fixture).await; + let before = policy(core).await; + core.execute_unprepared(&format!( + "CREATE OR REPLACE FUNCTION \"{}\".mst2_metadata_descriptor( + pid text,instance text,commit_id text,tree_id text) RETURNS bytea + LANGUAGE sql AS 'SELECT decode(''0000'',''hex'')'", + schema.replace('"', "\"\"") + )) + .await + .unwrap(); + assert!(test_upgrade_native_runtime(core).await.is_err()); + assert_eq!(policy(core).await, before); + assert_eq!(permanent_rows(&fixture).await, rows); + assert_eq!(retained_owners(&fixture).await, owners); +} diff --git a/src/api/router/snapshot_descriptor_wire_tests.rs b/src/api/router/snapshot_descriptor_wire_tests.rs new file mode 100644 index 00000000..f5d5536e --- /dev/null +++ b/src/api/router/snapshot_descriptor_wire_tests.rs @@ -0,0 +1,118 @@ +use mst2_codec::descriptor::ServingDescriptor; + +use super::*; + +async fn assert_postgres_descriptor_matches_codec(fixture: &Fixture, scope: &str) { + let resolved = success_json( + fixture + .app + .clone() + .oneshot( + Request::builder() + .method("POST") + .uri("/api/v2/snapshots/resolve") + .header("authorization", format!("Bearer {TOKEN}")) + .header("content-type", "application/json") + .body(Body::from( + json!({"target":{"kind":"latest"},"scope":scope}).to_string(), + )) + .unwrap(), + ) + .await + .unwrap(), + ) + .await; + let snapshot = resolved["descriptor"]["snapshot_id"].as_str().unwrap(); + let lease = resolved["lease_id"].as_str().unwrap(); + assert_eq!( + fixture + .state + .storage + .snapshot_metadata_family(lease, true) + .await + .unwrap(), + Some(SnapshotMetadataFamily::Rooted) + ); + let context = fixture + .state + .storage + .snapshot_context(snapshot, lease) + .await + .unwrap(); + let descriptor = &context.built.descriptor; + assert_eq!(descriptor.scope, scope); + let codec_bytes = descriptor.encode().unwrap(); + let schema = q_schema(fixture).await; + let quoted = format!("\"{}\"", schema.replace('"', "\"\"")); + let row = fixture + .state + .storage + .mono_storage() + .get_connection() + .query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + format!( + "SELECT {q}.mst2_metadata_descriptor(session.prepare_id,session.instance_id, + session.commit_oid,session.root_tree_oid) AS descriptor, + session.canonical_descriptor,session.metadata_root,session.snapshot_id + FROM {q}.mst2_qualified_session_incarnation session + JOIN {q}.mst2_qualified_lease_binding lease USING(snapshot_id,session_incarnation) + WHERE session.snapshot_id=$1 AND lease.lease_id=$2 AND session.state='READY' + AND lease.state='ACTIVE'", + q = quoted, + ), + [snapshot.into(), lease.into()], + )) + .await + .unwrap() + .unwrap(); + let database_bytes: Vec = row.try_get("", "descriptor").unwrap(); + assert_eq!(database_bytes, codec_bytes, "scope {scope:?}"); + assert_eq!( + row.try_get::>("", "canonical_descriptor").unwrap(), + codec_bytes + ); + assert_eq!( + ServingDescriptor::decode(&database_bytes).unwrap(), + *descriptor + ); + assert_eq!( + &database_bytes[56..58], + &u16::try_from(scope.len()).unwrap().to_le_bytes() + ); + assert_eq!(&database_bytes[58..58 + scope.len()], scope.as_bytes()); + assert_eq!( + row.try_get::>("", "metadata_root").unwrap(), + descriptor.metadata_root.to_vec() + ); + assert_eq!( + resolved["descriptor"]["metadata_root"], + format!("sha256:{}", hex::encode(descriptor.metadata_root)) + ); + let codec_snapshot = format!("sha256:{}", hex::encode(descriptor.snapshot_id().unwrap())); + assert_eq!(snapshot, codec_snapshot); + assert_eq!(context.built.snapshot_id, codec_snapshot); + assert_eq!( + row.try_get::("", "snapshot_id").unwrap(), + codec_snapshot + ); +} + +#[tokio::test] +async fn rooted_postgres_descriptor_matches_codec_bytes_for_root_ascii_long_and_utf8_scopes() { + let component = "a".repeat(247); + let fixture = Fixture::new_with_pg_config_directories_and_objects( + true, + 0, + &[(format!("{component}/é/file"), b"descriptor wire".to_vec())], + ) + .await; + let long_scope = format!("/project/{component}"); + let utf8_scope = format!("{long_scope}/é"); + assert_eq!(long_scope.len(), 256); + assert_eq!(utf8_scope.len(), 259); + assert_ne!(utf8_scope.len(), utf8_scope.chars().count()); + for scope in ["/", "/project", long_scope.as_str(), utf8_scope.as_str()] { + assert_postgres_descriptor_matches_codec(&fixture, scope).await; + } +} diff --git a/src/api/router/snapshot_generation_history_fixture.rs b/src/api/router/snapshot_generation_history_fixture.rs new file mode 100644 index 00000000..0f6e7005 --- /dev/null +++ b/src/api/router/snapshot_generation_history_fixture.rs @@ -0,0 +1,32 @@ +use sea_orm::{ConnectionTrait, DatabaseConnection, DbBackend, Statement}; + +pub(super) async fn restore_empty_g1_schema(db: &DatabaseConnection) { + super::generation_qualified_fixture::restore_empty_history_schema(db).await; + let count = db + .query_one_raw(Statement::from_string( + DbBackend::Postgres, + "SELECT count(*)::bigint AS n FROM mst2_metadata_current", + )) + .await + .unwrap() + .unwrap() + .try_get::("", "n") + .unwrap(); + assert_eq!( + count, 0, + "upgrade fixture has no generation-bound incarnation to erase" + ); + db.execute_unprepared( + "DROP TABLE mst2_metadata_current; + DROP FUNCTION mst2_metadata_current_guard(); + DROP INDEX idx_mst2_metadata_lifetime_node_generation; + ALTER TABLE mst2_metadata_lifetime DROP CONSTRAINT mst2_metadata_lifetime_pkey, + ADD PRIMARY KEY(page_id),ADD UNIQUE(node_id); + ALTER TABLE mst2_metadata_prepare DROP CONSTRAINT mst2_metadata_prepare_state_check, + DROP CONSTRAINT mst2_metadata_prepare_terminal_check, + ADD CONSTRAINT mst2_metadata_prepare_state_check CHECK (state IN ('PREPARING','COMMITTED')), + ADD CONSTRAINT mst2_metadata_prepare_check CHECK ((state='COMMITTED')=(committed_at IS NOT NULL)), + DROP COLUMN graph_domain,DROP COLUMN aborted_at,DROP COLUMN coverage_retired_at; + DELETE FROM seaql_migrations WHERE version='m20261007_000300_add_mst2_metadata_lifetime_history'" + ).await.unwrap(); +} diff --git a/src/api/router/snapshot_generation_qualified_fixture.rs b/src/api/router/snapshot_generation_qualified_fixture.rs new file mode 100644 index 00000000..e7020864 --- /dev/null +++ b/src/api/router/snapshot_generation_qualified_fixture.rs @@ -0,0 +1,56 @@ +use sea_orm::{ConnectionTrait, DatabaseConnection, DbBackend, Statement}; + +pub(super) async fn restore_empty_history_schema(db: &DatabaseConnection) { + let n:i64=db.query_one_raw(Statement::from_string(DbBackend::Postgres, + "SELECT (SELECT count(*) FROM mst2_metadata_current)+(SELECT count(*) FROM mst2_metadata_graph_node) + +(SELECT count(*) FROM mst2_metadata_graph_root)+(SELECT count(*) FROM mst2_metadata_graph_edge) + +(SELECT count(*) FROM mst2_metadata_gc_op) AS n")) + .await.unwrap().unwrap().try_get("","n").unwrap(); + assert_eq!( + n, 0, + "upgrade fixture has no qualified or current incarnation to erase" + ); + db.execute_unprepared( + "DO $$ DECLARE t text; tr text; BEGIN + FOREACH t IN ARRAY ARRAY['mst2_metadata_lifetime','mst2_metadata_current','mst2_metadata_payload', + 'mst2_metadata_prepare','mst2_metadata_prepare_page','mst2_retention_node','mst2_retention_edge', + 'mst2_retention_root','mst2_retention_gc_op','mst2_snapshot_context','mst2_snapshot_lease'] LOOP + FOREACH tr IN ARRAY ARRAY['mst2_metadata_statement_barrier','mst2_metadata_generic_domain_guard', + 'mst2_metadata_session_domain_guard','mst2_metadata_qualified_mapping_guard', + 'mst2_metadata_lifetime_insert_guard','mst2_metadata_lifetime_removed', + 'mst2_metadata_current_insert_guard','mst2_metadata_current_protected', + 'mst2_metadata_payload_fenced','mst2_metadata_payload_removed'] LOOP + EXECUTE format('DROP TRIGGER IF EXISTS %I ON %I',tr,t); + END LOOP; + END LOOP; + END $$; + DROP TABLE mst2_metadata_graph_root,mst2_metadata_graph_edge,mst2_metadata_graph_node,mst2_metadata_gc_op; + ALTER TABLE mst2_metadata_lifetime DROP COLUMN graph_domain; + ALTER TABLE mst2_metadata_prepare_page DROP CONSTRAINT mst2_metadata_prepare_page_prepare_id_page_id_generation_key; + ALTER TABLE mst2_metadata_prepare DROP CONSTRAINT mst2_metadata_prepare_prepare_id_storage_seal_key; + DROP INDEX idx_mst2_snapshot_context_metadata_root; + DO $$ DECLARE function_row record; BEGIN + FOR function_row IN SELECT proc.oid::regprocedure::text AS signature FROM pg_proc proc JOIN pg_namespace n ON n.oid=proc.pronamespace + WHERE n.nspname=current_schema() AND proc.proname LIKE 'mst2_metadata_%' + AND proc.proname NOT IN ('mst2_metadata_lifetime_guard','mst2_metadata_current_guard', + 'mst2_metadata_generation_seal_guard','mst2_metadata_generation_mapping_guard','mst2_metadata_payload_immutable') LOOP + EXECUTE 'DROP FUNCTION '||function_row.signature; + END LOOP; + END $$; + CREATE OR REPLACE FUNCTION mst2_metadata_current_guard() RETURNS trigger LANGUAGE plpgsql AS $$ + BEGIN RAISE EXCEPTION 'metadata current watermark requires generation-fenced collection'; END $$; + CREATE OR REPLACE FUNCTION mst2_metadata_lifetime_guard() RETURNS trigger LANGUAGE plpgsql AS $$ + BEGIN + IF TG_OP='DELETE' THEN RAISE EXCEPTION 'metadata lifetime watermark cannot be deleted'; END IF; + IF NEW.page_id IS DISTINCT FROM OLD.page_id OR NEW.node_id IS DISTINCT FROM OLD.node_id + OR NEW.generation IS DISTINCT FROM OLD.generation OR NEW.metadata_codec IS DISTINCT FROM OLD.metadata_codec + OR NEW.expected_size IS DISTINCT FROM OLD.expected_size THEN RAISE EXCEPTION 'metadata lifetime identity is immutable'; END IF; + IF NEW.state IS DISTINCT FROM OLD.state AND NOT (OLD.state='RESERVED' AND NEW.state='LIVE') THEN + RAISE EXCEPTION 'metadata lifetime transition requires generation collector'; END IF; + RETURN NEW; + END $$; + CREATE TRIGGER mst2_metadata_payload_immutable BEFORE UPDATE OR DELETE ON mst2_metadata_payload + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_payload_immutable(); + DELETE FROM seaql_migrations WHERE version='m20261007_000400_add_mst2_qualified_metadata_gc'" + ).await.unwrap(); +} diff --git a/src/api/router/snapshot_generation_upgrade_tests.rs b/src/api/router/snapshot_generation_upgrade_tests.rs new file mode 100644 index 00000000..28f4a203 --- /dev/null +++ b/src/api/router/snapshot_generation_upgrade_tests.rs @@ -0,0 +1,186 @@ +use sea_orm_migration::MigratorTrait; + +use super::*; +use crate::jupiter::migration::Migrator; + +async fn scalar(db: &sea_orm::DatabaseConnection, sql: &str) -> i64 { + db.query_one_raw(sea_orm::Statement::from_string( + sea_orm::DbBackend::Postgres, + sql, + )) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap() +} + +#[tokio::test] +async fn mst2_generation_additive_upgrade_preserves_legacy_v3_sid_lease_and_new_resolve() { + let fixture = Fixture::new_generic_history_with_pg_config(true).await; + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + assert_eq!( + scalar(db, "SELECT count(*) FROM mst2_metadata_lifetime").await, + 0 + ); + assert_eq!( + scalar( + db, + "SELECT count(*) FROM mst2_metadata_prepare WHERE storage_seal IS NOT NULL" + ) + .await, + 0 + ); + assert_eq!( + scalar( + db, + "SELECT count(*) FROM mst2_metadata_payload WHERE generation IS NOT NULL" + ) + .await, + 0 + ); + let pages = scalar(db, "SELECT count(*) FROM mst2_metadata_payload").await; + let original = success_json(fixture.send("GET", "descriptor", Body::empty()).await).await; + super::install_capability_fixture::restore_pre_capability_schema(db).await; + super::generation_history_fixture::restore_empty_g1_schema(db).await; + // The unchanged v3 installer created these durable rows. Remove only the + // empty additive schema in this isolated fixture to reproduce the previous + // deployed schema; the SID, lease, CAS bytes and all original guards survive. + db.execute_unprepared( + "DROP TRIGGER mst2_metadata_generation_mapping_guard ON mst2_metadata_prepare_page; + DROP TRIGGER mst2_metadata_generation_seal_guard ON mst2_metadata_prepare; + DROP FUNCTION mst2_metadata_generation_mapping_guard(); + DROP FUNCTION mst2_metadata_generation_seal_guard(); + ALTER TABLE mst2_metadata_prepare DROP COLUMN canonical_bindings,DROP COLUMN bindings_digest, + DROP COLUMN primary_scope,DROP COLUMN storage_seal; + ALTER TABLE mst2_metadata_prepare_page DROP COLUMN generation; + ALTER TABLE mst2_metadata_payload DROP COLUMN generation; + DROP TABLE mst2_metadata_lifetime; + DROP FUNCTION mst2_metadata_lifetime_guard(); + DELETE FROM seaql_migrations WHERE version='m20261007_000200_add_mst2_metadata_generations'" + ).await.unwrap(); + assert_eq!( + scalar( + db, + "SELECT count(*) FROM mst2_snapshot_lease WHERE state='ACTIVE'" + ) + .await, + 1 + ); + assert_eq!( + scalar(db, "SELECT count(*) FROM mst2_snapshot_context").await, + 1 + ); + assert_eq!( + success_json(fixture.send("GET", "descriptor", Body::empty()).await).await, + original + ); + Migrator::up(db, None).await.unwrap(); + assert_eq!( + scalar( + db, + "SELECT count(*) FROM mst2_metadata_payload WHERE generation IS NULL" + ) + .await, + pages + ); + assert_eq!( + scalar( + db, + "SELECT count(*) FROM mst2_metadata_prepare WHERE storage_seal IS NULL" + ) + .await, + 1 + ); + assert_eq!( + success_json(fixture.send("GET", "descriptor", Body::empty()).await).await, + original + ); + + let config = fixture.state.storage.config(); + let connection = crate::jupiter::storage::init::postgres_connection(&config.database) + .await + .unwrap(); + let storage = crate::jupiter::storage::init::with_generic_history_bootstrap( + crate::jupiter::storage::Storage::new_with_connection( + config, + Arc::new(connection), + fixture.state.storage.git_service.obj_storage.clone(), + ), + ) + .await + .unwrap(); + let state = MonoApiServiceState { + storage, + ..fixture.state.clone() + }; + let app = Router::new().nest( + "/api/v2", + crate::api::router::snapshot_router::generic_history_routers(state.clone()) + .with_state(state), + ); + let restored = success_json( + app.clone() + .oneshot(fixture.request("GET", "descriptor", Body::empty())) + .await + .unwrap(), + ) + .await; + assert_eq!(restored, original); + let next = success_json( + app.clone() + .oneshot( + Request::builder() + .method("POST") + .uri("/api/v2/snapshots/resolve") + .header("authorization", format!("Bearer {TOKEN}")) + .header("content-type", "application/json") + .body(Body::from( + json!({"target":{"kind":"latest"},"scope":"/project"}).to_string(), + )) + .unwrap(), + ) + .await + .unwrap(), + ) + .await; + assert_eq!(next["descriptor"]["snapshot_id"], fixture.snapshot); + assert_ne!(next["lease_id"], fixture.lease); + assert_eq!( + scalar( + db, + "SELECT count(*) FROM mst2_snapshot_lease WHERE state='ACTIVE'" + ) + .await, + 2 + ); + let content = app + .oneshot(fixture.request("GET", "blob?path=/file", Body::empty())) + .await + .unwrap(); + assert_eq!(content.status(), 200); + assert_eq!( + to_bytes(content.into_body(), usize::MAX) + .await + .unwrap() + .as_ref(), + fixture.raw.as_slice() + ); + assert_eq!( + scalar(db, "SELECT count(*) FROM mst2_metadata_lifetime").await, + 0 + ); + assert_eq!( + scalar(db, "SELECT count(*) FROM mst2_metadata_payload").await, + pages + ); + assert_eq!( + scalar( + db, + "SELECT count(*) FROM mst2_metadata_prepare WHERE storage_seal IS NULL" + ) + .await, + 1 + ); +} diff --git a/src/api/router/snapshot_install_capability_fixture.rs b/src/api/router/snapshot_install_capability_fixture.rs new file mode 100644 index 00000000..1806c7ab --- /dev/null +++ b/src/api/router/snapshot_install_capability_fixture.rs @@ -0,0 +1,33 @@ +use sea_orm::{ConnectionTrait, DatabaseConnection, DbBackend, Statement}; + +pub(super) async fn restore_pre_capability_schema(db: &DatabaseConnection) { + let bound:i64=db.query_one_raw(Statement::from_string(DbBackend::Postgres, + "SELECT (SELECT count(*) FROM mst2_metadata_prepare WHERE storage_seal IS NOT NULL + OR canonical_bindings IS NOT NULL OR bindings_digest IS NOT NULL OR primary_scope IS NOT NULL + OR graph_domain IS NOT NULL)+(SELECT count(*) FROM mst2_metadata_prepare_page WHERE generation IS NOT NULL) + +(SELECT count(*) FROM mst2_metadata_payload WHERE generation IS NOT NULL)")) + .await.unwrap().unwrap().try_get_by_index(0).unwrap(); + assert_eq!( + bound, 0, + "old-schema fixture only contains unbound legacy installations" + ); + // Reproduce the deployment before this additive migration. Preserve every + // original plan, member, byte, context, lease and protection root. + db.execute_unprepared( + "DROP TRIGGER mst2_00_install_capability_barrier ON mst2_metadata_prepare; + DROP TRIGGER mst2_00_install_capability_barrier ON mst2_metadata_prepare_page; + DROP TRIGGER mst2_install_capability_truncate_guard ON mst2_metadata_prepare; + DROP TRIGGER mst2_install_capability_truncate_guard ON mst2_metadata_prepare_page; + DROP TRIGGER mst2_install_capability_prepare_guard ON mst2_metadata_prepare; + DROP TRIGGER mst2_install_capability_mapping_guard ON mst2_metadata_prepare_page; + DROP TABLE mst2_metadata_install_seal; + DROP FUNCTION mst2_install_capability_barrier(); + DROP FUNCTION mst2_install_capability_truncate_guard(); + DROP FUNCTION mst2_install_capability_prepare_guard(); + DROP FUNCTION mst2_install_capability_mapping_guard(); + DROP FUNCTION mst2_install_capability_register(); + DELETE FROM seaql_migrations WHERE version='m20261007_000500_add_mst2_install_capability'", + ) + .await + .unwrap(); +} diff --git a/src/api/router/snapshot_lookup_metadata_tests.rs b/src/api/router/snapshot_lookup_metadata_tests.rs new file mode 100644 index 00000000..81c11cdb --- /dev/null +++ b/src/api/router/snapshot_lookup_metadata_tests.rs @@ -0,0 +1,271 @@ +use super::*; + +fn lookup_body(paths: &[&str]) -> Body { + Body::from(json!({"paths": paths}).to_string()) +} + +#[tokio::test] +async fn mst2_durable_http_lookup_uses_verified_metadata_after_service_rebuild_without_body_reads() +{ + let fixture = Fixture::new_with_pg_config(true).await; + let state = rebuilt(&fixture).await; + assert!(state.storage.native_snapshot_sessions.get().is_none()); + assert!(!Arc::ptr_eq( + &state.storage.native_projection_cache, + &fixture.state.storage.native_projection_cache + )); + let paths = [ + "/file", + "/alias", + "/executable", + "/empty", + "/link", + "/nested/file", + "/directory", + "/nested", + "/missing", + "/file/child", + "/link/child", + ]; + let value = success_json( + app(&state) + .oneshot(fixture.request("POST", "lookup", lookup_body(&paths))) + .await + .unwrap(), + ) + .await; + assert_eq!(value["snapshot_id"], fixture.snapshot); + let results = value["results"].as_array().unwrap(); + assert_eq!(results.len(), paths.len()); + for (result, path) in results.iter().zip(paths) { + assert_eq!(result["path"], path); + } + for (index, kind, size, content_digest) in [ + (0, "regular", fixture.raw.len(), fixture.digest_string()), + (1, "regular", fixture.raw.len(), fixture.digest_string()), + (2, "executable", fixture.raw.len(), fixture.digest_string()), + (3, "regular", 0, format!("sha256:{}", hex_of(&digest(&[])))), + ( + 4, + "symlink", + 4, + format!("sha256:{}", hex_of(&digest(b"file"))), + ), + (5, "regular", fixture.raw.len(), fixture.digest_string()), + ] { + assert_eq!(results[index]["status"], "found"); + assert_eq!( + results[index]["node"], + json!({ + "fs_kind": kind, + "name": paths[index].rsplit('/').next().unwrap(), + "size": size.to_string(), + "content_digest": content_digest, + }) + ); + } + let proofs = value["proof_pages"].as_array().unwrap(); + assert!(!proofs.is_empty()); + let mut proof_digests = std::collections::HashSet::new(); + for proof in proofs { + let bytes = STANDARD + .decode(proof["data_base64"].as_str().unwrap()) + .unwrap(); + mst2_codec::metapage::Page::decode(&bytes).unwrap(); + let digest = format!("sha256:{}", hex_of(&mst2_codec::metapage::page_id(&bytes))); + assert_eq!(proof["digest"], digest); + assert!(proof_digests.insert(digest)); + } + for index in [6, 7] { + let result = &results[index]; + assert_eq!(result["status"], "found"); + assert_eq!(result["node"]["fs_kind"], "directory"); + assert_eq!( + result["node"]["name"], + paths[index].rsplit('/').next().unwrap() + ); + assert_eq!(result["node"]["node_class"], "native_tree"); + assert_eq!(result["node"]["lifecycle"], "mutable"); + assert!(proof_digests.contains(result["node"]["directory_root"].as_str().unwrap())); + } + for (index, status) in [ + (8, "absent"), + (9, "not_directory"), + (10, "symlink_traversal"), + ] { + assert_eq!( + results[index], + json!({"path": paths[index], "status": status}) + ); + } + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn mst2_durable_http_lookup_missing_and_noncurrent_facts_never_fall_back_to_body_or_cache() { + let fixture = Fixture::new_with_pg_config(true).await; + fixture.map("/file").await; + let original = fixture.fact().await; + fixture.counts.reset(); + fixture.delete_fact().await; + error( + fixture + .send("POST", "lookup", lookup_body(&["/file"])) + .await, + 503, + "METADATA_NOT_READY", + true, + ) + .await; + fixture.counts.assert(0, 0); + for (domain, kind, generation) in [ + ("git", "blob", 1), + ("other", "blob", MST2_VERIFICATION_VERSION), + ("git", "tree", MST2_VERIFICATION_VERSION), + ] { + let mut fact = original.clone(); + fact.storage_domain = domain.to_string(); + fact.object_kind = kind.to_string(); + fact.verification_version = generation; + fixture.replace_fact(fact).await; + error( + fixture + .send("POST", "lookup", lookup_body(&["/alias"])) + .await, + 503, + "METADATA_NOT_READY", + true, + ) + .await; + fixture.counts.assert(0, 0); + } + fixture.replace_fact(original).await; + success_json( + fixture + .send("POST", "lookup", lookup_body(&["/file"])) + .await, + ) + .await; + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn mst2_durable_http_lookup_invalid_verified_facts_fail_before_body_reads() { + let fixture = Fixture::new_with_pg_config(true).await; + let original = fixture.fact().await; + for case in 0..5 { + let mut fact = original.clone(); + match case { + 0 => fact.state = "PENDING".to_string(), + 1 => fact.verification_version = MST2_VERIFICATION_VERSION + 1, + 2 => fact.size = -1, + 3 => fact.size = 8_796_093_022_209, + 4 => { + fact.raw_sha256.pop().unwrap(); + } + _ => unreachable!(), + } + fixture.replace_fact(fact).await; + error( + fixture + .send("POST", "lookup", lookup_body(&["/file"])) + .await, + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + fixture.counts.assert(0, 0); + } + fixture.replace_fact(original).await; + success_json( + fixture + .send("POST", "lookup", lookup_body(&["/file"])) + .await, + ) + .await; + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn mst2_durable_http_lookup_symlink_facts_outside_profile_fail_without_body_reads() { + let fixture = Fixture::new_with_pg_config(true).await; + let mono = fixture.state.storage.mono_storage(); + let main = mono.get_main_ref("/").await.unwrap().unwrap(); + let handler = MonoApiService::from(&fixture.state); + let root = handler.get_tree_by_hash(&main.ref_tree_hash).await.unwrap(); + let MetadataWalkOutcome::FoundFile { oid, .. } = + resolve_abs_metadata(&handler, &root, "/project/link") + .await + .unwrap() + else { + panic!("fixture link must be a fixed file"); + }; + let original = mono + .get_verified_blobs(vec![oid.clone()]) + .await + .unwrap() + .remove(&oid) + .unwrap(); + for size in [0, 4096] { + let mut fact = original.clone().into_active_model(); + fact.size = Set(size); + fact.update(mono.get_connection()).await.unwrap(); + error( + fixture + .send("POST", "lookup", lookup_body(&["/link"])) + .await, + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + fixture.counts.assert(0, 0); + } + let mut restored = original.into_active_model(); + restored.size = Set(4); + restored.update(mono.get_connection()).await.unwrap(); + let result = success_json( + fixture + .send("POST", "lookup", lookup_body(&["/link"])) + .await, + ) + .await; + assert_eq!(result["results"][0]["node"]["size"], "4"); + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn mst2_durable_http_lookup_verified_fact_database_failure_is_typed_without_body_reads() { + let fixture = Fixture::new_with_pg_config(true).await; + let mono = fixture.state.storage.mono_storage(); + mono.get_connection() + .execute_unprepared( + "ALTER TABLE mst2_verified_object RENAME TO mst2_verified_object_unavailable", + ) + .await + .unwrap(); + error( + fixture + .send("POST", "lookup", lookup_body(&["/file"])) + .await, + 500, + "INTERNAL", + true, + ) + .await; + fixture.counts.assert(0, 0); + mono.get_connection() + .execute_unprepared( + "ALTER TABLE mst2_verified_object_unavailable RENAME TO mst2_verified_object", + ) + .await + .unwrap(); + success_json( + fixture + .send("POST", "lookup", lookup_body(&["/file"])) + .await, + ) + .await; + fixture.counts.assert(0, 0); +} diff --git a/src/api/router/snapshot_native_runtime_upgrade_tests.rs b/src/api/router/snapshot_native_runtime_upgrade_tests.rs new file mode 100644 index 00000000..2fdee7d6 --- /dev/null +++ b/src/api/router/snapshot_native_runtime_upgrade_tests.rs @@ -0,0 +1,281 @@ +use super::*; +use crate::jupiter::{ + migration::{apply_migrations, test_upgrade_native_runtime}, + storage::qualified_metadata_family::reader_previous_fixture::restore_retention, +}; + +#[path = "snapshot_descriptor_upgrade_tests.rs"] +mod descriptor_wire; + +async fn retained_owners(fixture: &Fixture) -> Value { + let txn = transaction(fixture).await; + let value = txn.query_one_raw(Statement::from_string(DbBackend::Postgres, + "SELECT jsonb_build_object( + 'readers',(SELECT jsonb_agg(to_jsonb(x) ORDER BY operation_id,reader_issuance) FROM mst2_metadata_reader_operation x), + 'anchors',(SELECT jsonb_agg(to_jsonb(x) ORDER BY anchor_id) FROM mst2_metadata_root_anchor x), + 'issuance',(SELECT to_jsonb(x) FROM mst2_metadata_reader_issuance x), + 'functions',(SELECT jsonb_agg(jsonb_build_array(p.proname,p.oid::bigint) ORDER BY p.proname,p.oid) + FROM pg_catalog.pg_proc p JOIN pg_catalog.pg_namespace n ON n.oid=p.pronamespace + WHERE n.nspname=current_schema() OR (n.nspname=(SELECT c.nspname FROM mst2_metadata_family_identity i + JOIN pg_catalog.pg_namespace c ON c.oid=i.core_schema_oid LIMIT 1) + AND p.proname='mst2_route_family_registration_guard'))) AS owners")) + .await.unwrap().unwrap().try_get("", "owners").unwrap(); + txn.commit().await.unwrap(); + value +} + +async fn policy(core: &sea_orm::DatabaseConnection) -> Value { + core.query_one_raw(Statement::from_string( + DbBackend::Postgres, + "SELECT to_jsonb(p) FROM mst2_qualified_family_policy p WHERE singleton=1", + )) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap() +} + +async fn seed_live_and_terminal(fixture: &Fixture) { + let txn = transaction(fixture).await; + admission(&txn, fixture).await; + txn.commit().await.unwrap(); + let txn = transaction(fixture).await; + let terminal = admission(&txn, fixture).await; + finish(&txn, &terminal).await; + txn.commit().await.unwrap(); + assert_eq!( + q_count( + fixture, + "SELECT count(*) FROM {q}.mst2_metadata_reader_operation WHERE state='ACTIVE'" + ) + .await, + 1 + ); + assert_eq!( + q_count( + fixture, + "SELECT count(*) FROM {q}.mst2_metadata_reader_operation WHERE state='FINISHED'" + ) + .await, + 1 + ); + assert!( + q_count( + fixture, + "SELECT high_water FROM {q}.mst2_metadata_reader_issuance" + ) + .await + >= 2 + ); +} + +async fn assert_old_sid_works(fixture: &Fixture) { + let pinned = fixture + .state + .storage + .snapshot_context(&fixture.snapshot, &fixture.lease) + .await + .unwrap(); + let repository = RootedQualifiedMetadataRepository::open( + fixture.state.storage.mono_storage().get_connection(), + &fixture.state.storage.config().database, + ) + .await + .unwrap(); + let context = repository + .context(&fixture.snapshot, &fixture.lease, &pinned.built.instance_id) + .await + .unwrap(); + assert_eq!(context.built.descriptor, pinned.built.descriptor); + assert_eq!(context.commit_oid, pinned.commit_oid); + let lookup = repository + .lookup_metadata(&context, &["/file".to_owned()]) + .await + .unwrap(); + assert!(matches!( + lookup.results.as_slice(), + [RootedLookupStatus::File { .. }] + )); + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn captured_cc90_upgrade_preserves_nonzero_reader_issuance_all_owners_oids_and_old_sid() { + let fixture = Fixture::new_with_pg_config(true).await; + seed_live_and_terminal(&fixture).await; + let schema = q_schema(&fixture).await; + let mono = fixture.state.storage.mono_storage(); + let core = mono.get_connection(); + let rows = permanent_rows(&fixture).await; + let owners = retained_owners(&fixture).await; + restore_retention(core, &schema, true, false).await; + assert_eq!(retained_owners(&fixture).await, owners); + test_upgrade_native_runtime(core).await.unwrap(); + assert_eq!(permanent_rows(&fixture).await, rows); + assert_eq!(retained_owners(&fixture).await, owners); + let stamp = policy(core).await; + test_upgrade_native_runtime(core).await.unwrap(); + assert_eq!(policy(core).await, stamp); + assert_eq!(retained_owners(&fixture).await, owners); + assert_eq!(permanent_rows(&fixture).await, rows); + assert_old_sid_works(&fixture).await; + assert_eq!(permanent_rows(&fixture).await, rows); + provision_or_verify_rooted_qualified_family(core) + .await + .unwrap(); +} + +#[tokio::test] +async fn captured_cc90_interrupted_before_reader_migration_resumes_without_reader_ddl() { + let fixture = Fixture::new_with_pg_config(true).await; + seed_live_and_terminal(&fixture).await; + let schema = q_schema(&fixture).await; + let mono = fixture.state.storage.mono_storage(); + let core = mono.get_connection(); + restore_retention(core, &schema, true, false).await; + let rows = permanent_rows(&fixture).await; + let owners = retained_owners(&fixture).await; + core.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "DELETE FROM seaql_migrations WHERE version IN ($1,$2,$3)", + [ + "m20261008_000400_add_mst2_reader_retention".into(), + "m20261008_000500_fix_mst2_native_runtime".into(), + "m20261008_000600_fix_mst2_descriptor_wire".into(), + ], + )) + .await + .unwrap(); + apply_migrations(core, false).await.unwrap(); + assert_eq!(permanent_rows(&fixture).await, rows); + assert_eq!(retained_owners(&fixture).await, owners); + let count: i64 = core.query_one_raw(Statement::from_string(DbBackend::Postgres, + "SELECT count(*) FROM seaql_migrations WHERE version IN + ('m20261008_000400_add_mst2_reader_retention','m20261008_000500_fix_mst2_native_runtime', + 'm20261008_000600_fix_mst2_descriptor_wire')")) + .await.unwrap().unwrap().try_get_by_index(0).unwrap(); + assert_eq!(count, 3); + assert_old_sid_works(&fixture).await; +} + +async fn qless_upgrade(legacy: bool) { + let temp = tempfile::tempdir().unwrap(); + let (config, _schema) = test_db_config(temp.path()).await; + let core = crate::jupiter::storage::init::database_connection(&config) + .await + .unwrap(); + let old_schema: String = core + .query_one_raw(Statement::from_string( + DbBackend::Postgres, + "SELECT metadata_schema FROM mst2_metadata_namespace WHERE graph_domain='qualified-v1'", + )) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap(); + restore_retention(&core, &old_schema, false, legacy).await; + test_upgrade_native_runtime(&core).await.unwrap(); + let stamp = policy(&core).await; + test_upgrade_native_runtime(&core).await.unwrap(); + assert_eq!(policy(&core).await, stamp); + provision_or_verify_rooted_qualified_family(&core) + .await + .unwrap(); + provision_or_verify_rooted_qualified_family(&core) + .await + .unwrap(); + let row = core + .query_one_raw(Statement::from_string( + DbBackend::Postgres, + "SELECT metadata_schema FROM mst2_metadata_namespace WHERE graph_domain='qualified-v1'", + )) + .await + .unwrap() + .unwrap(); + let schema: String = row.try_get_by_index(0).unwrap(); + assert_ne!(schema, old_schema); + let water: i64 = core + .query_one_raw(Statement::from_string( + DbBackend::Postgres, + format!( + "SELECT high_water FROM \"{}\".mst2_metadata_reader_issuance", + schema.replace('"', "\"\"") + ), + )) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap(); + assert_eq!(water, 0); +} + +#[tokio::test] +async fn captured_cc90_qless_canonical_policy_upgrades_before_new_provisioning() { + qless_upgrade(false).await; +} + +#[tokio::test] +async fn captured_cc90_qless_exact_legacy_deparse_policy_upgrades_before_new_provisioning() { + qless_upgrade(true).await; +} + +#[tokio::test] +async fn captured_cc90_tampered_decoder_rejects_upgrade_without_rewriting_owners_or_policy() { + let fixture = Fixture::new_with_pg_config(true).await; + seed_live_and_terminal(&fixture).await; + let schema = q_schema(&fixture).await; + let mono = fixture.state.storage.mono_storage(); + let core = mono.get_connection(); + restore_retention(core, &schema, true, false).await; + let before = policy(core).await; + let owners = retained_owners(&fixture).await; + let rows = permanent_rows(&fixture).await; + core.execute_unprepared(&format!( + "CREATE OR REPLACE FUNCTION \"{}\".mst2_metadata_decode_rooted_plan(b bytea) RETURNS jsonb + LANGUAGE plpgsql IMMUTABLE STRICT AS $$ BEGIN RAISE EXCEPTION 'tampered decoder'; END $$", + schema.replace('"', "\"\""))).await.unwrap(); + assert!(test_upgrade_native_runtime(core).await.is_err()); + assert_eq!(policy(core).await, before); + assert_eq!(retained_owners(&fixture).await, owners); + assert_eq!(permanent_rows(&fixture).await, rows); +} + +#[tokio::test] +async fn captured_cc90_qless_arbitrary_policy_shape_is_never_resigned() { + for legacy in [false, true] { + let temp = tempfile::tempdir().unwrap(); + let (config, _schema) = test_db_config(temp.path()).await; + let core = crate::jupiter::storage::init::database_connection(&config) + .await + .unwrap(); + let schema: String = core.query_one_raw(Statement::from_string(DbBackend::Postgres, + "SELECT metadata_schema FROM mst2_metadata_namespace WHERE graph_domain='qualified-v1'")) + .await.unwrap().unwrap().try_get_by_index(0).unwrap(); + restore_retention(&core, &schema, false, legacy).await; + core.execute_unprepared("ALTER TABLE mst2_qualified_family_policy DISABLE TRIGGER mst2_route_family_policy_immutable; + UPDATE mst2_qualified_family_policy SET expected_shape=decode(repeat('a5',32),'hex') WHERE singleton=1; + ALTER TABLE mst2_qualified_family_policy ENABLE TRIGGER mst2_route_family_policy_immutable").await.unwrap(); + let before = policy(&core).await; + let schemas: Value = core.query_one_raw(Statement::from_string(DbBackend::Postgres, + "SELECT coalesce(jsonb_agg(n.nspname ORDER BY n.nspname),'[]'::jsonb) FROM pg_catalog.pg_namespace n + WHERE EXISTS(SELECT 1 FROM pg_catalog.pg_proc p WHERE p.pronamespace=n.oid + AND p.proname='mst2_metadata_dml_barrier' + AND strpos(p.prosrc,chr(34)||current_schema()||chr(34))>0)")) + .await.unwrap().unwrap().try_get_by_index(0).unwrap(); + assert!(test_upgrade_native_runtime(&core).await.is_err()); + assert_eq!(policy(&core).await, before); + let after: Value = core.query_one_raw(Statement::from_string(DbBackend::Postgres, + "SELECT coalesce(jsonb_agg(n.nspname ORDER BY n.nspname),'[]'::jsonb) FROM pg_catalog.pg_namespace n + WHERE EXISTS(SELECT 1 FROM pg_catalog.pg_proc p WHERE p.pronamespace=n.oid + AND p.proname='mst2_metadata_dml_barrier' + AND strpos(p.prosrc,chr(34)||current_schema()||chr(34))>0)")) + .await.unwrap().unwrap().try_get_by_index(0).unwrap(); + assert_eq!( + after, schemas, + "rejected migration must roll back every temporary template" + ); + } +} diff --git a/src/api/router/snapshot_objects_bounded_tests.rs b/src/api/router/snapshot_objects_bounded_tests.rs new file mode 100644 index 00000000..e7eb0bfa --- /dev/null +++ b/src/api/router/snapshot_objects_bounded_tests.rs @@ -0,0 +1,716 @@ +use std::io; + +use tokio::{sync::Notify, time::timeout}; + +use super::*; + +#[derive(Clone)] +pub(super) struct StreamFault { + pub(super) oid: String, + pub(super) kind: FaultKind, +} + +#[derive(Clone)] +pub(super) enum FaultKind { + Parts(Vec), + LateError(Bytes), + Oversized(Bytes, Arc), + Held { + raw: Bytes, + entered: Arc, + release: Arc, + drops: Arc, + }, + HeldThenError { + entered: Arc, + release: Arc, + drops: Arc, + }, + HeldFragment { + prefix: Bytes, + fragment: Option, + entered: Arc, + release: Arc, + tail_polls: Arc, + drops: Arc, + }, +} + +struct DropCount(Arc); + +impl Drop for DropCount { + fn drop(&mut self) { + self.0.fetch_add(1, Ordering::SeqCst); + } +} + +impl StreamFault { + pub(super) fn stream(self) -> ObjectByteStream { + match self.kind { + FaultKind::HeldFragment { + prefix, + fragment, + entered, + release, + tail_polls, + drops, + } => Box::pin(futures::stream::unfold( + ( + prefix, + fragment, + entered, + release, + tail_polls, + DropCount(drops), + 0u8, + ), + |(prefix, fragment, entered, release, tail_polls, owner, turn)| async move { + match turn { + 0 => Some(( + Ok(prefix.clone()), + (prefix, fragment, entered, release, tail_polls, owner, 1), + )), + 1 => { + entered.notify_one(); + release.notified().await; + fragment.clone().map(|part| { + ( + Ok(part), + (prefix, fragment, entered, release, tail_polls, owner, 2), + ) + }) + } + _ => { + tail_polls.fetch_add(1, Ordering::SeqCst); + std::future::pending().await + } + } + }, + )), + FaultKind::HeldThenError { + entered, + release, + drops, + } => Box::pin(futures::stream::unfold( + (true, entered, release, DropCount(drops)), + |(first, entered, release, owner)| async move { + if !first { + return None; + } + entered.notify_one(); + release.notified().await; + Some(( + Err(io::Error::other("held source read failed")), + (false, entered, release, owner), + )) + }, + )), + FaultKind::Parts(parts) => Box::pin(futures::stream::iter(parts.into_iter().map(Ok))), + FaultKind::LateError(raw) => Box::pin(futures::stream::iter([ + Ok(raw), + Err(io::Error::other("test late object stream failure")), + ])), + FaultKind::Oversized(raw, tail_polls) => Box::pin(futures::stream::unfold( + (Some(raw), tail_polls), + |(raw, tail_polls)| async move { + match raw { + Some(raw) => Some((Ok(raw), (None, tail_polls))), + None => { + tail_polls.fetch_add(1, Ordering::SeqCst); + Some(( + Err(io::Error::other("oversized tail must not be polled")), + (None, tail_polls), + )) + } + } + }, + )), + FaultKind::Held { + raw, + entered, + release, + drops, + } => Box::pin(futures::stream::unfold( + (raw, entered, release, DropCount(drops), 0u8), + |(raw, entered, release, drop_count, turn)| async move { + match turn { + 0 => Some((Ok(raw.slice(..1)), (raw, entered, release, drop_count, 1))), + 1 => { + entered.notify_one(); + release.notified().await; + Some((Ok(raw.slice(1..)), (raw, entered, release, drop_count, 2))) + } + _ => None, + } + }, + )), + } + } +} + +fn files(count: usize, size: usize) -> Vec<(String, Vec)> { + let seed = uuid::Uuid::new_v4(); + (0..count) + .map(|index| { + let mut raw = vec![index as u8; size]; + let prefix = format!("blob 3\0abc\0{seed}-{index}"); + raw[..prefix.len()].copy_from_slice(prefix.as_bytes()); + (format!("object-{index:03}"), raw) + }) + .collect() +} + +async fn fixture(objects: &[(String, Vec)]) -> Fixture { + Fixture::new_with_pg_config_directories_and_objects(false, 0, objects).await +} + +fn body(objects: &[(String, Vec)]) -> Body { + Body::from(request_bytes(objects)) +} + +fn request_bytes(objects: &[(String, Vec)]) -> Vec { + serde_json::to_vec(&json!({ + "items":objects.iter().map(|(name, raw)| json!({ + "path":format!("/{name}"), + "expected_digest":format!("sha256:{}", hex_of(&digest(raw))), + })).collect::>(), + "encoding":"identity", + })) + .unwrap() +} + +async fn object_oid(fixture: &Fixture, path: &str) -> String { + let handler = MonoApiService::from(&fixture.state); + let context = fixture + .state + .storage + .mono_storage() + .get_main_ref("/project") + .await + .unwrap() + .unwrap(); + let tree = handler + .get_tree_by_hash(&context.ref_tree_hash) + .await + .unwrap(); + match resolve_abs_metadata(&handler, &tree, path).await.unwrap() { + MetadataWalkOutcome::FoundFile { oid, .. } => oid, + other => panic!("object test path did not resolve: {other:?}"), + } +} + +async fn install_fault(fixture: &Fixture, path: &str, kind: FaultKind) { + let oid = object_oid(fixture, path).await; + *fixture.counts.object_fault.lock().unwrap() = Some(StreamFault { oid, kind }); +} + +async fn assert_objects(response: Response, request: &[u8], objects: &[(String, Vec)]) { + assert_eq!(response.status(), 200); + assert_eq!( + response.headers()["x-mega-request-digest"], + format!("sha256:{}", hex_of(&digest(request))) + ); + let encoded = to_bytes(response.into_body(), 10 * 1024 * 1024) + .await + .unwrap(); + let frames = parse_stream(&encoded).unwrap(); + let mut actual = Vec::new(); + let mut end = None; + for frame in frames { + match frame { + Frame::Object(payload) => { + assert!(end.is_none(), "END must be terminal"); + actual.extend(payload.objects); + } + Frame::End(record) => { + assert!(end.is_none()); + end = Some(record); + } + other => panic!("unexpected OBJECT frame: {other:?}"), + } + } + let mut expected = Vec::new(); + for (_, raw) in objects { + let id = digest(raw); + if !expected.iter().any(|(digest, _)| *digest == id) { + expected.push((id, raw.clone())); + } + } + assert_eq!(actual, expected); + let end = end.unwrap(); + assert_eq!(end.request_item_count, objects.len() as u32); + assert_eq!(end.unique_unit_count, expected.len() as u32); + assert_eq!( + end.logical_bytes, + expected + .iter() + .map(|(_, bytes)| bytes.len() as u64) + .sum::() + ); + assert_eq!(end.request_body_sha256, digest(request)); +} + +#[tokio::test] +async fn oversized_item_and_later_invalid_path_reject_entire_batch_before_body_io() { + let fixture = Fixture::new().await; + let link_digest = format!("sha256:{}", hex_of(&digest(b"file"))); + for last in [ + json!({"path":"/file","expected_digest":fixture.digest_string()}), + json!({"path":"/missing","expected_digest":link_digest}), + ] { + let oversized = last["path"] == "/file"; + let request = json!({"items":[ + {"path":"/link","expected_digest":link_digest},last, + ]}); + error( + fixture + .send("POST", "objects", Body::from(request.to_string())) + .await, + if oversized { 400 } else { 404 }, + if oversized { + "SCOPE_INVALID" + } else { + "PATH_NOT_FOUND" + }, + false, + ) + .await; + fixture.counts.assert(0, 0); + } +} + +#[tokio::test] +async fn unique_batch_over_eight_mib_rejects_before_any_body_io() { + let objects = files(33, 256 * 1024); + let fixture = fixture(&objects).await; + error( + fixture.send("POST", "objects", body(&objects)).await, + 400, + "SCOPE_INVALID", + false, + ) + .await; + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn cap_boundaries_and_empty_object_keep_exact_raw_bytes_and_end_counts() { + let mut objects = files(32, 256 * 1024); + objects.push(("empty".into(), Vec::new())); + let fixture = fixture(&objects[..32]).await; + let request = request_bytes(&objects); + assert_objects( + fixture + .send("POST", "objects", Body::from(request.clone())) + .await, + &request, + &objects, + ) + .await; + fixture.counts.assert(33, 8 * 1024 * 1024); +} + +#[tokio::test] +async fn every_alias_is_admitted_and_exact_oid_body_is_loaded_once() { + let raw = files(1, 8192).remove(0).1; + let objects: Vec<_> = (0..128) + .map(|i| (format!("alias-{i:03}"), raw.clone())) + .collect(); + let fixture = fixture(&objects).await; + let request = request_bytes(&objects); + assert_objects( + fixture + .send("POST", "objects", Body::from(request.clone())) + .await, + &request, + &objects, + ) + .await; + fixture.counts.assert(1, raw.len()); + fixture.counts.reset(); + let mut request: Value = serde_json::from_slice(&request).unwrap(); + request["items"][127]["expected_digest"] = json!(format!("sha256:{}", "00".repeat(32))); + error( + fixture + .send("POST", "objects", Body::from(request.to_string())) + .await, + 409, + "EXPECTED_DIGEST_MISMATCH", + false, + ) + .await; + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn conflicting_sizes_reject_before_io_and_distinct_oids_still_verify_each_body() { + let objects = files(2, 8192); + let fixture = fixture(&objects).await; + let oid = object_oid(&fixture, "/object-001").await; + let db = fixture + .state + .storage + .mono_storage() + .get_connection() + .clone(); + let fact = mst2_verified_object::Entity::find() + .filter(mst2_verified_object::Column::GitOid.eq(oid)) + .one(&db) + .await + .unwrap() + .unwrap(); + let mut fact = fact.into_active_model(); + fact.raw_sha256 = Set(digest(&objects[0].1).to_vec()); + fact.size = Set(8193); + let fact = fact.update(&db).await.unwrap(); + let request = json!({"items":[ + {"path":"/object-000","expected_digest":format!("sha256:{}", hex_of(&digest(&objects[0].1)))}, + {"path":"/object-001","expected_digest":format!("sha256:{}", hex_of(&digest(&objects[0].1)))}, + ]}); + error( + fixture + .send("POST", "objects", Body::from(request.to_string())) + .await, + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + fixture.counts.assert(0, 0); + let mut fact = fact.into_active_model(); + fact.size = Set(8192); + fact.update(&db).await.unwrap(); + error( + fixture + .send("POST", "objects", Body::from(request.to_string())) + .await, + 409, + "EXPECTED_DIGEST_MISMATCH", + false, + ) + .await; + fixture.counts.assert(2, 16384); +} + +#[tokio::test] +async fn missing_or_invalid_current_facts_fail_before_body_io() { + let fixture = Fixture::new().await; + let request = json!({"items":[{"path":"/file","expected_digest":fixture.digest_string()}]}); + let fact = fixture.fact().await; + fixture.delete_fact().await; + error( + fixture + .send("POST", "objects", Body::from(request.to_string())) + .await, + 503, + "METADATA_NOT_READY", + true, + ) + .await; + fixture.counts.assert(0, 0); + let mut invalid = fact; + invalid.raw_sha256 = vec![0; 31]; + fixture.replace_fact(invalid).await; + error( + fixture + .send("POST", "objects", Body::from(request.to_string())) + .await, + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn oversized_stream_chunk_is_rejected_without_copying_or_polling_tail() { + let objects = files(1, 8192); + let fixture = fixture(&objects).await; + let tail_polls = Arc::new(AtomicUsize::new(0)); + install_fault( + &fixture, + "/object-000", + FaultKind::Oversized(Bytes::from(vec![7; 8193]), tail_polls.clone()), + ) + .await; + error( + fixture.send("POST", "objects", body(&objects)).await, + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + fixture.counts.assert(1, 8193); + assert_eq!(tail_polls.load(Ordering::SeqCst), 0); +} + +#[tokio::test] +async fn truncated_wrong_sha_and_late_stream_error_never_produce_200() { + let objects = files(1, 8192); + let fixture = fixture(&objects).await; + for (fault, status, code, bytes) in [ + ( + FaultKind::Parts(vec![Bytes::copy_from_slice(&objects[0].1[..8191])]), + 502, + "INTEGRITY_ERROR", + 8191, + ), + ( + FaultKind::Parts(vec![Bytes::from(vec![0; 8192])]), + 409, + "EXPECTED_DIGEST_MISMATCH", + 8192, + ), + ( + FaultKind::LateError(Bytes::copy_from_slice(&objects[0].1)), + 500, + "INTERNAL", + 8192, + ), + ( + FaultKind::Parts(vec![ + Bytes::copy_from_slice(&objects[0].1), + Bytes::from_static(b"x"), + ]), + 502, + "INTEGRITY_ERROR", + 8193, + ), + ] { + fixture.counts.reset(); + install_fault(&fixture, "/object-000", fault).await; + error( + fixture.send("POST", "objects", body(&objects)).await, + status, + code, + code == "INTERNAL", + ) + .await; + fixture.counts.assert(1, bytes); + } +} + +#[tokio::test] +async fn a_later_object_stream_failure_keeps_earlier_verified_data_unpublished() { + let objects = files(2, 8192); + let fixture = fixture(&objects).await; + install_fault( + &fixture, + "/object-001", + FaultKind::LateError(Bytes::copy_from_slice(&objects[1].1)), + ) + .await; + error( + fixture.send("POST", "objects", body(&objects)).await, + 500, + "INTERNAL", + true, + ) + .await; + fixture.counts.assert(2, 16384); +} + +#[tokio::test] +async fn missing_actual_object_body_keeps_source_error_classification_and_can_retry() { + let objects = files(1, 8192); + let fixture = fixture(&objects).await; + let oid = object_oid(&fixture, "/object-000").await; + fixture + .state + .storage + .git_service + .obj_storage + .inner + .delete(&ObjectKey { + namespace: ObjectNamespace::Git, + key: oid.clone(), + }) + .await + .unwrap(); + error( + fixture.send("POST", "objects", body(&objects)).await, + 503, + "OBJECT_UNAVAILABLE", + false, + ) + .await; + fixture.counts.assert(1, 0); + fixture + .state + .storage + .git_service + .save_object_from_model(objects[0].1.clone(), &oid) + .await + .unwrap(); + fixture.counts.reset(); + let request = request_bytes(&objects); + assert_objects( + fixture + .send("POST", "objects", Body::from(request.clone())) + .await, + &request, + &objects, + ) + .await; + fixture.counts.assert(1, 8192); +} + +#[tokio::test] +async fn old_snapshot_objects_keep_fixed_oid_after_real_publication_advances() { + let objects = files(1, 8192); + let fixture = fixture(&objects).await; + let mono = fixture.state.storage.mono_storage(); + let old = mono.get_main_ref("/project").await.unwrap().unwrap(); + let old_commit = ObjectHash::from_hex_for_kind(HashKind::Sha1, &old.ref_commit_hash).unwrap(); + let oid = object_oid(&fixture, "/object-000").await; + let next_tree = tree(vec![item( + TreeItemMode::Blob, + ObjectHash::from_hex_for_kind(HashKind::Sha1, &oid).unwrap(), + "new-only", + )]); + let next_commit = Commit::from_tree_id_with_kind( + HashKind::Sha1, + next_tree.id, + vec![old_commit], + "OBJECT old snapshot must retain its fixed path", + ) + .unwrap(); + mono.save_mega_trees(vec![next_tree], next_commit.id, None) + .await + .unwrap(); + mono.save_mega_commits(vec![next_commit.clone()], None) + .await + .unwrap(); + publish_native_push(&fixture.state.storage, "/project", old_commit, &next_commit).await; + let head = mono + .read_native_publication_head( + fixture + .state + .storage + .config() + .mst2 + .instance_uuid + .as_deref() + .unwrap(), + ) + .await + .unwrap(); + assert_eq!(head.token.sequence, 2); + assert!(head.token.certificate.is_some()); + assert_ne!( + mono.get_main_ref("/project") + .await + .unwrap() + .unwrap() + .ref_tree_hash, + old.ref_tree_hash + ); + fixture.counts.reset(); + let request = request_bytes(&objects); + assert_objects( + fixture + .send("POST", "objects", Body::from(request.clone())) + .await, + &request, + &objects, + ) + .await; + fixture.counts.assert(1, 8192); +} + +#[tokio::test] +async fn cancelling_actual_object_request_drops_held_stream_and_retry_succeeds() { + let objects = files(1, 8192); + let fixture = fixture(&objects).await; + let entered = Arc::new(Notify::new()); + let release = Arc::new(Notify::new()); + let drops = Arc::new(AtomicUsize::new(0)); + install_fault( + &fixture, + "/object-000", + FaultKind::Held { + raw: Bytes::copy_from_slice(&objects[0].1), + entered: entered.clone(), + release, + drops: drops.clone(), + }, + ) + .await; + let task = tokio::spawn(fixture.app.clone().oneshot(fixture.request( + "POST", + "objects", + body(&objects), + ))); + timeout(Duration::from_secs(10), entered.notified()) + .await + .unwrap(); + assert!(!task.is_finished()); + fixture.counts.assert(1, 1); + task.abort(); + assert!(task.await.err().unwrap().is_cancelled()); + assert_eq!(drops.load(Ordering::SeqCst), 1); + *fixture.counts.object_fault.lock().unwrap() = None; + fixture.counts.reset(); + let request = request_bytes(&objects); + assert_objects( + fixture + .send("POST", "objects", Body::from(request.clone())) + .await, + &request, + &objects, + ) + .await; + fixture.counts.assert(1, 8192); +} + +#[tokio::test] +async fn lease_revoked_during_object_load_cannot_deliver_verified_frames() { + let objects = files(1, 8192); + let fixture = fixture(&objects).await; + let entered = Arc::new(Notify::new()); + let release = Arc::new(Notify::new()); + let drops = Arc::new(AtomicUsize::new(0)); + install_fault( + &fixture, + "/object-000", + FaultKind::Held { + raw: Bytes::copy_from_slice(&objects[0].1), + entered: entered.clone(), + release: release.clone(), + drops: drops.clone(), + }, + ) + .await; + let task = tokio::spawn(fixture.app.clone().oneshot(fixture.request( + "POST", + "objects", + body(&objects), + ))); + timeout(Duration::from_secs(10), entered.notified()) + .await + .unwrap(); + let revoked = fixture + .app + .clone() + .oneshot( + Request::builder() + .method("DELETE") + .uri(format!("/api/v2/snapshots/leases/{}", fixture.lease)) + .header("authorization", format!("Bearer {TOKEN}")) + .body(Body::empty()) + .unwrap(), + ) + .await + .unwrap(); + assert_eq!(revoked.status(), 200); + release.notify_one(); + let response = timeout(Duration::from_secs(10), task) + .await + .unwrap() + .unwrap() + .unwrap(); + error(response, 410, "LEASE_EXPIRED", false).await; + fixture.counts.assert(1, 8192); + assert_eq!(drops.load(Ordering::SeqCst), 1); +} diff --git a/src/api/router/snapshot_persisted_chunk_map_tests.rs b/src/api/router/snapshot_persisted_chunk_map_tests.rs new file mode 100644 index 00000000..23e1808b --- /dev/null +++ b/src/api/router/snapshot_persisted_chunk_map_tests.rs @@ -0,0 +1,1303 @@ +use sea_orm::{DatabaseConnection, DbBackend, IsolationLevel, Statement, TransactionTrait}; +use tokio::{sync::Notify, time::timeout}; + +use super::*; +use crate::{ + ceres::snapshot::{ + chunks::{ChunkMapSource, ChunkProjection}, + content_budget::MemoryBudget, + }, + jupiter::storage::native_chunk_map::PostgresChunkMapRepository, +}; + +fn statement(sql: &str, values: [sea_orm::Value; N]) -> Statement { + Statement::from_sql_and_values(DbBackend::Postgres, sql, values) +} + +async fn count(db: &DatabaseConnection, table: &str) -> i64 { + db.query_one_raw(statement( + &format!("SELECT count(*) AS count FROM {table}"), + [], + )) + .await + .unwrap() + .unwrap() + .try_get("", "count") + .unwrap() +} + +async fn reconstructed(fixture: &Fixture) -> MonoApiServiceState { + let mut config = (*fixture.state.storage.config()).clone(); + config.database.max_connection = 1; + config.database.min_connection = 1; + let config = Arc::new(config); + let connection = crate::jupiter::storage::init::postgres_connection(&config.database) + .await + .unwrap(); + let storage = crate::jupiter::storage::Storage::new_with_connection( + config, + Arc::new(connection), + fixture.state.storage.git_service.obj_storage.clone(), + ) + .await + .unwrap(); + MonoApiServiceState { + storage, + git_object_cache: Arc::new(GitObjectCache { + connection: fixture.state.git_object_cache.connection.clone(), + prefix: uuid::Uuid::new_v4().to_string(), + }), + ..fixture.state.clone() + } +} + +fn router(state: &MonoApiServiceState) -> Router { + Router::new().nest("/api/v2", routers(state.clone()).with_state(state.clone())) +} + +#[tokio::test] +async fn persisted_map_rebuilt_actual_http_uses_canonical_pages_and_only_requested_raw_range() { + let fixture = Fixture::new_with_pg_config(true).await; + let original = fixture.map("/file").await; + fixture.counts.assert(1, fixture.raw.len()); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 1); + let oracle = ChunkProjection::build(fixture.digest, fixture.raw.clone()).unwrap(); + assert_eq!( + original["map"]["map_id"], + format!("sha256:{}", hex_of(&oracle.map_id)) + ); + let state = reconstructed(&fixture).await; + assert!(state.storage.native_chunk_maps.get().is_none()); + let app = router(&state); + fixture.counts.reset(); + let map = success_json( + app.clone() + .oneshot(fixture.request("GET", "chunk-map?path=/alias", Body::empty())) + .await + .unwrap(), + ) + .await; + assert_eq!(map["map"], original["map"]); + let map_id = map["map"]["map_id"].as_str().unwrap(); + let page = success_json( + app.clone() + .oneshot(fixture.request( + "GET", + &format!("chunk-map/pages?path=/alias&map_id={map_id}&page_index=0"), + Body::empty(), + )) + .await + .unwrap(), + ) + .await; + let bytes = STANDARD + .decode(page["leaf_base64"].as_str().unwrap()) + .unwrap(); + let (leaf, proof) = oracle.leaf_and_proof(0).unwrap(); + assert_eq!(bytes, leaf.encode().unwrap()); + assert!(proof.is_empty()); + assert_eq!(page["proof"], json!([])); + fixture.counts.assert(0, 0); + assert!(fixture.counts.receipt_reads.load(Ordering::SeqCst) >= 2); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 0); + let body = fixture.chunk_body("/alias", map_id, "1").to_string(); + let response = app + .oneshot(fixture.request("POST", "chunks", Body::from(body.clone()))) + .await + .unwrap(); + assert_eq!(response.status(), 200); + let wire = to_bytes(response.into_body(), 2 * 1024 * 1024) + .await + .unwrap(); + let frames = parse_stream(&wire).unwrap(); + let [Frame::Chunk(chunk), Frame::End(end)] = frames.as_slice() else { + panic!("expected exact CHUNK and END"); + }; + assert_eq!(chunk.chunk_bytes, fixture.raw[CHUNK_SIZE as usize..]); + assert_eq!(chunk.chunk_index, 1); + assert_eq!(chunk.file_content_id, fixture.digest); + assert_eq!(chunk.map_id, oracle.map_id); + assert_eq!(end.request_item_count, 1); + assert_eq!(end.unique_unit_count, 1); + assert_eq!(end.logical_bytes, 113); + assert_eq!( + end.request_body_sha256, + <[u8; 32]>::from(Sha256::digest(body.as_bytes())) + ); + assert_eq!(fixture.counts.whole.load(Ordering::SeqCst), 0); + assert_eq!(fixture.counts.range.load(Ordering::SeqCst), 1); + assert_eq!(fixture.counts.bytes.load(Ordering::SeqCst), 113); +} + +#[tokio::test] +async fn persisted_map_current_fact_tuple_receipt_and_leaf_corruption_fail_closed_without_fallback() +{ + let fixture = Fixture::new().await; + let map = fixture.map("/file").await; + let original = fixture.fact().await; + let map_id = map["map"]["map_id"].as_str().unwrap(); + fixture.counts.reset(); + for case in 0..3 { + let mut fact = original.clone(); + match case { + 0 => fact.id += 1_000_000, + 1 => fact.created_at += chrono::Duration::seconds(1), + _ => fact.raw_sha256[0] ^= 1, + } + fixture.replace_fact(fact).await; + error( + fixture + .send("GET", "chunk-map?path=/file", Body::empty()) + .await, + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + fixture.counts.assert(0, 0); + } + fixture.replace_fact(original).await; + fixture + .counts + .receipt_read_failure + .store(true, Ordering::SeqCst); + error( + fixture + .send("GET", "chunk-map?path=/file", Body::empty()) + .await, + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + fixture + .counts + .receipt_read_failure + .store(false, Ordering::SeqCst); + *fixture.counts.receipt_read_corruption.lock().unwrap() = + Some(Bytes::from_static(b"forged DB proof")); + error( + fixture + .send("GET", "chunk-map?path=/file", Body::empty()) + .await, + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + *fixture.counts.receipt_read_corruption.lock().unwrap() = None; + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + let leaf: Vec = db + .query_one_raw(statement( + "SELECT payload FROM mst2_chunk_map_leaf WHERE map_id=$1 AND page_index=0", + [hex::decode(map_id.trim_start_matches("sha256:")) + .unwrap() + .into()], + )) + .await + .unwrap() + .unwrap() + .try_get("", "payload") + .unwrap(); + db.execute_unprepared("ALTER TABLE mst2_chunk_map_leaf DISABLE TRIGGER USER") + .await + .unwrap(); + let mut bad = leaf.clone(); + bad[16] ^= 1; + db.execute_raw(statement( + "UPDATE mst2_chunk_map_leaf SET payload=$1", + [bad.into()], + )) + .await + .unwrap(); + db.execute_unprepared("ALTER TABLE mst2_chunk_map_leaf ENABLE TRIGGER USER") + .await + .unwrap(); + for (method, suffix, body) in [ + ( + "GET", + format!("chunk-map/pages?path=/file&map_id={map_id}&page_index=0"), + Body::empty(), + ), + ( + "POST", + "chunks".to_string(), + Body::from(fixture.chunk_body("/file", map_id, "0").to_string()), + ), + ] { + error( + fixture.send(method, &suffix, body).await, + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + } + fixture.counts.assert(0, 0); + db.execute_unprepared("ALTER TABLE mst2_chunk_map_leaf DISABLE TRIGGER USER") + .await + .unwrap(); + db.execute_raw(statement( + "UPDATE mst2_chunk_map_leaf SET payload=$1", + [leaf.into()], + )) + .await + .unwrap(); + db.execute_unprepared("ALTER TABLE mst2_chunk_map_leaf ENABLE TRIGGER USER") + .await + .unwrap(); + assert_eq!(fixture.map("/file").await["map"], map["map"]); + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn ordinary_db_dml_cannot_forge_full_body_admission_or_mutate_admitted_indexes() { + let fixture = Fixture::new().await; + let source = ChunkMapSource::from_fact(fixture.fact().await, &fixture.oid).unwrap(); + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + let repository = PostgresChunkMapRepository::new(db.clone()).await.unwrap(); + let scope = repository.test_primary_scope(); + let source_bytes = source.canonical_bytes().unwrap(); + let leaf = ChunkLeaf { + page_index: 0, + chunk_sha256: vec![[9; 32]; 2], + }; + let root = leaf.leaf_hash().unwrap(); + let map = mst2_codec::chunkmap::ChunkMap::new(fixture.digest, fixture.raw.len() as u64, root) + .unwrap(); + let mut receipt = b"MST2-CHUNK-MAP-RECEIPT\0".to_vec(); + receipt.extend_from_slice(&(scope.len() as u32).to_be_bytes()); + receipt.extend_from_slice(scope); + receipt.extend_from_slice(&(source_bytes.len() as u32).to_be_bytes()); + receipt.extend_from_slice(&source_bytes); + let source_id: [u8; 32] = Sha256::digest(&receipt).into(); + receipt.extend_from_slice(&map.encode()); + let txn = db + .begin_with_config(Some(IsolationLevel::ReadCommitted), None) + .await + .unwrap(); + txn.execute_raw(statement( + "INSERT INTO mst2_chunk_map(map_id,descriptor,page_count,pages_root) VALUES($1,$2,1,$3)", + [ + map.map_id().to_vec().into(), + map.encode().into(), + root.to_vec().into(), + ], + )) + .await + .unwrap(); + txn.execute_raw(statement( + "INSERT INTO mst2_chunk_map_leaf(map_id,page_index,payload) VALUES($1,0,$2)", + [map.map_id().to_vec().into(), leaf.encode().unwrap().into()], + )) + .await + .unwrap(); + txn.execute_raw(statement( + "INSERT INTO mst2_chunk_map_node(map_id,first_page,page_count,digest) VALUES($1,0,1,$2)", + [map.map_id().to_vec().into(), root.to_vec().into()], + )) + .await + .unwrap(); + txn.execute_raw(statement("INSERT INTO mst2_chunk_map_source(storage_domain,git_oid,object_kind,fact_id,source_id,source_bytes,primary_scope,map_id,receipt_digest) VALUES('git',$1,'blob',$2,$3,$4,$5,$6,$7)", [fixture.oid.clone().into(),source.fact().id.into(),source_id.to_vec().into(),source_bytes.into(),scope.to_vec().into(),map.map_id().to_vec().into(),Sha256::digest(&receipt).to_vec().into()])).await.unwrap(); + txn.commit().await.unwrap(); + // All row checks and even a self-computed receipt digest pass. They + // still cannot create the trusted writer's independent object receipt. + fixture.counts.reset(); + error( + fixture + .send("GET", "chunk-map?path=/file", Body::empty()) + .await, + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + fixture.counts.assert(0, 0); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 0); + for table in [ + "mst2_chunk_map", + "mst2_chunk_map_leaf", + "mst2_chunk_map_node", + "mst2_chunk_map_source", + ] { + assert!( + db.execute_unprepared(&format!("DELETE FROM {table}")) + .await + .is_err() + ); + assert!( + db.execute_unprepared(&format!("TRUNCATE {table} CASCADE")) + .await + .is_err() + ); + } + assert!( + db.execute_unprepared("UPDATE mst2_chunk_map_source SET fact_id=fact_id") + .await + .is_err() + ); + assert!(db.execute_raw(statement("INSERT INTO mst2_chunk_map_node(map_id,first_page,page_count,digest) VALUES($1,1,1,$2)", [map.map_id().to_vec().into(),root.to_vec().into()])).await.is_err()); +} + +#[tokio::test] +async fn receipt_orphan_failure_and_cancelled_install_release_owned_credit_and_replay_atomically() { + for cancel in [false, true] { + let fixture = Fixture::new().await; + let mono = fixture.state.storage.mono_storage(); + let budget = MemoryBudget::new(8 * 1024 * 1024); + let repository = PostgresChunkMapRepository::new(mono.get_connection().clone()) + .await + .unwrap() + .with_test_budget(budget.clone()); + assert!( + fixture + .state + .storage + .native_chunk_maps + .set(repository) + .is_ok() + ); + if cancel { + let entered = Arc::new(Notify::new()); + let release = Arc::new(Notify::new()); + *fixture.counts.receipt_write_holds.lock().unwrap() = Some((entered.clone(), release)); + let app = fixture.app.clone(); + let request = fixture.request("GET", "chunk-map?path=/file", Body::empty()); + let task = tokio::spawn(async move { app.oneshot(request).await }); + timeout(Duration::from_secs(10), entered.notified()) + .await + .unwrap(); + assert!(budget.used() > 4 * 1024 * 1024); + assert_eq!( + count(mono.get_connection(), "mst2_chunk_map_source").await, + 0 + ); + task.abort(); + assert!(task.await.err().unwrap().is_cancelled()); + *fixture.counts.receipt_write_holds.lock().unwrap() = None; + } else { + fixture + .counts + .receipt_write_fail_after_create + .store(true, Ordering::SeqCst); + error( + fixture + .send("GET", "chunk-map?path=/file", Body::empty()) + .await, + 503, + "TEMPORARY_UNAVAILABLE", + true, + ) + .await; + fixture + .counts + .receipt_write_fail_after_create + .store(false, Ordering::SeqCst); + } + assert_eq!(budget.used(), 0); + fixture.counts.assert(1, fixture.raw.len()); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 1); + assert_eq!(count(mono.get_connection(), "mst2_chunk_map").await, 0); + fixture.counts.reset(); + fixture.map("/file").await; + fixture.counts.assert(1, fixture.raw.len()); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 1); + assert_eq!(count(mono.get_connection(), "mst2_chunk_map").await, 1); + assert_eq!( + count(mono.get_connection(), "mst2_chunk_map_source").await, + 1 + ); + assert_eq!(count(mono.get_connection(), "mst2_chunk_map_leaf").await, 1); + assert_eq!(count(mono.get_connection(), "mst2_chunk_map_node").await, 1); + fixture.counts.reset(); + fixture.map("/file").await; + fixture.counts.assert(0, 0); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 0); + assert_eq!(budget.used(), 0); + } +} + +#[tokio::test] +async fn map_json_transport_bytes_keep_owned_credit_until_the_last_clone_drops() { + let budget = MemoryBudget::new(4096); + let bytes = super::super::map_json_bytes( + &json!({"map_id":"sha256:owned"}), + budget.reserve(4096).unwrap(), + ) + .unwrap(); + let mut stream = Body::from(bytes).into_data_stream(); + let transport = stream.next().await.unwrap().unwrap(); + let clone = transport.clone(); + drop(transport); + drop(stream); + assert_eq!(budget.used(), 4096); + assert!(budget.reserve(1).is_err()); + drop(clone); + assert_eq!(budget.used(), 0); + assert!(budget.reserve(4096).is_ok()); +} + +#[test] +fn json_wire_limit_rejects_growth_and_refunds_credit() { + let budget = MemoryBudget::new(4096); + let value = json!({"path":"\u{1}".repeat(1000)}); + let error = super::super::map_json_bytes(&value, budget.reserve(4096).unwrap()) + .err() + .unwrap(); + assert_eq!(error.code, SnapshotErrorCode::Internal); + assert_eq!(budget.used(), 0); +} + +async fn observer(fixture: &Fixture) -> crate::ceres::snapshot::chunk_map_gate::InstallFlight { + let repository = fixture.state.storage.chunk_maps().await.unwrap(); + let source = ChunkMapSource::from_fact(fixture.fact().await, &fixture.oid).unwrap(); + crate::ceres::snapshot::chunk_map_gate::InstallFlight::acquire( + repository.source_identity(&source).unwrap(), + ) + .unwrap() +} + +async fn wait_owners( + flight: &crate::ceres::snapshot::chunk_map_gate::InstallFlight, + owners: usize, +) { + timeout(Duration::from_secs(10), async { + loop { + if flight.test_owner_count() == owners { + break; + } + tokio::task::yield_now().await; + } + }) + .await + .expect("actual HTTP callers did not reach the same-source install gate"); +} + +fn held_leader(fixture: &Fixture) -> (Arc, Arc, tokio::task::JoinHandle) { + let entered = Arc::new(Notify::new()); + let release = Arc::new(Notify::new()); + *fixture.counts.receipt_write_holds.lock().unwrap() = Some((entered.clone(), release.clone())); + let app = fixture.app.clone(); + let request = fixture.request("GET", "chunk-map?path=/file", Body::empty()); + let task = tokio::spawn(async move { app.oneshot(request).await.unwrap() }); + (entered, release, task) +} + +#[tokio::test] +async fn same_source_cold_actual_http_callers_share_one_full_pass_and_each_recheck_their_receipt() { + let fixture = Fixture::new().await; + let (entered, release, leader) = held_leader(&fixture); + timeout(Duration::from_secs(10), entered.notified()) + .await + .unwrap(); + let flight = observer(&fixture).await; + let mut joined = Vec::new(); + for _ in 0..6 { + let app = fixture.app.clone(); + let request = fixture.request("GET", "chunk-map?path=/alias", Body::empty()); + joined.push(tokio::spawn( + async move { app.oneshot(request).await.unwrap() }, + )); + } + let other_counts = Arc::new(ReadCounts::default()); + other_counts + .receipt_read_failure + .store(true, Ordering::SeqCst); + let mut other_state = fixture.state.clone(); + other_state.storage.git_service = GitService { + obj_storage: MegaObjectStorageWrapper::new(Arc::new(CountingStorage { + inner: fixture.state.storage.git_service.obj_storage.clone(), + counts: other_counts.clone(), + })), + }; + let other_app = router(&other_state); + let request = fixture.request("GET", "chunk-map?path=/file", Body::empty()); + let rejected = tokio::spawn(async move { other_app.oneshot(request).await.unwrap() }); + wait_owners(&flight, 9).await; // Leader, seven callers and this observer. + fixture.counts.assert(1, fixture.raw.len()); + release.notify_one(); + let expected = success_json(leader.await.unwrap()).await; + for task in joined { + assert_eq!( + success_json(task.await.unwrap()).await["map"], + expected["map"] + ); + } + error(rejected.await.unwrap(), 502, "INTEGRITY_ERROR", false).await; + assert_eq!(other_counts.receipt_reads.load(Ordering::SeqCst), 1); + assert_eq!(other_counts.whole.load(Ordering::SeqCst), 0); + fixture.counts.assert(1, fixture.raw.len()); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 1); + assert_eq!(fixture.counts.receipt_reads.load(Ordering::SeqCst), 8); + wait_owners(&flight, 1).await; +} + +#[tokio::test] +async fn failed_or_cancelled_leader_and_cancelled_waiter_leave_actual_http_retry_capacity() { + for mode in 0..3 { + let fixture = Fixture::new().await; + let budget = MemoryBudget::new(8 * 1024 * 1024); + let repository = PostgresChunkMapRepository::new( + fixture + .state + .storage + .mono_storage() + .get_connection() + .clone(), + ) + .await + .unwrap() + .with_test_budget(budget.clone()); + assert!( + fixture + .state + .storage + .native_chunk_maps + .set(repository) + .is_ok() + ); + let (entered, release, leader) = held_leader(&fixture); + timeout(Duration::from_secs(10), entered.notified()) + .await + .unwrap(); + let flight = observer(&fixture).await; + let app = fixture.app.clone(); + let request = fixture.request("GET", "chunk-map?path=/alias", Body::empty()); + let waiter = tokio::spawn(async move { app.oneshot(request).await.unwrap() }); + wait_owners(&flight, 3).await; + assert!(budget.used() > 4 * 1024 * 1024); + *fixture.counts.receipt_write_holds.lock().unwrap() = None; + match mode { + 0 => { + fixture + .counts + .receipt_write_fail_after_create + .store(true, Ordering::SeqCst); + release.notify_one(); + error(leader.await.unwrap(), 503, "TEMPORARY_UNAVAILABLE", true).await; + success_json(waiter.await.unwrap()).await; + } + 1 => { + leader.abort(); + assert!(leader.await.err().unwrap().is_cancelled()); + success_json(waiter.await.unwrap()).await; + } + _ => { + waiter.abort(); + assert!(waiter.await.err().unwrap().is_cancelled()); + wait_owners(&flight, 2).await; + release.notify_one(); + success_json(leader.await.unwrap()).await; + } + } + wait_owners(&flight, 1).await; + drop(flight); + assert_eq!(budget.used(), 0); + let passes = if mode == 2 { 1 } else { 2 }; + fixture.counts.assert(passes, passes * fixture.raw.len()); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), passes); + fixture.counts.reset(); + fixture.map("/file").await; + fixture.counts.assert(0, 0); + assert_eq!(budget.used(), 0); + } +} + +#[tokio::test] +async fn actual_http_long_escaped_legal_path_has_bounded_owned_json_and_empty_files_have_no_map() { + let component = "\u{1}".repeat(255); + let mut components = vec![component; 15]; + components.push("\u{1}".repeat(239)); + let name = components.join("/"); + let path = format!("/{name}"); + assert_eq!(path.len(), 4080); + crate::ceres::snapshot::view::validate_scope_relative_path(&path).unwrap(); + let fixture = Fixture::new_with_pg_config_directories_and_objects( + false, + 0, + &[(name, b"escaped path body".to_vec())], + ) + .await; + let encoded = url::form_urlencoded::byte_serialize(path.as_bytes()).collect::(); + for whole in [1, 0] { + fixture.counts.reset(); + let response = fixture + .send("GET", &format!("chunk-map?path={encoded}"), Body::empty()) + .await; + assert_eq!(response.status(), 200); + assert_eq!( + response.headers()["cache-control"], + "private, no-cache, no-transform" + ); + let bytes = to_bytes(response.into_body(), 64 * 1024).await.unwrap(); + assert!(bytes.len() > 4 * 1024); + let value: Value = serde_json::from_slice(&bytes).unwrap(); + assert_eq!(value["path"], path); + assert_eq!(value["map"]["file_size"], "17"); + fixture + .counts + .assert(whole, if whole == 0 { 0 } else { 17 }); + } + fixture.counts.reset(); + for suffix in ["chunk-map?path=/empty", "chunk-map/pages?path=/empty"] { + error( + fixture.send("GET", suffix, Body::empty()).await, + 400, + "SCOPE_INVALID", + false, + ) + .await; + } + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn actual_http_and_repository_initialization_ignore_poisoned_temp_fact_scope_and_map_shadows() +{ + let fixture = Fixture::new_with_pg_config(true).await; + let expected = fixture.map("/file").await; + let state = reconstructed(&fixture).await; + assert!(state.storage.native_chunk_maps.get().is_none()); + let mono = state.storage.mono_storage(); + let db = mono.get_connection(); + let schema = fixture + ._schema + .as_ref() + .unwrap() + .schema() + .replace('"', "\"\""); + let tables = [ + "mst2_verified_object", + "mst2_metadata_storage_scope", + "mst2_chunk_map", + "mst2_chunk_map_source", + "mst2_chunk_map_leaf", + "mst2_chunk_map_node", + ]; + for table in tables { + db.execute_unprepared(&format!("CREATE TEMP TABLE {table}(LIKE \"{schema}\".{table}); INSERT INTO pg_temp.{table} SELECT * FROM \"{schema}\".{table}")).await.unwrap(); + } + db.execute_unprepared("UPDATE pg_temp.mst2_verified_object SET raw_sha256=decode(repeat('09',32),'hex'); UPDATE pg_temp.mst2_metadata_storage_scope SET storage_uuid='temp-poison'; UPDATE pg_temp.mst2_chunk_map SET descriptor=decode('00','hex'); UPDATE pg_temp.mst2_chunk_map_leaf SET payload=decode('00','hex'); UPDATE pg_temp.mst2_chunk_map_node SET digest=decode(repeat('09',32),'hex')").await.unwrap(); + fixture.counts.reset(); + let app = router(&state); + let map = success_json( + app.clone() + .oneshot(fixture.request("GET", "chunk-map?path=/file", Body::empty())) + .await + .unwrap(), + ) + .await; + assert_eq!(map, expected); + let map_id = map["map"]["map_id"].as_str().unwrap(); + let page = success_json( + app.clone() + .oneshot(fixture.request( + "GET", + &format!("chunk-map/pages?path=/file&map_id={map_id}&page_index=0"), + Body::empty(), + )) + .await + .unwrap(), + ) + .await; + let oracle = ChunkProjection::build(fixture.digest, fixture.raw.clone()).unwrap(); + assert_eq!( + STANDARD + .decode(page["leaf_base64"].as_str().unwrap()) + .unwrap(), + oracle.leaf_and_proof(0).unwrap().0.encode().unwrap() + ); + let body = fixture.chunk_body("/file", map_id, "1").to_string(); + let response = app + .clone() + .oneshot(fixture.request("POST", "chunks", Body::from(body))) + .await + .unwrap(); + assert_eq!(response.status(), 200); + let wire = to_bytes(response.into_body(), 2 * 1024 * 1024) + .await + .unwrap(); + let frames = parse_stream(&wire).unwrap(); + let [Frame::Chunk(chunk), Frame::End(end)] = frames.as_slice() else { + panic!("expected CHUNK and END"); + }; + assert_eq!(chunk.chunk_bytes, fixture.raw[CHUNK_SIZE as usize..]); + assert_eq!(end.logical_bytes, 113); + assert_eq!(fixture.counts.whole.load(Ordering::SeqCst), 0); + assert_eq!(fixture.counts.range.load(Ordering::SeqCst), 1); + assert_eq!(fixture.counts.bytes.load(Ordering::SeqCst), 113); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 0); + // Real primary changes still fail despite an apparently healthy shadow. + let actual = fixture.state.storage.mono_storage(); + let source = ChunkMapSource::from_fact(fixture.fact().await, &fixture.oid).unwrap(); + let repository = state.storage.chunk_maps().await.unwrap(); + actual.get_connection().execute_unprepared("ALTER TABLE mst2_metadata_storage_scope DISABLE TRIGGER USER; UPDATE mst2_metadata_storage_scope SET storage_uuid='changed-real-primary'; ALTER TABLE mst2_metadata_storage_scope ENABLE TRIGGER USER").await.unwrap(); + assert_eq!( + repository + .read(&source, &state.storage.git_service.obj_storage) + .await + .err() + .unwrap() + .code, + crate::ceres::snapshot::error::SnapshotErrorCode::IntegrityError + ); + error( + app.oneshot(fixture.request("GET", "chunk-map?path=/file", Body::empty())) + .await + .unwrap(), + 502, + "INTEGRITY_ERROR", + false, + ) + .await; +} + +pub(super) async fn assert_three_page_proofs_and_selected_sibling_faults( + fixture: &Fixture, + map_id: &str, + digest: [u8; 32], + pattern: &[u8], +) { + use mst2_codec::chunkmap::{ChunkMap, ProofSide, leaf_proof, merkle_root}; + let full: [u8; 32] = Sha256::digest(pattern).into(); + let final_chunk: [u8; 32] = Sha256::digest(&pattern[..7]).into(); + let leaves = [ + ChunkLeaf { + page_index: 0, + chunk_sha256: vec![full; 256], + }, + ChunkLeaf { + page_index: 1, + chunk_sha256: vec![full; 256], + }, + ChunkLeaf { + page_index: 2, + chunk_sha256: vec![final_chunk], + }, + ]; + let hashes: Vec<_> = leaves + .iter() + .map(|leaf| leaf.leaf_hash().unwrap()) + .collect(); + let root = merkle_root(&hashes).unwrap(); + let oracle = ChunkMap::new(digest, 512 * CHUNK_SIZE as u64 + 7, root).unwrap(); + assert_eq!(map_id, format!("sha256:{}", hex_of(&oracle.map_id()))); + for index in 0..3 { + let page = success_json( + fixture + .send( + "GET", + &format!("chunk-map/pages?path=/file&map_id={map_id}&page_index={index}"), + Body::empty(), + ) + .await, + ) + .await; + assert_eq!( + STANDARD + .decode(page["leaf_base64"].as_str().unwrap()) + .unwrap(), + leaves[index as usize].encode().unwrap() + ); + let proof = leaf_proof(&hashes, index).unwrap(); + let expected: Vec<_> = proof.iter().map(|step| json!({ + "side": if step.side == ProofSide::Left { "left" } else { "right" }, + "sibling_pages": step.sibling_pages.to_string(), "digest": format!("sha256:{}", hex_of(&step.digest)), + })).collect(); + assert_eq!(page["proof"], json!(expected)); + verify_leaf( + 3, + index, + leaves[index as usize].leaf_hash().unwrap(), + &proof, + root, + ) + .unwrap(); + } + fixture.counts.assert(0, 0); + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + let id = oracle.map_id().to_vec(); + for missing in [false, true] { + db.execute_unprepared("ALTER TABLE mst2_chunk_map_node DISABLE TRIGGER USER") + .await + .unwrap(); + if missing { + db.execute_raw(statement( + "DELETE FROM mst2_chunk_map_node WHERE map_id=$1 AND first_page=2 AND page_count=1", + [id.clone().into()], + )) + .await + .unwrap(); + } else { + db.execute_raw(statement("UPDATE mst2_chunk_map_node SET digest=$2 WHERE map_id=$1 AND first_page=2 AND page_count=1", [id.clone().into(), vec![9u8; 32].into()])).await.unwrap(); + } + db.execute_unprepared("ALTER TABLE mst2_chunk_map_node ENABLE TRIGGER USER") + .await + .unwrap(); + for (method, suffix, body) in [ + ( + "GET", + format!("chunk-map/pages?path=/file&map_id={map_id}&page_index=0"), + Body::empty(), + ), + ( + "POST", + "chunks".to_string(), + Body::from(fixture.chunk_body("/file", map_id, "0").to_string()), + ), + ] { + error( + fixture.send(method, &suffix, body).await, + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + } + fixture.counts.assert(0, 0); + // Page 2 proves itself through [0,2); the unrelated damaged node is + // absent from its selected SQL and does not turn into a full-map scan. + let page = success_json( + fixture + .send( + "GET", + &format!("chunk-map/pages?path=/file&map_id={map_id}&page_index=2"), + Body::empty(), + ) + .await, + ) + .await; + assert_eq!( + STANDARD + .decode(page["leaf_base64"].as_str().unwrap()) + .unwrap(), + leaves[2].encode().unwrap() + ); + db.execute_unprepared("ALTER TABLE mst2_chunk_map_node DISABLE TRIGGER USER") + .await + .unwrap(); + if missing { + db.execute_raw(statement("INSERT INTO mst2_chunk_map_node(map_id,first_page,page_count,digest) VALUES($1,2,1,$2)", [id.clone().into(), hashes[2].to_vec().into()])).await.unwrap(); + } else { + db.execute_raw(statement("UPDATE mst2_chunk_map_node SET digest=$2 WHERE map_id=$1 AND first_page=2 AND page_count=1", [id.clone().into(), hashes[2].to_vec().into()])).await.unwrap(); + } + db.execute_unprepared("ALTER TABLE mst2_chunk_map_node ENABLE TRIGGER USER") + .await + .unwrap(); + } + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn failed_digest_failed_body_and_cancelled_cold_producer_allow_the_joined_current_source_to_retry() + { + use super::bounded_objects::{FaultKind, StreamFault}; + for mode in 0..3 { + let fixture = Fixture::new().await; + let budget = MemoryBudget::new(8 * 1024 * 1024); + let repository = PostgresChunkMapRepository::new( + fixture + .state + .storage + .mono_storage() + .get_connection() + .clone(), + ) + .await + .unwrap() + .with_test_budget(budget.clone()); + assert!( + fixture + .state + .storage + .native_chunk_maps + .set(repository) + .is_ok() + ); + let entered = Arc::new(Notify::new()); + let release = Arc::new(Notify::new()); + let drops = Arc::new(AtomicUsize::new(0)); + let kind = if mode == 1 { + FaultKind::HeldThenError { + entered: entered.clone(), + release: release.clone(), + drops: drops.clone(), + } + } else { + let mut raw = fixture.raw.clone(); + if mode == 0 { + raw[0] ^= 1; + } + FaultKind::Held { + raw: Bytes::from(raw), + entered: entered.clone(), + release: release.clone(), + drops: drops.clone(), + } + }; + *fixture.counts.object_fault.lock().unwrap() = Some(StreamFault { + oid: fixture.oid.clone(), + kind, + }); + let app = fixture.app.clone(); + let request = fixture.request("GET", "chunk-map?path=/file", Body::empty()); + let leader = tokio::spawn(async move { app.oneshot(request).await.unwrap() }); + timeout(Duration::from_secs(10), entered.notified()) + .await + .unwrap(); + let flight = observer(&fixture).await; + let app = fixture.app.clone(); + let request = fixture.request("GET", "chunk-map?path=/alias", Body::empty()); + let waiter = tokio::spawn(async move { app.oneshot(request).await.unwrap() }); + wait_owners(&flight, 3).await; + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 0); + assert!(budget.used() > 4 * 1024 * 1024); + *fixture.counts.object_fault.lock().unwrap() = None; + if mode == 2 { + leader.abort(); + assert!(leader.await.err().unwrap().is_cancelled()); + } else { + release.notify_one(); + error( + leader.await.unwrap(), + if mode == 0 { 502 } else { 503 }, + if mode == 0 { + "INTEGRITY_ERROR" + } else { + "OBJECT_UNAVAILABLE" + }, + false, + ) + .await; + } + let map = success_json(waiter.await.unwrap()).await; + assert_eq!( + map["map"]["file_content_id"], + format!("sha256:{}", hex_of(&fixture.digest)) + ); + assert_eq!(drops.load(Ordering::SeqCst), 1); + assert_eq!(fixture.counts.whole.load(Ordering::SeqCst), 2); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 1); + let failed_bytes = match mode { + 0 => fixture.raw.len(), + 1 => 0, + _ => 1, + }; + assert_eq!( + fixture.counts.bytes.load(Ordering::SeqCst), + fixture.raw.len() + failed_bytes + ); + wait_owners(&flight, 1).await; + drop(flight); + assert_eq!(budget.used(), 0); + fixture.counts.reset(); + fixture.map("/file").await; + fixture.counts.assert(0, 0); + } +} + +#[tokio::test] +async fn different_current_sources_enter_cold_installations_independently() { + let fixture = Fixture::new_with_pg_config_directories_and_objects( + false, + 0, + &[("other".into(), b"independent source".to_vec())], + ) + .await; + let (entered, release, leader) = held_leader(&fixture); + timeout(Duration::from_secs(10), entered.notified()) + .await + .unwrap(); + let app = fixture.app.clone(); + let request = fixture.request("GET", "chunk-map?path=/other", Body::empty()); + let other = tokio::spawn(async move { app.oneshot(request).await.unwrap() }); + timeout(Duration::from_secs(10), entered.notified()) + .await + .unwrap(); + assert_eq!(fixture.counts.whole.load(Ordering::SeqCst), 2); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 2); + release.notify_waiters(); + let map = success_json(leader.await.unwrap()).await; + let second = success_json(other.await.unwrap()).await; + assert_ne!(map["map"]["map_id"], second["map"]["map_id"]); + fixture.counts.assert(2, fixture.raw.len() + 18); +} + +#[tokio::test] +async fn persisted_descriptors_and_authenticated_pages_charge_until_the_last_live_reader_drops() { + let fixture = Fixture::new().await; + fixture.map("/file").await; + fixture.counts.reset(); + let budget = MemoryBudget::new(96 * 1024); + let repository = PostgresChunkMapRepository::new( + fixture + .state + .storage + .mono_storage() + .get_connection() + .clone(), + ) + .await + .unwrap() + .with_test_budget(budget.clone()); + let source = ChunkMapSource::from_fact(fixture.fact().await, &fixture.oid).unwrap(); + let map = repository + .read(&source, &fixture.state.storage.git_service.obj_storage) + .await + .unwrap() + .unwrap(); + let page = repository.selected_page(&map, 0).await.unwrap(); + let other_map = map.clone(); + let other_page = page.clone(); + drop(map); + drop(page); + assert_eq!(budget.used(), 96 * 1024); + assert_eq!( + budget.reserve(1).err().unwrap().code, + SnapshotErrorCode::TemporaryUnavailable + ); + other_page + .verify_chunk(&other_map.map, 1, &fixture.raw[CHUNK_SIZE as usize..]) + .unwrap(); + drop(other_map); + assert_eq!(budget.used(), 64 * 1024); + drop(other_page); + assert_eq!(budget.used(), 0); + let replay = repository + .read(&source, &fixture.state.storage.git_service.obj_storage) + .await + .unwrap() + .unwrap(); + assert_eq!(budget.used(), 32 * 1024); + drop(replay); + assert_eq!(budget.used(), 0); + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn actual_install_sql_failure_rolls_back_every_index_and_exact_retry_earns_admission_again() { + let fixture = Fixture::new().await; + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + let budget = MemoryBudget::new(8 * 1024 * 1024); + let repository = PostgresChunkMapRepository::new(db.clone()) + .await + .unwrap() + .with_test_budget(budget.clone()); + assert!( + fixture + .state + .storage + .native_chunk_maps + .set(repository) + .is_ok() + ); + db.execute_unprepared("CREATE FUNCTION chunk_map_install_test_failure() RETURNS trigger LANGUAGE plpgsql AS $test$ BEGIN RAISE EXCEPTION 'injected source-row failure after complete index insertion'; END $test$; CREATE TRIGGER chunk_map_install_test_failure BEFORE INSERT ON mst2_chunk_map_source FOR EACH ROW EXECUTE FUNCTION chunk_map_install_test_failure()").await.unwrap(); + error( + fixture + .send("GET", "chunk-map?path=/file", Body::empty()) + .await, + 503, + "TEMPORARY_UNAVAILABLE", + true, + ) + .await; + fixture.counts.assert(1, fixture.raw.len()); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 1); + assert_eq!(budget.used(), 0); + for table in [ + "mst2_chunk_map", + "mst2_chunk_map_leaf", + "mst2_chunk_map_node", + "mst2_chunk_map_source", + ] { + assert_eq!(count(db, table).await, 0); + } + db.execute_unprepared("DROP TRIGGER chunk_map_install_test_failure ON mst2_chunk_map_source; DROP FUNCTION chunk_map_install_test_failure()").await.unwrap(); + fixture.counts.reset(); + let map = fixture.map("/file").await; + fixture.counts.assert(1, fixture.raw.len()); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 1); + assert_eq!(budget.used(), 0); + for table in [ + "mst2_chunk_map", + "mst2_chunk_map_leaf", + "mst2_chunk_map_node", + "mst2_chunk_map_source", + ] { + assert_eq!(count(db, table).await, 1); + } + fixture.counts.reset(); + assert_eq!(fixture.map("/alias").await["map"], map["map"]); + fixture.counts.assert(0, 0); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 0); +} + +#[tokio::test] +async fn ordinary_incomplete_source_dml_cannot_commit_an_admitted_partial_map() { + let fixture = Fixture::new().await; + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + let source = ChunkMapSource::from_fact(fixture.fact().await, &fixture.oid).unwrap(); + let repository = PostgresChunkMapRepository::new(db.clone()).await.unwrap(); + let map = ChunkProjection::build(fixture.digest, fixture.raw.clone()).unwrap(); + let scope = repository.test_primary_scope(); + let txn = db + .begin_with_config(Some(IsolationLevel::ReadCommitted), None) + .await + .unwrap(); + txn.execute_raw(statement( + "INSERT INTO mst2_chunk_map(map_id,descriptor,page_count,pages_root) VALUES($1,$2,$3,$4)", + [ + map.map_id.to_vec().into(), + map.map.encode().into(), + (map.map.page_count as i32).into(), + map.map.pages_root.to_vec().into(), + ], + )) + .await + .unwrap(); + txn.execute_raw(statement("INSERT INTO mst2_chunk_map_source(storage_domain,git_oid,object_kind,fact_id,source_id,source_bytes,primary_scope,map_id,receipt_digest) VALUES('git',$1,'blob',$2,$3,$4,$5,$6,$7)", [fixture.oid.clone().into(), source.fact().id.into(), repository.source_identity(&source).unwrap().to_vec().into(), source.canonical_bytes().unwrap().into(), scope.to_vec().into(), map.map_id.to_vec().into(), vec![9u8;32].into()])).await.unwrap(); + assert!(txn.commit().await.is_err()); + for table in [ + "mst2_chunk_map", + "mst2_chunk_map_leaf", + "mst2_chunk_map_node", + "mst2_chunk_map_source", + ] { + assert_eq!(count(db, table).await, 0); + } + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn actual_map_and_page_json_revalidate_lease_before_the_first_transport_poll() { + for page in [false, true] { + let fixture = Fixture::new().await; + let map = fixture.map("/file").await; + let suffix = if page { + format!( + "chunk-map/pages?path=/file&map_id={}&page_index=0", + map["map"]["map_id"].as_str().unwrap() + ) + } else { + "chunk-map?path=/file".into() + }; + fixture.counts.reset(); + let response = fixture.send("GET", &suffix, Body::empty()).await; + assert_eq!(response.status(), 200); + let revoked = fixture + .app + .clone() + .oneshot( + Request::builder() + .method("DELETE") + .uri(format!("/api/v2/snapshots/leases/{}", fixture.lease)) + .header("authorization", format!("Bearer {TOKEN}")) + .body(Body::empty()) + .unwrap(), + ) + .await + .unwrap(); + assert_eq!(revoked.status(), 200); + let mut stream = response.into_body().into_data_stream(); + assert!(stream.next().await.unwrap().is_err()); + assert!(stream.next().await.is_none()); + fixture.counts.assert(0, 0); + } +} + +#[tokio::test] +async fn oversized_current_fact_digest_fails_before_source_or_receipt_io_and_recovers_canonical_bytes() + { + let fixture = Fixture::new().await; + let expected = fixture.map("/file").await; + let original = fixture.fact().await; + let mono = fixture.state.storage.mono_storage(); + mono.get_connection().execute_raw(statement("UPDATE mst2_verified_object SET raw_sha256=decode(repeat('ab',1048576),'hex') WHERE id=$1", [original.id.into()])).await.unwrap(); + fixture.counts.reset(); + error( + fixture + .send("GET", "chunk-map?path=/file", Body::empty()) + .await, + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + fixture.counts.assert(0, 0); + assert_eq!(fixture.counts.receipt_reads.load(Ordering::SeqCst), 0); + fixture.replace_fact(original).await; + assert_eq!(fixture.map("/file").await, expected); + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn warm_admission_requires_receipt_exact_bytes_size_and_final_eof_without_fallback() { + let fixture = Fixture::new().await; + let map = fixture.map("/file").await; + let repository = fixture.state.storage.chunk_maps().await.unwrap(); + let source = ChunkMapSource::from_fact(fixture.fact().await, &fixture.oid).unwrap(); + let key = ObjectKey { + namespace: ObjectNamespace::ChunkMapReceipt, + key: hex_of(&repository.source_identity(&source).unwrap()), + }; + let (mut stream, meta) = fixture + .state + .storage + .git_service + .obj_storage + .inner + .get_stream(&key) + .await + .unwrap(); + let mut original = Vec::new(); + while let Some(part) = stream.next().await { + original.extend_from_slice(&part.unwrap()); + } + assert_eq!(original.len() as i64, meta.size); + let mut wrong = original.clone(); + wrong[0] ^= 1; + let mut long = original.clone(); + long.push(0); + let cases = [ + original[..original.len() - 1].to_vec(), + long, + wrong, + original.clone(), + ]; + fixture + .counts + .receipt_read_meta_size + .store(meta.size, Ordering::SeqCst); + for (index, bytes) in cases.into_iter().enumerate() { + fixture.counts.reset(); + *fixture.counts.receipt_read_corruption.lock().unwrap() = Some(Bytes::from(bytes)); + fixture + .counts + .receipt_read_late_error + .store(index == 3, Ordering::SeqCst); + error( + fixture + .send("GET", "chunk-map?path=/file", Body::empty()) + .await, + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + fixture.counts.assert(0, 0); + assert_eq!(fixture.counts.receipt_reads.load(Ordering::SeqCst), 1); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 0); + } + *fixture.counts.receipt_read_corruption.lock().unwrap() = None; + fixture + .counts + .receipt_read_late_error + .store(false, Ordering::SeqCst); + assert_eq!(fixture.map("/file").await, map); + fixture.counts.assert(0, 0); +} diff --git a/src/api/router/snapshot_persisted_metadata_tests.rs b/src/api/router/snapshot_persisted_metadata_tests.rs new file mode 100644 index 00000000..ba95602d --- /dev/null +++ b/src/api/router/snapshot_persisted_metadata_tests.rs @@ -0,0 +1,904 @@ +use std::collections::BTreeSet; + +use mst2_codec::metapage::{Page, page_id}; + +use super::*; +use crate::jupiter::storage::native_snapshot_session::MetadataRouteRequest; + +fn metadata_body(items: Value, encoding: &str) -> Vec { + serde_json::to_vec_pretty(&json!({"encoding": encoding, "items": items})).unwrap() +} + +async fn metadata_bytes(fixture: &Fixture, app: Router, body: &[u8]) -> Vec { + let response = app + .oneshot(fixture.request("POST", "metadata/pages", Body::from(body.to_vec()))) + .await + .unwrap(); + assert_eq!(response.status(), 200); + assert_eq!(response.headers()["x-mega-snapshot-id"], fixture.snapshot); + to_bytes(response.into_body(), usize::MAX) + .await + .unwrap() + .to_vec() +} + +fn assert_metadata(bytes: &[u8], body: &[u8], count: u32, expected: &[([u8; 32], Vec)]) { + let frames = parse_stream(bytes).unwrap(); + let mut actual = Vec::new(); + for frame in &frames[..frames.len() - 1] { + let Frame::Meta(meta) = frame else { + panic!("metadata route emitted a non-META frame") + }; + assert!(meta.pages.len() <= mst2_codec::treeframe::META_MAX_PAGES); + assert!( + meta.pages + .iter() + .map(|(_, page)| 36 + page.len()) + .sum::() + <= mst2_codec::treeframe::META_MAX_RAW + ); + for (id, page) in &meta.pages { + assert_eq!(*id, page_id(page)); + Page::decode(page).unwrap(); + } + actual.extend_from_slice(&meta.pages); + } + assert_eq!(actual, expected); + let Frame::End(end) = frames.last().unwrap() else { + panic!("metadata route omitted END") + }; + assert_eq!(end.request_item_count, count); + assert_eq!(end.unique_unit_count, expected.len() as u32); + assert_eq!( + end.logical_bytes, + expected + .iter() + .map(|(_, page)| page.len() as u64) + .sum::() + ); + assert_eq!( + end.request_body_sha256, + <[u8; 32]>::from(Sha256::digest(body)) + ); +} + +#[tokio::test] +async fn mst2_persisted_meta_matches_canonical_routes_after_rebuild_and_advance_without_git_reads() +{ + let fixture = Fixture::new_generic_history_with_pg_config_and_directories(true, 140).await; + let context = fixture + .state + .storage + .snapshot_context(&fixture.snapshot, &fixture.lease) + .await + .unwrap(); + let handler = fixture + .state + .api_handler(std::path::Path::new("/")) + .await + .unwrap(); + let root_tree = handler + .get_tree_by_hash(&context.root_tree_oid) + .await + .unwrap(); + let built = crate::ceres::snapshot::pages::build_directory_page( + handler.as_ref(), + &root_tree, + "/project", + ) + .await + .unwrap(); + let (Page::Branch { children, .. }, _) = Page::decode(&built.page_bytes).unwrap() else { + panic!("wide fixture must use canonical radix pages") + }; + let label = children + .iter() + .find(|child| child.label == b'w') + .unwrap() + .label; + let specs = [ + ("/", vec![]), + ("/", vec![label]), + ("/wide-139", vec![]), + ("/nested", vec![]), + ("/directory", vec![]), + ("/", vec![label]), + ]; + let mut expected = Vec::new(); + let mut seen = BTreeSet::new(); + let mut items = Vec::new(); + for (path, route) in &specs { + let absolute = if *path == "/" { + "/project".to_owned() + } else { + format!("/project{path}") + }; + let directory = crate::ceres::snapshot::pages::build_directory_page( + handler.as_ref(), + &root_tree, + &absolute, + ) + .await + .unwrap(); + let pages = Page::pages_along_route(&directory.codec_entries, route).unwrap(); + let reached = page_id(pages.last().unwrap()); + items.push(json!({"directory_path": path, "route": route, + "expected_digest": format!("sha256:{}", hex::encode(reached))})); + for page in pages { + let id = page_id(&page); + if seen.insert(id) { + expected.push((id, page)); + } + } + } + advance(&fixture).await; + let state = rebuilt(&fixture).await; + assert!(state.storage.native_snapshot_sessions.get().is_none()); + let mono = fixture.state.storage.mono_storage(); + let held = mono.get_connection().begin().await.unwrap(); + held.execute_unprepared("LOCK TABLE mega_tree,mst2_verified_object IN ACCESS EXCLUSIVE MODE") + .await + .unwrap(); + fixture.counts.reset(); + for encoding in ["identity", "zstd"] { + let body = metadata_body(json!(items), encoding); + let bytes = tokio::time::timeout( + Duration::from_secs(4), + metadata_bytes(&fixture, app(&state), &body), + ) + .await + .expect("persisted META must not touch locked Git tree or verified-object tables"); + assert_metadata(&bytes, &body, specs.len() as u32, &expected); + } + let requests: Vec<_> = specs + .iter() + .map(|(path, route)| MetadataRouteRequest { + directory_path: path, + route, + expected_digest: None, + }) + .collect(); + let batch = state + .storage + .snapshot_sessions() + .await + .metadata_routes(&context, &requests) + .await + .unwrap(); + assert_eq!(batch.pages, expected); + assert_eq!(batch.work.page_queries, batch.work.pages_loaded); + assert!( + batch.work.pages_loaded < 20, + "route work must not scan the 140-directory DAG" + ); + assert!( + batch.work.walk_visits > batch.work.pages_loaded, + "duplicate/path visits reuse request pages" + ); + assert!( + batch.work.payload_bytes >= batch.pages.iter().map(|(_, page)| page.len() as u64).sum() + ); + held.rollback().await.unwrap(); + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn mst2_persisted_meta_absence_scope_digest_limits_and_release_oracles() { + let fixture = Fixture::new_generic_history_with_pg_config(true).await; + for (items, status, code) in [ + ( + json!([{"directory_path":"/missing"}]), + 404, + "PATH_NOT_FOUND", + ), + (json!([{"directory_path":"/file"}]), 409, "NOT_DIRECTORY"), + ( + json!([{"directory_path":"/link/child"}]), + 409, + "NOT_DIRECTORY", + ), + ( + json!([{"directory_path":"/nested", "route":[0]}]), + 404, + "PATH_NOT_FOUND", + ), + ( + json!([{"directory_path":"/../outside"}]), + 400, + "SCOPE_INVALID", + ), + ( + json!([{"directory_path":format!("/{}", vec!["a"; 256].join("/"))}]), + 400, + "SCOPE_INVALID", + ), + ( + json!([{"directory_path":"/outside"}]), + 404, + "PATH_NOT_FOUND", + ), + ( + json!([{"directory_path":"/", "expected_digest":format!("sha256:{}", "0".repeat(64))}]), + 409, + "EXPECTED_DIGEST_MISMATCH", + ), + (json!([]), 413, "LIMIT_EXCEEDED"), + ( + json!(vec![json!({"directory_path":"/"}); 65]), + 413, + "LIMIT_EXCEEDED", + ), + ] { + error( + fixture + .send( + "POST", + "metadata/pages", + Body::from(metadata_body(items, "identity")), + ) + .await, + status, + code, + false, + ) + .await; + } + let body = metadata_body(json!([{"directory_path":"/"}]), "identity"); + let response = fixture + .send("POST", "metadata/pages", Body::from(body)) + .await; + assert_eq!(response.status(), 200); + let mut stream = response.into_body().into_data_stream(); + let first = stream.next().await.unwrap().unwrap(); + let (Frame::Meta(meta), consumed) = mst2_codec::treeframe::parse_frame(&first).unwrap() else { + panic!("first persisted frame must be META") + }; + assert_eq!(consumed, first.len()); + assert!(!meta.pages.is_empty()); + lease_control(&fixture, &fixture.lease, "DELETE", false).await; + assert!(stream.next().await.unwrap().is_err()); + assert!(stream.next().await.is_none()); + assert!(matches!( + parse_stream(&first), + Err(mst2_codec::CodecError::BadOrdering( + "stream missing END/ERROR frame" + )) + )); + fixture.counts.assert(0, 0); +} + +async fn page_row(fixture: &Fixture, path: &str) -> ([u8; 32], Vec) { + let body = metadata_body(json!([{"directory_path":path}]), "identity"); + let bytes = metadata_bytes(fixture, fixture.app.clone(), &body).await; + let Frame::Meta(meta) = &parse_stream(&bytes).unwrap()[0] else { + panic!("META required") + }; + meta.pages[0].clone() +} + +async fn damage_page(fixture: &Fixture, id: [u8; 32], remove: bool) { + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + let modes = handoff_trigger_modes_for_test(db).await; + let txn = db.begin().await.unwrap(); + txn.execute_unprepared("SELECT pg_advisory_xact_lock(1296717362,hashtext(current_schema()))") + .await + .unwrap(); + txn.execute_unprepared( + "ALTER TABLE mst2_metadata_payload DISABLE TRIGGER mst2_metadata_payload_fenced", + ) + .await + .unwrap(); + if remove { + txn.execute_unprepared( + "ALTER TABLE mst2_metadata_payload DISABLE TRIGGER mst2_metadata_payload_removed", + ) + .await + .unwrap(); + } + let sql = if remove { + "DELETE FROM mst2_metadata_payload WHERE page_id=$1" + } else { + "UPDATE mst2_metadata_payload SET payload=set_byte(payload,0,0) WHERE page_id=$1" + }; + assert_eq!( + txn.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + sql, + [id.to_vec().into()] + )) + .await + .unwrap() + .rows_affected(), + 1 + ); + if remove { + txn.execute_unprepared( + "ALTER TABLE mst2_metadata_payload ENABLE TRIGGER mst2_metadata_payload_removed", + ) + .await + .unwrap(); + } + txn.execute_unprepared( + "ALTER TABLE mst2_metadata_payload ENABLE TRIGGER mst2_metadata_payload_fenced", + ) + .await + .unwrap(); + txn.commit().await.unwrap(); + assert_eq!(handoff_trigger_modes_for_test(db).await, modes); + assert!( + db.execute_unprepared("UPDATE mst2_metadata_payload SET payload=payload") + .await + .is_err() + ); +} + +#[tokio::test] +async fn mst2_persisted_meta_warm_missing_and_corrupt_pages_never_reproject_from_git() { + for remove in [true, false] { + let fixture = Fixture::new_generic_history_with_pg_config(true).await; + let (id, _) = page_row(&fixture, "/nested").await; + damage_page(&fixture, id, remove).await; + let body = metadata_body(json!([{"directory_path":"/nested"}]), "identity"); + let mono = fixture.state.storage.mono_storage(); + let held = mono.get_connection().begin().await.unwrap(); + held.execute_unprepared( + "LOCK TABLE mega_tree,mst2_verified_object IN ACCESS EXCLUSIVE MODE", + ) + .await + .unwrap(); + let response = tokio::time::timeout( + Duration::from_secs(4), + fixture.send("POST", "metadata/pages", Body::from(body)), + ) + .await + .expect("damaged persisted page must fail before any source read"); + error( + response, + if remove { 503 } else { 502 }, + if remove { + "OBJECT_UNAVAILABLE" + } else { + "INTEGRITY_ERROR" + }, + false, + ) + .await; + held.rollback().await.unwrap(); + fixture.counts.assert(0, 0); + } +} + +async fn wait_retention_waiter(txn: &sea_orm::DatabaseTransaction) { + tokio::time::timeout(Duration::from_secs(4), async { + loop { + txn.execute_unprepared("SELECT pg_stat_clear_snapshot()") + .await + .unwrap(); + if scalar( + txn, + "SELECT count(*) FROM pg_locks l + WHERE l.locktype='advisory' AND NOT l.granted + AND l.database=(SELECT oid FROM pg_database WHERE datname=current_database()) + AND l.pid<>pg_backend_pid() AND l.classid=1296717362::oid + AND l.objid=hashtext(current_schema())::oid AND l.objsubid=2", + ) + .await + > 0 + { + break; + } + tokio::task::yield_now().await; + } + }) + .await + .expect("META request did not wait for retention authority"); +} + +#[tokio::test] +async fn mst2_persisted_meta_rechecks_deadline_and_release_after_waiting_for_retention_lock() { + for expire in [true, false] { + let fixture = Fixture::new_generic_history_with_pg_config(true).await; + let mono = fixture.state.storage.mono_storage(); + let held = mono.get_connection().begin().await.unwrap(); + held.execute_unprepared( + "SELECT pg_advisory_xact_lock(1296717362,hashtext(current_schema()))", + ) + .await + .unwrap(); + let pending = { + let application = fixture.app.clone(); + let request = fixture.request( + "POST", + "metadata/pages", + Body::from(metadata_body( + json!([{"directory_path":"/nested"}]), + "identity", + )), + ); + tokio::spawn(async move { application.oneshot(request).await.unwrap() }) + }; + wait_retention_waiter(&held).await; + let sql = if expire { + "UPDATE mst2_snapshot_lease SET expires_at_unix=0 WHERE lease_id=$1" + } else { + "UPDATE mst2_snapshot_lease SET state='RELEASED' WHERE lease_id=$1" + }; + held.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + sql, + [fixture.lease.clone().into()], + )) + .await + .unwrap(); + held.commit().await.unwrap(); + error(pending.await.unwrap(), 410, "LEASE_EXPIRED", false).await; + fixture.counts.assert(0, 0); + } +} + +async fn bound_fixture() -> ( + Fixture, + crate::ceres::snapshot::retention_dag::MetadataPagePayload, +) { + let mut fixture = Fixture::new_generic_history_with_pg_config(true).await; + advance(&fixture).await; + let mono = fixture.state.storage.mono_storage(); + let head = mono + .read_native_publication_head( + fixture + .state + .storage + .config() + .mst2 + .instance_uuid + .as_deref() + .unwrap(), + ) + .await + .unwrap(); + let handler = fixture + .state + .api_handler(std::path::Path::new("/")) + .await + .unwrap(); + let tree = handler.get_tree_by_hash(&head.root.tree).await.unwrap(); + let prepared = crate::ceres::snapshot::pages::prepare_native_metadata_retention( + handler.as_ref(), + &tree, + "/project", + crate::ceres::snapshot::retention_dag::MetadataDagLimits::default(), + ) + .await + .unwrap(); + assert_eq!(prepared.dag().payloads().len(), 1); + let expected = prepared.dag().payloads()[0].clone(); + let repository = crate::jupiter::storage::native_metadata_install::generations::PostgresMetadataGenerationRepository::new( + mono.get_connection().clone()).await.unwrap(); + let intent = repository + .begin_intent("persisted-meta-bound-seed", &prepared) + .await + .unwrap(); + repository + .install_pages(&intent, prepared.dag().payloads()) + .await + .unwrap(); + repository.finalize(&intent).await.unwrap(); + let resolved = success_json( + fixture + .app + .clone() + .oneshot(resolve_request("/project")) + .await + .unwrap(), + ) + .await; + assert_ne!(resolved["descriptor"]["snapshot_id"], fixture.snapshot); + fixture.snapshot = resolved["descriptor"]["snapshot_id"] + .as_str() + .unwrap() + .to_owned(); + fixture.lease = resolved["lease_id"].as_str().unwrap().to_owned(); + let db = mono.get_connection(); + assert_eq!( + scalar( + db, + "SELECT count(*) FROM mst2_metadata_payload WHERE generation=1" + ) + .await, + 1 + ); + db.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "INSERT INTO mst2_metadata_lifetime(page_id,node_id,generation,state,metadata_codec,expected_size,graph_domain) + VALUES($1,$2,2,'RESERVED',1,$3,'generic-v1')", + [expected.id.to_vec().into(), format!("page:sha256:{}", hex::encode(expected.id)).into(), + (expected.size as i32).into()])).await.unwrap(); + let membership = db.query_one_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "SELECT pp.generation AS member_generation,c.generation AS current_generation + FROM mst2_snapshot_context s JOIN mst2_metadata_prepare_page pp ON pp.prepare_id=s.prepare_id + JOIN mst2_metadata_current c ON c.page_id=pp.page_id WHERE s.snapshot_id=$1", + [fixture.snapshot.clone().into()])).await.unwrap().unwrap(); + assert_eq!( + membership + .try_get::>("", "member_generation") + .unwrap(), + None + ); + assert_eq!( + membership.try_get::("", "current_generation").unwrap(), + 1 + ); + fixture.counts.reset(); + (fixture, expected) +} + +#[tokio::test] +async fn mst2_persisted_meta_serves_bound_generic_pages_with_null_members_and_extra_history() { + let (fixture, expected) = bound_fixture().await; + let body = metadata_body(json!([{"directory_path":"/"}]), "identity"); + for application in [fixture.app.clone(), app(&rebuilt(&fixture).await)] { + let bytes = metadata_bytes(&fixture, application, &body).await; + assert_metadata(&bytes, &body, 1, &[(expected.id, expected.bytes.clone())]); + } + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn mst2_persisted_meta_rejects_touched_graph_damage_and_prepare_membership_loss() { + for case in 0..5 { + let fixture = Fixture::new_generic_history_with_pg_config(true).await; + let (id, _) = page_row(&fixture, "/nested").await; + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + let modes = handoff_trigger_modes_for_test(db).await; + let txn = db.begin().await.unwrap(); + txn.execute_unprepared( + "SELECT pg_advisory_xact_lock(1296717362,hashtext(current_schema()))", + ) + .await + .unwrap(); + let node = format!("page:sha256:{}", hex::encode(id)); + let (sql, values) = match case { + 0 => ("UPDATE mst2_retention_node SET state='DELETING' WHERE node_id=$1", vec![node.into()]), + 1 => ("UPDATE mst2_retention_node SET bytes=bytes+1 WHERE node_id=$1", vec![node.into()]), + 2 => ("DELETE FROM mst2_retention_edge WHERE child_id=$1", vec![node.into()]), + 3 => ("INSERT INTO mst2_retention_gc_op(operation_id,node_id,operation,state,attempts,created_at) + VALUES('persisted-meta-tombstone',$1,'REMOVE','PENDING',0,now())", vec![node.into()]), + 4 => { + txn.execute_unprepared("ALTER TABLE mst2_metadata_prepare_page DISABLE TRIGGER mst2_install_capability_mapping_guard").await.unwrap(); + ("DELETE FROM mst2_metadata_prepare_page WHERE page_id=$1 AND prepare_id= + (SELECT prepare_id FROM mst2_snapshot_context WHERE snapshot_id=$2)", + vec![id.to_vec().into(), fixture.snapshot.clone().into()]) + } + _ => unreachable!(), + }; + assert!( + txn.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + sql, + values + )) + .await + .unwrap() + .rows_affected() + > 0 + ); + if case == 4 { + txn.execute_unprepared("ALTER TABLE mst2_metadata_prepare_page ENABLE TRIGGER mst2_install_capability_mapping_guard").await.unwrap(); + } + txn.commit().await.unwrap(); + assert_eq!(handoff_trigger_modes_for_test(db).await, modes); + let response = fixture + .send( + "POST", + "metadata/pages", + Body::from(metadata_body( + json!([{"directory_path":"/nested"}]), + "identity", + )), + ) + .await; + error( + response, + if case == 2 { 502 } else { 503 }, + if case == 2 { + "INTEGRITY_ERROR" + } else { + "OBJECT_UNAVAILABLE" + }, + false, + ) + .await; + fixture.counts.assert(0, 0); + } +} + +#[tokio::test] +async fn mst2_persisted_meta_reader_holds_protection_until_all_route_bytes_are_owned() { + use crate::jupiter::storage::native_snapshot_session::with_metadata_read_barriers; + + for release_lease in [false, true] { + let fixture = Fixture::new_generic_history_with_pg_config(true).await; + let expected = vec![ + page_row(&fixture, "/nested").await, + page_row(&fixture, "/directory").await, + ]; + let context = fixture + .state + .storage + .snapshot_context(&fixture.snapshot, &fixture.lease) + .await + .unwrap(); + let captured = Arc::new(Barrier::new(2)); + let resume = Arc::new(Barrier::new(2)); + let reading = { + let state = fixture.state.clone(); + let context = context.clone(); + let captured = captured.clone(); + let resume = resume.clone(); + tokio::spawn(async move { + let requests = [ + MetadataRouteRequest { + directory_path: "/nested", + route: &[], + expected_digest: None, + }, + MetadataRouteRequest { + directory_path: "/directory", + route: &[], + expected_digest: None, + }, + ]; + with_metadata_read_barriers( + captured, + resume, + state + .storage + .snapshot_sessions() + .await + .metadata_routes(&context, &requests), + ) + .await + }) + }; + tokio::time::timeout(Duration::from_secs(4), captured.wait()) + .await + .unwrap(); + let root = format!("page:{}", context.built.metadata_root); + let mut competing = { + let state = fixture.state.clone(); + let lease = fixture.lease.clone(); + let root = root.clone(); + tokio::spawn(async move { + if release_lease { + assert!(state.storage.snapshot_release(&lease).await.unwrap()); + None + } else { + Some( + PostgresRetentionRepository::new( + state.storage.mono_storage().get_connection().clone(), + ) + .mark_deleting("persisted-meta-reader-race", &root) + .await + .unwrap(), + ) + } + }) + }; + // The two service connections belong to the reader and competitor. + // Observe their lock wait without competing for either connection. + let mut observer_config = fixture.state.storage.config().database.clone(); + assert_eq!(observer_config.max_connection, 2); + observer_config.max_connection = 1; + observer_config.min_connection = 0; + let observer_connection = crate::jupiter::storage::init::with_generic_history_bootstrap( + crate::jupiter::storage::init::database_connection(&observer_config), + ) + .await + .unwrap(); + let observer = observer_connection.begin().await.unwrap(); + tokio::select! { + () = wait_retention_waiter(&observer) => {}, + outcome = &mut competing => { + panic!("retention competitor completed before waiting: release_lease={release_lease}, outcome={outcome:?}"); + }, + } + assert!(!reading.is_finished()); + assert!(!competing.is_finished()); + resume.wait().await; + let batch = reading.await.unwrap().unwrap(); + assert_eq!(batch.pages, expected); + assert_eq!( + batch.work.pages_loaded, 3, + "root and both directories are read before unlock" + ); + let outcome = competing.await.unwrap(); + observer.rollback().await.unwrap(); + if release_lease { + assert!(outcome.is_none()); + } else { + assert_eq!(outcome, Some(GcClaim::Unavailable)); + lease_control(&fixture, &fixture.lease, "DELETE", false).await; + } + let gc = PostgresRetentionRepository::new( + fixture + .state + .storage + .mono_storage() + .get_connection() + .clone(), + ); + assert_eq!( + gc.mark_deleting("persisted-meta-after-read", &root) + .await + .unwrap(), + GcClaim::Marked + ); + gc.complete_gc("persisted-meta-after-read").await.unwrap(); + let requests = [MetadataRouteRequest { + directory_path: "/nested", + route: &[], + expected_digest: None, + }]; + let rejected = fixture + .state + .storage + .snapshot_sessions() + .await + .metadata_routes(&context, &requests) + .await + .err() + .expect("released and collected root must fail before touching pages"); + assert_eq!(rejected.code, SnapshotErrorCode::LeaseExpired); + assert_eq!( + batch.pages, expected, + "collected graph does not invalidate owned route bytes" + ); + fixture.counts.assert(0, 0); + } +} + +async fn fault_binding(fixture: &Fixture, id: [u8; 32], case: usize, restore: bool) { + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + let modes = handoff_trigger_modes_for_test(db).await; + let txn = db.begin().await.unwrap(); + txn.execute_unprepared("SELECT pg_advisory_xact_lock(1296717362,hashtext(current_schema()))") + .await + .unwrap(); + let (table, guard, sql) = match case { + 0 => ( + "mst2_metadata_payload", + "mst2_metadata_payload_fenced", + if restore { + "UPDATE mst2_metadata_payload SET generation=1 WHERE page_id=$1" + } else { + "UPDATE mst2_metadata_payload SET generation=NULL WHERE page_id=$1" + }, + ), + 1 => ( + "mst2_metadata_current", + "mst2_metadata_current_guard", + if restore { + "UPDATE mst2_metadata_current SET generation=1 WHERE page_id=$1" + } else { + "UPDATE mst2_metadata_current SET generation=2 WHERE page_id=$1" + }, + ), + 2 => ( + "mst2_metadata_lifetime", + "mst2_metadata_lifetime_guard", + if restore { + "UPDATE mst2_metadata_lifetime SET state='LIVE' WHERE page_id=$1 AND generation=1" + } else { + "UPDATE mst2_metadata_lifetime SET state='DELETING' WHERE page_id=$1 AND generation=1" + }, + ), + 3 => ( + "mst2_metadata_lifetime", + "mst2_metadata_lifetime_guard", + if restore { + "UPDATE mst2_metadata_lifetime SET graph_domain='generic-v1' WHERE page_id=$1 AND generation=1" + } else { + "UPDATE mst2_metadata_lifetime SET graph_domain='qualified-v1' WHERE page_id=$1 AND generation=1" + }, + ), + 4 => ( + "mst2_metadata_lifetime", + "mst2_metadata_lifetime_guard", + if restore { + "UPDATE mst2_metadata_lifetime SET expected_size=expected_size-1 WHERE page_id=$1 AND generation=1" + } else { + "UPDATE mst2_metadata_lifetime SET expected_size=expected_size+1 WHERE page_id=$1 AND generation=1" + }, + ), + 5 => ( + "mst2_metadata_prepare_page", + "mst2_install_capability_mapping_guard", + if restore { + "UPDATE mst2_metadata_prepare_page SET generation=NULL WHERE page_id=$1 AND prepare_id= + (SELECT prepare_id FROM mst2_snapshot_context WHERE snapshot_id=$2)" + } else { + "UPDATE mst2_metadata_prepare_page SET generation=2 WHERE page_id=$1 AND prepare_id= + (SELECT prepare_id FROM mst2_snapshot_context WHERE snapshot_id=$2)" + }, + ), + _ => unreachable!(), + }; + txn.execute_unprepared(&format!("ALTER TABLE {table} DISABLE TRIGGER {guard}")) + .await + .unwrap(); + let values = if case == 5 { + vec![id.to_vec().into(), fixture.snapshot.clone().into()] + } else { + vec![id.to_vec().into()] + }; + assert_eq!( + txn.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + sql, + values + )) + .await + .unwrap() + .rows_affected(), + 1 + ); + if case == 1 { + let protection = txn + .query_one_raw(Statement::from_string( + DbBackend::Postgres, + "SELECT t.tgenabled::text AS mode,t.tgdeferrable AS deferrable,t.tginitdeferred AS deferred + FROM pg_trigger t WHERE t.tgrelid='mst2_metadata_current'::regclass + AND t.tgname='mst2_metadata_current_protected' AND NOT t.tgisinternal", + )) + .await + .unwrap() + .unwrap(); + assert!(matches!( + protection.try_get::("", "mode").unwrap().as_str(), + "O" | "A" + )); + assert!(protection.try_get::("", "deferrable").unwrap()); + assert!(protection.try_get::("", "deferred").unwrap()); + txn.execute_unprepared("SET CONSTRAINTS mst2_metadata_current_protected IMMEDIATE") + .await + .unwrap(); + txn.execute_unprepared("SET CONSTRAINTS mst2_metadata_current_protected DEFERRED") + .await + .unwrap(); + } + txn.execute_unprepared(&format!("ALTER TABLE {table} ENABLE TRIGGER {guard}")) + .await + .unwrap(); + txn.commit().await.unwrap(); + assert_eq!(handoff_trigger_modes_for_test(db).await, modes); +} + +#[tokio::test] +async fn mst2_persisted_meta_warm_reads_require_exact_generic_current_lifetime_and_membership() { + let (fixture, expected) = bound_fixture().await; + let body = metadata_body(json!([{"directory_path":"/"}]), "identity"); + let bytes = metadata_bytes(&fixture, fixture.app.clone(), &body).await; + assert_metadata(&bytes, &body, 1, &[(expected.id, expected.bytes.clone())]); + for case in 0..6 { + fault_binding(&fixture, expected.id, case, false).await; + error( + fixture + .send("POST", "metadata/pages", Body::from(body.clone())) + .await, + if matches!(case, 0 | 1 | 5) { 502 } else { 503 }, + if matches!(case, 0 | 1 | 5) { + "INTEGRITY_ERROR" + } else { + "OBJECT_UNAVAILABLE" + }, + false, + ) + .await; + fault_binding(&fixture, expected.id, case, true).await; + let bytes = metadata_bytes(&fixture, fixture.app.clone(), &body).await; + assert_metadata(&bytes, &body, 1, &[(expected.id, expected.bytes.clone())]); + } + fixture.counts.assert(0, 0); +} diff --git a/src/api/router/snapshot_raw_blob.rs b/src/api/router/snapshot_raw_blob.rs new file mode 100644 index 00000000..fa8fa3c7 --- /dev/null +++ b/src/api/router/snapshot_raw_blob.rs @@ -0,0 +1,416 @@ +//! Whole-file raw delivery from one current-OID stream, authenticated in chunks. + +use std::sync::Arc; + +use axum::{ + body::Body, + extract::{Path as AxumPath, Query, State}, + http::HeaderMap, + response::Response, +}; +use bytes::Bytes; +use futures::StreamExt; +use mst2_codec::chunkmap::{CHUNK_SIZE, CHUNKS_PER_PAGE}; +use serde::Deserialize; +use sha2::{Digest, Sha256}; + +use super::{ + REQUEST_HEADERS, content, ensure_enabled, internal, mst2_error_response, revalidate_access, + revalidate_request, +}; +use crate::{ + api::MonoApiServiceState, + ceres::snapshot::{ + content_budget::{ + MemoryBudget, MemoryLease, RANGE_WORK_BYTES, range_budget, response_budget, + }, + error::{SnapshotError, SnapshotErrorCode}, + pages::hex_of, + runtime::SnapshotContext, + view::validate_scope_relative_path, + }, + jupiter::storage::{ + Storage, + native_chunk_map::{AuthenticatedChunkPage, PersistedChunkMap}, + }, + orbit_api::object_storage::ObjectByteStream, +}; + +const PRODUCER_ITEM_MAX: usize = 8 * 1024 * 1024; + +#[derive(Deserialize, Debug)] +pub(super) struct BlobQuery { + path: String, + #[serde(default)] + expected_digest: Option, +} + +pub(super) struct BlobBudgets { + pub(super) scratch: Arc, + pub(super) response: Arc, +} + +impl Default for BlobBudgets { + fn default() -> Self { + Self { + scratch: range_budget().clone(), + response: response_budget().clone(), + } + } +} + +pub(super) async fn blob( + state: State, + path: AxumPath, + query: Query, + headers: HeaderMap, +) -> Response { + match blob_with_budgets(state, path, query, headers, BlobBudgets::default()).await { + Ok(response) | Err(response) => response, + } +} + +#[allow(clippy::result_large_err)] +pub(super) async fn blob_with_budgets( + state: State, + AxumPath(snapshot_id): AxumPath, + Query(query): Query, + headers: HeaderMap, + budgets: BlobBudgets, +) -> Result { + ensure_enabled(&state).map_err(mst2_error_response)?; + let context = super::request_context(&state, &snapshot_id).map_err(mst2_error_response)?; + validate_scope_relative_path(&query.path).map_err(mst2_error_response)?; + if headers.contains_key("range") { + return Err(mst2_error_response(SnapshotError::new( + SnapshotErrorCode::RangeNotSupported, + "raw blob reads are whole-file; use chunks for ranges", + ))); + } + let handler = state + .api_handler(std::path::Path::new("/")) + .await + .map_err(internal)?; + let file = content::resolve_snapshot_file_metadata( + &state, + &context, + &query.path, + query.expected_digest.as_deref(), + ) + .await?; + let scratch = budgets + .scratch + .reserve(if file.size == 0 { 0 } else { RANGE_WORK_BYTES }) + .map_err(mst2_error_response)?; + let first_credit = budgets + .response + .reserve(file.size.min(CHUNK_SIZE as u64) as usize) + .map_err(mst2_error_response)?; + revalidate_request(&state, &context) + .await + .map_err(mst2_error_response)?; + let map = if file.size == 0 { + if file.digest != <[u8; 32]>::from(Sha256::digest([])) { + return Err(mst2_error_response(integrity( + "empty raw source has a nonempty verified digest", + ))); + } + None + } else { + Some(content::project_resolved(handler.as_ref(), &file).await?) + }; + let storage = handler.get_context(); + let page = if let Some(map) = map.as_ref() { + Some( + storage + .chunk_maps() + .await + .map_err(mst2_error_response)? + .selected_page(map, 0) + .await + .map_err(mst2_error_response)?, + ) + } else { + None + }; + if let Some(map) = map.as_ref() { + map.ensure_live().await.map_err(mst2_error_response)?; + } + let open = handler.get_raw_blob_stream_with_meta(&file.oid); + let opened = if let Some(map) = map.as_ref() { + map.await_backend(open).await.map_err(mst2_error_response)? + } else { + open.await + }; + let (input, meta) = + opened.map_err(|error| mst2_error_response(content::content_read_error(error)))?; + if let Some(map) = map.as_ref() { + map.ensure_live().await.map_err(mst2_error_response)?; + } + if u64::try_from(meta.size).ok() != Some(file.size) { + return Err(mst2_error_response(integrity( + "raw source physical size disagrees with its fixed verified fact", + ))); + } + let headers = REQUEST_HEADERS.try_with(Clone::clone).map_err(|_| { + mst2_error_response(SnapshotError::new( + SnapshotErrorCode::Unauthenticated, + "request authentication context missing", + )) + })?; + revalidate_access(&state, &context, &headers) + .await + .map_err(mst2_error_response)?; + let digest = file.digest; + let size = file.size; + let kind = content::fs_kind_str(file.fs_kind); + let mut reader = RawBlobReader { + state: state.0, + context, + headers, + storage, + map, + page, + input: Some(input), + pending: Bytes::new(), + pending_offset: 0, + received: 0, + index: 0, + size, + digest, + hash: Sha256::new(), + first_credit: Some(first_credit), + response_budget: budgets.response, + _scratch: scratch, + }; + let body = if size == 0 { + reader.require_eof().await.map_err(mst2_error_response)?; + reader.validate().await.map_err(mst2_error_response)?; + Body::empty() + } else { + let stream = futures::stream::try_unfold(reader, |mut reader| async move { + let Some(bytes) = reader.next_chunk().await? else { + return Ok::<_, SnapshotError>(None); + }; + Ok(Some((bytes, reader))) + }); + Body::from_stream(stream) + }; + Response::builder() + .header("etag", format!("\"sha256:{}\"", hex_of(&digest))) + .header("content-length", size.to_string()) + .header("x-mega-content-size", size.to_string()) + .header("x-mega-fs-kind", kind) + .header("cache-control", "private, no-cache, no-transform") + .header("vary", "Authorization, Accept") + .body(body) + .map_err(|_| mst2_error_response(internal("raw blob response build failed"))) +} + +struct RawBlobReader { + state: MonoApiServiceState, + context: SnapshotContext, + headers: HeaderMap, + storage: Storage, + map: Option>, + page: Option>, + input: Option, + pending: Bytes, + pending_offset: usize, + received: u64, + index: u64, + size: u64, + digest: [u8; 32], + hash: Sha256, + first_credit: Option, + response_budget: Arc, + _scratch: MemoryLease, +} + +struct RawBlobChunk { + bytes: Vec, + _credit: MemoryLease, + _map: Arc, +} + +impl AsRef<[u8]> for RawBlobChunk { + fn as_ref(&self) -> &[u8] { + &self.bytes + } +} + +impl RawBlobReader { + async fn validate(&mut self) -> Result<(), SnapshotError> { + revalidate_access(&self.state, &self.context, &self.headers).await?; + if let Some(map) = self.map.as_ref() { + map.ensure_live().await?; + } + Ok(()) + } + + async fn next_chunk(&mut self) -> Result, SnapshotError> { + self.validate().await?; + let map = self + .map + .as_ref() + .ok_or_else(|| integrity("nonempty raw source has no authenticated map"))?; + if self.index == map.map.chunk_count { + return Ok(None); + } + let length = map + .map + .chunk_len(self.index) + .map_err(|_| integrity("raw source chunk index disagrees with its map"))? + as usize; + let page_index = self.index / CHUNKS_PER_PAGE as u64; + if self.page.as_ref().map(|p| p.leaf.page_index) != Some(page_index) { + drop(self.page.take()); + self.page = Some( + self.storage + .chunk_maps() + .await? + .selected_page(map, page_index) + .await?, + ); + self.validate().await?; + } + let credit = if let Some(credit) = self.first_credit.take() { + credit + } else { + self.response_budget.reserve(length)? + }; + let mut raw = Vec::new(); + raw.try_reserve_exact(length).map_err(|_| { + SnapshotError::new( + SnapshotErrorCode::TemporaryUnavailable, + "raw chunk allocation could not be admitted", + ) + })?; + let mut empty_parts = 0; + while raw.len() < length { + if self.pending_offset == self.pending.len() { + self.pending = Bytes::new(); + self.pending_offset = 0; + let Some(part) = self.poll_source().await? else { + return Err(integrity("raw source is truncated before its final chunk")); + }; + if part.len() as u64 > self.size - self.received { + return Err(integrity("raw source exceeds its fixed verified size")); + } + if part.len() > PRODUCER_ITEM_MAX { + return Err(SnapshotError::new( + SnapshotErrorCode::TemporaryUnavailable, + "raw producer item exceeds its consumer processing limit", + )); + } + self.received += part.len() as u64; + if part.is_empty() { + empty_parts += 1; + if empty_parts == 32 { + tokio::task::yield_now().await; + self.validate().await?; + empty_parts = 0; + } + continue; + } + self.pending = part; + } + let amount = (length - raw.len()).min(self.pending.len() - self.pending_offset); + let end = self.pending_offset + amount; + let bytes = &self.pending[self.pending_offset..end]; + self.hash.update(bytes); + raw.extend_from_slice(bytes); + self.pending_offset = end; + } + let map = self + .map + .as_ref() + .ok_or_else(|| integrity("raw source authenticated map disappeared"))?; + self.page + .as_ref() + .ok_or_else(|| integrity("raw source authenticated page disappeared"))? + .verify_chunk(&map.map, self.index, &raw)?; + self.index += 1; + if self.index == map.map.chunk_count { + self.require_eof().await?; + let actual: [u8; 32] = self.hash.clone().finalize().into(); + if actual != self.digest { + return Err(integrity( + "raw source whole-file digest disagrees with its fixed fact", + )); + } + } + self.validate().await?; + if let Some(map) = self.map.as_ref() { + map.record_progress().await?; + } + if raw.capacity() > credit.bytes { + return Err(integrity("raw chunk allocation exceeds its owned credit")); + } + Ok(Some(Bytes::from_owner(RawBlobChunk { + bytes: raw, + _credit: credit, + _map: self + .map + .as_ref() + .cloned() + .ok_or_else(|| integrity("raw source authenticated map disappeared"))?, + }))) + } + + async fn poll_source(&mut self) -> Result, SnapshotError> { + self.validate().await?; + let input = self + .input + .as_mut() + .ok_or_else(|| integrity("raw source stream was already closed"))?; + let result = if let Some(map) = self.map.as_ref() { + map.await_backend(input.next()).await? + } else { + input.next().await + }; + self.validate().await?; + let part = result.transpose().map_err(|error| { + tracing::warn!(%error, "fixed-view raw stream failed"); + SnapshotError::new( + SnapshotErrorCode::ObjectUnavailable, + "fixed-view raw stream failed", + ) + })?; + if part.as_ref().is_some_and(|bytes| !bytes.is_empty()) + && let Some(map) = self.map.as_ref() + { + map.record_progress().await?; + } + Ok(part) + } + + async fn require_eof(&mut self) -> Result<(), SnapshotError> { + if self.received != self.size || self.pending_offset != self.pending.len() { + return Err(integrity( + "raw source final length disagrees with its fixed fact", + )); + } + let mut empty_parts = 0; + while let Some(part) = self.poll_source().await? { + if !part.is_empty() { + return Err(integrity( + "raw source has trailing bytes after its fixed size", + )); + } + empty_parts += 1; + if empty_parts == 32 { + tokio::task::yield_now().await; + self.validate().await?; + empty_parts = 0; + } + } + drop(self.input.take()); + self.pending = Bytes::new(); + Ok(()) + } +} + +fn integrity(message: &str) -> SnapshotError { + SnapshotError::new(SnapshotErrorCode::IntegrityError, message) +} diff --git a/src/api/router/snapshot_raw_blob_tests.rs b/src/api/router/snapshot_raw_blob_tests.rs new file mode 100644 index 00000000..79a38c79 --- /dev/null +++ b/src/api/router/snapshot_raw_blob_tests.rs @@ -0,0 +1,791 @@ +use axum::{ + extract::{Path as AxumPath, Query, State}, + http::HeaderMap, + routing::get, +}; +use tokio::{sync::Notify, time::timeout}; + +use super::*; +use crate::{ + api::router::snapshot_router::{raw_blob as serving, snapshot_auth_middleware}, + ceres::snapshot::content_budget::{MemoryBudget, RANGE_WORK_BYTES}, +}; + +pub(super) fn budgeted_app( + fixture: &Fixture, + response_budget: &Arc, + scratch_budget: &Arc, +) -> Router { + let response_budget = response_budget.clone(); + let scratch_budget = scratch_budget.clone(); + Router::new() + .route( + "/api/v2/snapshots/{snapshot_id}/blob", + get( + move |state: State, + path: AxumPath, + query: Query, + headers: HeaderMap| { + serving::blob_with_budgets( + state, + path, + query, + headers, + serving::BlobBudgets { + response: response_budget.clone(), + scratch: scratch_budget.clone(), + }, + ) + }, + ), + ) + .route_layer(axum::middleware::from_fn_with_state( + fixture.state.clone(), + snapshot_auth_middleware, + )) + .with_state(fixture.state.clone()) +} + +fn fault(fixture: &Fixture, kind: bounded_objects::FaultKind) { + *fixture.counts.object_fault.lock().unwrap() = Some(bounded_objects::StreamFault { + oid: fixture.oid.clone(), + kind, + }); +} + +async fn path_oid(fixture: &Fixture, path: &str) -> String { + use crate::ceres::snapshot::pages::{MetadataWalkOutcome, resolve_abs_metadata}; + let handler = fixture + .state + .api_handler(std::path::Path::new("/")) + .await + .unwrap(); + let context = fixture + .state + .storage + .snapshot_context(&fixture.snapshot, &fixture.lease) + .await + .unwrap(); + let root = handler + .get_tree_by_hash(&context.root_tree_oid) + .await + .unwrap(); + match resolve_abs_metadata(handler.as_ref(), &root, &format!("/project{path}")) + .await + .unwrap() + { + MetadataWalkOutcome::FoundFile { oid, .. } => oid, + other => panic!("raw fixture path must be a fixed file: {other:?}"), + } +} + +async fn release_lease(fixture: &Fixture) { + let response = fixture + .app + .clone() + .oneshot( + Request::builder() + .method("DELETE") + .uri(format!("/api/v2/snapshots/leases/{}", fixture.lease)) + .header("authorization", format!("Bearer {TOKEN}")) + .body(Body::empty()) + .unwrap(), + ) + .await + .unwrap(); + assert_eq!(response.status(), 200); +} + +#[tokio::test] +async fn cold_raw_get_earns_source_proof_then_serves_one_current_stream_with_exact_headers() { + let fixture = Fixture::new().await; + let response = fixture.send("GET", "blob?path=/file", Body::empty()).await; + assert_eq!(response.status(), 200); + assert_eq!( + response.headers()["content-length"], + fixture.raw.len().to_string() + ); + assert_eq!( + response.headers()["x-mega-content-size"], + fixture.raw.len().to_string() + ); + assert_eq!( + response.headers()["etag"], + format!("\"{}\"", fixture.digest_string()) + ); + assert_eq!(response.headers()["x-mega-fs-kind"], "regular"); + assert_eq!(response.headers()["vary"], "Authorization, Accept"); + assert_eq!( + response.headers()["cache-control"], + "private, no-cache, no-transform" + ); + assert_eq!(fixture.counts.whole.load(Ordering::SeqCst), 2); + assert_eq!( + fixture.counts.bytes.load(Ordering::SeqCst), + fixture.raw.len() + ); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 1); + assert_eq!( + to_bytes(response.into_body(), 2 * CHUNK_SIZE as usize) + .await + .unwrap() + .as_ref(), + fixture.raw + ); + fixture.counts.assert(2, 2 * fixture.raw.len()); + fixture.counts.reset(); + let response = fixture.send("GET", "blob?path=/alias", Body::empty()).await; + assert_eq!(response.status(), 200); + assert_eq!(fixture.counts.bytes.load(Ordering::SeqCst), 0); + assert_eq!( + to_bytes(response.into_body(), 2 * CHUNK_SIZE as usize) + .await + .unwrap() + .as_ref(), + fixture.raw + ); + fixture.counts.assert(1, fixture.raw.len()); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 0); + assert_eq!(fixture.counts.receipt_reads.load(Ordering::SeqCst), 1); +} + +#[tokio::test] +async fn raw_memory_pressure_rejects_before_cold_proof_or_delivery_source_io() { + let fixture = Fixture::new().await; + for occupy_scratch in [false, true] { + fixture.counts.reset(); + let response_budget = MemoryBudget::new(CHUNK_SIZE as usize); + let scratch_budget = MemoryBudget::new(RANGE_WORK_BYTES); + let held = if occupy_scratch { + scratch_budget.reserve(RANGE_WORK_BYTES).unwrap() + } else { + response_budget.reserve(CHUNK_SIZE as usize).unwrap() + }; + let response = budgeted_app(&fixture, &response_budget, &scratch_budget) + .oneshot(fixture.request("GET", "blob?path=/file", Body::empty())) + .await + .unwrap(); + error(response, 503, "TEMPORARY_UNAVAILABLE", true).await; + fixture.counts.assert(0, 0); + assert_eq!(fixture.counts.receipt_reads.load(Ordering::SeqCst), 0); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 0); + assert_eq!( + scratch_budget.used(), + if occupy_scratch { RANGE_WORK_BYTES } else { 0 } + ); + assert_eq!( + response_budget.used(), + if occupy_scratch { + 0 + } else { + CHUNK_SIZE as usize + } + ); + drop(held); + assert_eq!(response_budget.used(), 0); + assert_eq!(scratch_budget.used(), 0); + } +} + +#[tokio::test] +async fn raw_missing_or_forged_receipt_and_real_stored_corruption_fail_without_rebuilding() { + let fixture = Fixture::new().await; + fixture.map("/file").await; + for missing in [true, false] { + fixture.counts.reset(); + fixture + .counts + .receipt_read_failure + .store(missing, Ordering::SeqCst); + *fixture.counts.receipt_read_corruption.lock().unwrap() = + (!missing).then(|| Bytes::from_static(b"forged receipt")); + error( + fixture.send("GET", "blob?path=/file", Body::empty()).await, + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + fixture.counts.assert(0, 0); + assert_eq!(fixture.counts.receipt_reads.load(Ordering::SeqCst), 1); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 0); + } + fixture + .counts + .receipt_read_failure + .store(false, Ordering::SeqCst); + *fixture.counts.receipt_read_corruption.lock().unwrap() = None; + let mut wrong = fixture.raw.clone(); + wrong[0] ^= 1; + fixture.write_raw(wrong).await; + let response = fixture.send("GET", "blob?path=/file", Body::empty()).await; + assert_eq!(response.status(), 200); + let mut stream = response.into_body().into_data_stream(); + assert!(stream.next().await.unwrap().is_err()); + assert!(stream.next().await.is_none()); + assert_eq!(fixture.counts.whole.load(Ordering::SeqCst), 1); + assert_eq!(fixture.counts.range.load(Ordering::SeqCst), 0); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 0); + fixture.write_raw(fixture.raw.clone()).await; + let response = fixture.send("GET", "blob?path=/file", Body::empty()).await; + assert_eq!(response.status(), 200); + assert_eq!( + to_bytes(response.into_body(), 2 * CHUNK_SIZE as usize) + .await + .unwrap() + .as_ref(), + fixture.raw + ); + fixture.counts.assert(1, fixture.raw.len()); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 0); +} + +#[tokio::test] +async fn warm_fragmented_actual_raw_body_polls_one_chunk_and_transport_clones_keep_exact_credit() { + let fixture = Fixture::new().await; + fixture.map("/file").await; + fixture.counts.reset(); + fault( + &fixture, + bounded_objects::FaultKind::Parts(vec![ + Bytes::new(), + Bytes::copy_from_slice(&fixture.raw[..13]), + Bytes::copy_from_slice(&fixture.raw[13..CHUNK_SIZE as usize - 7]), + Bytes::copy_from_slice(&fixture.raw[CHUNK_SIZE as usize - 7..CHUNK_SIZE as usize]), + Bytes::copy_from_slice(&fixture.raw[CHUNK_SIZE as usize..]), + Bytes::new(), + ]), + ); + let response_budget = MemoryBudget::new(fixture.raw.len()); + let scratch_budget = MemoryBudget::new(RANGE_WORK_BYTES); + let response = budgeted_app(&fixture, &response_budget, &scratch_budget) + .oneshot(fixture.request("GET", "blob?path=/file", Body::empty())) + .await + .unwrap(); + assert_eq!(response.status(), 200); + fixture.counts.assert(1, 0); + assert_eq!(response_budget.used(), CHUNK_SIZE as usize); + assert_eq!(scratch_budget.used(), RANGE_WORK_BYTES); + let mut stream = response.into_body().into_data_stream(); + let first = stream.next().await.unwrap().unwrap(); + assert_eq!(first.as_ref(), &fixture.raw[..CHUNK_SIZE as usize]); + fixture.counts.assert(1, CHUNK_SIZE as usize); + let transport = first.clone(); + drop(first); + assert_eq!(response_budget.used(), CHUNK_SIZE as usize); + let final_bytes = stream.next().await.unwrap().unwrap(); + assert_eq!(final_bytes.as_ref(), &fixture.raw[CHUNK_SIZE as usize..]); + assert_eq!(response_budget.used(), fixture.raw.len()); + assert!(stream.next().await.is_none()); + assert_eq!(scratch_budget.used(), 0); + assert_eq!(response_budget.used(), fixture.raw.len()); + drop(stream); + drop(final_bytes); + assert_eq!(response_budget.used(), CHUNK_SIZE as usize); + drop(transport); + assert_eq!(response_budget.used(), 0); + fixture.counts.assert(1, fixture.raw.len()); +} + +#[tokio::test] +async fn raw_transport_quota_rejects_next_chunk_before_more_source_io_and_errors_terminally() { + let fixture = Fixture::new().await; + fixture.map("/file").await; + fixture.counts.reset(); + fault( + &fixture, + bounded_objects::FaultKind::Parts(vec![ + Bytes::copy_from_slice(&fixture.raw[..CHUNK_SIZE as usize]), + Bytes::copy_from_slice(&fixture.raw[CHUNK_SIZE as usize..]), + ]), + ); + let response_budget = MemoryBudget::new(CHUNK_SIZE as usize); + let scratch_budget = MemoryBudget::new(RANGE_WORK_BYTES); + let response = budgeted_app(&fixture, &response_budget, &scratch_budget) + .oneshot(fixture.request("GET", "blob?path=/file", Body::empty())) + .await + .unwrap(); + assert_eq!(response.status(), 200); + let mut stream = response.into_body().into_data_stream(); + let transport = stream.next().await.unwrap().unwrap(); + assert_eq!(transport.as_ref(), &fixture.raw[..CHUNK_SIZE as usize]); + assert!(stream.next().await.unwrap().is_err()); + assert!(stream.next().await.is_none()); + fixture.counts.assert(1, CHUNK_SIZE as usize); + assert_eq!(scratch_budget.used(), 0); + assert_eq!(response_budget.used(), CHUNK_SIZE as usize); + drop(transport); + assert_eq!(response_budget.used(), 0); +} + +#[tokio::test] +async fn warm_raw_corruption_growth_truncation_and_late_error_never_yield_the_last_bytes() { + let fixture = Fixture::new().await; + fixture.map("/file").await; + let mut wrong = fixture.raw.clone(); + wrong[0] ^= 1; + let mut grown = fixture.raw.clone(); + grown.push(0); + let cases = [ + ( + bounded_objects::FaultKind::Parts(vec![Bytes::from(wrong)]), + 0, + ), + ( + bounded_objects::FaultKind::Parts(vec![Bytes::from(grown)]), + 0, + ), + ( + bounded_objects::FaultKind::Parts(vec![Bytes::copy_from_slice( + &fixture.raw[..fixture.raw.len() - 1], + )]), + CHUNK_SIZE as usize, + ), + ( + bounded_objects::FaultKind::LateError(Bytes::copy_from_slice(&fixture.raw)), + CHUNK_SIZE as usize, + ), + ]; + for (kind, delivered) in cases { + fixture.counts.reset(); + fault(&fixture, kind); + let response_budget = MemoryBudget::new(2 * CHUNK_SIZE as usize); + let scratch_budget = MemoryBudget::new(RANGE_WORK_BYTES); + let response = budgeted_app(&fixture, &response_budget, &scratch_budget) + .oneshot(fixture.request("GET", "blob?path=/file", Body::empty())) + .await + .unwrap(); + assert_eq!(response.status(), 200); + let mut stream = response.into_body().into_data_stream(); + let mut bytes = Vec::new(); + loop { + match stream.next().await { + Some(Ok(part)) => bytes.extend_from_slice(&part), + Some(Err(_)) => break, + None => panic!("corrupt current raw source must fail the actual response body"), + } + } + assert_eq!(bytes.len(), delivered); + assert_eq!(bytes, fixture.raw[..delivered]); + assert!(stream.next().await.is_none()); + assert_eq!(response_budget.used(), 0); + assert_eq!(scratch_budget.used(), 0); + assert_eq!(fixture.counts.whole.load(Ordering::SeqCst), 1); + assert_eq!(fixture.counts.range.load(Ordering::SeqCst), 0); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 0); + } + *fixture.counts.object_fault.lock().unwrap() = None; + let response = fixture.send("GET", "blob?path=/file", Body::empty()).await; + assert_eq!( + to_bytes(response.into_body(), 2 * CHUNK_SIZE as usize) + .await + .unwrap() + .as_ref(), + fixture.raw + ); +} + +#[tokio::test] +async fn raw_drop_cancel_and_revocation_during_held_io_drop_producer_and_all_owned_credit() { + for mode in 0..3 { + let fixture = Fixture::new().await; + fixture.map("/file").await; + fixture.counts.reset(); + let entered = Arc::new(Notify::new()); + let release = Arc::new(Notify::new()); + let drops = Arc::new(AtomicUsize::new(0)); + fault( + &fixture, + bounded_objects::FaultKind::Held { + raw: Bytes::copy_from_slice(&fixture.raw), + entered: entered.clone(), + release: release.clone(), + drops: drops.clone(), + }, + ); + let response_budget = MemoryBudget::new(2 * CHUNK_SIZE as usize); + let scratch_budget = MemoryBudget::new(RANGE_WORK_BYTES); + let response = budgeted_app(&fixture, &response_budget, &scratch_budget) + .oneshot(fixture.request("GET", "blob?path=/file", Body::empty())) + .await + .unwrap(); + assert_eq!(response.status(), 200); + fixture.counts.assert(1, 0); + if mode == 0 { + drop(response); + } else { + let mut stream = response.into_body().into_data_stream(); + let task = tokio::spawn(async move { + let first = stream.next().await; + (first, stream) + }); + timeout(Duration::from_secs(10), entered.notified()) + .await + .unwrap(); + fixture.counts.assert(1, 1); + if mode == 1 { + task.abort(); + assert!(task.await.err().unwrap().is_cancelled()); + } else { + release_lease(&fixture).await; + release.notify_one(); + let (first, mut stream) = task.await.unwrap(); + assert!(first.unwrap().is_err()); + assert!(stream.next().await.is_none()); + } + } + assert_eq!(drops.load(Ordering::SeqCst), 1); + assert_eq!(response_budget.used(), 0); + assert_eq!(scratch_budget.used(), 0); + } +} + +#[tokio::test] +async fn raw_first_poll_rechecks_released_lease_before_consuming_body_bytes() { + let fixture = Fixture::new().await; + fixture.map("/file").await; + fixture.counts.reset(); + let response_budget = MemoryBudget::new(2 * CHUNK_SIZE as usize); + let scratch_budget = MemoryBudget::new(RANGE_WORK_BYTES); + let response = budgeted_app(&fixture, &response_budget, &scratch_budget) + .oneshot(fixture.request("GET", "blob?path=/file", Body::empty())) + .await + .unwrap(); + assert_eq!(response.status(), 200); + release_lease(&fixture).await; + let mut stream = response.into_body().into_data_stream(); + assert!(stream.next().await.unwrap().is_err()); + assert!(stream.next().await.is_none()); + fixture.counts.assert(1, 0); + assert_eq!(response_budget.used(), 0); + assert_eq!(scratch_budget.used(), 0); +} + +#[tokio::test] +async fn raw_post_await_revocation_rejects_fragments_and_eof_before_another_backend_poll() { + for case in 0..4 { + let fixture = Fixture::new().await; + fixture.map("/file").await; + fixture.counts.reset(); + let entered = Arc::new(Notify::new()); + let release = Arc::new(Notify::new()); + let tail_polls = Arc::new(AtomicUsize::new(0)); + let drops = Arc::new(AtomicUsize::new(0)); + let prefix = if case < 2 { + Bytes::copy_from_slice(&fixture.raw[..1]) + } else { + Bytes::copy_from_slice(&fixture.raw) + }; + let prefix_length = prefix.len(); + let fragment = match case { + 0 => Some(Bytes::copy_from_slice(&fixture.raw[1..2])), + 1 | 2 => Some(Bytes::new()), + _ => None, + }; + fault( + &fixture, + bounded_objects::FaultKind::HeldFragment { + prefix, + fragment, + entered: entered.clone(), + release: release.clone(), + tail_polls: tail_polls.clone(), + drops: drops.clone(), + }, + ); + let response_budget = MemoryBudget::new(2 * CHUNK_SIZE as usize); + let scratch_budget = MemoryBudget::new(RANGE_WORK_BYTES); + let response = budgeted_app(&fixture, &response_budget, &scratch_budget) + .oneshot(fixture.request("GET", "blob?path=/file", Body::empty())) + .await + .unwrap(); + assert_eq!(response.status(), 200); + let mut stream = response.into_body().into_data_stream(); + let mut task = tokio::spawn(async move { + let mut delivered = Vec::new(); + loop { + match stream.next().await { + Some(Ok(bytes)) => delivered.extend_from_slice(&bytes), + Some(Err(error)) => return (delivered, error, stream), + None => panic!("revoked source must fail the actual body"), + } + } + }); + timeout(Duration::from_secs(10), entered.notified()) + .await + .unwrap(); + fixture.counts.assert(1, prefix_length); + release_lease(&fixture).await; + release.notify_one(); + let finished = timeout(Duration::from_secs(10), &mut task).await; + if finished.is_err() { + task.abort(); + let _ = task.await; + panic!("revoked fragmented source continued into another held backend poll"); + } + let (delivered, _, mut stream) = finished.unwrap().unwrap(); + assert_eq!( + delivered, + fixture.raw[..if case < 2 { 0 } else { CHUNK_SIZE as usize }] + ); + assert!(stream.next().await.is_none()); + assert_eq!(tail_polls.load(Ordering::SeqCst), 0); + assert_eq!(drops.load(Ordering::SeqCst), 1); + assert_eq!(response_budget.used(), 0); + assert_eq!(scratch_budget.used(), 0); + } +} + +#[tokio::test] +async fn empty_raw_post_await_revocation_fails_before_another_eof_poll_or_empty_success() { + for fragment in [Some(Bytes::new()), None] { + let fixture = Fixture::new().await; + let oid = path_oid(&fixture, "/empty").await; + fixture.counts.reset(); + let entered = Arc::new(Notify::new()); + let release = Arc::new(Notify::new()); + let tail_polls = Arc::new(AtomicUsize::new(0)); + let drops = Arc::new(AtomicUsize::new(0)); + *fixture.counts.object_fault.lock().unwrap() = Some(bounded_objects::StreamFault { + oid, + kind: bounded_objects::FaultKind::HeldFragment { + prefix: Bytes::new(), + fragment, + entered: entered.clone(), + release: release.clone(), + tail_polls: tail_polls.clone(), + drops: drops.clone(), + }, + }); + let response_budget = MemoryBudget::new(CHUNK_SIZE as usize); + let scratch_budget = MemoryBudget::new(RANGE_WORK_BYTES); + let app = budgeted_app(&fixture, &response_budget, &scratch_budget); + let request = fixture.request("GET", "blob?path=/empty", Body::empty()); + let mut task = tokio::spawn(async move { app.oneshot(request).await.unwrap() }); + timeout(Duration::from_secs(10), entered.notified()) + .await + .unwrap(); + release_lease(&fixture).await; + release.notify_one(); + let finished = timeout(Duration::from_secs(10), &mut task).await; + if finished.is_err() { + task.abort(); + let _ = task.await; + panic!("revoked empty source continued into another held backend poll"); + } + error(finished.unwrap().unwrap(), 410, "LEASE_EXPIRED", false).await; + assert_eq!(tail_polls.load(Ordering::SeqCst), 0); + assert_eq!(drops.load(Ordering::SeqCst), 1); + assert_eq!(response_budget.used(), 0); + assert_eq!(scratch_budget.used(), 0); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 0); + } +} + +#[tokio::test] +async fn raw_physical_size_and_current_fact_fail_before_body_and_path_errors_keep_formal_statuses() +{ + let fixture = Fixture::new().await; + fixture.map("/file").await; + fixture.counts.reset(); + for (path, status, code) in [ + ("/absent", 404, "PATH_NOT_FOUND"), + ("/outside", 404, "PATH_NOT_FOUND"), + ("/directory", 409, "NOT_DIRECTORY"), + ("/file/child", 409, "NOT_DIRECTORY"), + ("/link/child", 409, "SYMLINK_TRAVERSAL"), + ("/../outside", 400, "SCOPE_INVALID"), + ] { + error( + fixture + .send("GET", &format!("blob?path={path}"), Body::empty()) + .await, + status, + code, + false, + ) + .await; + fixture.counts.assert(0, 0); + } + let mut request = fixture.request("GET", "blob?path=/file", Body::empty()); + request + .headers_mut() + .insert("range", "bytes=0-9".parse().unwrap()); + error( + fixture.app.clone().oneshot(request).await.unwrap(), + 400, + "RANGE_NOT_SUPPORTED", + false, + ) + .await; + error( + fixture + .send( + "GET", + &format!("blob?path=/file&expected_digest=sha256:{}", "0".repeat(64)), + Body::empty(), + ) + .await, + 409, + "EXPECTED_DIGEST_MISMATCH", + false, + ) + .await; + fixture.counts.assert(0, 0); + *fixture.counts.object_size_override.lock().unwrap() = Some(fixture.raw.len() as i64 + 1); + error( + fixture.send("GET", "blob?path=/file", Body::empty()).await, + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + fixture.counts.assert(1, 0); + *fixture.counts.object_size_override.lock().unwrap() = None; + let original = fixture.fact().await; + let mut changed = original.clone(); + changed.created_at += chrono::Duration::seconds(1); + fixture.replace_fact(changed).await; + fixture.counts.reset(); + error( + fixture.send("GET", "blob?path=/file", Body::empty()).await, + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + fixture.counts.assert(0, 0); + fixture.replace_fact(original).await; + for (path, kind, bytes) in [ + ("/empty", "regular", &b""[..]), + ("/link", "symlink", &b"file"[..]), + ("/executable", "executable", fixture.raw.as_slice()), + ] { + let response = fixture + .send("GET", &format!("blob?path={path}"), Body::empty()) + .await; + assert_eq!(response.status(), 200); + assert_eq!(response.headers()["x-mega-fs-kind"], kind); + assert_eq!( + response.headers()["content-length"], + bytes.len().to_string() + ); + assert_eq!( + to_bytes(response.into_body(), 2 * CHUNK_SIZE as usize) + .await + .unwrap() + .as_ref(), + bytes + ); + } +} + +#[tokio::test] +async fn raw_visible_producer_item_cap_rejects_before_copy_hash_or_tail_poll() { + let raw = vec![7; 8 * 1024 * 1024 + 1]; + let fixture = Fixture::new_with_pg_config_directories_and_objects( + false, + 0, + &[("large".into(), raw.clone())], + ) + .await; + let oid = path_oid(&fixture, "/large").await; + *fixture.counts.object_fault.lock().unwrap() = Some(bounded_objects::StreamFault { + oid: oid.clone(), + kind: bounded_objects::FaultKind::Parts( + raw.chunks(CHUNK_SIZE as usize) + .map(Bytes::copy_from_slice) + .collect(), + ), + }); + fixture.map("/large").await; + let tails = Arc::new(AtomicUsize::new(0)); + *fixture.counts.object_fault.lock().unwrap() = Some(bounded_objects::StreamFault { + oid, + kind: bounded_objects::FaultKind::Oversized(Bytes::from(raw), tails.clone()), + }); + fixture.counts.reset(); + let response_budget = MemoryBudget::new(2 * CHUNK_SIZE as usize); + let scratch_budget = MemoryBudget::new(RANGE_WORK_BYTES); + let response = budgeted_app(&fixture, &response_budget, &scratch_budget) + .oneshot(fixture.request("GET", "blob?path=/large", Body::empty())) + .await + .unwrap(); + assert_eq!(response.status(), 200); + let mut stream = response.into_body().into_data_stream(); + assert!(stream.next().await.unwrap().is_err()); + assert!(stream.next().await.is_none()); + assert_eq!(tails.load(Ordering::SeqCst), 0); + assert_eq!(response_budget.used(), 0); + assert_eq!(scratch_budget.used(), 0); + assert_eq!(fixture.counts.whole.load(Ordering::SeqCst), 1); + assert_eq!(fixture.counts.range.load(Ordering::SeqCst), 0); + assert_eq!( + fixture.counts.bytes.load(Ordering::SeqCst), + 8 * 1024 * 1024 + 1 + ); +} + +#[tokio::test] +async fn empty_raw_source_requires_physical_zero_exact_eof_and_current_empty_digest() { + let fixture = Fixture::new().await; + let oid = path_oid(&fixture, "/empty").await; + for (kind, status, code) in [ + ( + bounded_objects::FaultKind::Parts(vec![Bytes::from_static(b"growth")]), + 502, + "INTEGRITY_ERROR", + ), + ( + bounded_objects::FaultKind::LateError(Bytes::new()), + 503, + "OBJECT_UNAVAILABLE", + ), + ] { + *fixture.counts.object_fault.lock().unwrap() = Some(bounded_objects::StreamFault { + oid: oid.clone(), + kind, + }); + fixture.counts.reset(); + error( + fixture.send("GET", "blob?path=/empty", Body::empty()).await, + status, + code, + false, + ) + .await; + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 0); + } + *fixture.counts.object_fault.lock().unwrap() = None; + fixture.counts.reset(); + let mono = fixture.state.storage.mono_storage(); + let original = mono + .get_verified_blobs(vec![oid.clone()]) + .await + .unwrap() + .remove(&oid) + .unwrap(); + let mut changed = original.clone().into_active_model(); + changed.raw_sha256 = Set(vec![9; 32]); + changed.update(mono.get_connection()).await.unwrap(); + error( + fixture.send("GET", "blob?path=/empty", Body::empty()).await, + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + fixture.counts.assert(0, 0); + original + .into_active_model() + .reset_all() + .update(mono.get_connection()) + .await + .unwrap(); + let response = fixture.send("GET", "blob?path=/empty", Body::empty()).await; + assert_eq!(response.status(), 200); + assert_eq!(response.headers()["content-length"], "0"); + assert!(to_bytes(response.into_body(), 1).await.unwrap().is_empty()); + fixture.counts.assert(1, 0); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 0); +} diff --git a/src/api/router/snapshot_reader_retention_tests.rs b/src/api/router/snapshot_reader_retention_tests.rs new file mode 100644 index 00000000..52890bd7 --- /dev/null +++ b/src/api/router/snapshot_reader_retention_tests.rs @@ -0,0 +1,458 @@ +use sea_orm::DatabaseTransaction; + +use super::*; + +#[path = "snapshot_reader_retention_upgrade_tests.rs"] +mod upgrade; + +async fn transaction(fixture: &Fixture) -> DatabaseTransaction { + let schema = q_schema(fixture).await; + let mono = fixture.state.storage.mono_storage(); + let txn = mono.get_connection().begin().await.unwrap(); + txn.execute_unprepared(&format!( + "SET LOCAL search_path=\"{}\",pg_catalog,pg_temp", + schema.replace('"', "\"\"") + )) + .await + .unwrap(); + txn +} + +async fn admission(txn: &DatabaseTransaction, fixture: &Fixture) -> (String, i64) { + let reader = txn.query_one_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "SELECT operation_id::text,reader_issuance FROM mst2_metadata_begin_reader($1,$2, + (SELECT instance_id FROM mst2_qualified_session_incarnation WHERE snapshot_id=$1 AND state='READY' LIMIT 1))", + [fixture.snapshot.clone().into(),fixture.lease.clone().into()])).await.unwrap().unwrap(); + ( + reader.try_get("", "operation_id").unwrap(), + reader.try_get("", "reader_issuance").unwrap(), + ) +} + +async fn finish(txn: &DatabaseTransaction, ticket: &(String, i64)) { + txn.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT mst2_metadata_finish_reader($1::uuid,$2::bigint)", + [ticket.0.clone().into(), ticket.1.into()], + )) + .await + .unwrap(); +} + +async fn prune(txn: &DatabaseTransaction, maximum: i32) -> i64 { + txn.query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT mst2_metadata_prune_readers($1)", + [maximum.into()], + )) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap() +} + +#[tokio::test] +async fn committed_metadata_requests_retire_owners_without_retiring_source_history() { + let fixture = Fixture::new_with_pg_config(true).await; + let initial = q_count( + &fixture, + "SELECT high_water FROM {q}.mst2_metadata_reader_issuance", + ) + .await; + let certificate_count = q_count( + &fixture, + "SELECT count(*) FROM {q}.mst2_metadata_page_certificate", + ) + .await; + let source_count = q_count( + &fixture, + "SELECT count(*) FROM {q}.mst2_metadata_source_root_attestation", + ) + .await; + for _ in 0..96 { + success_json( + fixture + .app + .clone() + .oneshot(fixture.request( + "POST", + "lookup", + Body::from(json!({"paths":["/file"]}).to_string()), + )) + .await + .unwrap(), + ) + .await; + } + assert_eq!( + q_count( + &fixture, + "SELECT high_water FROM {q}.mst2_metadata_reader_issuance" + ) + .await, + initial + 96 + ); + assert_eq!( + q_count( + &fixture, + "SELECT count(*) FROM {q}.mst2_metadata_reader_operation" + ) + .await, + 1 + ); + assert_eq!( + q_count( + &fixture, + "SELECT count(*) FROM {q}.mst2_metadata_reader_operation WHERE state='ACTIVE'" + ) + .await, + 0 + ); + assert_eq!(q_count(&fixture,"SELECT count(*) FROM {q}.mst2_metadata_root_anchor WHERE anchor_kind IN ('REQUEST','READER')").await,0); + assert_eq!( + q_count( + &fixture, + "SELECT count(*) FROM {q}.mst2_metadata_page_certificate" + ) + .await, + certificate_count + ); + assert_eq!( + q_count( + &fixture, + "SELECT count(*) FROM {q}.mst2_metadata_source_root_attestation" + ) + .await, + source_count + ); + success_json( + fixture + .app + .clone() + .oneshot(fixture.request("GET", "descriptor", Body::empty())) + .await + .unwrap(), + ) + .await; + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn terminal_owner_survives_its_deferred_completion_and_prunes_in_a_later_transaction() { + let fixture = Fixture::new_with_pg_config(true).await; + let txn = transaction(&fixture).await; + let ticket = admission(&txn, &fixture).await; + txn.commit().await.unwrap(); + for forbidden in [ + "DELETE FROM mst2_metadata_reader_operation", + "UPDATE mst2_metadata_reader_operation SET state='EXPIRED'", + "UPDATE mst2_metadata_reader_issuance SET high_water=0", + "DELETE FROM mst2_metadata_reader_issuance", + "TRUNCATE mst2_metadata_reader_issuance", + "SELECT mst2_metadata_prune_readers(65)", + ] { + let txn = transaction(&fixture).await; + assert!(txn.execute_unprepared(forbidden).await.is_err()); + txn.rollback().await.unwrap(); + } + let txn = transaction(&fixture).await; + finish(&txn, &ticket).await; + assert_eq!(prune(&txn, 64).await, 0); + txn.commit().await.unwrap(); + let txn = transaction(&fixture).await; + // A terminal no-op must not enqueue another deferred row lookup. + txn.execute_unprepared("UPDATE mst2_metadata_reader_operation SET state=state") + .await + .unwrap(); + assert_eq!(prune(&txn, 0).await, 0); + assert_eq!(prune(&txn, 64).await, 1); + txn.commit().await.unwrap(); + assert_eq!( + q_count( + &fixture, + "SELECT count(*) FROM {q}.mst2_metadata_reader_operation" + ) + .await, + 0 + ); + assert_eq!( + q_count( + &fixture, + "SELECT high_water FROM {q}.mst2_metadata_reader_issuance" + ) + .await, + ticket.1 + ); +} + +#[tokio::test] +async fn reader_pruning_respects_the_64_owner_budget_with_a_committed_backlog() { + let fixture = Fixture::new_with_pg_config(true).await; + let initial = q_count( + &fixture, + "SELECT high_water FROM {q}.mst2_metadata_reader_issuance", + ) + .await; + let txn = transaction(&fixture).await; + // Complete each owner before admitting the next. Their terminal transaction + // fence retains the entire backlog until its deferred checks have committed. + for _ in 0..65 { + let ticket = admission(&txn, &fixture).await; + finish(&txn, &ticket).await; + } + assert_eq!(prune(&txn, 64).await, 0); + txn.commit().await.unwrap(); + assert_eq!( + q_count( + &fixture, + "SELECT count(*) FROM {q}.mst2_metadata_reader_operation WHERE state='FINISHED'" + ) + .await, + 65 + ); + assert_eq!( + q_count( + &fixture, + "SELECT count(*) FROM {q}.mst2_metadata_reader_operation WHERE state='ACTIVE'" + ) + .await, + 0 + ); + assert_eq!(q_count(&fixture,"SELECT count(*) FROM {q}.mst2_metadata_root_anchor WHERE anchor_kind IN ('REQUEST','READER')").await,0); + assert_eq!( + q_count( + &fixture, + "SELECT high_water FROM {q}.mst2_metadata_reader_issuance" + ) + .await, + initial + 65 + ); + let txn = transaction(&fixture).await; + assert_eq!(prune(&txn, 64).await, 64); + txn.commit().await.unwrap(); + assert_eq!( + q_count( + &fixture, + "SELECT count(*) FROM {q}.mst2_metadata_reader_operation" + ) + .await, + 1 + ); + let txn = transaction(&fixture).await; + assert_eq!(prune(&txn, 64).await, 1); + txn.commit().await.unwrap(); + assert_eq!( + q_count( + &fixture, + "SELECT count(*) FROM {q}.mst2_metadata_reader_operation" + ) + .await, + 0 + ); + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn stale_actual_reader_cannot_read_or_finish_a_reissued_uuid() { + let fixture = Fixture::new_with_pg_config(true).await; + let admitted = Arc::new(Barrier::new(2)); + let resume = Arc::new(Barrier::new(2)); + let request = fixture.request( + "POST", + "metadata/pages", + Body::from( + json!({"encoding":"identity","items":[{"directory_path":"/nested"}]}).to_string(), + ), + ); + let app = fixture.app.clone(); + let pending = tokio::spawn(with_rooted_reader_barriers( + admitted.clone(), + resume.clone(), + async move { app.oneshot(request).await.unwrap() }, + )); + tokio::time::timeout(Duration::from_secs(4), admitted.wait()) + .await + .unwrap(); + let txn = transaction(&fixture).await; + let owner=txn.query_one_raw(Statement::from_string(DbBackend::Postgres, + "SELECT operation_id::text,reader_issuance,to_jsonb(r) AS owner FROM mst2_metadata_reader_operation r WHERE state='ACTIVE'")) + .await.unwrap().unwrap(); + let old = ( + owner.try_get::("", "operation_id").unwrap(), + owner.try_get::("", "reader_issuance").unwrap(), + ); + let old_row: Value = owner.try_get("", "owner").unwrap(); + let anchors:Value=txn.query_one_raw(Statement::from_string(DbBackend::Postgres, + "SELECT jsonb_agg(to_jsonb(a)) AS anchors FROM mst2_metadata_root_anchor a WHERE anchor_kind IN ('REQUEST','READER')")) + .await.unwrap().unwrap().try_get("","anchors").unwrap(); + finish(&txn, &old).await; + txn.commit().await.unwrap(); + let txn = transaction(&fixture).await; + assert_eq!(prune(&txn, 64).await, 1); + txn.commit().await.unwrap(); + let txn = transaction(&fixture).await; + assert!(txn.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "INSERT INTO mst2_metadata_reader_operation SELECT * FROM jsonb_populate_record(NULL::mst2_metadata_reader_operation,$1::jsonb)", + [old_row.clone().into()])).await.is_err()); + txn.rollback().await.unwrap(); + let next = old.1 + 1; + let txn = transaction(&fixture).await; + txn.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "INSERT INTO mst2_metadata_reader_operation SELECT * FROM jsonb_populate_record(NULL::mst2_metadata_reader_operation, + $1::jsonb||jsonb_build_object('reader_issuance',$2::bigint))",[old_row.into(),next.into()])).await.unwrap(); + txn.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "INSERT INTO mst2_metadata_root_anchor SELECT (jsonb_populate_record(NULL::mst2_metadata_root_anchor, + value||jsonb_build_object('reader_issuance',$2::bigint))).* FROM jsonb_array_elements($1::jsonb)", + [anchors.into(),next.into()])).await.unwrap(); + txn.commit().await.unwrap(); + let txn = transaction(&fixture).await; + let source = txn + .query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT a.attestation_id::text,a.root_page,a.root_generation,a.root_certificate_digest + FROM mst2_metadata_source_root_attestation a JOIN mst2_metadata_reader_operation r + ON r.root_page=a.root_page AND r.root_generation=a.root_generation + WHERE r.operation_id=$1::uuid AND r.reader_issuance=$2 LIMIT 1", + [old.0.clone().into(), next.into()], + )) + .await + .unwrap() + .unwrap(); + let source_id: String = source.try_get("", "attestation_id").unwrap(); + let source_root: Vec = source.try_get("", "root_page").unwrap(); + let source_generation: i64 = source.try_get("", "root_generation").unwrap(); + let source_certificate: Vec = source.try_get("", "root_certificate_digest").unwrap(); + assert!(txn.query_all_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "SELECT * FROM mst2_metadata_read_source_entries($1::uuid,$2::bigint,$3::uuid,$4,$5,$6,'[]'::jsonb)", + [old.0.clone().into(),old.1.into(),source_id.clone().into(),source_root.clone().into(),source_generation.into(),source_certificate.clone().into()])).await.is_err()); + txn.rollback().await.unwrap(); + let txn = transaction(&fixture).await; + assert!(txn.query_all_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "SELECT * FROM mst2_metadata_read_source_entries($1::uuid,$2::bigint,$3::uuid,$4,$5,$6,'[]'::jsonb)", + [old.0.clone().into(),next.into(),source_id.into(),source_root.into(),source_generation.into(),source_certificate.into()])).await.unwrap().is_empty()); + txn.commit().await.unwrap(); + resume.wait().await; + let response = tokio::time::timeout(Duration::from_secs(4), pending) + .await + .unwrap() + .unwrap(); + assert!(response.status().is_server_error()); + assert_eq!( + q_count( + &fixture, + "SELECT count(*) FROM {q}.mst2_metadata_reader_operation WHERE state='ACTIVE'" + ) + .await, + 1 + ); + assert_eq!(q_count(&fixture,"SELECT count(*) FROM {q}.mst2_metadata_root_anchor WHERE anchor_kind IN ('REQUEST','READER')").await,2); + let txn = transaction(&fixture).await; + finish(&txn, &(old.0, next)).await; + txn.commit().await.unwrap(); + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn rolled_back_reader_issuance_is_unpublished_and_extreme_issuance_fails_closed() { + let fixture = Fixture::new_with_pg_config(true).await; + let initial = q_count( + &fixture, + "SELECT high_water FROM {q}.mst2_metadata_reader_issuance", + ) + .await; + let txn = transaction(&fixture).await; + let unpublished = admission(&txn, &fixture).await; + txn.rollback().await.unwrap(); + assert_eq!( + q_count( + &fixture, + "SELECT high_water FROM {q}.mst2_metadata_reader_issuance" + ) + .await, + initial + ); + assert_eq!( + q_count( + &fixture, + "SELECT count(*) FROM {q}.mst2_metadata_reader_operation" + ) + .await, + 0 + ); + let txn = transaction(&fixture).await; + let published = admission(&txn, &fixture).await; + assert_eq!(published.1, unpublished.1); + txn.commit().await.unwrap(); + let txn = transaction(&fixture).await; + assert_eq!( + txn.query_one_raw(Statement::from_string( + DbBackend::Postgres, + "SELECT mst2_metadata_next_reader_issuance(9223372036854775806)" + )) + .await + .unwrap() + .unwrap() + .try_get_by_index::(0) + .unwrap(), + i64::MAX + ); + assert!( + txn.query_one_raw(Statement::from_string( + DbBackend::Postgres, + "SELECT mst2_metadata_next_reader_issuance(9223372036854775807)" + )) + .await + .is_err() + ); + txn.rollback().await.unwrap(); + assert_eq!( + q_count( + &fixture, + "SELECT high_water FROM {q}.mst2_metadata_reader_issuance" + ) + .await, + published.1 + ); + let txn = transaction(&fixture).await; + finish(&txn, &published).await; + txn.commit().await.unwrap(); +} + +#[tokio::test] +#[ignore = "capacity soak: run explicitly against dedicated PostgreSQL"] +async fn reader_history_remains_bounded_after_more_than_65536_committed_http_requests() { + let fixture = Fixture::new_with_pg_config(true).await; + for _ in 0..65_537 { + success_json( + fixture + .app + .clone() + .oneshot(fixture.request( + "POST", + "lookup", + Body::from(json!({"paths":["/file"]}).to_string()), + )) + .await + .unwrap(), + ) + .await; + } + assert_eq!( + q_count( + &fixture, + "SELECT count(*) FROM {q}.mst2_metadata_reader_operation" + ) + .await, + 1 + ); + assert!( + q_count( + &fixture, + "SELECT high_water FROM {q}.mst2_metadata_reader_issuance" + ) + .await + > 65_536 + ); + fixture.counts.assert(0, 0); +} diff --git a/src/api/router/snapshot_reader_retention_upgrade_tests.rs b/src/api/router/snapshot_reader_retention_upgrade_tests.rs new file mode 100644 index 00000000..54a6eaef --- /dev/null +++ b/src/api/router/snapshot_reader_retention_upgrade_tests.rs @@ -0,0 +1,202 @@ +use super::*; +use crate::jupiter::{ + migration::test_upgrade_reader_retention, + storage::qualified_metadata_family::{ + RootedLookupStatus, RootedQualifiedMetadataRepository, + provision_or_verify_rooted_qualified_family, reader_previous_fixture::restore_previous, + }, +}; + +#[path = "snapshot_native_runtime_upgrade_tests.rs"] +mod native_runtime; + +async fn permanent_rows(fixture: &Fixture) -> Value { + let txn = transaction(fixture).await; + let rows=txn.query_one_raw(Statement::from_string(DbBackend::Postgres, + "SELECT jsonb_build_object( + 'sessions',(SELECT jsonb_agg(to_jsonb(x) ORDER BY snapshot_id,session_incarnation) FROM mst2_qualified_session_incarnation x), + 'leases',(SELECT jsonb_agg(to_jsonb(x) ORDER BY lease_id) FROM mst2_qualified_lease_binding x), + 'sources',(SELECT jsonb_agg(to_jsonb(x) ORDER BY attestation_id) FROM mst2_metadata_source_root_attestation x), + 'certificates',(SELECT jsonb_agg(to_jsonb(x) ORDER BY page_id,generation) FROM mst2_metadata_page_certificate x)) AS rows")) + .await.unwrap().unwrap().try_get("","rows").unwrap(); + txn.commit().await.unwrap(); + rows +} + +#[tokio::test] +async fn current_reader_retention_migration_verifies_fresh_family_without_rewriting_history() { + let fixture = Fixture::new_with_pg_config(true).await; + let before = permanent_rows(&fixture).await; + let high_water = q_count( + &fixture, + "SELECT high_water FROM {q}.mst2_metadata_reader_issuance", + ) + .await; + let mono = fixture.state.storage.mono_storage(); + let core = mono.get_connection(); + test_upgrade_reader_retention(core).await.unwrap(); + assert_eq!(permanent_rows(&fixture).await, before); + assert_eq!( + q_count( + &fixture, + "SELECT high_water FROM {q}.mst2_metadata_reader_issuance" + ) + .await, + high_water + ); + provision_or_verify_rooted_qualified_family(core) + .await + .unwrap(); + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn captured_previous_q_upgrade_preserves_old_sid_source_proofs_and_active_legacy_reader() { + let fixture = Fixture::new_with_pg_config(true).await; + let pinned = fixture + .state + .storage + .snapshot_context(&fixture.snapshot, &fixture.lease) + .await + .unwrap(); + let before = permanent_rows(&fixture).await; + let txn = transaction(&fixture).await; + let active = admission(&txn, &fixture).await; + txn.commit().await.unwrap(); + let schema = q_schema(&fixture).await; + let mono = fixture.state.storage.mono_storage(); + let core = mono.get_connection(); + restore_previous(core, &schema, true).await; + assert_eq!(permanent_rows(&fixture).await, before); + test_upgrade_reader_retention(core).await.unwrap(); + assert_eq!(permanent_rows(&fixture).await, before); + assert_eq!(q_count(&fixture,"SELECT count(*) FROM {q}.mst2_metadata_reader_operation WHERE reader_issuance=0 AND state='ACTIVE'").await,1); + assert_eq!(q_count(&fixture,"SELECT count(*) FROM {q}.mst2_metadata_root_anchor WHERE reader_issuance=0 AND anchor_kind IN ('REQUEST','READER')").await,2); + let restarted = + RootedQualifiedMetadataRepository::open(core, &fixture.state.storage.config().database) + .await + .unwrap(); + let fixed = restarted + .context(&fixture.snapshot, &fixture.lease, &pinned.built.instance_id) + .await + .unwrap(); + assert_eq!(fixed.built.descriptor, pinned.built.descriptor); + assert_eq!(fixed.commit_oid, pinned.commit_oid); + let lookup = restarted + .lookup_metadata(&fixed, &["/file".to_owned()]) + .await + .unwrap(); + assert!(matches!( + lookup.results.as_slice(), + [RootedLookupStatus::File { .. }] + )); + assert_eq!(permanent_rows(&fixture).await, before); + assert_eq!(q_count(&fixture,"SELECT count(*) FROM {q}.mst2_metadata_reader_operation WHERE reader_issuance=0 AND state='ACTIVE'").await,1); + let txn = transaction(&fixture).await; + finish(&txn, &(active.0.clone(), 0)).await; + txn.commit().await.unwrap(); + let txn = transaction(&fixture).await; + assert!(prune(&txn, 64).await >= 1); + txn.commit().await.unwrap(); + assert_eq!( + q_count( + &fixture, + "SELECT count(*) FROM {q}.mst2_metadata_reader_operation WHERE reader_issuance=0" + ) + .await, + 0 + ); + provision_or_verify_rooted_qualified_family(core) + .await + .unwrap(); + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn captured_previous_core_without_q_upgrades_before_new_provisioning() { + let temp = tempfile::tempdir().unwrap(); + let (config, _schema) = test_db_config(temp.path()).await; + let core = crate::jupiter::storage::init::database_connection(&config) + .await + .unwrap(); + let old_schema: String = core + .query_one_raw(Statement::from_string( + DbBackend::Postgres, + "SELECT metadata_schema FROM mst2_metadata_namespace WHERE graph_domain='qualified-v1'", + )) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap(); + restore_previous(&core, &old_schema, false).await; + test_upgrade_reader_retention(&core).await.unwrap(); + provision_or_verify_rooted_qualified_family(&core) + .await + .unwrap(); + let schema: String = core + .query_one_raw(Statement::from_string( + DbBackend::Postgres, + "SELECT metadata_schema FROM mst2_metadata_namespace WHERE graph_domain='qualified-v1'", + )) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap(); + assert_ne!(schema, old_schema); + let water: i64 = core + .query_one_raw(Statement::from_string( + DbBackend::Postgres, + format!( + "SELECT high_water FROM \"{}\".mst2_metadata_reader_issuance", + schema.replace('"', "\"\"") + ), + )) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap(); + assert_eq!(water, 0); + provision_or_verify_rooted_qualified_family(&core) + .await + .unwrap(); +} + +#[tokio::test] +async fn captured_previous_tampered_q_rejects_upgrade_without_partial_issuance_schema() { + let fixture = Fixture::new_with_pg_config(true).await; + let before = permanent_rows(&fixture).await; + let schema = q_schema(&fixture).await; + let mono = fixture.state.storage.mono_storage(); + let core = mono.get_connection(); + restore_previous(core, &schema, true).await; + core.execute_unprepared(&format!( + "ALTER TABLE \"{}\".mst2_metadata_reader_operation ADD COLUMN unexpected integer", + schema.replace('"', "\"\"") + )) + .await + .unwrap(); + assert!(test_upgrade_reader_retention(core).await.is_err()); + assert_eq!(permanent_rows(&fixture).await, before); + let exists: bool = core + .query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT to_regclass($1) IS NULL AS absent", + [format!("{schema}.mst2_metadata_reader_issuance").into()], + )) + .await + .unwrap() + .unwrap() + .try_get("", "absent") + .unwrap(); + assert!(exists); + let implementation:String=core.query_one_raw(Statement::from_string(DbBackend::Postgres, + "SELECT encode(implementation_fingerprint,'hex') FROM mst2_qualified_family_policy WHERE singleton=1")) + .await.unwrap().unwrap().try_get_by_index(0).unwrap(); + assert_eq!( + implementation, + "6d7095dc052e60bf6de21a987b72e23fdfbfe819355f0e847d2a7f3c1cbd3f27" + ); +} diff --git a/src/api/router/snapshot_request.rs b/src/api/router/snapshot_request.rs new file mode 100644 index 00000000..ec43af70 --- /dev/null +++ b/src/api/router/snapshot_request.rs @@ -0,0 +1,114 @@ +//! Bounded raw JSON input for the MST/2 POST surface (spec 14). + +use std::{collections::HashSet, fmt, time::Duration}; + +use axum::{ + extract::{FromRequest, Request}, + http::StatusCode, +}; +use bytes::Bytes; +use serde::{ + Deserialize, Deserializer, + de::{self, MapAccess, SeqAccess, Visitor}, +}; + +use crate::ceres::snapshot::error::{SnapshotError, SnapshotErrorCode}; + +/// One overall read deadline, including a body which keeps trickling bytes. +/// This bounds input collection only, not handler work or response streams. +pub(super) const JSON_REQUEST_TIMEOUT: Duration = Duration::from_secs(10); + +/// Preserve the original bytes for TreeFrame request-body digests. The router +/// supplies DefaultBodyLimit; its rejection is converted to the MST envelope. +pub(super) struct Mst2Bytes(pub(super) Bytes); + +/// Decode keys before comparing them, including escaped spellings. This +/// separate pass also catches duplicate optional fields whose first value is +/// null, which a derived DTO can otherwise treat as an absent field. +pub(super) fn validate_json_keys(body: &[u8]) -> Result<(), serde_json::Error> { + serde_json::from_slice::(body).map(|_| ()) +} + +struct UniqueKeys; + +impl<'de> Deserialize<'de> for UniqueKeys { + fn deserialize>(deserializer: D) -> Result { + deserializer.deserialize_any(UniqueKeyVisitor) + } +} + +struct UniqueKeyVisitor; + +impl<'de> Visitor<'de> for UniqueKeyVisitor { + type Value = UniqueKeys; + + fn expecting(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result { + formatter.write_str("JSON without duplicate object keys") + } + + fn visit_bool(self, _: bool) -> Result { + Ok(UniqueKeys) + } + fn visit_i64(self, _: i64) -> Result { + Ok(UniqueKeys) + } + fn visit_u64(self, _: u64) -> Result { + Ok(UniqueKeys) + } + fn visit_f64(self, _: f64) -> Result { + Ok(UniqueKeys) + } + fn visit_str(self, _: &str) -> Result { + Ok(UniqueKeys) + } + fn visit_unit(self) -> Result { + Ok(UniqueKeys) + } + + fn visit_seq>(self, mut sequence: A) -> Result { + while sequence.next_element::()?.is_some() {} + Ok(UniqueKeys) + } + + fn visit_map>(self, mut object: A) -> Result { + let mut keys = HashSet::new(); + while let Some(key) = object.next_key::()? { + if !keys.insert(key) { + return Err(de::Error::custom("duplicate JSON object key")); + } + object.next_value::()?; + } + Ok(UniqueKeys) + } +} + +impl FromRequest for Mst2Bytes +where + S: Send + Sync, +{ + type Rejection = SnapshotError; + + async fn from_request(req: Request, state: &S) -> Result { + match tokio::time::timeout(JSON_REQUEST_TIMEOUT, Bytes::from_request(req, state)).await { + Ok(Ok(body)) => Ok(Self(body)), + Ok(Err(error)) => { + let (code, message) = if error.status() == StatusCode::PAYLOAD_TOO_LARGE { + ( + SnapshotErrorCode::LimitExceeded, + "request body over the spec 14 limit", + ) + } else { + ( + SnapshotErrorCode::InvalidRequest, + "could not read request body", + ) + }; + Err(SnapshotError::new(code, message)) + } + Err(_) => Err(SnapshotError::new( + SnapshotErrorCode::TemporaryUnavailable, + "request body read deadline exceeded", + )), + } + } +} diff --git a/src/api/router/snapshot_request_tests.rs b/src/api/router/snapshot_request_tests.rs new file mode 100644 index 00000000..fc577c85 --- /dev/null +++ b/src/api/router/snapshot_request_tests.rs @@ -0,0 +1,353 @@ +use std::{ + convert::Infallible, + pin::Pin, + sync::{ + Arc, + atomic::{AtomicBool, AtomicUsize, Ordering}, + }, + task::{Context, Poll}, + time::Duration, +}; + +use axum::{ + Router, + body::{Body, to_bytes}, + http::Request, + middleware, + response::Response, + routing::get, +}; +use bytes::Bytes; +use futures::{Stream, stream}; +use serde_json::Value; +use tower::ServiceExt; + +use super::{JSON_REQUEST_LIMIT, request::JSON_REQUEST_TIMEOUT, routers}; +use crate::{ + api::{MonoApiServiceState, oauth::api_store::BrowserSessionStore}, + ceres::{ + api_service::cache::GitObjectCache, + snapshot::{descriptor::build, runtime::runtime, view::SnapshotView}, + }, + config::testing::isolated_config, + jupiter::tests::{test_redis_manager, test_storage_with_config}, + server::trace_context::{TraceContext, inject_trace_context}, +}; + +struct Fixture { + app: Router, + snapshot_id: String, + lease_id: String, + expires: u64, + _temp: tempfile::TempDir, +} + +impl Fixture { + async fn new() -> Self { + let temp = tempfile::TempDir::new().unwrap(); + let mut config = isolated_config(temp.path().join("config")); + config.mst2.enabled = true; + config.mst2.instance_uuid = Some(uuid::Uuid::new_v4().to_string()); + config.mst2.auth_token = Some("mst2-input-test".to_string()); + let storage = test_storage_with_config(temp.path(), config).await; + let state = MonoApiServiceState { + entity_store: storage.entity_store.clone(), + storage, + session_store: BrowserSessionStore::Anonymous, + git_object_cache: Arc::new(GitObjectCache { + connection: test_redis_manager().await, + prefix: String::new(), + }), + listen_addr: "127.0.0.1:0".to_string(), + }; + let view = SnapshotView::from_commit(&"1".repeat(40), &"2".repeat(40)); + let built = build(&state.storage.config().mst2, &view, "/", [3; 32]).unwrap(); + let ctx = runtime() + .insert_context(built, &view.commit_oid, &view.root_tree_oid, 60) + .expect("fixture lease registration must succeed"); + // Reproduce the server's nest-after-layer order: MST must establish + // its own request context even when the earlier layer does not run. + let app = Router::new() + .route("/outside", get(|| async { "ok" })) + .layer(middleware::from_fn(inject_trace_context)) + .nest("/api/v2", routers(state.clone()).with_state(state)); + Self { + app, + snapshot_id: ctx.built.snapshot_id, + lease_id: ctx.lease_id, + expires: ctx.lease_expires_at_unix, + _temp: temp, + } + } + + fn request(&self, suffix: &str, body: Body) -> Request { + Request::builder() + .method("POST") + .uri(format!("/api/v2/snapshots/{suffix}")) + .header("authorization", "Bearer mst2-input-test") + .header("x-mega-snapshot-lease", &self.lease_id) + .header("content-type", "application/json") + .body(body) + .unwrap() + } + + fn renew_path(&self) -> String { + format!("leases/{}/renew", self.lease_id) + } + + fn assert_lease_unchanged(&self) { + let ctx = runtime().context(&self.snapshot_id).unwrap(); + assert_eq!(ctx.lease_id, self.lease_id); + assert_eq!(ctx.lease_expires_at_unix, self.expires); + } +} + +async fn assert_error(response: Response, status: u16, code: &str, retryable: bool) -> Value { + assert_eq!(response.status().as_u16(), status); + assert_eq!(response.headers()["content-type"], "application/json"); + let request_id = response.headers()["x-request-id"] + .to_str() + .unwrap() + .to_string(); + assert!(!request_id.is_empty()); + let body = to_bytes(response.into_body(), 16 * 1024).await.unwrap(); + let value: Value = serde_json::from_slice(&body).unwrap(); + assert_eq!(value["error"]["code"], code); + assert_eq!(value["error"]["request_id"], request_id); + assert_eq!(value["error"]["retryable"], retryable); + assert!(!value["error"]["message"].as_str().unwrap().is_empty()); + value +} + +fn chunked(bytes: Vec) -> Body { + let chunks: Vec> = bytes + .chunks(4096) + .map(|chunk| Ok(Bytes::copy_from_slice(chunk))) + .collect(); + Body::from_stream(stream::iter(chunks)) +} + +#[tokio::test] +async fn mst2_json_actual_chunked_limit_has_typed_errors_on_every_post() { + let fixture = Fixture::new().await; + let paths = [ + "resolve".to_string(), + fixture.renew_path(), + format!("{}/lookup", fixture.snapshot_id), + format!("{}/metadata/pages", fixture.snapshot_id), + format!("{}/objects", fixture.snapshot_id), + format!("{}/chunks", fixture.snapshot_id), + ]; + for path in paths { + for declared in [None, Some("2")] { + let mut request = fixture.request(&path, chunked(vec![b' '; JSON_REQUEST_LIMIT + 1])); + if let Some(length) = declared { + request + .headers_mut() + .insert("content-length", length.parse().unwrap()); + } + let response = fixture.app.clone().oneshot(request).await.unwrap(); + assert_error(response, 413, "LIMIT_EXCEEDED", false).await; + fixture.assert_lease_unchanged(); + } + } + + let mut body = br#"{"lease_seconds":120}"#.to_vec(); + body.resize(JSON_REQUEST_LIMIT, b' '); + let response = fixture + .app + .clone() + .oneshot(fixture.request(&fixture.renew_path(), chunked(body))) + .await + .unwrap(); + assert_eq!(response.status(), 200); + assert!(!response.headers()["x-request-id"].is_empty()); + let renewed = runtime().context(&fixture.snapshot_id).unwrap(); + assert!(renewed.lease_expires_at_unix > fixture.expires); +} + +#[tokio::test] +async fn mst2_json_declared_oversize_is_rejected_without_reading_body() { + let fixture = Fixture::new().await; + let polls = Arc::new(AtomicUsize::new(0)); + let observed = polls.clone(); + let body = Body::from_stream(stream::poll_fn(move |_| { + observed.fetch_add(1, Ordering::SeqCst); + Poll::>>::Pending + })); + let mut request = fixture.request(&fixture.renew_path(), body); + request.headers_mut().insert( + "content-length", + (JSON_REQUEST_LIMIT + 1).to_string().parse().unwrap(), + ); + let response = fixture.app.clone().oneshot(request).await.unwrap(); + assert_error(response, 413, "LIMIT_EXCEEDED", false).await; + assert_eq!(polls.load(Ordering::SeqCst), 0); + fixture.assert_lease_unchanged(); +} + +#[tokio::test] +async fn mst2_json_closed_dtos_duplicates_and_malformed_input_do_not_renew() { + let fixture = Fixture::new().await; + let renew = fixture.renew_path(); + let lookup = format!("{}/lookup", fixture.snapshot_id); + let metadata = format!("{}/metadata/pages", fixture.snapshot_id); + let objects = format!("{}/objects", fixture.snapshot_id); + let chunks = format!("{}/chunks", fixture.snapshot_id); + let cases: Vec<(&str, &[u8])> = vec![ + ("resolve", br#"{"target":{"kind":"latest"},"scope":"/","scope":"/"}"#), + ("resolve", br#"{"target":{"kind":"latest","kind":"latest"}}"#), + ("resolve", br#"{"target":{"kind":"latest","extra":true}}"#), + ("resolve", br#"{"target":{"kind":"latest"},"extra":true}"#), + (&renew, br#"{"lease_seconds":3600,"lease_seconds":120}"#), + (&renew, br#"{"lease_seconds":3600,"\u006cease_seconds":120}"#), + (&renew, br#"{"lease_seconds":null,"lease_seconds":120}"#), + (&renew, br#"{"lease_seconds":null,"\u006cease_seconds":120}"#), + (&renew, br#"{"lease_seconds":null,"lease_seconds":null}"#), + (&renew, br#"{"lease_seconds":3600,"extra":true}"#), + (&renew, br#"{"lease_seconds":"120"}"#), + (&renew, br#"{"lease_seconds":1.5}"#), + (&renew, br#"{"lease_seconds":-1}"#), + (&renew, br#"{"lease_seconds":NaN}"#), + (&renew, br#"{"lease_seconds":Infinity}"#), + (&renew, br#"{"lease_seconds":18446744073709551616}"#), + (&renew, br#"{"lease_seconds":"#), + (&renew, b"{\"lease_seconds\":\xff}"), + (&renew, b"{\"lease_seconds\":3600} {}"), + (&renew, b"{\"lease_seconds\":3600} x"), + (&lookup, br#"{"paths":[],"paths":[]}"#), + (&lookup, br#"{"paths":[],"extra":true}"#), + (&metadata, br#"{"items":[],"items":[]}"#), + (&metadata, br#"{"items":[],"extra":true}"#), + (&metadata, br#"{"items":[{"directory_path":"/","route":[],"route":[]}]}"#), + (&metadata, br#"{"items":[{"directory_path":"/","extra":true}]}"#), + (&objects, br#"{"items":[],"items":[]}"#), + (&objects, br#"{"items":[],"extra":true}"#), + (&objects, br#"{"items":[{"path":"/a","path":"/b","expected_digest":"x"}]}"#), + (&objects, br#"{"items":[{"path":"/a","expected_digest":"x","extra":true}]}"#), + (&chunks, br#"{"items":[],"items":[]}"#), + (&chunks, br#"{"items":[],"extra":true}"#), + (&chunks, br#"{"items":[{"path":"/a","expected_digest":"x","map_id":"x","chunk_index":"0","chunk_index":"1"}]}"#), + (&chunks, br#"{"items":[{"path":"/a","expected_digest":"x","map_id":"x","chunk_index":"0","extra":true}]}"#), + ]; + for (path, body) in cases { + let response = fixture + .app + .clone() + .oneshot(fixture.request(path, Body::from(body.to_vec()))) + .await + .unwrap(); + assert_error(response, 400, "INVALID_REQUEST", false).await; + fixture.assert_lease_unchanged(); + } + + for body in [Body::empty(), Body::from("{} \r\n\t")] { + let response = fixture + .app + .clone() + .oneshot(fixture.request(&renew, body)) + .await + .unwrap(); + assert_eq!(response.status(), 200); + assert!(!response.headers()["x-request-id"].is_empty()); + } +} + +struct SlowBody { + interval: Option, + chunks: Arc, + dropped: Arc, +} + +impl Stream for SlowBody { + type Item = Result; + + fn poll_next(self: Pin<&mut Self>, cx: &mut Context<'_>) -> Poll> { + let this = self.get_mut(); + match &mut this.interval { + Some(interval) => match interval.poll_tick(cx) { + Poll::Ready(_) => { + this.chunks.fetch_add(1, Ordering::SeqCst); + Poll::Ready(Some(Ok(Bytes::from_static(b" ")))) + } + Poll::Pending => Poll::Pending, + }, + None => Poll::Pending, + } + } +} + +impl Drop for SlowBody { + fn drop(&mut self) { + self.dropped.store(true, Ordering::SeqCst); + } +} + +#[tokio::test] +async fn mst2_json_real_post_pending_and_trickle_bodies_hit_overall_deadline() { + let fixture = Fixture::new().await; + let exercise = |trickle: bool| { + let app = fixture.app.clone(); + let dropped = Arc::new(AtomicBool::new(false)); + let chunks = Arc::new(AtomicUsize::new(0)); + let body = Body::from_stream(SlowBody { + interval: trickle.then(|| tokio::time::interval(Duration::from_millis(50))), + chunks: chunks.clone(), + dropped: dropped.clone(), + }); + let request = fixture.request(&fixture.renew_path(), body); + async move { + let response = tokio::time::timeout( + JSON_REQUEST_TIMEOUT + Duration::from_secs(5), + app.oneshot(request), + ) + .await + .expect("MST input deadline must terminate the request") + .unwrap(); + assert_error(response, 503, "TEMPORARY_UNAVAILABLE", true).await; + assert!(dropped.load(Ordering::SeqCst)); + if trickle { + assert!(chunks.load(Ordering::SeqCst) > 1); + } else { + assert_eq!(chunks.load(Ordering::SeqCst), 0); + } + } + }; + tokio::join!(exercise(false), exercise(true)); + fixture.assert_lease_unchanged(); + let response = fixture + .app + .clone() + .oneshot(fixture.request(&fixture.renew_path(), Body::from("{}"))) + .await + .unwrap(); + assert_eq!(response.status(), 200); +} + +#[tokio::test] +async fn mst2_json_body_read_failure_is_typed_and_preserves_request_context() { + let fixture = Fixture::new().await; + let body = Body::from_stream(stream::once(async { + Err::(std::io::Error::other("private backend key must not leak")) + })); + let mut request = fixture.request(&fixture.renew_path(), body); + request + .headers_mut() + .insert("x-request-id", "different-inbound-id".parse().unwrap()); + request.extensions_mut().insert(TraceContext { + trace_id: Arc::from("existing-trace-id"), + }); + let response = fixture.app.clone().oneshot(request).await.unwrap(); + let value = assert_error(response, 400, "INVALID_REQUEST", false).await; + assert_eq!(value["error"]["request_id"], "existing-trace-id"); + assert_eq!(value["error"]["message"], "could not read request body"); + fixture.assert_lease_unchanged(); + + let mut request = fixture.request(&fixture.renew_path(), Body::from("{")); + request + .headers_mut() + .insert("x-request-id", "accepted-inbound-id".parse().unwrap()); + let response = fixture.app.clone().oneshot(request).await.unwrap(); + let value = assert_error(response, 400, "INVALID_REQUEST", false).await; + assert_eq!(value["error"]["request_id"], "accepted-inbound-id"); +} diff --git a/src/api/router/snapshot_rooted_metadata.rs b/src/api/router/snapshot_rooted_metadata.rs new file mode 100644 index 00000000..159685d4 --- /dev/null +++ b/src/api/router/snapshot_rooted_metadata.rs @@ -0,0 +1,189 @@ +//! JSON metadata endpoints use the same certified, protected fixed-root reader. + +use mst2_codec::metapage::{Entry, EntryKind}; + +use super::*; +use crate::{ + ceres::snapshot::runtime::SnapshotContext, + jupiter::storage::qualified_metadata_family::RootedLookupStatus, +}; + +fn entry_json(entry: &Entry) -> Result { + let name = std::str::from_utf8(&entry.name).map_err(internal)?; + let kind = match entry.kind { + EntryKind::Regular => "regular", + EntryKind::Executable => "executable", + EntryKind::Symlink => "symlink", + EntryKind::Directory => "directory", + }; + let mut value = json!({"name":name,"fs_kind":kind}); + if entry.is_dir() { + value["directory_root"] = json!(format!("sha256:{}", hex_of(&entry.child_root))); + value["node_class"] = json!("native_tree"); + value["lifecycle"] = json!("mutable"); + } else { + value["size"] = json!(entry.size.to_string()); + value["content_digest"] = json!(format!("sha256:{}", hex_of(&entry.content_id))); + } + Ok(value) +} + +fn cursor_name( + sid: &str, + path: &str, + query: &DirectoryQuery, +) -> Result, SnapshotError> { + let Some(cursor) = query.cursor.as_deref() else { + return Ok(None); + }; + let invalid = || { + SnapshotError::new( + SnapshotErrorCode::CursorInvalid, + "cursor bound to different parameters or invalid", + ) + }; + let (payload, signature) = cursor.rsplit_once('.').ok_or_else(invalid)?; + if runtime().sign_cursor(payload) != signature { + return Err(invalid()); + } + let bytes = base64::engine::general_purpose::STANDARD + .decode(payload) + .map_err(|_| invalid())?; + let value: serde_json::Value = serde_json::from_slice(&bytes).map_err(|_| invalid())?; + if value["s"].as_str() != Some(sid) + || value["p"].as_str() != Some(path) + || value["l"].as_u64() != Some(query.limit as u64) + { + return Err(invalid()); + } + let name = value["a"].as_str().ok_or_else(invalid)?; + if name.is_empty() || name.len() > 255 || name.contains('/') || name.contains('\0') { + return Err(invalid()); + } + Ok(Some(name.into())) +} + +#[allow(clippy::result_large_err)] +pub(super) async fn directory_response( + state: &MonoApiServiceState, + ctx: &SnapshotContext, + sid: &str, + query: &DirectoryQuery, +) -> Result { + let absolute = abs_view_path(&ctx.built.descriptor.scope, &query.path); + let last = cursor_name(sid, &absolute, query)?; + let repository = state + .storage + .rooted_qualified_metadata_writer() + .await + .map_err(internal)?; + let window = repository + .directory_window(ctx, &query.path, last.as_deref(), query.limit as usize) + .await + .map_err(mst2_error_response)?; + let entries = window + .entries + .iter() + .map(entry_json) + .collect::, _>>()?; + let next = if window.has_more { + let name = window.entries.last().ok_or_else(|| { + mst2_error_response(internal( + "qualified directory returned an empty continuation", + )) + })?; + let payload = json!({"s":sid,"p":absolute,"l":query.limit,"a":std::str::from_utf8(&name.name).map_err(internal)?}); + let encoded = base64_of(&serde_json::to_vec(&payload).map_err(internal)?); + Some(format!("{encoded}.{}", runtime().sign_cursor(&encoded))) + } else { + None + }; + let ancestors = if query.ancestors.as_deref() == Some("chain") { + window + .ancestors + .iter() + .filter(|(path, _)| path != &query.path) + .map(|(path, root)| { + json!({"path":path, + "directory_root":format!("sha256:{}",hex_of(root)),"node_class":"native_tree"}) + }) + .collect::>() + } else { + Vec::new() + }; + revalidate_request(state, ctx) + .await + .map_err(mst2_error_response)?; + let body = json!({"snapshot_id":sid,"path":query.path,"metadata_root":ctx.built.metadata_root, + "directory_root":format!("sha256:{}",hex_of(&window.directory_root)),"node_class":"native_tree","lifecycle":"mutable", + "range_start_exclusive":last,"entries":entries,"entry_count":window.entry_count.to_string(),"next_cursor":next, + "proof_pages":window.proof_pages.iter().map(|(root,bytes)|json!({"digest":format!("sha256:{}",hex_of(root)), + "data_base64":base64_of(bytes)})).collect::>(),"ancestor_chain":ancestors}); + let mut response = Json(body).into_response(); + let tag = format!( + "\"{}:{}:{}:{}\"", + &sid[..16.min(sid.len())], + hex_of(&window.directory_root), + query.limit, + query.cursor.as_deref().unwrap_or("") + ); + if let Ok(value) = HeaderValue::from_str(&tag) { + response.headers_mut().insert("etag", value); + } + response.headers_mut().insert( + "cache-control", + HeaderValue::from_static("private, no-cache, no-transform"), + ); + Ok(response) +} + +#[allow(clippy::result_large_err)] +pub(super) async fn lookup_response( + state: &MonoApiServiceState, + ctx: &SnapshotContext, + sid: &str, + request: &LookupRequest, +) -> Result { + let repository = state + .storage + .rooted_qualified_metadata_writer() + .await + .map_err(internal)?; + let batch = repository + .lookup_metadata(ctx, &request.paths) + .await + .map_err(mst2_error_response)?; + let mut results = Vec::with_capacity(batch.results.len()); + for (path, status) in request.paths.iter().zip(batch.results) { + let mut item = json!({"path":path}); + match status { + RootedLookupStatus::Directory(root) => { + let mut node = json!({"fs_kind":"directory","directory_root":format!("sha256:{}",hex_of(&root)), + "node_class":"native_tree","lifecycle":"mutable"}); + if path != "/" { + node["name"] = json!(path.rsplit('/').next()); + } + item["status"] = json!("found"); + item["node"] = node; + } + RootedLookupStatus::File { entry, .. } => { + item["status"] = json!("found"); + item["node"] = entry_json(&entry)?; + } + RootedLookupStatus::Absent => item["status"] = json!("absent"), + RootedLookupStatus::NotDirectory { symlink } => { + item["status"] = json!(if symlink { + "symlink_traversal" + } else { + "not_directory" + }) + } + } + results.push(item); + } + revalidate_request(state, ctx) + .await + .map_err(mst2_error_response)?; + Ok(Json(json!({"snapshot_id":sid,"results":results,"proof_pages":batch.proof_pages.iter().map(|(root,bytes)|json!({ + "digest":format!("sha256:{}",hex_of(root)),"data_base64":base64_of(bytes)})).collect::>()})).into_response()) +} diff --git a/src/api/router/snapshot_rooted_metadata_tests.rs b/src/api/router/snapshot_rooted_metadata_tests.rs new file mode 100644 index 00000000..33179e3d --- /dev/null +++ b/src/api/router/snapshot_rooted_metadata_tests.rs @@ -0,0 +1,790 @@ +use sea_orm::{DbBackend, Statement, TransactionTrait}; +use tokio::sync::Barrier; + +use super::*; +use crate::jupiter::storage::qualified_metadata_family::{ + SnapshotMetadataFamily, with_rooted_reader_barriers, with_rooted_source_fact_barriers, + with_rooted_source_temporary_shadow, +}; + +#[path = "snapshot_reader_retention_tests.rs"] +mod reader_retention; + +#[path = "snapshot_descriptor_wire_tests.rs"] +mod descriptor_wire; + +async fn q_schema(fixture: &Fixture) -> String { + fixture + .state + .storage + .mono_storage() + .get_connection() + .query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT namespace.metadata_schema FROM mst2_snapshot_storage_route route + JOIN mst2_metadata_namespace namespace USING(namespace_uuid) WHERE route.snapshot_id=$1 + AND namespace.graph_domain='qualified-v1'", + [fixture.snapshot.clone().into()], + )) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap() +} + +async fn q_count(fixture: &Fixture, query: &str) -> i64 { + let schema = q_schema(fixture).await; + let query = query.replace("{q}", &format!("\"{}\"", schema.replace('"', "\"\""))); + fixture + .state + .storage + .mono_storage() + .get_connection() + .query_one_raw(Statement::from_string(DbBackend::Postgres, query)) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap() +} + +async fn resolve(fixture: &Fixture) -> Value { + success_json( + fixture + .app + .clone() + .oneshot( + Request::builder() + .method("POST") + .uri("/api/v2/snapshots/resolve") + .header("authorization", format!("Bearer {TOKEN}")) + .header("content-type", "application/json") + .body(Body::from( + json!({"target":{"kind":"latest"},"scope":"/project"}).to_string(), + )) + .unwrap(), + ) + .await + .unwrap(), + ) + .await +} + +async fn release(fixture: &Fixture, lease: &str) -> Value { + success_json( + fixture + .app + .clone() + .oneshot( + Request::builder() + .method("DELETE") + .uri(format!("/api/v2/snapshots/leases/{lease}")) + .header("authorization", format!("Bearer {TOKEN}")) + .body(Body::empty()) + .unwrap(), + ) + .await + .unwrap(), + ) + .await +} + +async fn collect_unowned(fixture: &Fixture) { + let writer = fixture + .state + .storage + .rooted_qualified_metadata_writer() + .await + .unwrap(); + for _ in 0..8 { + let work = writer.maintenance_tick(64).await.unwrap(); + assert!(work.collector_enabled && work.examined <= 64); + if q_count(fixture, "SELECT count(*) FROM {q}.mst2_metadata_payload").await == 0 { + return; + } + } + panic!("bounded maintenance failed to collect the small unowned fixture"); +} + +#[tokio::test] +async fn default_rooted_http_handoff_renew_release_and_fresh_incarnation_keep_exact_ownership() { + let fixture = Fixture::new_with_pg_config(true).await; + assert_eq!( + fixture + .state + .storage + .snapshot_metadata_family(&fixture.lease, true) + .await + .unwrap(), + Some(SnapshotMetadataFamily::Rooted) + ); + assert_eq!( + q_count( + &fixture, + "SELECT count(*) FROM {q}.mst2_qualified_session_incarnation WHERE state='READY'" + ) + .await, + 1 + ); + let first_generation = q_count( + &fixture, + "SELECT root_generation FROM {q}.mst2_qualified_session_incarnation WHERE state='READY'", + ) + .await; + assert_eq!(q_count(&fixture,"SELECT count(*) FROM {q}.mst2_metadata_root_anchor WHERE anchor_kind IN ('PREPARE','REUSE')").await,0); + assert_eq!(q_count(&fixture,"SELECT count(*) FROM {q}.mst2_metadata_prepare WHERE plan_kind='ROOTED' AND state='COMMITTED' AND coverage_retired_at IS NOT NULL").await,1); + let original_payloads = + q_count(&fixture, "SELECT count(*) FROM {q}.mst2_metadata_payload").await; + let next = resolve(&fixture).await; + assert_eq!(next["descriptor"]["snapshot_id"], fixture.snapshot); + let second = next["lease_id"].as_str().unwrap(); + assert_ne!(second, fixture.lease); + assert_eq!( + q_count(&fixture, "SELECT count(*) FROM {q}.mst2_metadata_prepare").await, + 1 + ); + assert_eq!( + q_count(&fixture, "SELECT count(*) FROM {q}.mst2_metadata_payload").await, + original_payloads + ); + let previous_deadline = fixture + .state + .storage + .snapshot_context(&fixture.snapshot, second) + .await + .unwrap() + .lease_expires_at_unix; + let renewed = fixture + .state + .storage + .snapshot_renew(second, 1) + .await + .unwrap(); + assert_eq!(renewed.snapshot_id, fixture.snapshot); + assert!(renewed.expires_at_unix >= previous_deadline); + assert_eq!(release(&fixture, &fixture.lease).await["released"], true); + assert_eq!(release(&fixture, &fixture.lease).await["released"], false); + assert_eq!(q_count(&fixture,"SELECT count(*) FROM {q}.mst2_qualified_lease_binding WHERE state='RELEASED' AND lease_epoch=2").await,1); + let mut request = fixture.request("GET", "descriptor", Body::empty()); + request + .headers_mut() + .insert("x-mega-snapshot-lease", second.parse().unwrap()); + success_json(fixture.app.clone().oneshot(request).await.unwrap()).await; + assert_eq!(release(&fixture, second).await["released"], true); + assert_eq!( + q_count( + &fixture, + "SELECT count(*) FROM {q}.mst2_metadata_root_anchor" + ) + .await, + 0 + ); + assert_eq!( + q_count( + &fixture, + "SELECT count(*) FROM {q}.mst2_qualified_session_incarnation WHERE state='RETIRED'" + ) + .await, + 1 + ); + collect_unowned(&fixture).await; + assert_eq!( + q_count( + &fixture, + "SELECT count(*) FROM {q}.mst2_metadata_graph_node" + ) + .await, + 0 + ); + assert!( + q_count( + &fixture, + "SELECT count(*) FROM {q}.mst2_metadata_page_certificate" + ) + .await + > 0 + ); + let fresh = resolve(&fixture).await; + assert_eq!(fresh["descriptor"]["snapshot_id"], fixture.snapshot); + assert_ne!(fresh["lease_id"], next["lease_id"]); + assert_eq!( + q_count( + &fixture, + "SELECT count(*) FROM {q}.mst2_qualified_session_incarnation" + ) + .await, + 2 + ); + assert_eq!( + q_count( + &fixture, + "SELECT count(*) FROM {q}.mst2_qualified_session_incarnation WHERE state='READY'" + ) + .await, + 1 + ); + assert_eq!( + q_count(&fixture, "SELECT count(*) FROM {q}.mst2_metadata_payload").await, + original_payloads + ); + assert_eq!( + q_count( + &fixture, + "SELECT root_generation FROM {q}.mst2_qualified_session_incarnation WHERE state='READY'" + ) + .await, + first_generation + 1 + ); + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn rooted_wide_directory_windows_and_lookup_survive_rebuild_with_valid_proofs() { + let fixture = Fixture::new_with_pg_config_and_directories(true, 140).await; + let config = fixture.state.storage.config(); + let connection = crate::jupiter::storage::init::postgres_connection(&config.database) + .await + .unwrap(); + let storage = crate::jupiter::storage::Storage::new_with_connection( + config, + Arc::new(connection), + fixture.state.storage.git_service.obj_storage.clone(), + ) + .await + .unwrap(); + let state = MonoApiServiceState { + storage, + ..fixture.state.clone() + }; + let app = Router::new().nest("/api/v2", routers(state.clone()).with_state(state)); + let mut cursor = None; + let mut names = Vec::new(); + for _ in 0..12 { + let mut suffix = "directory?path=/&limit=17".to_string(); + if let Some(current) = &cursor { + suffix.push_str(&format!("&cursor={current}")); + } + let response = app + .clone() + .oneshot(fixture.request("GET", &suffix, Body::empty())) + .await + .unwrap(); + let window = success_json(response).await; + assert_eq!(window["entry_count"], "147"); + let entries = window["entries"].as_array().unwrap(); + assert!(!entries.is_empty() && entries.len() <= 17); + names.extend( + entries + .iter() + .map(|entry| entry["name"].as_str().unwrap().to_owned()), + ); + for proof in window["proof_pages"].as_array().unwrap() { + let bytes = STANDARD + .decode(proof["data_base64"].as_str().unwrap()) + .unwrap(); + mst2_codec::metapage::Page::decode(&bytes).unwrap(); + assert_eq!( + proof["digest"], + format!("sha256:{}", hex_of(&mst2_codec::metapage::page_id(&bytes))) + ); + } + cursor = window["next_cursor"].as_str().map(str::to_owned); + if cursor.is_none() { + break; + } + } + assert!(cursor.is_none()); + assert_eq!(names.len(), 147); + assert!( + names + .windows(2) + .all(|pair| pair[0].as_bytes() < pair[1].as_bytes()) + ); + let lookup=success_json(app.oneshot(fixture.request("POST","lookup", + Body::from(json!({"paths":["/nested/file","/wide-139/file-139","/missing","/file/child","/link/child"]}).to_string()))) + .await.unwrap()).await; + let statuses: Vec<_> = lookup["results"] + .as_array() + .unwrap() + .iter() + .map(|row| row["status"].as_str().unwrap()) + .collect(); + assert_eq!( + statuses, + [ + "found", + "found", + "absent", + "not_directory", + "symlink_traversal" + ] + ); + assert_eq!( + q_count( + &fixture, + "SELECT count(*) FROM {q}.mst2_metadata_reader_operation WHERE state='ACTIVE'" + ) + .await, + 0 + ); + assert_eq!(q_count(&fixture,"SELECT count(*) FROM {q}.mst2_metadata_root_anchor WHERE anchor_kind IN ('REQUEST','READER')").await,0); + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn rooted_reader_release_race_keeps_independent_roots_until_owned_buffers_finish() { + let fixture = Fixture::new_with_pg_config(true).await; + let admitted = Arc::new(Barrier::new(2)); + let resume = Arc::new(Barrier::new(2)); + let request = fixture.request( + "POST", + "metadata/pages", + Body::from( + json!({"encoding":"identity","items":[{"directory_path":"/nested"}]}).to_string(), + ), + ); + let app = fixture.app.clone(); + let pending = tokio::spawn(with_rooted_reader_barriers( + admitted.clone(), + resume.clone(), + async move { app.oneshot(request).await.unwrap() }, + )); + tokio::time::timeout(Duration::from_secs(4), admitted.wait()) + .await + .unwrap(); + assert_eq!(q_count(&fixture,"SELECT count(*) FROM {q}.mst2_metadata_root_anchor WHERE anchor_kind IN ('REQUEST','READER')").await,2); + let work = fixture + .state + .storage + .rooted_qualified_metadata_writer() + .await + .unwrap() + .maintenance_tick(64) + .await + .unwrap(); + assert_eq!(work.payload_pages_removed, 0); + assert_eq!(release(&fixture, &fixture.lease).await["released"], true); + assert_eq!(q_count(&fixture,"SELECT count(*) FROM {q}.mst2_metadata_root_anchor WHERE anchor_kind IN ('REQUEST','READER')").await,2); + let work = fixture + .state + .storage + .rooted_qualified_metadata_writer() + .await + .unwrap() + .maintenance_tick(64) + .await + .unwrap(); + assert_eq!(work.payload_pages_removed, 0); + assert_eq!( + q_count( + &fixture, + "SELECT count(*) FROM {q}.mst2_metadata_root_anchor WHERE anchor_kind='LEASE'" + ) + .await, + 1 + ); + resume.wait().await; + error( + tokio::time::timeout(Duration::from_secs(4), pending) + .await + .unwrap() + .unwrap(), + 410, + "LEASE_EXPIRED", + false, + ) + .await; + assert_eq!( + q_count( + &fixture, + "SELECT count(*) FROM {q}.mst2_metadata_reader_operation WHERE state='FINISHED'" + ) + .await, + 1 + ); + assert_eq!( + q_count( + &fixture, + "SELECT count(*) FROM {q}.mst2_metadata_root_anchor" + ) + .await, + 0 + ); + collect_unowned(&fixture).await; + fixture.counts.assert(0, 0); +} + +async fn source_oid(fixture: &Fixture) -> String { + let schema = q_schema(fixture).await; + let quoted = format!("\"{}\"", schema.replace('"', "\"\"")); + fixture.state.storage.mono_storage().get_connection().query_one_raw(Statement::from_string(DbBackend::Postgres, + format!("SELECT split_part(a.tagged_tree_oid,':',2) FROM {quoted}.mst2_qualified_session_incarnation s + JOIN {quoted}.mst2_metadata_source_root_attestation a ON a.attestation_id=s.attestation_id WHERE s.state='READY'"))) + .await.unwrap().unwrap().try_get_by_index(0).unwrap() +} + +async fn source_revision(fixture: &Fixture, oid: &str) -> String { + fixture + .state + .storage + .mono_storage() + .get_connection() + .query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT revision::text FROM mst2_rooted_source_tree_revision WHERE tree_id=$1", + [oid.into()], + )) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap() +} + +#[tokio::test] +async fn actual_q_body_callers_preserve_unchanged_source_revision_and_ignore_temp_shadow() { + let fixture = Fixture::new_with_pg_config_and_directories(true, 140).await; + fixture.map("/file").await; + fixture.counts.reset(); + let oid = source_oid(&fixture).await; + let revision = source_revision(&fixture, &oid).await; + fixture + .state + .storage + .mono_storage() + .get_connection() + .execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "UPDATE mega_tree SET sub_trees=sub_trees,pack_offset=pack_offset WHERE tree_id=$1", + [oid.clone().into()], + )) + .await + .unwrap(); + assert_eq!(source_revision(&fixture, &oid).await, revision); + let response = + with_rooted_source_temporary_shadow(fixture.send("HEAD", "blob?path=/file", Body::empty())) + .await; + assert_eq!(response.status(), 200); + assert_eq!( + response.headers()["x-mega-content-size"], + fixture.raw.len().to_string() + ); + fixture.counts.assert(0, 0); + let response = + with_rooted_source_temporary_shadow(fixture.send("GET", "blob?path=/file", Body::empty())) + .await; + assert_eq!(response.status(), 200); + assert_eq!( + to_bytes(response.into_body(), fixture.raw.len() + 1) + .await + .unwrap() + .as_ref(), + fixture.raw + ); + assert_eq!( + q_count( + &fixture, + "SELECT count(*) FROM {q}.mst2_metadata_reader_operation WHERE state='ACTIVE'" + ) + .await, + 0 + ); +} + +#[tokio::test] +async fn actual_q_body_callers_reject_changed_or_deleted_source_before_opening_a_body() { + for delete in [false, true] { + let fixture = Fixture::new_with_pg_config(true).await; + let oid = source_oid(&fixture).await; + let revision = source_revision(&fixture, &oid).await; + fixture + .state + .storage + .mono_storage() + .get_connection() + .execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + if delete { + "DELETE FROM mega_tree WHERE tree_id=$1" + } else { + "UPDATE mega_tree SET sub_trees=decode('00','hex') WHERE tree_id=$1" + }, + [oid.clone().into()], + )) + .await + .unwrap(); + assert_ne!(source_revision(&fixture, &oid).await, revision); + let response = fixture.send("GET", "blob?path=/file", Body::empty()).await; + assert!(response.status().is_client_error() || response.status().is_server_error()); + fixture.counts.assert(0, 0); + } +} + +#[tokio::test] +async fn actual_q_body_callers_recheck_only_returned_current_file_facts() { + let fixture = Fixture::new_with_pg_config(true).await; + fixture.state.storage.mono_storage().get_connection().execute_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "UPDATE mst2_verified_object SET verification_version=1 WHERE storage_domain='git' AND object_kind='blob' AND git_oid=$1", + [fixture.oid.clone().into()])).await.unwrap(); + let untouched = fixture + .send("HEAD", "blob?path=/empty", Body::empty()) + .await; + assert_eq!(untouched.status(), 200); + assert_eq!(untouched.headers()["x-mega-content-size"], "0"); + error( + fixture.send("GET", "blob?path=/file", Body::empty()).await, + 503, + "METADATA_NOT_READY", + false, + ) + .await; + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn actual_q_body_caller_source_mutation_after_reader_admission_cannot_serve_stale_content() { + let fixture = Fixture::new_with_pg_config(true).await; + let oid = source_oid(&fixture).await; + let admitted = Arc::new(Barrier::new(2)); + let resume = Arc::new(Barrier::new(2)); + let app = fixture.app.clone(); + let request = fixture.request("GET", "blob?path=/file", Body::empty()); + let pending = tokio::spawn(with_rooted_reader_barriers( + admitted.clone(), + resume.clone(), + async move { app.oneshot(request).await.unwrap() }, + )); + tokio::time::timeout(Duration::from_secs(4), admitted.wait()) + .await + .unwrap(); + fixture + .state + .storage + .mono_storage() + .get_connection() + .execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "UPDATE mega_tree SET sub_trees=decode('00','hex') WHERE tree_id=$1", + [oid.into()], + )) + .await + .unwrap(); + resume.wait().await; + let response = tokio::time::timeout(Duration::from_secs(4), pending) + .await + .unwrap() + .unwrap(); + assert!(response.status().is_client_error() || response.status().is_server_error()); + assert_eq!(q_count(&fixture,"SELECT count(*) FROM {q}.mst2_metadata_root_anchor WHERE anchor_kind IN ('REQUEST','READER')").await,0); + assert_eq!( + q_count( + &fixture, + "SELECT count(*) FROM {q}.mst2_metadata_reader_operation WHERE state='FINISHED'" + ) + .await, + 1 + ); + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn actual_q_body_caller_retries_a_busy_current_file_fact_without_opening_a_body() { + let fixture = Fixture::new_with_pg_config(true).await; + let writer = fixture + .state + .storage + .mono_storage() + .get_connection() + .begin() + .await + .unwrap(); + writer.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "UPDATE mst2_verified_object SET raw_sha256=raw_sha256 WHERE storage_domain='git' AND object_kind='blob' AND git_oid=$1", + [fixture.oid.clone().into()], + )).await.unwrap(); + error( + fixture.send("GET", "blob?path=/file", Body::empty()).await, + 503, + "TEMPORARY_UNAVAILABLE", + true, + ) + .await; + assert_eq!(q_count(&fixture,"SELECT count(*) FROM {q}.mst2_metadata_root_anchor WHERE anchor_kind IN ('REQUEST','READER')").await,0); + fixture.counts.assert(0, 0); + writer.rollback().await.unwrap(); + assert_eq!( + fixture + .send("HEAD", "blob?path=/file", Body::empty()) + .await + .status(), + 200 + ); + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn actual_q_body_caller_holds_the_selected_current_fact_until_the_read_transaction_finishes() +{ + let fixture = Fixture::new_with_pg_config(true).await; + let admitted = Arc::new(Barrier::new(2)); + let resume = Arc::new(Barrier::new(2)); + let app = fixture.app.clone(); + let request = fixture.request("HEAD", "blob?path=/file", Body::empty()); + let pending = tokio::spawn(with_rooted_source_fact_barriers( + admitted.clone(), + resume.clone(), + async move { app.oneshot(request).await.unwrap() }, + )); + tokio::time::timeout(Duration::from_secs(4), admitted.wait()) + .await + .unwrap(); + let core_writer = fixture + .state + .storage + .mono_storage() + .get_connection() + .clone(); + let oid = fixture.oid.clone(); + let (ready, received) = tokio::sync::oneshot::channel(); + let update = tokio::spawn(async move { + let txn = core_writer.begin().await.unwrap(); + let pid: i32 = txn + .query_one_raw(Statement::from_string( + DbBackend::Postgres, + "SELECT pg_backend_pid()", + )) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap(); + ready.send(pid).unwrap(); + txn.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "UPDATE mst2_verified_object SET raw_sha256=raw_sha256 WHERE storage_domain='git' AND object_kind='blob' AND git_oid=$1", + [oid.into()])).await.unwrap(); + txn.commit().await.unwrap(); + }); + let pid = received.await.unwrap(); + tokio::time::timeout(Duration::from_secs(4),async { + loop { + let waiting:bool=fixture.state.storage.mono_storage().get_connection().query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres,"SELECT coalesce(wait_event_type='Lock',false) FROM pg_stat_activity WHERE pid=$1",[pid.into()])) + .await.unwrap().unwrap().try_get_by_index(0).unwrap(); + if waiting {break;} tokio::task::yield_now().await; + } + }).await.unwrap(); + assert!(!update.is_finished()); + resume.wait().await; + let response = tokio::time::timeout(Duration::from_secs(4), pending) + .await + .unwrap() + .unwrap(); + assert_eq!(response.status(), 200); + tokio::time::timeout(Duration::from_secs(4), update) + .await + .unwrap() + .unwrap(); + assert_eq!(q_count(&fixture,"SELECT count(*) FROM {q}.mst2_metadata_root_anchor WHERE anchor_kind IN ('REQUEST','READER')").await,0); + fixture.counts.assert(0, 0); +} + +async fn durable_lease_state(fixture: &Fixture) -> Value { + let schema = q_schema(fixture).await; + let quoted = format!("\"{}\"", schema.replace('"', "\"\"")); + fixture.state.storage.mono_storage().get_connection().query_one_raw(Statement::from_string(DbBackend::Postgres, + format!("SELECT jsonb_build_object( + 'sessions',(SELECT jsonb_agg(to_jsonb(s) ORDER BY s.snapshot_id,s.session_incarnation) FROM {quoted}.mst2_qualified_session_incarnation s), + 'leases',(SELECT jsonb_agg(to_jsonb(l) ORDER BY l.lease_id) FROM {quoted}.mst2_qualified_lease_binding l), + 'anchors',(SELECT jsonb_agg(to_jsonb(a) ORDER BY a.anchor_id) FROM {quoted}.mst2_metadata_root_anchor a), + 'roots',(SELECT jsonb_agg(to_jsonb(r) ORDER BY r.prepare_id,r.page_id,r.generation) FROM {quoted}.mst2_metadata_graph_root r), + 'routes',(SELECT jsonb_agg(to_jsonb(r) ORDER BY r.lease_id) FROM mst2_lease_storage_route r))"))) + .await.unwrap().unwrap().try_get_by_index(0).unwrap() +} + +#[tokio::test] +async fn actual_q_lease_http_and_direct_handoff_retry_source_lock_without_partial_durable_mutation() +{ + let fixture = Fixture::new_with_pg_config(true).await; + let storage = &fixture.state.storage; + let context = storage + .snapshot_context(&fixture.snapshot, &fixture.lease) + .await + .unwrap(); + let config = storage.config(); + let instance = config.mst2.instance_uuid.as_deref().unwrap(); + let head = storage + .mono_storage() + .read_native_publication_head(instance) + .await + .unwrap(); + let repository = storage.rooted_qualified_metadata_writer().await.unwrap(); + let before = durable_lease_state(&fixture).await; + let revision = source_revision(&fixture, &source_oid(&fixture).await).await; + let writer = storage + .mono_storage() + .get_connection() + .begin() + .await + .unwrap(); + let locked=writer.query_one_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "SELECT revision::text FROM mst2_rooted_source_tree_revision WHERE revision=$1::uuid FOR UPDATE", + [revision.clone().into()])).await.unwrap().unwrap(); + assert_eq!(locked.try_get_by_index::(0).unwrap(), revision); + let direct_errors = tokio::time::timeout(Duration::from_secs(4), async { + [ + repository + .open_session(&head, &context.built, None, 600) + .await + .unwrap_err(), + repository + .renew(&fixture.lease, 600, instance) + .await + .unwrap_err(), + repository.release(&fixture.lease).await.unwrap_err(), + ] + }) + .await + .unwrap(); + for error in direct_errors { + assert_eq!(error.code, SnapshotErrorCode::TemporaryUnavailable); + } + assert_eq!(durable_lease_state(&fixture).await, before); + for (method, uri) in [ + ("POST", "/api/v2/snapshots/resolve".to_owned()), + ( + "POST", + format!("/api/v2/snapshots/leases/{}/renew", fixture.lease), + ), + ( + "DELETE", + format!("/api/v2/snapshots/leases/{}", fixture.lease), + ), + ] { + let body = if uri.ends_with("/resolve") { + Body::from(json!({"target":{"kind":"latest"},"scope":"/project"}).to_string()) + } else { + Body::empty() + }; + let request = Request::builder() + .method(method) + .uri(uri) + .header("authorization", format!("Bearer {TOKEN}")) + .header("content-type", "application/json") + .body(body) + .unwrap(); + let response = + tokio::time::timeout(Duration::from_secs(4), fixture.app.clone().oneshot(request)) + .await + .unwrap() + .unwrap(); + error(response, 503, "TEMPORARY_UNAVAILABLE", true).await; + assert_eq!(durable_lease_state(&fixture).await, before); + } + writer.rollback().await.unwrap(); + assert_eq!(durable_lease_state(&fixture).await, before); + let next = resolve(&fixture).await; + assert_ne!(next["lease_id"], fixture.lease); + assert_eq!(release(&fixture, &fixture.lease).await["released"], true); +} diff --git a/src/api/router/snapshot_router.rs b/src/api/router/snapshot_router.rs index 556897e3..cb593e72 100644 --- a/src/api/router/snapshot_router.rs +++ b/src/api/router/snapshot_router.rs @@ -15,19 +15,27 @@ use axum::{ }; use base64::Engine; use bytes::Bytes; +use git_internal::hash::{ObjectHash, get_hash_kind}; use mst2_codec::descriptor; +use request::Mst2Bytes; use serde::Deserialize; use serde_json::json; +use sha2::{Digest, Sha256}; use crate::{ api::MonoApiServiceState, ceres::snapshot::{ descriptor::build as build_descriptor, error::{SnapshotError, SnapshotErrorCode}, - pages::{WalkOutcome, base64_of, build_directory_page, hex_of, proof_pages, resolve_abs}, + pages::{ + MetadataWalkOutcome, base64_of, build_directory_page, build_directory_page_with_work, + hex_of, proof_pages, resolve_abs_metadata, + }, + projection_observation::{NativeResolveSource, ResolvedProjection}, runtime::{now_unix, runtime}, view::{SnapshotView, validate_scope_relative_path}, }, + orbit_api::factory::MegaObjectStorageWrapper, }; pub fn routers(api_state: MonoApiServiceState) -> Router { @@ -40,7 +48,7 @@ pub fn routers(api_state: MonoApiServiceState) -> Router { .route("/snapshots/{snapshot_id}/directory", get(directory)) .route( "/snapshots/{snapshot_id}/blob", - get(blob).head(content::blob_head), + get(raw_blob::blob).head(content::blob_head), ) .route("/snapshots/{snapshot_id}/lookup", post(lookup)) .route( @@ -69,17 +77,185 @@ pub fn routers(api_state: MonoApiServiceState) -> Router { )) } +/// Explicit fixture for the historical G authority and corruption contracts. +/// Production resolve always installs the rooted family for a new SID. +#[cfg(test)] +pub(crate) fn generic_history_routers( + api_state: MonoApiServiceState, +) -> Router { + routers(api_state).layer(axum::middleware::from_fn( + |request: axum::extract::Request, next: axum::middleware::Next| async move { + GENERIC_HISTORY_RESOLVE.scope(true, next.run(request)).await + }, + )) +} + #[path = "snapshot_content.rs"] mod content; +#[path = "snapshot_raw_blob.rs"] +mod raw_blob; + +#[path = "snapshot_request.rs"] +mod request; + +#[cfg(test)] +#[path = "snapshot_request_tests.rs"] +mod request_tests; + /// Spec 14 §4: JSON request bytes hard limit. pub(crate) const JSON_REQUEST_LIMIT: usize = 131_072; +/// The media type is part of the MST/2 TreeFrame wire contract (spec 06 +/// §1). Keep it in one place so every frame-producing endpoint has the same +/// response representation. +pub(crate) const TREEFRAME_MEDIA_TYPE: &str = "application/vnd.mega.treeframe;version=2"; + +/// Build a TreeFrame response with the protocol identity headers. The request +/// digest covers the exact bytes that were parsed, including JSON whitespace +/// and key ordering, so callers must pass the original body. +pub(crate) fn treeframe_response( + snapshot_id: &str, + request_body: &[u8], + body: Vec, +) -> Result { + treeframe_response_body(snapshot_id, request_body, axum::body::Body::from(body)) +} + +fn treeframe_response_body( + snapshot_id: &str, + request_body: &[u8], + body: axum::body::Body, +) -> Result { + let request_digest: [u8; 32] = Sha256::digest(request_body).into(); + Response::builder() + .header("content-type", TREEFRAME_MEDIA_TYPE) + .header("x-mega-snapshot-id", snapshot_id) + .header( + "x-mega-request-digest", + format!("sha256:{}", hex_of(&request_digest)), + ) + .header("cache-control", "private, no-cache, no-transform") + .header("vary", "Authorization, Accept") + .body(body) + .map_err(|e| internal(format!("TreeFrame response build failed: {e}"))) +} + +fn guarded_treeframe_response( + state: &MonoApiServiceState, + context: &crate::ceres::snapshot::runtime::SnapshotContext, + snapshot_id: &str, + request_body: &[u8], + frames: Vec>, +) -> Result { + guarded_treeframe_response_with_budget(state, context, snapshot_id, request_body, frames, None) +} + +fn guarded_treeframe_response_with_budget( + state: &MonoApiServiceState, + context: &crate::ceres::snapshot::runtime::SnapshotContext, + snapshot_id: &str, + request_body: &[u8], + frames: Vec>, + memory: Option, +) -> Result { + // Field order also keeps admission credits until frame allocations drop + // on failures before ownership has moved into individual Bytes. + struct FrameAllocation { + frames: Vec>, + memory: Option, + } + let mut allocation = FrameAllocation { frames, memory }; + if let Some(lease) = &allocation.memory { + let allocated = allocation + .frames + .iter() + .try_fold(0usize, |total, frame| total.checked_add(frame.capacity())); + if allocated.is_none_or(|allocated| allocated > lease.bytes) { + return Err(internal("encoded frames exceed their memory reservation")); + } + } + let headers = REQUEST_HEADERS.try_with(Clone::clone).map_err(|_| { + SnapshotError::new( + SnapshotErrorCode::Unauthenticated, + "request authentication context missing", + ) + })?; + let memory = allocation.memory.take().map(std::sync::Arc::new); + let units = std::mem::take(&mut allocation.frames) + .into_iter() + .map(|bytes| match &memory { + Some(lease) => { + Bytes::from_owner(crate::ceres::snapshot::content_budget::BudgetedFrame { + bytes, + lease: lease.clone(), + }) + } + None => Bytes::from(bytes), + }) + .collect::>(); + let stream = futures::stream::unfold( + (units, state.clone(), context.clone(), headers), + |(mut units, state, context, headers)| async move { + let unit = units.pop_front()?; + if let Err(error) = revalidate_access(&state, &context, &headers).await { + units.clear(); + return Some((Err(error), (units, state, context, headers))); + } + Some((Ok(unit), (units, state, context, headers))) + }, + ); + treeframe_response_body( + snapshot_id, + request_body, + axum::body::Body::from_stream(stream), + ) +} + tokio::task_local! { /// Per-request id for the error envelope (spec 14 §5). Sourced from the /// global `TraceContext` so the envelope, the `X-Request-Id` response /// header and log spans all carry the same id. static REQUEST_ID: String; + static REQUEST_CONTEXT: Option; + static REQUEST_HEADERS: HeaderMap; +} + +#[cfg(test)] +tokio::task_local! { + static NATIVE_RESOLVE_BARRIERS: (std::sync::Arc, std::sync::Arc); + static NATIVE_HANDOFF_BARRIERS: (std::sync::Arc, std::sync::Arc); + static REJECT_NATIVE_OBSERVATION_SOURCE: bool; + static GENERIC_HISTORY_RESOLVE: bool; +} + +#[cfg(test)] +pub(crate) async fn with_rejected_native_observation_source( + future: F, +) -> F::Output { + REJECT_NATIVE_OBSERVATION_SOURCE.scope(true, future).await +} + +#[cfg(test)] +pub(crate) async fn with_native_resolve_barriers( + captured: std::sync::Arc, + release: std::sync::Arc, + future: F, +) -> F::Output { + NATIVE_RESOLVE_BARRIERS + .scope((captured, release), future) + .await +} + +#[cfg(test)] +pub(crate) async fn with_native_handoff_barriers( + prepared: std::sync::Arc, + release: std::sync::Arc, + future: F, +) -> F::Output { + NATIVE_HANDOFF_BARRIERS + .scope((prepared, release), future) + .await } /// The request id of the in-flight request, for error envelopes. @@ -93,32 +269,71 @@ const LEASE_HEADER: &str = "x-mega-snapshot-lease"; /// every endpoint except `capabilities` requires `Authorization: Bearer /// ` when the deployment configured one, and snapshot-bound endpoints /// must additionally present the lease they resolved -/// (`X-Mega-Snapshot-Lease`), validated against the in-memory lease table — +/// (`X-Mega-Snapshot-Lease`), validated against the session authority — /// knowing the snapshot id alone is not a capability. async fn snapshot_auth_middleware( State(state): State, - req: axum::extract::Request, + mut req: axum::extract::Request, next: axum::middleware::Next, ) -> Response { - // Same id the global trace layer echoes on responses and logs. + // Nested routes may be added after the server's trace layer. Reuse an + // existing context or establish one here, including rejection responses. let id = req .extensions() .get::() - .map(|c| c.trace_id.to_string()) - .unwrap_or_default(); - REQUEST_ID - .scope(id, async { - if let Some(res) = auth_error(&state, req.headers(), req.uri().path()) { - return res; - } - next.run(req).await + .map(|c| c.trace_id.clone()) + .unwrap_or_else(|| crate::server::trace_context::resolve_trace_id(req.headers())); + req.extensions_mut() + .insert(crate::server::trace_context::TraceContext { + trace_id: id.clone(), + }); + let mut response = REQUEST_ID + .scope(id.to_string(), async { + let headers = req.headers().clone(); + let path = req.uri().path().to_owned(); + let context = match authenticate_request(&state, &headers, &path).await { + Ok(context) => context, + Err(error) => return mst2_error_response(error), + }; + REQUEST_HEADERS + .scope( + headers.clone(), + REQUEST_CONTEXT.scope(context.clone(), async { + let response = next.run(req).await; + if response.status().is_success() + || response.status() == StatusCode::NOT_MODIFIED + { + match context { + Some(context) => { + if let Err(error) = + revalidate_access(&state, &context, &headers).await + { + return mst2_error_response(error); + } + } + None => { + if let Err(error) = + authenticate_request(&state, &headers, &path).await + { + return mst2_error_response(error); + } + } + } + } + response + }), + ) + .await }) - .await + .await; + if let Ok(value) = HeaderValue::from_str(&id) { + response.headers_mut().insert("x-request-id", value); + } + response } -/// Spec 14 §4 enforcement with the MST/2 error envelope (DefaultBodyLimit's -/// own rejection is plain-text). Content-Length is checked here; a lying -/// chunked body still trips DefaultBodyLimit inside the extractor. +/// Reject an oversized declared length before consuming any body. Actual +/// bytes and the overall read deadline are checked by `Mst2Bytes`. async fn reject_oversize_body( req: axum::extract::Request, next: axum::middleware::Next, @@ -142,12 +357,14 @@ async fn reject_oversize_body( /// (spec 14 §5 INVALID_REQUEST). Size is enforced by the router layers. #[allow(clippy::result_large_err)] pub(crate) fn parse_json_body(body: &Bytes) -> Result { - serde_json::from_slice(body).map_err(|e| { - mst2_error_response(SnapshotError::new( - SnapshotErrorCode::InvalidRequest, - format!("malformed request body: {e}"), - )) - }) + request::validate_json_keys(body) + .and_then(|()| serde_json::from_slice(body)) + .map_err(|e| { + mst2_error_response(SnapshotError::new( + SnapshotErrorCode::InvalidRequest, + format!("malformed request body: {e}"), + )) + }) } /// Bearer-token check shared by the auth middleware; the parsing rule is the @@ -160,17 +377,18 @@ fn bearer_ok(headers: &HeaderMap, token: &str) -> bool { .is_some_and(|cred| cred == token) } -fn unauthenticated(message: &'static str) -> Response { - mst2_error_response(SnapshotError::new( - SnapshotErrorCode::Unauthenticated, - message, - )) +fn unauthenticated(message: &'static str) -> SnapshotError { + SnapshotError::new(SnapshotErrorCode::Unauthenticated, message) } /// Auth decision for one request path (see [`snapshot_auth_middleware`]). /// `capabilities` stays open; lease routes need only the bearer; a /// snapshot-bound route (`/snapshots/sha256:…/…`) also needs its lease. -fn auth_error(state: &MonoApiServiceState, headers: &HeaderMap, path: &str) -> Option { +async fn authenticate_request( + state: &MonoApiServiceState, + headers: &HeaderMap, + path: &str, +) -> Result, SnapshotError> { let config = state.storage.config(); let token = config.mst2.auth_token.as_deref(); let path = path.split('?').next().unwrap_or(path); @@ -178,32 +396,104 @@ fn auth_error(state: &MonoApiServiceState, headers: &HeaderMap, path: &str) -> O // middleware runs — accept both the stripped and full forms. let rest = path .strip_prefix("/api/v2/snapshots/") - .or_else(|| path.strip_prefix("/snapshots/"))?; + .or_else(|| path.strip_prefix("/snapshots/")); + let Some(rest) = rest else { + return Ok(None); + }; if rest == "capabilities" { - return None; + return Ok(None); } if let Some(token) = token && !bearer_ok(headers, token) { - return Some(unauthenticated("missing or invalid bearer credentials")); + return Err(unauthenticated("missing or invalid bearer credentials")); } let mut segments = rest.split('/'); match segments.next() { // Lease management names the lease in the path, not a snapshot. - Some("resolve") | Some("leases") | Some("capabilities") | None => None, + Some("resolve") | Some("leases") | Some("capabilities") | None => Ok(None), Some(snapshot_id) => { let lease = headers.get(LEASE_HEADER).and_then(|v| v.to_str().ok()); match lease { - None => Some(unauthenticated("missing X-Mega-Snapshot-Lease header")), - Some(lease) => runtime() - .validate_lease(snapshot_id, lease) - .err() - .map(mst2_error_response), + None => Err(unauthenticated("missing X-Mega-Snapshot-Lease header")), + Some(lease) => state + .storage + .snapshot_context(snapshot_id, lease) + .await + .map(Some), } } } } +fn request_context( + state: &MonoApiServiceState, + snapshot_id: &str, +) -> Result { + if let Ok(Some(context)) = REQUEST_CONTEXT.try_with(Clone::clone) + && context.built.snapshot_id == snapshot_id + { + return Ok(context); + } + if !state.storage.config().mst2.publication_enabled { + return runtime().context(snapshot_id); + } + Err(SnapshotError::new( + SnapshotErrorCode::Unauthenticated, + "validated snapshot session missing", + )) +} + +async fn revalidate_access( + state: &MonoApiServiceState, + context: &crate::ceres::snapshot::runtime::SnapshotContext, + headers: &HeaderMap, +) -> Result<(), SnapshotError> { + let config = state.storage.config(); + if !config.mst2.enabled { + return Err(SnapshotError::new( + SnapshotErrorCode::SnapshotNotReady, + "snapshot surface disabled", + )); + } + if let Some(token) = config.mst2.auth_token.as_deref() + && !bearer_ok(headers, token) + { + return Err(SnapshotError::new( + SnapshotErrorCode::Unauthenticated, + "bearer credentials changed", + )); + } + let current = state + .storage + .snapshot_context(&context.built.snapshot_id, &context.lease_id) + .await?; + if current.built.descriptor != context.built.descriptor + || current.commit_oid != context.commit_oid + || current.root_tree_oid != context.root_tree_oid + || current.authorization_epoch != context.authorization_epoch + { + return Err(SnapshotError::new( + SnapshotErrorCode::IntegrityError, + "fixed session changed during request", + )); + } + Ok(()) +} + +async fn revalidate_request( + state: &MonoApiServiceState, + context: &crate::ceres::snapshot::runtime::SnapshotContext, +) -> Result<(), SnapshotError> { + let headers = REQUEST_HEADERS.try_with(Clone::clone).map_err(|_| { + SnapshotError::new( + SnapshotErrorCode::Unauthenticated, + "request authentication context missing", + ) + })?; + revalidate_access(state, context, &headers).await +} + fn mst2_error_response(err: SnapshotError) -> Response { let status = StatusCode::from_u16(err.code.http_status()).unwrap_or(StatusCode::INTERNAL_SERVER_ERROR); @@ -216,7 +506,10 @@ fn mst2_error_response(err: SnapshotError) -> Response { "request_id": current_request_id(), "retryable": matches!( err.code, - SnapshotErrorCode::SnapshotNotReady | SnapshotErrorCode::Internal + SnapshotErrorCode::SnapshotNotReady + | SnapshotErrorCode::MetadataNotReady + | SnapshotErrorCode::TemporaryUnavailable + | SnapshotErrorCode::Internal ), } })), @@ -230,39 +523,61 @@ impl From for Response { } } +impl IntoResponse for SnapshotError { + fn into_response(self) -> Response { + mst2_error_response(self) + } +} + /// MegaError → snapshot error: storage/lease failures must surface as errors, /// never as absence (spec 00 SYS-04). fn internal(e: E) -> SnapshotError { SnapshotError::new(SnapshotErrorCode::Internal, e.to_string()) } -async fn capabilities() -> Json { - // Honest capability set for this build (spec 04 §3): only what this slice - // serves is true; everything else stays false until accepted. +async fn capabilities(State(state): State) -> Json { + capabilities_document(&state.storage.git_service.obj_storage) +} + +fn capabilities_document(backend: &MegaObjectStorageWrapper) -> Json { + // Full delivery must work from a cold source, including receipt admission. + // Existing warm receipts do not establish this backend-wide contract. + let full_delivery = backend.supports_chunk_map_retention(); Json(json!({ "protocol_versions": [2], "metadata_codecs": [1], - "frame_encodings": ["identity", "zstd"], + "frame_encodings": ["identity"], "features": { - "resolve": true, + "strict_publication": true, "directory": true, - "leases": true, "lookup": true, "metadata_pages": true, - "raw_blob": true, - "objects": true, - "chunk_reads": true, - "full_hydration": false, + "raw_blob": full_delivery, + "small_objects": true, + "chunk_reads": full_delivery, + "full_hydration": full_delivery, + "region_hints": false, "offline_export": false, - "bindings": false, - "immutable_release": false, }, "limits": { + "max_file_bytes": "8796093022208", + "max_path_bytes": 4096, + "max_path_components": 256, "metadata_page_bytes": 16384, "metadata_leaf_entries": 128, - "max_directory_page_limit": 256, + "max_json_request_bytes": 131072, + "max_json_response_bytes": 1048576, + "max_directory_entries": 256, + "max_request_items": 128, + "max_metadata_items": 64, + "small_object_bytes": 262144, + "small_batch_bytes": 8388608, + "object_frame_raw_bytes": 1048576, + "chunk_frame_raw_bytes": 1048652, + "frame_wire_bytes": 2097152, + "zstd_window_bytes": 8388608, "chunk_size": 1048576, - "small_object_bytes": 262144 + "chunk_batch_bytes": 134217728 } })) } @@ -319,7 +634,10 @@ fn ensure_enabled(state: &MonoApiServiceState) -> Result<(), SnapshotError> { // `lfs_router::enforce_lfs_access` (where the error is the rare arm and is // boxed), there is nothing to gain here, so the lint is allowed outright. #[allow(clippy::result_large_err)] -async fn resolve(state: State, body: Bytes) -> Result { +async fn resolve( + state: State, + Mst2Bytes(body): Mst2Bytes, +) -> Result { ensure_enabled(&state).map_err(mst2_error_response)?; let req: ResolveRequest = parse_json_body(&body)?; // Unknown target kinds are client errors, never a silent fallback to @@ -349,21 +667,105 @@ async fn resolve(state: State, body: Bytes) -> Result Some(source), + Err(_) => { + if let Some(sink) = &state.storage.projection_observation_sink { + sink.reject_binding(); + } + tracing::warn!("native resolve observation source rejected"); + None + } + }; + selected_native_head = Some(head.clone()); + ( + head.root.commit, + head.root.tree, + head.token.sequence.to_string(), + head.token.epoch.to_string(), + ) + } else { + let main = state + .storage + .mono_storage() + .get_main_ref("/") + .await + .map_err(internal)? + .ok_or_else(|| { + mst2_error_response(SnapshotError::new( + SnapshotErrorCode::SnapshotNotReady, + "monorepo main ref missing", + )) + })?; + let sequence = runtime() + .publication_sequence(&main.ref_commit_hash) + .to_string(); + ( + main.ref_commit_hash, + main.ref_tree_hash, + sequence, + "1".to_owned(), + ) + }; + + #[cfg(test)] + if let Ok((captured, release)) = NATIVE_RESOLVE_BARRIERS.try_with(|value| value.clone()) { + captured.wait().await; + release.wait().await; + } let view = SnapshotView::from_commit(&commit_oid, &tree_oid); if let Some(want) = &req.target.view_id { let kind_matches = req.target.kind == "view"; @@ -385,47 +787,193 @@ async fn resolve(state: State, body: Bytes) -> Result = None; + let (metadata_root, projection_work, prepared_rooted) = if let Some((root, work, _)) = + prepared_generic.as_ref() + { + (*root, Some(work.clone()), None) + } else if selected_native_head.is_some() { + let repository = state .storage - .mono_storage() - .publication_sequence(&namespace) + .rooted_qualified_metadata_writer() .await .map_err(internal)?; - durable.to_string() + let prepared = + crate::ceres::snapshot::rooted_metadata_projection::prepare_rooted_native_metadata( + handler.as_ref(), + &root_tree, + &req.scope, + repository, + ) + .await + .map_err(mst2_error_response)?; + (prepared.plan.root, None, Some(prepared)) } else { - runtime().publication_sequence(&commit_oid).to_string() + let (page, work) = build_directory_page_with_work(handler.as_ref(), &root_tree, &req.scope) + .await + .map_err(mst2_error_response)?; + (page.page_id, Some(work), None) }; + let projection_elapsed = projection_started.elapsed(); + let built = build_descriptor(&config.mst2, &view, &req.scope, metadata_root) + .map_err(mst2_error_response)?; + let ctx = if let Some(head) = selected_native_head.as_ref() { + use crate::jupiter::storage::qualified_metadata_family::SnapshotMetadataFamily; + let family = state + .storage + .snapshot_metadata_family(&built.snapshot_id, false) + .await + .map_err(mst2_error_response)?; + if family == Some(SnapshotMetadataFamily::Generic) || prepared_generic.is_some() { + // A permanent SID route retains its original physical family. + let existing = state + .storage + .snapshot_sessions() + .await + .open(head, &built, None, req.lease_seconds) + .await + .map_err(mst2_error_response)?; + if let Some(existing) = existing { + existing + } else { + let (_, _, prepared) = prepared_generic.as_ref().ok_or_else(|| { + mst2_error_response(internal("existing generic route has no durable session")) + })?; + let sessions = state.storage.snapshot_sessions().await; + let receipt = sessions + .install(&built, prepared) + .await + .map_err(mst2_error_response)?; + #[cfg(test)] + if let Ok((prepared, release)) = NATIVE_HANDOFF_BARRIERS.try_with(Clone::clone) { + prepared.wait().await; + release.wait().await; + } + sessions + .open(head, &built, Some(&receipt), req.lease_seconds) + .await + .map_err(mst2_error_response)? + .ok_or_else(|| { + mst2_error_response(internal( + "generic history fixture handoff returned no context", + )) + })? + } + } else { + let repository = state + .storage + .rooted_qualified_metadata_writer() + .await + .map_err(internal)?; + if let Some(context) = repository + .open_session(head, &built, None, req.lease_seconds) + .await + .map_err(mst2_error_response)? + { + context + } else { + let prepared = prepared_rooted.as_ref().ok_or_else(|| { + mst2_error_response(internal("rooted resolve has no source projection")) + })?; + let receipt = repository + .install(&built, prepared) + .await + .map_err(crate::jupiter::storage::native_snapshot_session::install_error) + .map_err(mst2_error_response)?; + #[cfg(test)] + if let Ok((prepared, release)) = + NATIVE_HANDOFF_BARRIERS.try_with(|value| value.clone()) + { + prepared.wait().await; + release.wait().await; + } + repository + .open_session(head, &built, Some(&receipt), req.lease_seconds) + .await + .map_err(mst2_error_response)? + .ok_or_else(|| { + mst2_error_response(internal("durable session handoff returned no context")) + })? + } + } + } else { + runtime() + .insert_context(built.clone(), &commit_oid, &tree_oid, req.lease_seconds) + .map_err(mst2_error_response)? + }; + + if let Some(source) = native_source { + let request_id = current_request_id(); + let resolved = ResolvedProjection { + descriptor: &ctx.built.descriptor, + snapshot_id: &ctx.built.snapshot_id, + metadata_root: &ctx.built.metadata_root, + context_commit: &ctx.commit_oid, + context_root_tree: &ctx.root_tree_oid, + fixed_root_tree: root_tree.id, + requested_scope: &req.scope, + request_id: &request_id, + }; + let observation = if let Some(prepared) = prepared_rooted { + source.observe_rooted(resolved, prepared.work, projection_elapsed) + } else { + source.observe( + resolved, + projection_work.unwrap_or_default(), + projection_elapsed, + ) + }; + match observation { + Ok(observation) => { + if let Some(sink) = &state.storage.projection_observation_sink { + let _ = sink.enqueue(&observation); + } + observation.emit(); + } + Err(_) => { + if let Some(sink) = &state.storage.projection_observation_sink { + sink.reject_binding(); + } + tracing::warn!("native resolve observation context rejected"); + } + } + } + + revalidate_request(&state, &ctx) + .await + .map_err(mst2_error_response)?; let body = json!({ "descriptor": descriptor_json(&built), - "publication_sequence": seq.to_string(), - "writer_epoch": "1", + "publication_sequence": sequence, + "writer_epoch": writer_epoch, "lease_id": ctx.lease_id, "lease_expires_at": crate::ceres::snapshot::runtime::rfc3339(ctx.lease_expires_at_unix), "authorization_epoch": "1", @@ -459,9 +1007,7 @@ async fn descriptor_get( AxumPath(snapshot_id): AxumPath, ) -> Result { ensure_enabled(&state).map_err(mst2_error_response)?; - let ctx = runtime() - .context(&snapshot_id) - .map_err(mst2_error_response)?; + let ctx = request_context(&state, &snapshot_id).map_err(mst2_error_response)?; // Reading the descriptor back never touches latest (spec 04 §2). let body = json!({ "snapshot_id": snapshot_id, @@ -483,21 +1029,18 @@ struct RenewRequest { async fn lease_renew( state: State, AxumPath(lease_id): AxumPath, - body: Bytes, + Mst2Bytes(body): Mst2Bytes, ) -> Result { ensure_enabled(&state).map_err(mst2_error_response)?; let req: RenewRequest = if body.is_empty() { RenewRequest::default() } else { - serde_json::from_slice(&body).map_err(|e| { - mst2_error_response(SnapshotError::new( - SnapshotErrorCode::ScopeInvalid, - format!("malformed renew body: {e}"), - )) - })? + parse_json_body(&body)? }; - let renewed = runtime() - .renew_lease(&lease_id, req.lease_seconds.unwrap_or(600)) + let renewed = state + .storage + .snapshot_renew(&lease_id, req.lease_seconds.unwrap_or(600)) + .await .map_err(mst2_error_response)?; // Renewal never changes the version or authorization (spec 04 §4). let body = json!({ @@ -517,7 +1060,11 @@ async fn lease_release( ensure_enabled(&state).map_err(mst2_error_response)?; // Idempotent: releasing an unknown/already-released lease still succeeds // (spec 04 §2). This never deletes Git content. - let released = runtime().release_lease(&lease_id); + let released = state + .storage + .snapshot_release(&lease_id) + .await + .map_err(mst2_error_response)?; Ok(Json(json!({ "lease_id": lease_id, "released": released })).into_response()) } @@ -533,6 +1080,9 @@ struct DirectoryQuery { ancestors: Option, } +#[path = "snapshot_rooted_metadata.rs"] +mod rooted_metadata; + fn default_limit() -> u32 { 128 } @@ -548,9 +1098,7 @@ async fn directory( Query(q): Query, ) -> Result { ensure_enabled(&state).map_err(mst2_error_response)?; - let ctx = runtime() - .context(&snapshot_id) - .map_err(mst2_error_response)?; + let ctx = request_context(&state, &snapshot_id).map_err(mst2_error_response)?; validate_scope_relative_path(&q.path).map_err(mst2_error_response)?; if !(1..=256).contains(&q.limit) { return Err(mst2_error_response(SnapshotError::new( @@ -558,6 +1106,18 @@ async fn directory( "limit must be 1..256", ))); } + if state.storage.config().mst2.publication_enabled + && state + .storage + .snapshot_metadata_family(&ctx.lease_id, true) + .await + .map_err(mst2_error_response)? + == Some( + crate::jupiter::storage::qualified_metadata_family::SnapshotMetadataFamily::Rooted, + ) + { + return rooted_metadata::directory_response(&state, &ctx, &snapshot_id, &q).await; + } let handler = state .api_handler(std::path::Path::new("/")) @@ -740,98 +1300,6 @@ async fn directory( Ok(resp) } -#[derive(Deserialize, Debug)] -struct BlobQuery { - path: String, - #[serde(default)] - expected_digest: Option, -} - -#[allow(clippy::result_large_err, clippy::too_many_lines)] -async fn blob( - state: State, - AxumPath(snapshot_id): AxumPath, - Query(q): Query, - headers: HeaderMap, -) -> Result { - ensure_enabled(&state).map_err(mst2_error_response)?; - let ctx = runtime() - .context(&snapshot_id) - .map_err(mst2_error_response)?; - validate_scope_relative_path(&q.path).map_err(mst2_error_response)?; - if headers.contains_key("range") { - // Spec 04 section 9: raw blob has no Range semantics this profile. - return Err(mst2_error_response(SnapshotError::new( - SnapshotErrorCode::RangeNotSupported, - "raw blob reads are whole-file; use chunks for ranges", - ))); - } - - let handler = state - .api_handler(std::path::Path::new("/")) - .await - .map_err(internal)?; - let root_tree = handler - .get_tree_by_hash(&ctx.root_tree_oid) - .await - .map_err(internal)?; - let abs_path = abs_view_path(&ctx.built.descriptor.scope, &q.path); - - match resolve_abs(handler.as_ref(), &root_tree, &abs_path) - .await - .map_err(mst2_error_response)? - { - WalkOutcome::FoundFile { - fs_kind, - raw, - digest, - .. - } => { - if let Some(expected) = &q.expected_digest - && expected != &format!("sha256:{}", hex_of(&digest)) - { - return Err(mst2_error_response(SnapshotError::new( - SnapshotErrorCode::DigestMismatch, - "content does not match expected_digest", - ))); - } - let fs_kind_str = match fs_kind { - crate::ceres::snapshot::resolver::FsKind::Regular => "regular", - crate::ceres::snapshot::resolver::FsKind::Executable => "executable", - crate::ceres::snapshot::resolver::FsKind::Symlink => "symlink", - crate::ceres::snapshot::resolver::FsKind::Directory => "directory", - }; - Response::builder() - .header("etag", format!("\"sha256:{}\"", hex_of(&digest))) - .header("cache-control", "private, no-cache, no-transform") - .header("x-mega-fs-kind", fs_kind_str) - .body(axum::body::Body::from(Bytes::from(raw))) - .map_err(|e| { - mst2_error_response(SnapshotError::new( - SnapshotErrorCode::Internal, - format!("body build failed: {e}"), - )) - }) - } - WalkOutcome::FoundDir => Err(mst2_error_response(SnapshotError::new( - SnapshotErrorCode::NotDirectory, - "path is a directory", - ))), - WalkOutcome::Absent => Err(mst2_error_response(SnapshotError::new( - SnapshotErrorCode::PathNotFound, - "path absent in the fixed view", - ))), - WalkOutcome::NotDirectory { symlink } => Err(mst2_error_response(SnapshotError::new( - if symlink { - SnapshotErrorCode::SymlinkTraversal - } else { - SnapshotErrorCode::NotDirectory - }, - "intermediate component is not a directory", - ))), - } -} - /// Scope-relative request path -> absolute view path (spec 04 section 1). fn abs_view_path(scope: &str, path: &str) -> String { if path == "/" { @@ -854,7 +1322,7 @@ async fn lookup( state: State, AxumPath(snapshot_id): AxumPath, _headers: HeaderMap, - body: Bytes, + Mst2Bytes(body): Mst2Bytes, ) -> Result { ensure_enabled(&state).map_err(mst2_error_response)?; let req: LookupRequest = parse_json_body(&body)?; @@ -865,13 +1333,24 @@ async fn lookup( ))); } + let ctx = request_context(&state, &snapshot_id).map_err(mst2_error_response)?; + if state.storage.config().mst2.publication_enabled + && state + .storage + .snapshot_metadata_family(&ctx.lease_id, true) + .await + .map_err(mst2_error_response)? + == Some( + crate::jupiter::storage::qualified_metadata_family::SnapshotMetadataFamily::Rooted, + ) + { + return rooted_metadata::lookup_response(&state, &ctx, &snapshot_id, &req).await; + } + let handler = state .api_handler(std::path::Path::new("/")) .await .map_err(internal)?; - let ctx = runtime() - .context(&snapshot_id) - .map_err(mst2_error_response)?; let root_tree = handler .get_tree_by_hash(&ctx.root_tree_oid) .await @@ -883,11 +1362,11 @@ async fn lookup( validate_scope_relative_path(path).map_err(mst2_error_response)?; let abs_path = abs_view_path(&ctx.built.descriptor.scope, path); let mut entry = json!({"path": path}); - match resolve_abs(handler.as_ref(), &root_tree, &abs_path) + match resolve_abs_metadata(handler.as_ref(), &root_tree, &abs_path) .await .map_err(mst2_error_response)? { - WalkOutcome::FoundDir => { + MetadataWalkOutcome::FoundDir => { let built = build_directory_page(handler.as_ref(), &root_tree, &abs_path) .await .map_err(mst2_error_response)?; @@ -904,25 +1383,23 @@ async fn lookup( entry["node"] = node; deepest_dirs.push(abs_path); } - WalkOutcome::FoundFile { - fs_kind, - size, - digest, - .. - } => { + MetadataWalkOutcome::FoundFile { fs_kind, oid } => { + let fact = + content::verified_file_metadata(handler.as_ref(), fs_kind, oid, path, None) + .await?; entry["status"] = json!("found"); let mut node = json!({"fs_kind": fs_kind.as_str()}); if let Some(name) = path.rsplit('/').next() { node["name"] = json!(name); } - node["size"] = json!(size.to_string()); - node["content_digest"] = json!(format!("sha256:{}", hex_of(&digest))); + node["size"] = json!(fact.size.to_string()); + node["content_digest"] = json!(format!("sha256:{}", hex_of(&fact.digest))); entry["node"] = node; } - WalkOutcome::Absent => { + MetadataWalkOutcome::Absent => { entry["status"] = json!("absent"); } - WalkOutcome::NotDirectory { symlink } => { + MetadataWalkOutcome::NotDirectory { symlink } => { entry["status"] = if symlink { json!("symlink_traversal") } else { @@ -1011,7 +1488,7 @@ async fn metadata_pages( state: State, AxumPath(snapshot_id): AxumPath, _headers: HeaderMap, - body: Bytes, + Mst2Bytes(body): Mst2Bytes, ) -> Result { ensure_enabled(&state).map_err(mst2_error_response)?; let req: MetadataPagesRequest = parse_json_body(&body)?; @@ -1024,66 +1501,90 @@ async fn metadata_pages( for item in &req.items { validate_scope_relative_path(&item.directory_path).map_err(mst2_error_response)?; } - let ctx = runtime() - .context(&snapshot_id) - .map_err(mst2_error_response)?; + let ctx = request_context(&state, &snapshot_id).map_err(mst2_error_response)?; - let handler = state - .api_handler(std::path::Path::new("/")) - .await - .map_err(internal)?; - let root_tree = handler - .get_tree_by_hash(&ctx.root_tree_oid) - .await - .map_err(internal)?; - let scope = &ctx.built.descriptor.scope; - - // Unique pages across all items, in first-seen order. - let mut unique: Vec<([u8; 32], Vec)> = Vec::new(); - let mut seen: Vec<[u8; 32]> = Vec::new(); - let mut logical_bytes: u64 = 0; - for item in &req.items { - let abs_path = abs_view_path(scope, &item.directory_path); - let built = build_directory_page(handler.as_ref(), &root_tree, &abs_path) + let unique = if state.storage.config().mst2.publication_enabled { + use crate::jupiter::storage::native_snapshot_session::MetadataRouteRequest; + let items: Vec<_> = req + .items + .iter() + .map(|item| MetadataRouteRequest { + directory_path: &item.directory_path, + route: &item.route, + expected_digest: item.expected_digest.as_deref(), + }) + .collect(); + let batch = state + .storage + .snapshot_metadata_routes(&ctx, &items) .await .map_err(mst2_error_response)?; - let pages = - mst2_codec::metapage::Page::pages_along_route(&built.codec_entries, &item.route) - .map_err(|e| match e { - // A label the fixed view does not have is proven absence. - mst2_codec::CodecError::BadOrdering(m) => SnapshotError::new( - SnapshotErrorCode::PathNotFound, - format!("{}: route does not resolve ({m})", item.directory_path), - ), - other => SnapshotError::new( - SnapshotErrorCode::Internal, - format!("{}: route walk failed ({other})", item.directory_path), + tracing::debug!( + page_queries = batch.work.page_queries, + pages_loaded = batch.work.pages_loaded, + payload_bytes = batch.work.payload_bytes, + walk_visits = batch.work.walk_visits, + edge_references_checked = batch.work.edge_references_checked, + "served persisted generic metadata routes" + ); + batch.pages + } else { + let handler = state + .api_handler(std::path::Path::new("/")) + .await + .map_err(internal)?; + let root_tree = handler + .get_tree_by_hash(&ctx.root_tree_oid) + .await + .map_err(internal)?; + let scope = &ctx.built.descriptor.scope; + let mut unique: Vec<([u8; 32], Vec)> = Vec::new(); + let mut seen: Vec<[u8; 32]> = Vec::new(); + for item in &req.items { + let abs_path = abs_view_path(scope, &item.directory_path); + let built = build_directory_page(handler.as_ref(), &root_tree, &abs_path) + .await + .map_err(mst2_error_response)?; + let pages = + mst2_codec::metapage::Page::pages_along_route(&built.codec_entries, &item.route) + .map_err(|e| match e { + // A label the fixed view does not have is proven absence. + mst2_codec::CodecError::BadOrdering(m) => SnapshotError::new( + SnapshotErrorCode::PathNotFound, + format!("{}: route does not resolve ({m})", item.directory_path), + ), + other => SnapshotError::new( + SnapshotErrorCode::Internal, + format!("{}: route walk failed ({other})", item.directory_path), + ), + })?; + let last = pages + .last() + .expect("pages_along_route returns at least the root page"); + let last_id = mst2_codec::metapage::page_id(last); + if let Some(expected) = &item.expected_digest + && expected != &format!("sha256:{}", hex_of(&last_id)) + { + return Err(mst2_error_response(SnapshotError::new( + SnapshotErrorCode::DigestMismatch, + format!( + "{}: route does not reach expected_digest", + item.directory_path ), - })?; - let last = pages - .last() - .expect("pages_along_route returns at least the root page"); - let last_id = mst2_codec::metapage::page_id(last); - if let Some(expected) = &item.expected_digest - && expected != &format!("sha256:{}", hex_of(&last_id)) - { - return Err(mst2_error_response(SnapshotError::new( - SnapshotErrorCode::DigestMismatch, - format!( - "{}: route does not reach expected_digest", - item.directory_path - ), - ))); - } - for page in pages { - let id = mst2_codec::metapage::page_id(&page); - if !seen.contains(&id) { - logical_bytes += page.len() as u64; - seen.push(id); - unique.push((id, page)); + ))); + } + for page in pages { + let id = mst2_codec::metapage::page_id(&page); + if !seen.contains(&id) { + seen.push(id); + unique.push((id, page)); + } } } - } + unique + }; + let page_count = unique.len(); + let logical_bytes = unique.iter().map(|(_, page)| page.len() as u64).sum(); // Frames hold at most 64 pages and at most 1 MiB of raw payload (spec 06), // so a wide route set becomes several META frames rather than one @@ -1097,18 +1598,18 @@ async fn metadata_pages( .unwrap_or(crate::ceres::snapshot::frame_stream::Encoding::Identity); use crate::ceres::snapshot::frame_stream::FrameStream; let mut stream = FrameStream::new(1, encoding); - let mut out: Vec = Vec::new(); + let mut out: Vec> = Vec::new(); let mut frame: Vec<([u8; 32], Vec)> = Vec::new(); let mut frame_raw: usize = 0; let mut flush = |frame: &mut Vec<([u8; 32], Vec)>, raw: &mut usize, - out: &mut Vec| + out: &mut Vec>| -> Result<(), SnapshotError> { if frame.is_empty() { return Ok(()); } let bytes = stream.meta(std::mem::take(frame))?; - out.extend_from_slice(&bytes); + out.push(bytes); *raw = 0; Ok(()) }; @@ -1130,15 +1631,82 @@ async fn metadata_pages( let request_body_sha256: [u8; 32] = sha2::Digest::finalize(hasher).into(); let end = stream.end( req.items.len() as u32, - u32::try_from(seen.len()).unwrap_or(u32::MAX), + u32::try_from(page_count).unwrap_or(u32::MAX), logical_bytes, request_body_sha256, ); - out.extend_from_slice(&end); + out.push(end); - Response::builder() - .header("content-type", "application/octet-stream") - .header("cache-control", "private, no-cache, no-transform") - .body(axum::body::Body::from(Bytes::from(out))) - .map_err(|e| mst2_error_response(internal(format!("body build failed: {e}")))) + guarded_treeframe_response(&state, &ctx, &snapshot_id, &body, out).map_err(mst2_error_response) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn capabilities_advertise_the_canonical_full_delivery_contract() { + let backend = crate::jupiter::storage::object_storage::mock_object_storage(); + let Json(value) = capabilities_document(&backend); + assert_eq!( + value, + json!({ + "protocol_versions": [2], + "metadata_codecs": [1], + "frame_encodings": ["identity"], + "features": { + "strict_publication": true, + "directory": true, + "lookup": true, + "metadata_pages": true, + "raw_blob": true, + "small_objects": true, + "chunk_reads": true, + "full_hydration": true, + "region_hints": false, + "offline_export": false, + }, + "limits": { + "max_file_bytes": "8796093022208", + "max_path_bytes": 4096, + "max_path_components": 256, + "metadata_page_bytes": 16384, + "metadata_leaf_entries": 128, + "max_json_request_bytes": 131072, + "max_json_response_bytes": 1048576, + "max_directory_entries": 256, + "max_request_items": 128, + "max_metadata_items": 64, + "small_object_bytes": 262144, + "small_batch_bytes": 8388608, + "object_frame_raw_bytes": 1048576, + "chunk_frame_raw_bytes": 1048652, + "frame_wire_bytes": 2097152, + "zstd_window_bytes": 8388608, + "chunk_size": 1048576, + "chunk_batch_bytes": 134217728 + } + }) + ); + } + + #[test] + fn treeframe_response_emits_protocol_identity_headers() { + let snapshot_id = "sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa"; + let request_body = br#"{"items":[], "encoding":"identity"}"#; + let response = treeframe_response(snapshot_id, request_body, vec![1, 2, 3]).unwrap(); + + assert_eq!(response.headers()["content-type"], TREEFRAME_MEDIA_TYPE); + assert_eq!(response.headers()["x-mega-snapshot-id"], snapshot_id); + let digest: [u8; 32] = Sha256::digest(request_body).into(); + assert_eq!( + response.headers()["x-mega-request-digest"], + format!("sha256:{}", hex_of(&digest)) + ); + assert_eq!( + response.headers()["cache-control"], + "private, no-cache, no-transform" + ); + assert_eq!(response.headers()["vary"], "Authorization, Accept"); + } } diff --git a/src/api/router/snapshot_session_tests.rs b/src/api/router/snapshot_session_tests.rs new file mode 100644 index 00000000..e7f4f734 --- /dev/null +++ b/src/api/router/snapshot_session_tests.rs @@ -0,0 +1,1337 @@ +use sea_orm::{ConnectionTrait, DatabaseConnection, DbBackend, Statement, TransactionTrait}; +use tokio::sync::Barrier; + +use super::*; +use crate::{ + api::router::snapshot_router::{with_native_handoff_barriers, with_native_resolve_barriers}, + jupiter::storage::{ + mst2_retention::{GcClaim, PostgresRetentionRepository}, + push_queue_storage::PushQueueStorage, + }, +}; + +fn resolve_request(scope: &str) -> Request { + Request::builder() + .method("POST") + .uri("/api/v2/snapshots/resolve") + .header("authorization", format!("Bearer {TOKEN}")) + .header("content-type", "application/json") + .body(Body::from( + json!({"target":{"kind":"latest"},"scope":scope}).to_string(), + )) + .unwrap() +} + +async fn scalar(db: &C, sql: &str) -> i64 { + db.query_one_raw(Statement::from_string(DbBackend::Postgres, sql.to_owned())) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap() +} + +async fn root_metadata( + fixture: &Fixture, +) -> crate::ceres::snapshot::pages::PreparedNativeMetadataRetention { + let mono = fixture.state.storage.mono_storage(); + let head = mono + .read_native_publication_head( + fixture + .state + .storage + .config() + .mst2 + .instance_uuid + .as_deref() + .unwrap(), + ) + .await + .unwrap(); + let handler = fixture + .state + .api_handler(std::path::Path::new("/")) + .await + .unwrap(); + let tree = handler.get_tree_by_hash(&head.root.tree).await.unwrap(); + crate::ceres::snapshot::pages::prepare_native_metadata_retention( + handler.as_ref(), + &tree, + "/", + crate::ceres::snapshot::retention_dag::MetadataDagLimits::default(), + ) + .await + .unwrap() +} + +async fn stored_metadata_ids(db: &DatabaseConnection) -> std::collections::BTreeSet<[u8; 32]> { + db.query_all_raw(Statement::from_string( + DbBackend::Postgres, + "SELECT page_id FROM mst2_metadata_payload", + )) + .await + .unwrap() + .into_iter() + .map(|row| { + let id: Vec = row.try_get_by_index(0).unwrap(); + id.try_into().unwrap() + }) + .collect() +} + +async fn rebuilt(fixture: &Fixture) -> MonoApiServiceState { + let config = fixture.state.storage.config(); + let connection = crate::jupiter::storage::init::postgres_connection(&config.database) + .await + .unwrap(); + let assembly = crate::jupiter::storage::Storage::new_with_connection( + config, + Arc::new(connection), + fixture.state.storage.git_service.obj_storage.clone(), + ); + let storage = if fixture.generic_history { + crate::jupiter::storage::init::with_generic_history_bootstrap(assembly).await + } else { + assembly.await + } + .unwrap(); + MonoApiServiceState { + storage, + git_object_cache: Arc::new(GitObjectCache { + connection: fixture.state.git_object_cache.connection.clone(), + prefix: uuid::Uuid::new_v4().to_string(), + }), + ..fixture.state.clone() + } +} + +fn app(state: &MonoApiServiceState) -> Router { + Router::new().nest( + "/api/v2", + crate::api::router::snapshot_router::generic_history_routers(state.clone()) + .with_state(state.clone()), + ) +} + +async fn lease_control(fixture: &Fixture, lease: &str, method: &str, renew: bool) -> Value { + let suffix = if renew { "/renew" } else { "" }; + success_json( + fixture + .app + .clone() + .oneshot( + Request::builder() + .method(method) + .uri(format!("/api/v2/snapshots/leases/{lease}{suffix}")) + .header("authorization", format!("Bearer {TOKEN}")) + .header("content-type", "application/json") + .body(Body::empty()) + .unwrap(), + ) + .await + .unwrap(), + ) + .await +} + +async fn advance(fixture: &Fixture) { + let mono = fixture.state.storage.mono_storage(); + let old = mono.get_main_ref("/project").await.unwrap().unwrap(); + let old_oid = ObjectHash::from_hex_for_kind(HashKind::Sha1, &old.ref_commit_hash).unwrap(); + let oid = ObjectHash::from_hex_for_kind(HashKind::Sha1, &fixture.oid).unwrap(); + let next_tree = tree(vec![item(TreeItemMode::Blob, oid, "new-only")]); + let next = Commit::from_tree_id_with_kind( + HashKind::Sha1, + next_tree.id, + vec![old_oid], + "durable HTTP next view", + ) + .unwrap(); + mono.save_mega_trees(vec![next_tree], next.id, None) + .await + .unwrap(); + mono.save_mega_commits(vec![next.clone()], None) + .await + .unwrap(); + publish_native_push(&fixture.state.storage, "/project", old_oid, &next).await; +} + +#[tokio::test] +async fn mst2_durable_http_resolve_installs_complete_dag_and_warm_leases_share_it() { + let fixture = Fixture::new_generic_history_with_pg_config(true).await; + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + let pages = scalar(db, "SELECT count(*) FROM mst2_metadata_payload").await; + assert!( + pages >= 3, + "scope includes nested and empty directory pages" + ); + assert_eq!( + scalar(db, "SELECT count(*) FROM mst2_metadata_prepare_page").await, + pages + ); + assert_eq!( + scalar(db, "SELECT count(*) FROM mst2_snapshot_context").await, + 1 + ); + assert_eq!( + scalar( + db, + "SELECT count(*) FROM mst2_snapshot_lease WHERE state='ACTIVE'" + ) + .await, + 1 + ); + assert_eq!( + scalar( + db, + "SELECT count(*) FROM mst2_retention_root WHERE root_kind='prepare'" + ) + .await, + 0 + ); + assert_eq!( + scalar( + db, + "SELECT count(*) FROM mst2_retention_root WHERE root_kind='lease'" + ) + .await, + 1 + ); + assert_eq!( + scalar( + db, + "SELECT count(*) FROM mst2_retention_root WHERE root_kind='pin'" + ) + .await, + 1 + ); + let edges = scalar(db, "SELECT count(*) FROM mst2_retention_edge").await; + let refs = scalar( + db, + "SELECT sum(incoming_refs)::bigint FROM mst2_retention_node", + ) + .await; + assert_eq!(edges, refs); + fixture.counts.reset(); + let again = success_json( + fixture + .app + .clone() + .oneshot(resolve_request("/project")) + .await + .unwrap(), + ) + .await; + assert_eq!(again["descriptor"]["snapshot_id"], fixture.snapshot); + assert_ne!(again["lease_id"], fixture.lease); + fixture.counts.assert(0, 0); + assert_eq!( + scalar(db, "SELECT count(*) FROM mst2_metadata_payload").await, + pages + ); + assert_eq!( + scalar(db, "SELECT count(*) FROM mst2_metadata_prepare").await, + 1 + ); + assert_eq!( + scalar(db, "SELECT count(*) FROM mst2_snapshot_context").await, + 1 + ); + assert_eq!( + scalar( + db, + "SELECT count(*) FROM mst2_retention_root WHERE root_kind='lease'" + ) + .await, + 2 + ); + assert_eq!( + scalar( + db, + "SELECT sum(incoming_refs)::bigint FROM mst2_retention_node" + ) + .await, + refs + ); + + let state = rebuilt(&fixture).await; + assert!(state.storage.native_snapshot_sessions.get().is_none()); + assert!(!Arc::ptr_eq( + &state.storage.native_projection_cache, + &fixture.state.storage.native_projection_cache + )); + let replacement = app(&state); + let descriptor = success_json( + replacement + .clone() + .oneshot(fixture.request("GET", "descriptor", Body::empty())) + .await + .unwrap(), + ) + .await; + assert_eq!(descriptor["snapshot_id"], fixture.snapshot); + assert_eq!(descriptor["lease_id"], fixture.lease); + assert_eq!(descriptor["descriptor"], again["descriptor"]); + for (method, path, body) in [ + ("GET", "directory?path=/nested", Body::empty()), + ( + "POST", + "lookup", + Body::from(json!({"paths":["/nested/file"]}).to_string()), + ), + ( + "POST", + "metadata/pages", + Body::from(json!({"items":[{"directory_path":"/nested"}]}).to_string()), + ), + ("HEAD", "blob?path=/file", Body::empty()), + ] { + let response = replacement + .clone() + .oneshot(fixture.request(method, path, body)) + .await + .unwrap(); + assert_eq!(response.status(), 200); + to_bytes(response.into_body(), usize::MAX).await.unwrap(); + } + fixture.counts.assert(0, 0); + let response = replacement + .oneshot(fixture.request("GET", "blob?path=/file", Body::empty())) + .await + .unwrap(); + assert_eq!(response.status(), 200); + assert_eq!( + to_bytes(response.into_body(), usize::MAX) + .await + .unwrap() + .as_ref(), + fixture.raw.as_slice() + ); + fixture.counts.assert(2, 2 * fixture.raw.len()); + assert_eq!(fixture.counts.receipt_writes.load(Ordering::SeqCst), 1); +} + +#[tokio::test] +async fn mst2_durable_http_old_sid_and_original_lease_survive_real_publication_and_fresh_service() { + let fixture = Fixture::new_generic_history_with_pg_config(true).await; + let old = success_json(fixture.send("GET", "descriptor", Body::empty()).await).await; + advance(&fixture).await; + let next = success_json( + fixture + .app + .clone() + .oneshot(resolve_request("/project")) + .await + .unwrap(), + ) + .await; + assert_eq!(next["publication_sequence"], "2"); + assert_ne!(next["descriptor"]["snapshot_id"], fixture.snapshot); + let state = rebuilt(&fixture).await; + let replacement = app(&state); + let restored = success_json( + replacement + .clone() + .oneshot(fixture.request("GET", "descriptor", Body::empty())) + .await + .unwrap(), + ) + .await; + assert_eq!(restored, old); + let response = replacement + .clone() + .oneshot(fixture.request("GET", "blob?path=/file", Body::empty())) + .await + .unwrap(); + assert_eq!(response.status(), 200); + assert_eq!( + to_bytes(response.into_body(), usize::MAX) + .await + .unwrap() + .as_ref(), + fixture.raw.as_slice() + ); + error( + replacement + .oneshot(fixture.request("GET", "chunk-map?path=/new-only", Body::empty())) + .await + .unwrap(), + 404, + "PATH_NOT_FOUND", + false, + ) + .await; + let new_request = Request::builder() + .uri(format!( + "/api/v2/snapshots/{}/blob?path=/new-only", + next["descriptor"]["snapshot_id"].as_str().unwrap() + )) + .header("authorization", format!("Bearer {TOKEN}")) + .header("x-mega-snapshot-lease", next["lease_id"].as_str().unwrap()) + .body(Body::empty()) + .unwrap(); + let response = app(&state).oneshot(new_request).await.unwrap(); + assert_eq!(response.status(), 200); + assert_eq!( + to_bytes(response.into_body(), usize::MAX) + .await + .unwrap() + .as_ref(), + fixture.raw.as_slice() + ); +} + +#[tokio::test] +async fn mst2_durable_http_renew_release_expiry_and_last_lease_retire_protection() { + let fixture = Fixture::new_generic_history_with_pg_config(true).await; + let another = success_json( + fixture + .app + .clone() + .oneshot(resolve_request("/project")) + .await + .unwrap(), + ) + .await; + let other = another["lease_id"].as_str().unwrap(); + let renewed = lease_control(&fixture, &fixture.lease, "POST", true).await; + assert_eq!(renewed["snapshot_id"], fixture.snapshot); + let state = rebuilt(&fixture).await; + let restored = success_json( + app(&state) + .oneshot(fixture.request("GET", "descriptor", Body::empty())) + .await + .unwrap(), + ) + .await; + assert_eq!(restored["lease_expires_at"], renewed["lease_expires_at"]); + let first = lease_control(&fixture, &fixture.lease, "DELETE", false).await; + let second = lease_control(&fixture, &fixture.lease, "DELETE", false).await; + assert_eq!(first["released"], true); + assert_eq!(second["released"], false); + error( + app(&state) + .oneshot(fixture.request("GET", "descriptor", Body::empty())) + .await + .unwrap(), + 410, + "LEASE_EXPIRED", + false, + ) + .await; + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + assert_eq!( + scalar( + db, + "SELECT count(*) FROM mst2_retention_root WHERE root_kind='lease'" + ) + .await, + 1 + ); + assert_eq!( + scalar( + db, + "SELECT count(*) FROM mst2_retention_root WHERE root_kind='pin'" + ) + .await, + 1 + ); + let mut request = fixture.request("HEAD", "blob?path=/file", Body::empty()); + request + .headers_mut() + .insert("x-mega-snapshot-lease", other.parse().unwrap()); + assert_eq!(app(&state).oneshot(request).await.unwrap().status(), 200); + db.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "UPDATE mst2_snapshot_lease SET expires_at_unix=0 WHERE lease_id=$1", + [other.into()], + )) + .await + .unwrap(); + let renew = Request::builder() + .method("POST") + .uri(format!("/api/v2/snapshots/leases/{other}/renew")) + .header("authorization", format!("Bearer {TOKEN}")) + .body(Body::empty()) + .unwrap(); + error( + app(&state).oneshot(renew).await.unwrap(), + 410, + "LEASE_EXPIRED", + false, + ) + .await; + assert_eq!( + scalar( + db, + "SELECT count(*) FROM mst2_snapshot_lease WHERE state='EXPIRED'" + ) + .await, + 1 + ); + assert_eq!( + scalar(db, "SELECT count(*) FROM mst2_retention_root").await, + 0 + ); + let stored = + crate::callisto::mst2_snapshot_context::Entity::find_by_id(fixture.snapshot.clone()) + .one(db) + .await + .unwrap() + .unwrap(); + let node = format!("page:sha256:{}", hex::encode(stored.metadata_root)); + assert_eq!( + PostgresRetentionRepository::new(db.clone()) + .mark_deleting("last-lease-gc", &node) + .await + .unwrap(), + GcClaim::Marked + ); + assert_eq!( + scalar(db, "SELECT count(*) FROM mst2_metadata_payload").await, + scalar(db, "SELECT count(*) FROM mst2_metadata_prepare_page").await, + "release does not delete payload or Git bytes" + ); + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn mst2_durable_http_live_to_deleting_winner_rejects_paused_resolve_without_leak() { + let fixture = Fixture::new_generic_history_with_pg_config(true).await; + let captured = Arc::new(Barrier::new(2)); + let release = Arc::new(Barrier::new(2)); + let resolving = { + let app = fixture.app.clone(); + let captured = captured.clone(); + let release = release.clone(); + tokio::spawn(async move { + with_native_resolve_barriers( + captured, + release, + app.oneshot(resolve_request("/project")), + ) + .await + .unwrap() + }) + }; + captured.wait().await; + lease_control(&fixture, &fixture.lease, "DELETE", false).await; + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + let context = + crate::callisto::mst2_snapshot_context::Entity::find_by_id(fixture.snapshot.clone()) + .one(db) + .await + .unwrap() + .unwrap(); + let node = format!("page:sha256:{}", hex::encode(context.metadata_root)); + let txn = db.begin().await.unwrap(); + assert_eq!( + PostgresRetentionRepository::mark_deleting_in_txn(&txn, "http-gc-winner", &node) + .await + .unwrap(), + GcClaim::Marked + ); + let txn_id = scalar(db, "SELECT count(*) FROM mst2_snapshot_lease").await; + release.wait().await; + txn.commit().await.unwrap(); + error(resolving.await.unwrap(), 503, "OBJECT_UNAVAILABLE", false).await; + assert_eq!( + scalar(db, "SELECT count(*) FROM mst2_snapshot_lease").await, + txn_id + ); + assert_eq!( + scalar(db, "SELECT count(*) FROM mst2_retention_root").await, + 0 + ); +} + +#[tokio::test] +async fn mst2_durable_http_lease_winner_blocks_gc_until_the_last_release() { + let fixture = Fixture::new_generic_history_with_pg_config(true).await; + let another = success_json( + fixture + .app + .clone() + .oneshot(resolve_request("/project")) + .await + .unwrap(), + ) + .await; + let other = another["lease_id"].as_str().unwrap(); + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + let stored = + crate::callisto::mst2_snapshot_context::Entity::find_by_id(fixture.snapshot.clone()) + .one(db) + .await + .unwrap() + .unwrap(); + let node = format!("page:sha256:{}", hex::encode(stored.metadata_root)); + let retention = PostgresRetentionRepository::new(db.clone()); + assert_eq!( + retention + .mark_deleting("lease-winner-first", &node) + .await + .unwrap(), + GcClaim::Unavailable + ); + lease_control(&fixture, &fixture.lease, "DELETE", false).await; + assert_eq!( + retention + .mark_deleting("lease-winner-second", &node) + .await + .unwrap(), + GcClaim::Unavailable + ); + lease_control(&fixture, other, "DELETE", false).await; + assert_eq!( + scalar(db, "SELECT count(*) FROM mst2_retention_root").await, + 0 + ); + assert_eq!( + retention + .mark_deleting("lease-winner-final", &node) + .await + .unwrap(), + GcClaim::Marked + ); + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn mst2_durable_http_renew_waits_for_lock_before_checking_database_deadline() { + let fixture = Fixture::new_generic_history_with_pg_config(true).await; + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + let held = db.begin().await.unwrap(); + held.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT pg_advisory_xact_lock($1,hashtext(current_schema()))", + [crate::jupiter::storage::mst2_retention::RETENTION_LOCK_KEY.into()], + )) + .await + .unwrap(); + let renewing = { + let app = fixture.app.clone(); + let lease = fixture.lease.clone(); + tokio::spawn(async move { + app.oneshot( + Request::builder() + .method("POST") + .uri(format!("/api/v2/snapshots/leases/{lease}/renew")) + .header("authorization", format!("Bearer {TOKEN}")) + .body(Body::empty()) + .unwrap(), + ) + .await + .unwrap() + }) + }; + tokio::time::timeout(Duration::from_secs(30), async { + loop { + held.execute_raw(Statement::from_string( + DbBackend::Postgres, + "SELECT pg_stat_clear_snapshot()".to_owned(), + )) + .await + .unwrap(); + let blocked = scalar( + &held, + "SELECT count(*) FROM pg_locks l JOIN pg_stat_activity a ON a.pid=l.pid + WHERE l.locktype='advisory' AND NOT l.granted AND a.datname=current_database() + AND a.application_name=current_schema()", + ) + .await; + if blocked > 0 { + break; + } + tokio::task::yield_now().await; + } + }) + .await + .expect("HTTP renewal did not wait on the retention lock"); + held.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "UPDATE mst2_snapshot_lease SET expires_at_unix=0 WHERE lease_id=$1", + [fixture.lease.clone().into()], + )) + .await + .unwrap(); + held.commit().await.unwrap(); + error(renewing.await.unwrap(), 410, "LEASE_EXPIRED", false).await; + assert_eq!( + scalar( + db, + "SELECT count(*) FROM mst2_snapshot_lease WHERE state='EXPIRED'" + ) + .await, + 1 + ); + assert_eq!( + scalar(db, "SELECT count(*) FROM mst2_retention_root").await, + 0 + ); + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn mst2_durable_http_publication_advance_rejects_mixed_resolve_then_retries_current() { + let fixture = Fixture::new_generic_history_with_pg_config(true).await; + let captured = Arc::new(Barrier::new(2)); + let release = Arc::new(Barrier::new(2)); + let resolving = { + let app = fixture.app.clone(); + let captured = captured.clone(); + let release = release.clone(); + tokio::spawn(async move { + with_native_resolve_barriers( + captured, + release, + app.oneshot(resolve_request("/project")), + ) + .await + .unwrap() + }) + }; + captured.wait().await; + advance(&fixture).await; + release.wait().await; + error(resolving.await.unwrap(), 503, "SNAPSHOT_NOT_READY", true).await; + let mono = fixture.state.storage.mono_storage(); + assert_eq!( + scalar( + mono.get_connection(), + "SELECT count(*) FROM mst2_snapshot_lease" + ) + .await, + 1 + ); + let next = success_json( + fixture + .app + .clone() + .oneshot(resolve_request("/project")) + .await + .unwrap(), + ) + .await; + assert_eq!(next["publication_sequence"], "2"); + assert_ne!(next["descriptor"]["snapshot_id"], fixture.snapshot); + let state = rebuilt(&fixture).await; + assert_eq!( + app(&state) + .oneshot(fixture.request("HEAD", "blob?path=/file", Body::empty())) + .await + .unwrap() + .status(), + 200 + ); +} + +#[tokio::test] +async fn mst2_durable_http_install_fault_never_hands_off_or_returns_success() { + let fixture = Fixture::new_generic_history_with_pg_config(true).await; + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + let prepared = root_metadata(&fixture).await; + let existing = stored_metadata_ids(db).await; + let missing: Vec<_> = prepared + .dag() + .payloads() + .iter() + .filter(|page| !existing.contains(&page.id)) + .map(|page| page.id) + .collect(); + assert_eq!(missing, [prepared.dag().root()]); + db.execute_unprepared( + "CREATE FUNCTION reject_http_page() RETURNS trigger LANGUAGE plpgsql AS $$ + BEGIN RAISE EXCEPTION 'forced HTTP install interruption'; END $$; + CREATE TRIGGER reject_http_page BEFORE INSERT ON mst2_metadata_payload + FOR EACH ROW EXECUTE FUNCTION reject_http_page() ", + ) + .await + .unwrap(); + error( + fixture + .app + .clone() + .oneshot(resolve_request("/")) + .await + .unwrap(), + 503, + "TEMPORARY_UNAVAILABLE", + true, + ) + .await; + assert_eq!(stored_metadata_ids(db).await, existing); + assert_eq!( + scalar(db, "SELECT count(*) FROM mst2_snapshot_context").await, + 1 + ); + assert_eq!( + scalar(db, "SELECT count(*) FROM mst2_snapshot_lease").await, + 1 + ); + assert_eq!( + scalar( + db, + "SELECT count(*) FROM mst2_retention_root WHERE root_kind='prepare'" + ) + .await, + 0 + ); + assert_eq!( + scalar( + db, + "SELECT count(*) FROM mst2_metadata_prepare WHERE state='PREPARING'" + ) + .await, + 1 + ); + assert_eq!( + fixture + .send("HEAD", "blob?path=/file", Body::empty()) + .await + .status(), + 200 + ); + db.execute_unprepared( + "DROP TRIGGER reject_http_page ON mst2_metadata_payload; DROP FUNCTION reject_http_page() ", + ) + .await + .unwrap(); + let success = success_json( + fixture + .app + .clone() + .oneshot(resolve_request("/")) + .await + .unwrap(), + ) + .await; + assert_eq!(success["descriptor"]["scope"], "/"); + assert_eq!( + stored_metadata_ids(db).await, + prepared + .dag() + .payloads() + .iter() + .map(|page| page.id) + .collect() + ); + assert_eq!( + scalar( + db, + "SELECT count(*) FROM mst2_metadata_prepare WHERE state='PREPARING'" + ) + .await, + 0 + ); + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn mst2_durable_http_frame_and_raw_delivery_recheck_after_release() { + for raw in [false, true] { + let fixture = Fixture::new_generic_history_with_pg_config(true).await; + let response = if raw { + fixture.send("GET", "blob?path=/file", Body::empty()).await + } else { + fixture + .send( + "POST", + "metadata/pages", + Body::from(json!({"items":[{"directory_path":"/"}]}).to_string()), + ) + .await + }; + assert_eq!(response.status(), 200); + let mut stream = response.into_body().into_data_stream(); + let first = stream.next().await.unwrap().unwrap(); + assert!(!first.is_empty()); + if raw { + assert_eq!(first.len(), 1_048_576); + } + lease_control(&fixture, &fixture.lease, "DELETE", false).await; + assert!( + stream.next().await.unwrap().is_err(), + "revocation must suppress the next raw block or END frame" + ); + assert!(stream.next().await.is_none()); + // Cold raw first earns an immutable receipt, then opens the separate + // authenticated delivery stream. Revocation still suppresses its tail. + fixture.counts.assert( + if raw { 2 } else { 0 }, + if raw { + fixture.raw.len() + first.len() + } else { + 0 + }, + ); + } +} + +#[tokio::test] +async fn mst2_durable_http_current_state_wrong_lease_and_corrupt_source_reject_warm_reads() { + let fixture = Fixture::new_generic_history_with_pg_config(true).await; + let warm = success_json(fixture.send("GET", "descriptor", Body::empty()).await).await; + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + for (sid, lease) in [ + (fixture.snapshot.clone(), uuid::Uuid::new_v4().to_string()), + (format!("sha256:{}", "1".repeat(64)), fixture.lease.clone()), + ] { + let request = Request::builder() + .method("GET") + .uri(format!("/api/v2/snapshots/{sid}/descriptor")) + .header("authorization", format!("Bearer {TOKEN}")) + .header("x-mega-snapshot-lease", lease) + .header("if-none-match", format!("\"{}\"", fixture.digest_string())) + .body(Body::empty()) + .unwrap(); + error( + fixture.app.clone().oneshot(request).await.unwrap(), + 410, + "LEASE_EXPIRED", + false, + ) + .await; + } + db.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "UPDATE mst2_snapshot_context SET authorization_epoch=2 WHERE snapshot_id=$1", + [fixture.snapshot.clone().into()], + )) + .await + .unwrap(); + let request = fixture.request("GET", "descriptor", Body::empty()); + error( + fixture.app.clone().oneshot(request).await.unwrap(), + 403, + "SCOPE_FORBIDDEN", + false, + ) + .await; + db.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "UPDATE mst2_snapshot_context SET authorization_epoch=1 WHERE snapshot_id=$1", + [fixture.snapshot.clone().into()], + )) + .await + .unwrap(); + db.execute_unprepared("UPDATE mst2_native_publication SET writer_epoch=2") + .await + .unwrap(); + error( + fixture.send("GET", "descriptor", Body::empty()).await, + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + let state = rebuilt(&fixture).await; + error( + app(&state) + .oneshot(fixture.request("GET", "descriptor", Body::empty())) + .await + .unwrap(), + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + assert_eq!(warm["snapshot_id"], fixture.snapshot); + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn mst2_durable_http_large_dag_batches_handoff_without_scanning_pages_or_edges() { + let fixture = Fixture::new_generic_history_with_pg_config_and_directories(true, 80).await; + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + let projected = root_metadata(&fixture).await; + let existing = stored_metadata_ids(db).await; + assert!(existing.len() > 64); + let missing: Vec<_> = projected + .dag() + .payloads() + .iter() + .filter(|page| !existing.contains(&page.id)) + .map(|page| page.id) + .collect(); + assert_eq!(missing, [projected.dag().root()]); + let missing_batches = projected + .dag() + .payloads() + .chunks(64) + .filter(|batch| batch.iter().any(|page| !existing.contains(&page.id))) + .count() as i64; + assert_eq!(missing_batches, 1); + db.execute_unprepared( + "CREATE TABLE http_install_batch_count(singleton integer PRIMARY KEY,batches bigint NOT NULL,rows bigint NOT NULL); + INSERT INTO http_install_batch_count VALUES(1,0,0); + CREATE FUNCTION count_http_install_batch() RETURNS trigger LANGUAGE plpgsql AS $$ + BEGIN UPDATE http_install_batch_count SET batches=batches+1 WHERE singleton=1; RETURN NULL; END $$; + CREATE TRIGGER count_http_install_batch AFTER INSERT ON mst2_metadata_payload + FOR EACH STATEMENT EXECUTE FUNCTION count_http_install_batch(); + CREATE FUNCTION count_http_install_row() RETURNS trigger LANGUAGE plpgsql AS $$ + BEGIN UPDATE http_install_batch_count SET rows=rows+1 WHERE singleton=1; RETURN NULL; END $$; + CREATE TRIGGER count_http_install_row AFTER INSERT ON mst2_metadata_payload + FOR EACH ROW EXECUTE FUNCTION count_http_install_row()", + ).await.unwrap(); + let prepared = Arc::new(Barrier::new(2)); + let release = Arc::new(Barrier::new(2)); + let resolving = { + let app = fixture.app.clone(); + let prepared = prepared.clone(); + let release = release.clone(); + tokio::spawn(async move { + with_native_handoff_barriers(prepared, release, app.oneshot(resolve_request("/"))) + .await + .unwrap() + }) + }; + prepared.wait().await; + let members = scalar( + db, + "SELECT count(*) FROM mst2_metadata_prepare_page pp JOIN mst2_metadata_prepare p + ON p.prepare_id=pp.prepare_id WHERE p.scope='/'", + ) + .await; + assert!(members > 64); + assert_eq!(members as usize, projected.dag().payloads().len()); + assert_eq!( + scalar(db, "SELECT rows FROM http_install_batch_count").await, + missing.len() as i64 + ); + assert_eq!( + stored_metadata_ids(db).await, + projected + .dag() + .payloads() + .iter() + .map(|page| page.id) + .collect() + ); + assert_eq!( + scalar(db, "SELECT batches FROM http_install_batch_count").await, + missing_batches + ); + let writer = db.begin().await.unwrap(); + assert!( + PushQueueStorage::try_mono_write_lock(&writer) + .await + .unwrap(), + "full projection and installation completed without retaining the mono writer lock" + ); + writer.rollback().await.unwrap(); + let blocked_scans = db.begin().await.unwrap(); + blocked_scans.execute_unprepared( + "LOCK TABLE mst2_metadata_prepare_page,mst2_metadata_payload,mst2_retention_edge IN ACCESS EXCLUSIVE MODE", + ).await.unwrap(); + release.wait().await; + let response = tokio::time::timeout(Duration::from_secs(5), resolving) + .await + .expect("handoff must not read page membership, payload, or edge tables") + .unwrap(); + let resolved = success_json(response).await; + let writer = db.begin().await.unwrap(); + assert!( + PushQueueStorage::try_mono_write_lock(&writer) + .await + .unwrap() + ); + writer.rollback().await.unwrap(); + blocked_scans.rollback().await.unwrap(); + assert_eq!(resolved["descriptor"]["scope"], "/"); + assert_eq!( + scalar(db, "SELECT batches FROM http_install_batch_count").await, + missing_batches + ); + fixture.counts.assert(0, 0); + let warm = success_json( + fixture + .app + .clone() + .oneshot(resolve_request("/")) + .await + .unwrap(), + ) + .await; + assert_eq!( + warm["descriptor"]["snapshot_id"], + resolved["descriptor"]["snapshot_id"] + ); + assert_eq!( + scalar(db, "SELECT batches FROM http_install_batch_count").await, + missing_batches + ); + advance(&fixture).await; + let state = rebuilt(&fixture).await; + let directory = success_json( + app(&state) + .oneshot(fixture.request("GET", "directory?path=/wide-079", Body::empty())) + .await + .unwrap(), + ) + .await; + assert_eq!(directory["entries"][0]["name"], "file-079"); + assert_eq!( + app(&state) + .oneshot(fixture.request("HEAD", "blob?path=/wide-079/file-079", Body::empty())) + .await + .unwrap() + .status(), + 200 + ); + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn mst2_durable_http_handoff_rejects_changed_plan_or_receipt_summary() { + for mutation in [ + "canonical_plan=canonical_plan||decode('00','hex')", + "projection_revision=projection_revision+1", + "total_bytes=total_bytes+1", + ] { + let fixture = Fixture::new_generic_history_with_pg_config_and_directories(true, 80).await; + let prepared = Arc::new(Barrier::new(2)); + let release = Arc::new(Barrier::new(2)); + let resolving = { + let app = fixture.app.clone(); + let prepared = prepared.clone(); + let release = release.clone(); + tokio::spawn(async move { + with_native_handoff_barriers(prepared, release, app.oneshot(resolve_request("/"))) + .await + .unwrap() + }) + }; + prepared.wait().await; + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + corrupt_registered_handoff_plan_for_test(db, mutation).await; + release.wait().await; + error(resolving.await.unwrap(), 502, "INTEGRITY_ERROR", false).await; + assert_eq!( + scalar(db, "SELECT count(*) FROM mst2_snapshot_context").await, + 1 + ); + assert_eq!( + scalar(db, "SELECT count(*) FROM mst2_snapshot_lease").await, + 1 + ); + assert_eq!( + scalar( + db, + "SELECT count(*) FROM mst2_retention_root WHERE root_kind='lease'" + ) + .await, + 1 + ); + assert!( + scalar( + db, + "SELECT count(*) FROM mst2_retention_root WHERE root_kind='prepare'" + ) + .await + > 64 + ); + assert_eq!( + fixture + .send("HEAD", "blob?path=/file", Body::empty()) + .await + .status(), + 200 + ); + fixture.counts.assert(0, 0); + } +} + +async fn corrupt_registered_handoff_plan_for_test(db: &DatabaseConnection, mutation: &str) { + let guard_modes = handoff_trigger_modes_for_test(db).await; + assert!(guard_modes.iter().any(|(table, trigger, mode)| { + table == "mst2_metadata_prepare" + && trigger == "mst2_install_capability_prepare_guard" + && mode == "O" + })); + let prepares = db + .query_all_raw(Statement::from_string( + DbBackend::Postgres, + "SELECT prepare_id FROM mst2_metadata_prepare WHERE scope='/'", + )) + .await + .unwrap(); + assert_eq!(prepares.len(), 1); + let prepare_id: String = prepares[0].try_get("", "prepare_id").unwrap(); + let protected_update = || { + Statement::from_sql_and_values( + DbBackend::Postgres, + "UPDATE mst2_metadata_prepare SET total_bytes=total_bytes+1 WHERE prepare_id=$1", + [prepare_id.clone().into()], + ) + }; + assert!( + db.execute_raw(protected_update()) + .await + .unwrap_err() + .to_string() + .contains("registered metadata preparation identity is immutable") + ); + // The isolated fault injection restores this exact production guard in the + // same transaction. Acquire the statement barrier before ALTER's table lock. + let txn = db + .begin_with_config(Some(sea_orm::IsolationLevel::ReadCommitted), None) + .await + .unwrap(); + txn.execute_unprepared("SELECT pg_advisory_xact_lock(1296717362,hashtext(current_schema()))") + .await + .unwrap(); + txn.execute_unprepared( + "ALTER TABLE mst2_metadata_prepare DISABLE TRIGGER mst2_install_capability_prepare_guard", + ) + .await + .unwrap(); + let changed = txn + .execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + format!("UPDATE mst2_metadata_prepare SET {mutation} WHERE prepare_id=$1"), + [prepare_id.clone().into()], + )) + .await + .unwrap(); + assert_eq!(changed.rows_affected(), 1); + txn.execute_unprepared( + "ALTER TABLE mst2_metadata_prepare ENABLE TRIGGER mst2_install_capability_prepare_guard", + ) + .await + .unwrap(); + txn.commit().await.unwrap(); + assert_eq!(handoff_trigger_modes_for_test(db).await, guard_modes); + assert!( + db.execute_raw(protected_update()) + .await + .unwrap_err() + .to_string() + .contains("registered metadata preparation identity is immutable") + ); +} + +async fn handoff_trigger_modes_for_test(db: &DatabaseConnection) -> Vec<(String, String, String)> { + db.query_all_raw(Statement::from_string( + DbBackend::Postgres, + "SELECT c.relname,t.tgname,t.tgenabled::text AS mode FROM pg_trigger t + JOIN pg_class c ON c.oid=t.tgrelid JOIN pg_namespace n ON n.oid=c.relnamespace + WHERE n.nspname=current_schema() AND NOT t.tgisinternal ORDER BY c.relname,t.tgname", + )) + .await + .unwrap() + .into_iter() + .map(|row| { + ( + row.try_get("", "relname").unwrap(), + row.try_get("", "tgname").unwrap(), + row.try_get("", "mode").unwrap(), + ) + }) + .collect() +} + +async fn overwrite_primary_scope_for_test(db: &DatabaseConnection, storage_uuid: String) { + // Fault injection owns this isolated schema. Restore the production + // immutable trigger before commit; failed injection rolls back its DDL. + let txn = db.begin().await.unwrap(); + txn.execute_unprepared( + "ALTER TABLE mst2_metadata_storage_scope DISABLE TRIGGER mst2_metadata_scope_immutable", + ) + .await + .unwrap(); + let changed = txn + .execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "UPDATE mst2_metadata_storage_scope SET storage_uuid=$1 WHERE singleton=1", + [storage_uuid.into()], + )) + .await + .unwrap(); + assert_eq!(changed.rows_affected(), 1); + txn.execute_unprepared( + "ALTER TABLE mst2_metadata_storage_scope ENABLE TRIGGER mst2_metadata_scope_immutable", + ) + .await + .unwrap(); + txn.commit().await.unwrap(); +} + +#[tokio::test] +async fn mst2_durable_http_warm_reads_renew_and_frame_delivery_reject_primary_scope_drift() { + let fixture = Fixture::new_generic_history_with_pg_config(true).await; + success_json(fixture.send("GET", "descriptor", Body::empty()).await).await; + let response = fixture + .send( + "POST", + "metadata/pages", + Body::from(json!({"items":[{"directory_path":"/nested"}]}).to_string()), + ) + .await; + assert_eq!(response.status(), 200); + let mut body = response.into_body().into_data_stream(); + assert!(body.next().await.unwrap().is_ok()); + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + let original: String = db + .query_one_raw(Statement::from_string( + DbBackend::Postgres, + "SELECT storage_uuid FROM mst2_metadata_storage_scope WHERE singleton=1".to_owned(), + )) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap(); + overwrite_primary_scope_for_test(db, uuid::Uuid::new_v4().to_string()).await; + // Family selection rejects physical authority drift as INTEGRITY_ERROR. + error( + fixture.send("GET", "descriptor", Body::empty()).await, + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + error( + fixture + .app + .clone() + .oneshot( + Request::builder() + .method("POST") + .uri(format!("/api/v2/snapshots/leases/{}/renew", fixture.lease)) + .header("authorization", format!("Bearer {TOKEN}")) + .body(Body::empty()) + .unwrap(), + ) + .await + .unwrap(), + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + assert!( + body.next().await.unwrap().is_err(), + "cached validation cannot emit END after scope drift" + ); + assert!(body.next().await.is_none()); + overwrite_primary_scope_for_test(db, original).await; + success_json(fixture.send("GET", "descriptor", Body::empty()).await).await; + fixture.counts.assert(0, 0); +} + +#[path = "snapshot_lookup_metadata_tests.rs"] +mod metadata_lookup; + +#[path = "snapshot_persisted_metadata_tests.rs"] +mod persisted_metadata; diff --git a/src/api/router/snapshot_storage_route_fixture.rs b/src/api/router/snapshot_storage_route_fixture.rs new file mode 100644 index 00000000..71fb3204 --- /dev/null +++ b/src/api/router/snapshot_storage_route_fixture.rs @@ -0,0 +1,189 @@ +use sea_orm::{ConnectionTrait, DatabaseConnection, DbBackend, Statement, TransactionTrait}; + +async fn trigger_modes(db: &C) -> String { + db.query_one_raw(Statement::from_string( + DbBackend::Postgres, + "SELECT coalesce(string_agg(c.relname||':'||t.tgname||':'||t.tgenabled::text,',' + ORDER BY c.relname,t.tgname),'') AS modes FROM pg_trigger t + JOIN pg_class c ON c.oid=t.tgrelid JOIN pg_namespace n ON n.oid=c.relnamespace + WHERE n.nspname=current_schema() AND NOT t.tgisinternal", + )) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap() +} + +pub(super) async fn restore_pre_route_schema(db: &DatabaseConnection) { + let txn = db.begin().await.unwrap(); + txn.execute_unprepared("SELECT mst2_route_enter(current_schema())") + .await + .unwrap(); + let untouched = txn.query_one_raw(Statement::from_string(DbBackend::Postgres, + "SELECT coalesce(string_agg(c.relname||':'||t.tgname||':'||t.tgenabled::text,',' ORDER BY c.relname,t.tgname),'') + FROM pg_trigger t JOIN pg_class c ON c.oid=t.tgrelid + JOIN pg_namespace n ON n.oid=c.relnamespace JOIN pg_proc p ON p.oid=t.tgfoid + WHERE n.nspname=current_schema() AND NOT t.tgisinternal AND left(p.proname,11)<>'mst2_route_'", + )).await.unwrap().unwrap().try_get_by_index::(0).unwrap(); + // Remove only this additive layer in an isolated deployment-upgrade fixture. + // The original contexts, leases, plans, payloads and protection stay intact. + txn.execute_unprepared( + "DO $$ DECLARE r record; functions text; BEGIN + IF EXISTS(SELECT 1 FROM mst2_metadata_namespace WHERE graph_domain='qualified-v1') THEN + RAISE EXCEPTION 'route upgrade fixture cannot tear down a provisioned qualified family'; + END IF; + IF EXISTS(SELECT 1 FROM mst2_rooted_source_tree_revision) THEN + RAISE EXCEPTION 'route upgrade fixture cannot erase actual qualified source history'; + END IF; + DROP FUNCTION mst2_metadata_has_generic_overlap(bytea); + FOR r IN SELECT c.relname,t.tgname FROM pg_trigger t + JOIN pg_class c ON c.oid=t.tgrelid JOIN pg_namespace n ON n.oid=c.relnamespace + JOIN pg_proc p ON p.oid=t.tgfoid + WHERE n.nspname=current_schema() AND NOT t.tgisinternal AND left(p.proname,11)='mst2_route_' + LOOP EXECUTE format('DROP TRIGGER %I ON %I.%I',r.tgname,current_schema(),r.relname); END LOOP; + DROP TABLE mst2_lease_storage_route,mst2_generic_session_storage_binding,mst2_snapshot_storage_route,mst2_metadata_namespace,mst2_qualified_family_policy,mst2_rooted_source_tree_revision; + FOR r IN SELECT conname FROM pg_constraint + WHERE conrelid='mst2_snapshot_context'::regclass AND contype='u' + AND pg_get_constraintdef(oid)='UNIQUE (snapshot_id, prepare_id, metadata_root)' + LOOP EXECUTE format('ALTER TABLE mst2_snapshot_context DROP CONSTRAINT %I',r.conname); END LOOP; + SELECT string_agg(format('%I.%I(%s)',n.nspname,p.proname,pg_get_function_identity_arguments(p.oid)),',') + INTO functions FROM pg_proc p JOIN pg_namespace n ON n.oid=p.pronamespace + WHERE n.nspname=current_schema() AND left(p.proname,11)='mst2_route_'; + IF functions IS NULL THEN RAISE EXCEPTION 'route upgrade fixture has no route functions'; END IF; + EXECUTE 'DROP FUNCTION '||functions; + END $$; + DELETE FROM seaql_migrations WHERE version IN ( + 'm20261007_000600_add_mst2_storage_routes','m20261008_000200_add_mst2_rooted_qualified_family')", + ) + .await + .unwrap(); + assert_eq!(trigger_modes(&txn).await, untouched); + txn.commit().await.unwrap(); +} + +pub(super) async fn reject_half_lease_commit_for_test( + db: &DatabaseConnection, + lease_id: &str, + insert_sql: &str, +) { + let before = trigger_modes(db).await; + let txn = db.begin().await.unwrap(); + txn.execute_unprepared("SELECT mst2_route_enter(current_schema())") + .await + .unwrap(); + // Simulate an omitted derived write; all statement, identity and deferred + // completeness guards remain active. Failed commit rolls back the trigger + // change; pending deferred events forbid ALTER TABLE before commit. + txn.execute_unprepared( + "ALTER TABLE mst2_snapshot_lease DISABLE TRIGGER mst2_route_lease_insert", + ) + .await + .unwrap(); + assert_eq!( + txn.execute_unprepared(insert_sql) + .await + .unwrap() + .rows_affected(), + 1 + ); + let half = txn + .query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT (SELECT count(*) FROM mst2_snapshot_lease WHERE lease_id=$1)::bigint AS actual, + (SELECT count(*) FROM mst2_lease_storage_route WHERE lease_id=$1)::bigint AS routes", + [lease_id.into()], + )) + .await + .unwrap() + .unwrap(); + assert_eq!(half.try_get::("", "actual").unwrap(), 1); + assert_eq!(half.try_get::("", "routes").unwrap(), 0); + let paused = before.replacen( + "mst2_snapshot_lease:mst2_route_lease_insert:O", + "mst2_snapshot_lease:mst2_route_lease_insert:D", + 1, + ); + assert_ne!(paused, before); + assert_eq!(trigger_modes(&txn).await, paused); + let complete = txn + .query_one_raw(Statement::from_string( + DbBackend::Postgres, + "SELECT t.tgenabled::text AS mode,t.tgdeferrable AS deferrable,t.tginitdeferred AS deferred + FROM pg_catalog.pg_trigger t WHERE t.tgrelid='mst2_snapshot_lease'::regclass + AND t.tgname='mst2_route_complete' AND NOT t.tgisinternal", + )) + .await + .unwrap() + .unwrap(); + let mode: String = complete.try_get("", "mode").unwrap(); + assert!(matches!(mode.as_str(), "O" | "A")); + assert!(before.contains(&format!("mst2_snapshot_lease:mst2_route_complete:{mode}"))); + assert!(complete.try_get::("", "deferrable").unwrap()); + assert!(complete.try_get::("", "deferred").unwrap()); + let rejected = txn.commit().await.unwrap_err(); + assert!( + rejected + .to_string() + .contains("storage route lease committed without exact routing"), + "{rejected}" + ); + assert_eq!(trigger_modes(db).await, before); +} + +pub(super) async fn overwrite_lease_route_incarnation_for_test( + db: &DatabaseConnection, + lease_id: &str, +) { + let sql = "UPDATE mst2_lease_storage_route SET session_incarnation=$2::uuid WHERE lease_id=$1"; + let incarnation = uuid::Uuid::new_v4().to_string(); + let mutation = || { + Statement::from_sql_and_values( + DbBackend::Postgres, + sql, + [lease_id.into(), incarnation.clone().into()], + ) + }; + assert!( + db.execute_raw(mutation()) + .await + .unwrap_err() + .to_string() + .contains("immutable") + ); + let before = trigger_modes(db).await; + let txn = db.begin().await.unwrap(); + txn.execute_unprepared("SELECT mst2_route_enter(current_schema())") + .await + .unwrap(); + txn.execute_unprepared( + "ALTER TABLE mst2_lease_storage_route DISABLE TRIGGER mst2_route_immutable", + ) + .await + .unwrap(); + assert_eq!( + txn.execute_raw(mutation()).await.unwrap().rows_affected(), + 1 + ); + txn.execute_unprepared( + "ALTER TABLE mst2_lease_storage_route ENABLE TRIGGER mst2_route_immutable", + ) + .await + .unwrap(); + assert_eq!(trigger_modes(&txn).await, before); + txn.commit().await.unwrap(); + assert_eq!(trigger_modes(db).await, before); + let forbidden = uuid::Uuid::new_v4().to_string(); + assert_ne!(forbidden, incarnation); + assert!( + db.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + sql, + [lease_id.into(), forbidden.into()], + )) + .await + .unwrap_err() + .to_string() + .contains("immutable") + ); +} diff --git a/src/api/router/snapshot_storage_route_tests.rs b/src/api/router/snapshot_storage_route_tests.rs new file mode 100644 index 00000000..e5a0ad42 --- /dev/null +++ b/src/api/router/snapshot_storage_route_tests.rs @@ -0,0 +1,1201 @@ +use mst2_codec::{ + descriptor::ServingDescriptor, + metapage::{Entry, EntryKind, Page, page_id}, +}; +use sea_orm::{ + ConnectionTrait, DatabaseConnection, DatabaseTransaction, DbBackend, IsolationLevel, Statement, + TransactionTrait, +}; +use sea_orm_migration::MigratorTrait; + +use super::*; +use crate::{ + ceres::snapshot::{ + pages::PreparedNativeMetadataRetention, + retention::RetentionRoot, + retention_dag::{MetadataDagBuilder, MetadataDagLimits}, + }, + jupiter::{ + migration::Migrator, + storage::{ + mst2_retention::{PostgresRetentionRepository, RETENTION_LOCK_KEY}, + native_metadata_install::generations::qualified::PostgresQualifiedMetadataRepository, + push_queue_storage::MONO_WRITE_LOCK_KEY1, + }, + }, +}; + +const ROUTE_LOCK_KEY: i32 = 1_296_718_001; +const ROUTE_TABLES: [&str; 4] = [ + "mst2_metadata_namespace", + "mst2_snapshot_storage_route", + "mst2_generic_session_storage_binding", + "mst2_lease_storage_route", +]; + +fn statement(sql: &str, values: impl IntoIterator) -> Statement { + Statement::from_sql_and_values(DbBackend::Postgres, sql, values) +} + +async fn scalar(db: &C, sql: &str) -> i64 { + db.query_one_raw(Statement::from_string(DbBackend::Postgres, sql)) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap() +} + +async fn json_sql(db: &C, sql: &str) -> Value { + let value: String = db + .query_one_raw(Statement::from_string(DbBackend::Postgres, sql)) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap(); + serde_json::from_str(&value).unwrap() +} + +async fn routes(db: &C) -> Value { + json_sql( + db, + "SELECT jsonb_build_object( + 'namespace',(SELECT jsonb_agg(to_jsonb(n) ORDER BY namespace_uuid) FROM mst2_metadata_namespace n), + 'snapshot',(SELECT jsonb_agg(to_jsonb(r) ORDER BY snapshot_id) FROM mst2_snapshot_storage_route r), + 'binding',(SELECT jsonb_agg(to_jsonb(b) ORDER BY snapshot_id) FROM mst2_generic_session_storage_binding b), + 'lease',(SELECT jsonb_agg(to_jsonb(l) ORDER BY lease_id) FROM mst2_lease_storage_route l))::text", + ) + .await +} + +async fn sources(db: &C) -> Value { + json_sql( + db, + "SELECT jsonb_build_object( + 'context',(SELECT jsonb_agg(to_jsonb(s) ORDER BY snapshot_id) FROM mst2_snapshot_context s), + 'lease',(SELECT jsonb_agg(to_jsonb(l) ORDER BY lease_id) FROM mst2_snapshot_lease l), + 'prepare',(SELECT jsonb_agg(to_jsonb(p) ORDER BY prepare_id) FROM mst2_metadata_prepare p), + 'member',(SELECT jsonb_agg(to_jsonb(m) ORDER BY prepare_id,page_id) FROM mst2_metadata_prepare_page m), + 'payload',(SELECT jsonb_agg(to_jsonb(p) ORDER BY page_id) FROM mst2_metadata_payload p), + 'root',(SELECT jsonb_agg(to_jsonb(r) ORDER BY root_key,node_id) FROM mst2_retention_root r))::text", + ) + .await +} + +async fn roots(db: &C) -> Value { + json_sql( + db, + "SELECT coalesce(jsonb_agg(to_jsonb(r) ORDER BY root_key,node_id),'[]'::jsonb)::text FROM mst2_retention_root r", + ) + .await +} + +async fn domain_boundary_digests(db: &DatabaseConnection) -> Value { + let mut inventory = serde_json::Map::new(); + for (table, keys) in [ + ("mst2_snapshot_context", "r.snapshot_id"), + ("mst2_snapshot_lease", "r.lease_id"), + ("mst2_metadata_prepare", "r.prepare_id"), + ("mst2_metadata_prepare_page", "r.prepare_id,r.page_id"), + ("mst2_metadata_payload", "r.page_id"), + ("mst2_metadata_lifetime", "r.page_id,r.generation"), + ("mst2_metadata_current", "r.page_id"), + ("mst2_metadata_graph_node", "r.page_id,r.generation"), + ( + "mst2_metadata_graph_edge", + "r.parent_page,r.parent_generation,r.child_page,r.child_generation", + ), + ( + "mst2_metadata_graph_root", + "r.prepare_id,r.page_id,r.generation", + ), + ("mst2_metadata_gc_op", "r.operation_id"), + ("mst2_retention_node", "r.node_id"), + ("mst2_retention_edge", "r.parent_id,r.child_id"), + ("mst2_retention_root", "r.node_id,r.root_key"), + ("mst2_retention_gc_op", "r.operation_id"), + ("mst2_metadata_install_seal", "r.prepare_id"), + ("mst2_metadata_storage_scope", "r.singleton"), + ("mst2_metadata_namespace", "r.namespace_uuid"), + ("mst2_snapshot_storage_route", "r.snapshot_id"), + ( + "mst2_generic_session_storage_binding", + "r.session_incarnation", + ), + ("mst2_lease_storage_route", "r.lease_id"), + ] { + // Keep plans, binding bytes and payloads in PostgreSQL; only their + // keyed row digests leave the database for the rollback oracle. + inventory.insert( + table.to_owned(), + json_sql( + db, + &format!( + "SELECT coalesce(jsonb_agg(jsonb_build_object('key',jsonb_build_array({keys}), + 'sha256',encode(sha256(convert_to(to_jsonb(r)::text,'UTF8')),'hex')) + ORDER BY {keys}),'[]'::jsonb)::text FROM {table} r", + ), + ) + .await, + ); + } + Value::Object(inventory) +} + +fn resolve_request() -> Request { + Request::builder() + .method("POST") + .uri("/api/v2/snapshots/resolve") + .header("authorization", format!("Bearer {TOKEN}")) + .header("content-type", "application/json") + .body(Body::from( + json!({"target":{"kind":"latest"},"scope":"/project"}).to_string(), + )) + .unwrap() +} + +async fn lease_control(fixture: &Fixture, lease: &str, renew: bool) -> Response { + fixture + .app + .clone() + .oneshot( + Request::builder() + .method(if renew { "POST" } else { "DELETE" }) + .uri(format!( + "/api/v2/snapshots/leases/{lease}{}", + if renew { "/renew" } else { "" } + )) + .header("authorization", format!("Bearer {TOKEN}")) + .body(Body::empty()) + .unwrap(), + ) + .await + .unwrap() +} + +async fn rebuilt(fixture: &Fixture) -> Router { + let state = rebuilt_state(fixture).await; + Router::new().nest( + "/api/v2", + crate::api::router::snapshot_router::generic_history_routers(state.clone()) + .with_state(state), + ) +} + +async fn rebuilt_state(fixture: &Fixture) -> MonoApiServiceState { + let config = fixture.state.storage.config(); + let mut database = config.database.clone(); + database.max_connection = 1; + database.min_connection = 1; + let connection = crate::jupiter::storage::init::postgres_connection(&database) + .await + .unwrap(); + let storage = crate::jupiter::storage::init::with_generic_history_bootstrap( + crate::jupiter::storage::Storage::new_with_connection( + config, + Arc::new(connection), + fixture.state.storage.git_service.obj_storage.clone(), + ), + ) + .await + .unwrap(); + MonoApiServiceState { + storage, + git_object_cache: Arc::new(GitObjectCache { + connection: fixture.state.git_object_cache.connection.clone(), + prefix: uuid::Uuid::new_v4().to_string(), + }), + ..fixture.state.clone() + } +} + +async fn assert_exact_bindings(db: &C, leases: i64) { + assert_eq!( + scalar(db, "SELECT count(*) FROM mst2_metadata_namespace").await, + 1 + ); + assert_eq!( + scalar(db, "SELECT count(*) FROM mst2_snapshot_storage_route").await, + 1 + ); + assert_eq!( + scalar( + db, + "SELECT count(*) FROM mst2_generic_session_storage_binding" + ) + .await, + 1 + ); + assert_eq!( + scalar(db, "SELECT count(*) FROM mst2_lease_storage_route").await, + leases + ); + assert_eq!( + scalar( + db, + "SELECT count(*) FROM mst2_snapshot_context s + JOIN mst2_metadata_prepare p ON p.prepare_id=s.prepare_id + JOIN mst2_snapshot_storage_route r ON r.snapshot_id=s.snapshot_id + JOIN mst2_generic_session_storage_binding b ON b.snapshot_id=s.snapshot_id + JOIN mst2_metadata_namespace n ON n.namespace_uuid=r.namespace_uuid + WHERE r.canonical_descriptor=s.canonical_descriptor + AND r.instance_id=s.instance_id AND r.commit_oid=s.commit_oid + AND r.root_tree_oid=s.root_tree_oid AND r.metadata_root=s.metadata_root + AND b.namespace_uuid=r.namespace_uuid AND b.prepare_id=s.prepare_id + AND b.metadata_root=s.metadata_root AND p.metadata_root=s.metadata_root + AND r.source_profile=jsonb_build_object('source_domain',p.source_domain, + 'tagged_root_tree_oid',p.tagged_root_tree_oid,'scope',p.scope, + 'schema_version',p.schema_version,'metadata_codec',p.metadata_codec, + 'materialization_policy',p.materialization_policy,'fs_semantics',p.fs_semantics, + 'access_projection',p.access_projection,'verification_revision',p.verification_revision, + 'projection_revision',p.projection_revision)", + ) + .await, + 1, + ); + assert_eq!( + scalar( + db, + "SELECT count(*) FROM mst2_snapshot_lease l + JOIN mst2_lease_storage_route r ON r.lease_id=l.lease_id + JOIN mst2_generic_session_storage_binding b ON b.snapshot_id=l.snapshot_id + WHERE r.snapshot_id=l.snapshot_id AND r.namespace_uuid=b.namespace_uuid + AND r.session_incarnation=b.session_incarnation AND r.prepare_id=b.prepare_id + AND r.metadata_root=b.metadata_root AND r.authorization_epoch=l.authorization_epoch + AND r.publication_sequence=l.publication_sequence AND r.writer_epoch=l.writer_epoch + AND r.certificate_receipt_id=l.certificate_receipt_id", + ) + .await, + leases, + ); + let stored = routes(db).await; + uuid::Uuid::parse_str( + stored["binding"][0]["session_incarnation"] + .as_str() + .unwrap(), + ) + .unwrap(); + let namespace = &stored["namespace"][0]; + assert_eq!(namespace["graph_domain"], "generic-v1"); + assert_eq!(namespace["family_identity"], "v3-generic-session-1"); + assert_eq!(namespace["admission_state"], "G_ADMITTED_Q_CLOSED"); + assert_eq!(namespace["collector_state"], "CLOSED"); + assert_eq!(scalar(db, + "SELECT count(*) FROM mst2_metadata_namespace n JOIN mst2_metadata_storage_scope s ON s.singleton=1 + JOIN pg_namespace c ON c.nspname=current_schema() JOIN pg_database d ON d.datname=current_database() + WHERE n.core_schema=c.nspname AND n.core_schema_oid=c.oid + AND n.metadata_schema=c.nspname AND n.metadata_schema_oid=c.oid + AND n.database_name=d.datname AND n.database_oid=d.oid AND n.storage_uuid=s.storage_uuid + AND n.mono_lock_key2=hashtext(current_schema()) + AND n.server_address IS NOT DISTINCT FROM inet_server_addr()::text + AND n.server_port IS NOT DISTINCT FROM inet_server_port()", + ).await, 1); + assert_eq!( + scalar( + db, + "SELECT count(*) FROM information_schema.columns + WHERE table_schema=current_schema() AND table_name='mst2_snapshot_storage_route' + AND column_name IN ('prepare_id','generation','root_generation','session_incarnation')", + ) + .await, + 0, + "the permanent SID route must not freeze a physical incarnation", + ); + assert_eq!( + scalar( + db, + "SELECT count(*) FROM pg_constraint WHERE conrelid='mst2_lease_storage_route'::regclass + AND contype='f' AND confrelid='mst2_generic_session_storage_binding'::regclass", + ) + .await, + 0, + "future namespace-specific incarnations must not have a universal G ledger FK" + ); +} + +async fn rejected(db: &DatabaseConnection, sql: &str) { + let txn = db.begin().await.unwrap(); + match txn.execute_unprepared(sql).await { + Ok(_) => { + assert!(txn.commit().await.is_err(), "mutation committed: {sql}"); + } + Err(error) => { + assert!( + error.to_string().contains("storage route"), + "wrong rejection for {sql}: {error}" + ); + txn.rollback().await.unwrap(); + } + } +} + +#[tokio::test] +async fn mst2_generic_storage_routes_actual_resolve_warm_and_fresh_service_keep_exact_tuple() { + let fixture = Fixture::new_generic_history_with_pg_config(true).await; + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + assert_exact_bindings(db, 1).await; + let initial = routes(db).await; + assert_eq!(initial["snapshot"][0]["snapshot_id"], fixture.snapshot); + assert_eq!(initial["lease"][0]["lease_id"], fixture.lease); + let original = success_json(fixture.send("GET", "descriptor", Body::empty()).await).await; + let warm = success_json( + fixture + .app + .clone() + .oneshot(resolve_request()) + .await + .unwrap(), + ) + .await; + assert_eq!(warm["descriptor"]["snapshot_id"], fixture.snapshot); + assert_ne!(warm["lease_id"], fixture.lease); + assert_exact_bindings(db, 2).await; + let after = routes(db).await; + for key in ["namespace", "snapshot", "binding"] { + assert_eq!(after[key], initial[key], "warm resolve changed {key}"); + } + let cold = rebuilt(&fixture).await; + assert_eq!( + success_json( + cold.clone() + .oneshot(fixture.request("GET", "descriptor", Body::empty())) + .await + .unwrap() + ) + .await, + original, + ); + assert_eq!( + cold.oneshot(fixture.request("HEAD", "blob?path=/file", Body::empty())) + .await + .unwrap() + .status(), + 200, + ); + assert_eq!(routes(db).await, after); + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn mst2_generic_storage_routes_every_member_delete_and_truncate_are_immutable() { + let fixture = Fixture::new_generic_history_with_pg_config(true).await; + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + let original = routes(db).await; + let source = sources(db).await; + for table in ROUTE_TABLES { + let columns = db + .query_all_raw(statement( + "SELECT column_name,udt_name FROM information_schema.columns + WHERE table_schema=current_schema() AND table_name=$1 ORDER BY ordinal_position", + [table.into()], + )) + .await + .unwrap(); + assert!(!columns.is_empty(), "missing route relation: {table}"); + for column in columns { + let name: String = column.try_get("", "column_name").unwrap(); + let kind: String = column.try_get("", "udt_name").unwrap(); + let quoted = format!("\"{}\"", name.replace('"', "\"\"")); + let mutation = match kind.as_str() { + "uuid" => format!("'{}'::uuid", uuid::Uuid::new_v4()), + "text" | "varchar" | "bpchar" => format!("coalesce({quoted},'')||'-mutated'"), + "int2" | "int4" | "int8" => format!("coalesce({quoted},0)+1"), + "oid" => format!("({quoted}::bigint+1)::oid"), + "bytea" => format!("set_byte({quoted},0,(get_byte({quoted},0)+1)%256)"), + "jsonb" => format!("{quoted}||jsonb_build_object('_mutation',true)"), + "timestamptz" => format!("{quoted}+interval '1 second'"), + "bool" => format!("NOT {quoted}"), + other => panic!("uncovered immutable route member {table}.{name}: {other}"), + }; + rejected(db, &format!("UPDATE {table} SET {quoted}={mutation}")).await; + assert_eq!(routes(db).await, original, "{table}.{name}"); + } + rejected(db, &format!("DELETE FROM {table}")).await; + rejected(db, &format!("TRUNCATE {table} CASCADE")).await; + assert_eq!(routes(db).await, original, "{table}"); + } + for table in ["mst2_snapshot_context", "mst2_snapshot_lease"] { + rejected(db, &format!("DELETE FROM {table}")).await; + rejected(db, &format!("TRUNCATE {table} CASCADE")).await; + } + assert_eq!(sources(db).await, source); + fixture.counts.assert(0, 0); +} + +fn lease_insert(lease: &str) -> String { + format!( + "INSERT INTO mst2_snapshot_lease(lease_id,snapshot_id,authorization_epoch, + publication_sequence,writer_epoch,certificate_receipt_id,expires_at_unix,state) + SELECT '{lease}',snapshot_id,authorization_epoch,publication_sequence,writer_epoch, + certificate_receipt_id,floor(extract(epoch FROM clock_timestamp()))::bigint+3600,'ACTIVE' + FROM mst2_snapshot_context", + ) +} + +#[tokio::test] +async fn mst2_generic_storage_routes_raw_lease_derives_atomically_and_half_rows_roll_back() { + let fixture = Fixture::new_generic_history_with_pg_config(true).await; + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + let original = routes(db).await; + let source = sources(db).await; + let lease = uuid::Uuid::new_v4().to_string(); + let txn = db.begin().await.unwrap(); + assert_eq!( + txn.execute_unprepared(&lease_insert(&lease)) + .await + .unwrap() + .rows_affected(), + 1 + ); + assert_exact_bindings(&txn, 2).await; + txn.rollback().await.unwrap(); + assert_eq!(routes(db).await, original); + assert_eq!(sources(db).await, source); + super::storage_route_fixture::reject_half_lease_commit_for_test( + db, + &lease, + &lease_insert(&lease), + ) + .await; + assert_eq!(routes(db).await, original); + assert_eq!(sources(db).await, source); + for table in [ + "mst2_snapshot_storage_route", + "mst2_generic_session_storage_binding", + ] { + assert_eq!(routes(db).await, original); + rejected( + db, + &format!( + "INSERT INTO {table} SELECT (jsonb_populate_record(NULL::{table}, + to_jsonb(r)||jsonb_build_object('snapshot_id','sha256:{}'))).* FROM {table} r", + hex::encode([7; 32]), + ), + ) + .await; + } + rejected(db, &format!( + "INSERT INTO mst2_lease_storage_route + SELECT (jsonb_populate_record(NULL::mst2_lease_storage_route, + to_jsonb(r)||jsonb_build_object('lease_id','{lease}'))).* FROM mst2_lease_storage_route r", + )).await; + let txn = db.begin().await.unwrap(); + let error = txn.execute_unprepared( + "INSERT INTO mst2_metadata_namespace SELECT (jsonb_populate_record(NULL::mst2_metadata_namespace, + to_jsonb(n)||jsonb_build_object('namespace_uuid','00000000-0000-4000-8000-000000000001', + 'graph_domain','qualified-v1'))).* FROM mst2_metadata_namespace n", + ).await.expect_err("unadmitted qualified namespace must be rejected before commit"); + assert!( + error.to_string().contains( + "qualified namespace registration has no exact admitted rooted family scope and lock set" + ), + "wrong qualified namespace rejection: {error}" + ); + txn.rollback().await.unwrap(); + assert_eq!(routes(db).await, original); + assert_eq!(sources(db).await, source); +} + +#[tokio::test] +async fn mst2_generic_storage_routes_renew_terminal_and_unknown_release_preserve_route_and_roots() { + let fixture = Fixture::new_generic_history_with_pg_config(true).await; + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + let original = routes(db).await; + let unknown = uuid::Uuid::new_v4().to_string(); + let root: Vec = db + .query_one_raw(Statement::from_string( + DbBackend::Postgres, + "SELECT metadata_root FROM mst2_snapshot_context", + )) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap(); + let txn = db.begin().await.unwrap(); + PostgresRetentionRepository::acquire_existing_roots_in_txn( + &txn, + &format!("page:sha256:{}", hex::encode(root)), + &[RetentionRoot::Lease(unknown.clone())], + ) + .await + .unwrap(); + txn.commit().await.unwrap(); + let protected = roots(db).await; + assert_eq!( + success_json(lease_control(&fixture, &unknown, false).await).await["released"], + false + ); + assert_eq!( + roots(db).await, + protected, + "unknown release must not remove an unrelated protection root" + ); + assert_eq!(routes(db).await, original); + let renewed = success_json(lease_control(&fixture, &fixture.lease, true).await).await; + assert_eq!(renewed["lease_id"], fixture.lease); + assert_eq!(renewed["snapshot_id"], fixture.snapshot); + assert_eq!(routes(db).await, original); + assert_eq!( + success_json(lease_control(&fixture, &fixture.lease, false).await).await["released"], + true + ); + let terminal_roots = roots(db).await; + assert_eq!(routes(db).await, original); + assert_eq!( + success_json(lease_control(&fixture, &fixture.lease, false).await).await["released"], + false + ); + assert_eq!(roots(db).await, terminal_roots); + error( + lease_control(&fixture, &fixture.lease, true).await, + 410, + "LEASE_EXPIRED", + false, + ) + .await; + error( + fixture.send("GET", "descriptor", Body::empty()).await, + 410, + "LEASE_EXPIRED", + false, + ) + .await; + assert_eq!( + fixture + .send("HEAD", "blob?path=/file", Body::empty()) + .await + .status(), + 410 + ); + assert_eq!(routes(db).await, original); + assert_eq!(roots(db).await, terminal_roots); + assert_exact_bindings(db, 1).await; + fixture.counts.assert(0, 0); +} + +async fn wait_for_advisory_waiter(held: &DatabaseTransaction, key: i32) -> i64 { + tokio::time::timeout(Duration::from_secs(30), async { + loop { + held.execute_unprepared("SELECT pg_stat_clear_snapshot()").await.unwrap(); + let row = held.query_one_raw(statement( + "SELECT l.pid::bigint AS pid FROM pg_locks l JOIN pg_stat_activity a ON a.pid=l.pid + WHERE l.locktype='advisory' AND l.classid=$1::bigint::oid AND l.objid=hashtext(current_schema())::oid + AND l.objsubid=2 AND NOT l.granted AND a.datname=current_database() + AND a.application_name=current_schema() LIMIT 1", + [i64::from(key).into()], + )).await.unwrap(); + if let Some(row) = row { break row.try_get("", "pid").unwrap(); } + tokio::task::yield_now().await; + } + }).await.expect("raw lease statement did not wait on the expected advisory barrier") +} + +async fn lock_count(held: &DatabaseTransaction, pid: i64, key: i32) -> i64 { + held.query_one_raw(statement( + "SELECT count(*)::bigint FROM pg_locks WHERE pid::bigint=$1 AND locktype='advisory' + AND classid=$2::bigint::oid AND objid=hashtext(current_schema())::oid AND objsubid=2 AND granted", + [pid.into(), i64::from(key).into()], + )).await.unwrap().unwrap().try_get_by_index(0).unwrap() +} + +#[tokio::test] +async fn mst2_generic_storage_routes_raw_statement_waits_mono_then_route_then_retention() { + let fixture = Fixture::new_generic_history_with_pg_config(true).await; + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + for held_key in [MONO_WRITE_LOCK_KEY1, ROUTE_LOCK_KEY, RETENTION_LOCK_KEY] { + let held = db.begin().await.unwrap(); + held.execute_raw(statement( + "SELECT pg_advisory_xact_lock($1,hashtext(current_schema()))", + [held_key.into()], + )) + .await + .unwrap(); + let writing = { + let db = db.clone(); + let lease = fixture.lease.clone(); + tokio::spawn(async move { + let txn = db.begin().await.unwrap(); + let result = txn.execute_unprepared(&format!( + "{} ON CONFLICT(lease_id) DO UPDATE SET expires_at_unix=EXCLUDED.expires_at_unix", + lease_insert(&lease), + )).await; + txn.rollback().await.unwrap(); + result + }) + }; + let pid = wait_for_advisory_waiter(&held, held_key).await; + assert_eq!( + lock_count(&held, pid, MONO_WRITE_LOCK_KEY1).await, + if held_key == MONO_WRITE_LOCK_KEY1 { + 0 + } else { + 1 + } + ); + assert_eq!( + lock_count(&held, pid, ROUTE_LOCK_KEY).await, + if held_key == RETENTION_LOCK_KEY { 1 } else { 0 } + ); + assert_eq!(lock_count(&held, pid, RETENTION_LOCK_KEY).await, 0); + assert!( + held.query_one_raw(statement( + "SELECT lease_id FROM mst2_snapshot_lease WHERE lease_id=$1 FOR UPDATE NOWAIT", + [fixture.lease.clone().into()], + )) + .await + .unwrap() + .is_some(), + "the blocked statement locked its existing lease before the barrier" + ); + assert_eq!( + held.query_one_raw(statement( + "SELECT count(*)::bigint FROM pg_locks WHERE pid::bigint=$1 AND locktype='tuple' + AND relation IN ('mst2_snapshot_context'::regclass,'mst2_snapshot_lease'::regclass)", + [pid.into()], + )) + .await + .unwrap() + .unwrap() + .try_get_by_index::(0) + .unwrap(), + 0 + ); + held.rollback().await.unwrap(); + assert_eq!(writing.await.unwrap().unwrap().rows_affected(), 1); + assert_exact_bindings(db, 1).await; + } + for (key, message) in [ + ( + RETENTION_LOCK_KEY, + "cannot acquire core locks after retention", + ), + (ROUTE_LOCK_KEY, "lock was acquired before core mono"), + ] { + for lock in ["pg_advisory_xact_lock", "pg_advisory_xact_lock_shared"] { + let held = db.begin().await.unwrap(); + held.execute_raw(statement( + &format!("SELECT {lock}($1,hashtext(current_schema()))"), + [key.into()], + )) + .await + .unwrap(); + let result = tokio::time::timeout( + Duration::from_secs(5), + held.execute_unprepared(&lease_insert(&uuid::Uuid::new_v4().to_string())), + ) + .await + .expect("reverse-order statement waited instead of rejecting"); + let error = result.unwrap_err(); + assert!( + error.to_string().contains(message), + "wrong reverse-order {lock} rejection: {error}" + ); + held.rollback().await.unwrap(); + } + } + assert_exact_bindings(db, 1).await; +} + +#[tokio::test] +async fn mst2_generic_storage_routes_schema_isolation_wrong_caller_rr_and_temp_shadow_fail_closed() +{ + let fixture = Fixture::new_generic_history_with_pg_config(true).await; + let alien = Fixture::new_generic_history_with_pg_config(true).await; + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + let original = routes(db).await; + let other = routes(alien.state.storage.mono_storage().get_connection()).await; + assert_ne!( + original["namespace"][0]["namespace_uuid"], + other["namespace"][0]["namespace_uuid"] + ); + let transplanted = original["snapshot"][0].clone(); + let alien_db = alien.state.storage.mono_storage(); + let txn = alien_db.get_connection().begin().await.unwrap(); + let transplant = txn.execute_raw(statement( + "INSERT INTO mst2_snapshot_storage_route SELECT (jsonb_populate_record(NULL::mst2_snapshot_storage_route, + $1::jsonb||jsonb_build_object('namespace_uuid',(SELECT namespace_uuid FROM mst2_metadata_namespace)))).*", + [transplanted.to_string().into()], + )).await.unwrap_err(); + assert!(transplant.to_string().contains("actual generic context")); + txn.rollback().await.unwrap(); + let schema = fixture._schema.as_ref().unwrap().schema(); + let qualified = format!("\"{}\".mst2_route_enter", schema.replace('"', "\"\"")); + let txn = db + .begin_with_config(Some(IsolationLevel::RepeatableRead), None) + .await + .unwrap(); + assert!( + txn.execute_unprepared(&lease_insert(&uuid::Uuid::new_v4().to_string())) + .await + .is_err() + ); + txn.rollback().await.unwrap(); + let txn = db.begin().await.unwrap(); + txn.execute_unprepared("SET LOCAL search_path=pg_catalog") + .await + .unwrap(); + assert!( + txn.execute_raw(statement( + &format!("SELECT {qualified}($1)"), + ["pg_catalog".into()] + )) + .await + .is_err() + ); + txn.rollback().await.unwrap(); + let txn = db.begin().await.unwrap(); + assert!( + txn.execute_raw(statement( + &format!("SELECT {qualified}($1)"), + [alien._schema.as_ref().unwrap().schema().into()] + )) + .await + .is_err() + ); + txn.rollback().await.unwrap(); + let txn = db.begin().await.unwrap(); + txn.execute_unprepared( + "CREATE TEMP TABLE mst2_metadata_namespace(namespace_uuid uuid); + CREATE TEMP TABLE mst2_snapshot_storage_route(snapshot_id text); + CREATE TEMP TABLE mst2_generic_session_storage_binding(snapshot_id text); + CREATE TEMP TABLE mst2_lease_storage_route(lease_id text)", + ) + .await + .unwrap(); + txn.execute_raw(statement( + &format!("SELECT {qualified}($1)"), + [schema.into()], + )) + .await + .unwrap(); + assert_eq!( + scalar( + &txn, + &format!( + "SELECT count(*) FROM \"{}\".mst2_metadata_namespace", + schema.replace('"', "\"\"") + ) + ) + .await, + 1 + ); + assert_eq!( + scalar(&txn, "SELECT count(*) FROM pg_temp.mst2_metadata_namespace").await, + 0 + ); + txn.rollback().await.unwrap(); + error( + rebuilt(&alien) + .await + .oneshot(fixture.request("GET", "descriptor", Body::empty())) + .await + .unwrap(), + 410, + "LEASE_EXPIRED", + false, + ) + .await; + assert_eq!(routes(db).await, original); + assert_eq!( + routes(alien.state.storage.mono_storage().get_connection()).await, + other + ); + fixture.counts.assert(0, 0); + alien.counts.assert(0, 0); +} + +#[tokio::test] +async fn mst2_generic_storage_routes_additive_backfill_keeps_active_terminal_sources_bytes_and_roots() + { + let fixture = Fixture::new_generic_history_with_pg_config(true).await; + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + let warm = success_json( + fixture + .app + .clone() + .oneshot(resolve_request()) + .await + .unwrap(), + ) + .await; + let second = warm["lease_id"].as_str().unwrap(); + assert_eq!( + success_json(lease_control(&fixture, second, false).await).await["released"], + true + ); + assert_exact_bindings(db, 2).await; + let source = sources(db).await; + let original = routes(db).await; + let descriptor = success_json(fixture.send("GET", "descriptor", Body::empty()).await).await; + super::storage_route_fixture::restore_pre_route_schema(db).await; + assert_eq!(sources(db).await, source); + Migrator::up(db, None).await.unwrap(); + assert_eq!(sources(db).await, source); + assert_exact_bindings(db, 2).await; + let restored = routes(db).await; + for key in [ + "canonical_descriptor", + "instance_id", + "commit_oid", + "root_tree_oid", + "metadata_root", + "source_profile", + ] { + assert_eq!( + restored["snapshot"][0][key], original["snapshot"][0][key], + "backfill changed {key}" + ); + } + assert_eq!( + success_json( + rebuilt(&fixture) + .await + .oneshot(fixture.request("GET", "descriptor", Body::empty()),) + .await + .unwrap() + ) + .await, + descriptor + ); + assert_eq!( + success_json(lease_control(&fixture, second, false).await).await["released"], + false + ); + assert_eq!(sources(db).await, source); + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn mst2_generic_storage_routes_actual_http_ignores_temp_source_and_ledger_shadows() { + let fixture = Fixture::new_generic_history_with_pg_config(true).await; + let state = rebuilt_state(&fixture).await; + let app = Router::new().nest( + "/api/v2", + crate::api::router::snapshot_router::generic_history_routers(state.clone()) + .with_state(state.clone()), + ); + let mono = state.storage.mono_storage(); + let db = mono.get_connection(); + let original = routes(db).await; + assert_eq!( + app.clone() + .oneshot(fixture.request("HEAD", "blob?path=/file", Body::empty())) + .await + .unwrap() + .status(), + 200 + ); + let schema = fixture + ._schema + .as_ref() + .unwrap() + .schema() + .replace('"', "\"\""); + db.execute_unprepared(&format!( + "CREATE TEMP TABLE mst2_snapshot_context(LIKE \"{schema}\".mst2_snapshot_context); + CREATE TEMP TABLE mst2_snapshot_lease(LIKE \"{schema}\".mst2_snapshot_lease); + CREATE TEMP TABLE mst2_metadata_namespace(namespace_uuid uuid); + CREATE TEMP TABLE mst2_snapshot_storage_route(snapshot_id text); + CREATE TEMP TABLE mst2_generic_session_storage_binding(snapshot_id text); + CREATE TEMP TABLE mst2_lease_storage_route(lease_id text)", + )) + .await + .unwrap(); + assert_eq!( + app.clone() + .oneshot(fixture.request("HEAD", "blob?path=/file", Body::empty())) + .await + .unwrap() + .status(), + 200 + ); + let renewed = success_json( + app.oneshot( + Request::builder() + .method("POST") + .uri(format!("/api/v2/snapshots/leases/{}/renew", fixture.lease)) + .header("authorization", format!("Bearer {TOKEN}")) + .body(Body::empty()) + .unwrap(), + ) + .await + .unwrap(), + ) + .await; + assert_eq!(renewed["lease_id"], fixture.lease); + assert_eq!(renewed["snapshot_id"], fixture.snapshot); + for table in [ + "mst2_snapshot_context", + "mst2_snapshot_lease", + "mst2_metadata_namespace", + "mst2_snapshot_storage_route", + "mst2_generic_session_storage_binding", + "mst2_lease_storage_route", + ] { + assert_eq!( + scalar(db, &format!("SELECT count(*) FROM pg_temp.{table}")).await, + 0 + ); + } + let real = fixture.state.storage.mono_storage(); + assert_eq!(routes(real.get_connection()).await, original); + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn mst2_generic_storage_routes_temp_prepare_shadow_rejects_qualified_context_and_registered_generic_rebind() + { + let fixture = Fixture::new_generic_history_with_pg_config(true).await; + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + let original_routes = routes(db).await; + let unique = uuid::Uuid::new_v4().to_string(); + let entries = [Entry::file( + EntryKind::Regular, + unique.as_bytes(), + 3, + digest(unique.as_bytes()), + )]; + let child = Page::build(&entries).unwrap(); + let roots = [Entry::dir(b"qualified-route-boundary", page_id(&child))]; + let root = Page::build(&roots).unwrap(); + let mut builder = MetadataDagBuilder::new(MetadataDagLimits::default()); + builder.add_directory(&child, &entries).unwrap(); + builder.add_directory(&root, &roots).unwrap(); + let prepared = PreparedNativeMetadataRetention::test_installation( + Arc::new(builder.finish(page_id(&root)).unwrap()), + "/", + ); + let tree = "a".repeat(40); + assert_eq!(prepared.fixed_root_tree_oid(), format!("sha1:{tree}")); + let qualified = PostgresQualifiedMetadataRepository::new(db.clone()) + .await + .unwrap(); + let intent = qualified + .begin_intent(&format!("qualified-route-boundary:{unique}"), &prepared) + .await + .unwrap(); + qualified + .install_pages(&intent, prepared.dag().payloads()) + .await + .unwrap(); + let receipt = qualified.finalize(&intent).await.unwrap(); + assert_eq!(receipt.metadata_root(), prepared.dag().root()); + let committed = db.query_one_raw(statement( + "SELECT state,graph_domain,metadata_root,tagged_root_tree_oid FROM mst2_metadata_prepare WHERE prepare_id=$1", + [intent.prepare_id().into()], + )).await.unwrap().unwrap(); + assert_eq!( + committed.try_get::("", "state").unwrap(), + "COMMITTED" + ); + assert_eq!( + committed.try_get::("", "graph_domain").unwrap(), + "qualified-v1" + ); + assert_eq!( + committed.try_get::>("", "metadata_root").unwrap(), + receipt.metadata_root().to_vec() + ); + assert_eq!( + committed + .try_get::("", "tagged_root_tree_oid") + .unwrap(), + format!("sha1:{tree}") + ); + assert_eq!(scalar(db, + "SELECT count(*) FROM mst2_metadata_lifetime WHERE graph_domain='qualified-v1' AND state='LIVE'", + ).await, prepared.dag().payloads().len() as i64); + assert_eq!(routes(db).await, original_routes); + let before = domain_boundary_digests(db).await; + let descriptor = ServingDescriptor { + instance_uuid: *uuid::Uuid::parse_str( + fixture + .state + .storage + .config() + .mst2 + .instance_uuid + .as_deref() + .unwrap(), + ) + .unwrap() + .as_bytes(), + namespace_view_id: digest(unique.as_bytes()), + scope: prepared.scope().to_owned(), + metadata_root: receipt.metadata_root(), + }; + let sid = format!("sha256:{}", hex::encode(descriptor.snapshot_id().unwrap())); + assert_ne!(sid, fixture.snapshot); + let schema = format!( + "\"{}\"", + fixture + ._schema + .as_ref() + .unwrap() + .schema() + .replace('"', "\"\"") + ); + let shadow = + format!("CREATE TEMP TABLE mst2_metadata_prepare(LIKE {schema}.mst2_metadata_prepare)"); + let txn = db.begin().await.unwrap(); + txn.execute_unprepared(&shadow).await.unwrap(); + txn.execute_unprepared(&format!("SET LOCAL search_path={schema},pg_catalog")) + .await + .unwrap(); + assert_eq!( + scalar(&txn, "SELECT count(*) FROM mst2_metadata_prepare").await, + 0 + ); + assert_eq!(scalar(&txn, "SELECT CASE WHEN to_regclass('mst2_metadata_prepare')='pg_temp.mst2_metadata_prepare'::regclass THEN 1::bigint ELSE 0::bigint END").await, 1); + let rejected = txn.execute_raw(statement(&format!( + "INSERT INTO {schema}.mst2_snapshot_context(snapshot_id,canonical_descriptor,instance_id,commit_oid, + root_tree_oid,metadata_root,prepare_id,publication_sequence,writer_epoch,certificate_receipt_id,authorization_epoch,state) + SELECT $1,$2,instance_id,commit_oid,$3,$4,$5,publication_sequence,writer_epoch,certificate_receipt_id,authorization_epoch,'READY' + FROM {schema}.mst2_snapshot_context WHERE snapshot_id=$6", + ), [sid.into(),descriptor.encode().unwrap().into(),tree.into(),receipt.metadata_root().to_vec().into(), + intent.prepare_id().into(),fixture.snapshot.clone().into()], + )).await.unwrap_err(); + assert!( + rejected + .to_string() + .contains("storage route is not derived from its actual generic context"), + "{rejected}" + ); + txn.rollback().await.unwrap(); + assert_eq!( + domain_boundary_digests(db).await, + before, + "rejected Q adoption changed source, bytes, graph roots or route inventory" + ); + let generic = db + .query_one_raw(statement( + &format!( + "SELECT p.prepare_id,p.graph_domain FROM {schema}.mst2_metadata_prepare p + JOIN {schema}.mst2_metadata_install_seal i ON i.prepare_id=p.prepare_id + JOIN {schema}.mst2_snapshot_context s ON s.prepare_id=p.prepare_id WHERE s.snapshot_id=$1", + ), + [fixture.snapshot.clone().into()], + )) + .await + .unwrap() + .unwrap(); + let prepare_id: String = generic.try_get("", "prepare_id").unwrap(); + let domain: Option = generic.try_get("", "graph_domain").unwrap(); + assert!( + domain + .as_deref() + .is_none_or(|domain| domain == "generic-v1") + ); + let txn = db.begin().await.unwrap(); + txn.execute_unprepared(&shadow).await.unwrap(); + assert_eq!( + scalar(&txn, "SELECT count(*) FROM mst2_metadata_prepare").await, + 0 + ); + let rejected = txn.execute_raw(statement(&format!( + "UPDATE {schema}.mst2_metadata_prepare SET graph_domain='qualified-v1' WHERE prepare_id=$1", + ), [prepare_id.clone().into()])).await.unwrap_err(); + let error = rejected.to_string(); + assert!( + error.contains("registered metadata preparation identity is immutable") + || error.contains("metadata generation seal cannot be rebound"), + "{error}" + ); + txn.rollback().await.unwrap(); + let after_domain: Option = db.query_one_raw(statement(&format!( + "SELECT graph_domain FROM {schema}.mst2_metadata_prepare WHERE prepare_id=$1", + ), [prepare_id.into()])).await.unwrap().unwrap().try_get_by_index(0).unwrap(); + assert_eq!(after_domain, domain); + assert_eq!( + domain_boundary_digests(db).await, + before, + "registered G rebind changed source, bytes, graph roots or route inventory" + ); + assert_eq!(routes(db).await, original_routes); + assert_exact_bindings(db, 1).await; + assert_eq!( + fixture + .send("HEAD", "blob?path=/file", Body::empty()) + .await + .status(), + 200 + ); + fixture.counts.assert(0, 0); +} + +#[tokio::test] +async fn mst2_generic_storage_routes_corrupt_original_incarnation_rejects_reads_renew_release_without_repair() + { + let fixture = Fixture::new_generic_history_with_pg_config(true).await; + let mono = fixture.state.storage.mono_storage(); + let db = mono.get_connection(); + let source = sources(db).await; + let original = routes(db).await; + super::storage_route_fixture::overwrite_lease_route_incarnation_for_test(db, &fixture.lease) + .await; + let corrupted = routes(db).await; + assert_ne!(corrupted["lease"], original["lease"]); + for key in ["namespace", "snapshot", "binding"] { + assert_eq!(corrupted[key], original[key]); + } + error( + fixture.send("GET", "descriptor", Body::empty()).await, + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + assert_eq!( + fixture + .send("HEAD", "blob?path=/file", Body::empty()) + .await + .status(), + 502 + ); + error( + lease_control(&fixture, &fixture.lease, true).await, + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + error( + lease_control(&fixture, &fixture.lease, false).await, + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + error( + rebuilt(&fixture) + .await + .oneshot(fixture.request("GET", "descriptor", Body::empty())) + .await + .unwrap(), + 502, + "INTEGRITY_ERROR", + false, + ) + .await; + assert_eq!( + routes(db).await, + corrupted, + "route corruption must not be silently repaired" + ); + assert_eq!( + sources(db).await, + source, + "failed access must not change leases, plans, bytes or roots" + ); + fixture.counts.assert(0, 0); +} diff --git a/src/callisto/mod.rs b/src/callisto/mod.rs index dab5b7c5..8bdddc74 100644 --- a/src/callisto/mod.rs +++ b/src/callisto/mod.rs @@ -66,8 +66,20 @@ pub mod mega_view_root_chain_scan; pub mod mega_webhook; pub mod mega_webhook_delivery; pub mod mega_webhook_event_type; +pub mod mst2_metadata_current; +pub mod mst2_metadata_lifetime; +pub mod mst2_metadata_payload; +pub mod mst2_metadata_prepare; +pub mod mst2_metadata_prepare_page; +pub mod mst2_native_head; +pub mod mst2_native_publication; pub mod mst2_publication; pub mod mst2_publication_outbox; +pub mod mst2_queue_noop_receipt; +pub mod mst2_retention_edge; +pub mod mst2_retention_gc_op; +pub mod mst2_retention_node; +pub mod mst2_retention_root; pub mod mst2_verified_object; pub mod notification_event_types; pub mod oci_blob_ref; @@ -82,3 +94,6 @@ pub mod ssh_keys; pub mod user_notification_preferences; pub mod user_notification_settings; pub mod vault; + +pub mod mst2_snapshot_context; +pub mod mst2_snapshot_lease; diff --git a/src/callisto/mst2_metadata_current.rs b/src/callisto/mst2_metadata_current.rs new file mode 100644 index 00000000..0c759a1c --- /dev/null +++ b/src/callisto/mst2_metadata_current.rs @@ -0,0 +1,13 @@ +use sea_orm::entity::prelude::*; + +#[derive(Clone, Debug, PartialEq, DeriveEntityModel, Eq)] +#[sea_orm(table_name = "mst2_metadata_current")] +pub struct Model { + #[sea_orm(primary_key, auto_increment = false)] + pub page_id: Vec, + pub generation: i64, +} + +#[derive(Copy, Clone, Debug, EnumIter, DeriveRelation)] +pub enum Relation {} +impl ActiveModelBehavior for ActiveModel {} diff --git a/src/callisto/mst2_metadata_lifetime.rs b/src/callisto/mst2_metadata_lifetime.rs new file mode 100644 index 00000000..ba0fa4bc --- /dev/null +++ b/src/callisto/mst2_metadata_lifetime.rs @@ -0,0 +1,19 @@ +use sea_orm::entity::prelude::*; + +#[derive(Clone, Debug, PartialEq, DeriveEntityModel, Eq)] +#[sea_orm(table_name = "mst2_metadata_lifetime")] +pub struct Model { + #[sea_orm(primary_key, auto_increment = false)] + pub page_id: Vec, + pub node_id: String, + #[sea_orm(primary_key, auto_increment = false)] + pub generation: i64, + pub state: String, + pub metadata_codec: i16, + pub expected_size: i32, + pub graph_domain: String, +} + +#[derive(Copy, Clone, Debug, EnumIter, DeriveRelation)] +pub enum Relation {} +impl ActiveModelBehavior for ActiveModel {} diff --git a/src/callisto/mst2_metadata_payload.rs b/src/callisto/mst2_metadata_payload.rs new file mode 100644 index 00000000..9ec1a65b --- /dev/null +++ b/src/callisto/mst2_metadata_payload.rs @@ -0,0 +1,17 @@ +use sea_orm::entity::prelude::*; + +#[derive(Clone, Debug, PartialEq, DeriveEntityModel, Eq)] +#[sea_orm(table_name = "mst2_metadata_payload")] +pub struct Model { + #[sea_orm(primary_key, auto_increment = false)] + pub page_id: Vec, + pub generation: Option, + pub metadata_codec: i16, + pub byte_size: i32, + pub payload: Vec, + pub created_at: DateTimeWithTimeZone, +} + +#[derive(Copy, Clone, Debug, EnumIter, DeriveRelation)] +pub enum Relation {} +impl ActiveModelBehavior for ActiveModel {} diff --git a/src/callisto/mst2_metadata_prepare.rs b/src/callisto/mst2_metadata_prepare.rs new file mode 100644 index 00000000..c586cf0d --- /dev/null +++ b/src/callisto/mst2_metadata_prepare.rs @@ -0,0 +1,39 @@ +use sea_orm::entity::prelude::*; + +#[derive(Clone, Debug, PartialEq, DeriveEntityModel, Eq)] +#[sea_orm(table_name = "mst2_metadata_prepare")] +pub struct Model { + #[sea_orm(primary_key, auto_increment = false)] + pub prepare_id: String, + pub operation_id: String, + pub manifest_digest: Vec, + pub canonical_plan: Vec, + pub canonical_bindings: Option>, + pub bindings_digest: Option>, + pub primary_scope: Option>, + pub storage_seal: Option>, + pub graph_domain: Option, + pub source_domain: String, + pub tagged_root_tree_oid: String, + pub scope: String, + pub schema_version: i16, + pub metadata_codec: i16, + pub materialization_policy: i16, + pub fs_semantics: i16, + pub access_projection: i16, + pub verification_revision: i32, + pub projection_revision: i16, + pub metadata_root: Vec, + pub node_count: i32, + pub edge_count: i32, + pub total_bytes: i64, + pub state: String, + pub created_at: DateTimeWithTimeZone, + pub committed_at: Option, + pub aborted_at: Option, + pub coverage_retired_at: Option, +} + +#[derive(Copy, Clone, Debug, EnumIter, DeriveRelation)] +pub enum Relation {} +impl ActiveModelBehavior for ActiveModel {} diff --git a/src/callisto/mst2_metadata_prepare_page.rs b/src/callisto/mst2_metadata_prepare_page.rs new file mode 100644 index 00000000..7946022e --- /dev/null +++ b/src/callisto/mst2_metadata_prepare_page.rs @@ -0,0 +1,16 @@ +use sea_orm::entity::prelude::*; + +#[derive(Clone, Debug, PartialEq, DeriveEntityModel, Eq)] +#[sea_orm(table_name = "mst2_metadata_prepare_page")] +pub struct Model { + #[sea_orm(primary_key, auto_increment = false)] + pub prepare_id: String, + #[sea_orm(primary_key, auto_increment = false)] + pub page_id: Vec, + pub expected_size: i32, + pub generation: Option, +} + +#[derive(Copy, Clone, Debug, EnumIter, DeriveRelation)] +pub enum Relation {} +impl ActiveModelBehavior for ActiveModel {} diff --git a/src/callisto/mst2_native_head.rs b/src/callisto/mst2_native_head.rs new file mode 100644 index 00000000..68a26e00 --- /dev/null +++ b/src/callisto/mst2_native_head.rs @@ -0,0 +1,20 @@ +use sea_orm::entity::prelude::*; + +#[derive(Clone, Debug, PartialEq, DeriveEntityModel, Eq)] +#[sea_orm(table_name = "mst2_native_head")] +pub struct Model { + #[sea_orm(primary_key, auto_increment = false)] + pub namespace: String, + pub instance_id: String, + pub sequence: i64, + pub writer_epoch: i64, + pub root_commit: String, + pub root_tree: String, + pub state: String, + pub certificate_receipt_id: Option, +} + +#[derive(Copy, Clone, Debug, EnumIter, DeriveRelation)] +pub enum Relation {} + +impl ActiveModelBehavior for ActiveModel {} diff --git a/src/callisto/mst2_native_publication.rs b/src/callisto/mst2_native_publication.rs new file mode 100644 index 00000000..449be1db --- /dev/null +++ b/src/callisto/mst2_native_publication.rs @@ -0,0 +1,27 @@ +use sea_orm::entity::prelude::*; + +#[derive(Clone, Debug, PartialEq, DeriveEntityModel, Eq)] +#[sea_orm(table_name = "mst2_native_publication")] +pub struct Model { + #[sea_orm(primary_key, auto_increment = false)] + pub receipt_id: i64, + pub namespace: String, + pub instance_id: String, + pub sequence: i64, + pub writer_epoch: i64, + pub old_root_commit: String, + pub old_root_tree: String, + pub root_commit: String, + pub root_tree: String, + pub origin_path: String, + pub origin_ref: String, + pub old_path_commit: Option, + pub old_path_tree: Option, + pub path_commit: String, + pub path_tree: String, +} + +#[derive(Copy, Clone, Debug, EnumIter, DeriveRelation)] +pub enum Relation {} + +impl ActiveModelBehavior for ActiveModel {} diff --git a/src/callisto/mst2_publication.rs b/src/callisto/mst2_publication.rs index ff69518e..603b6509 100644 --- a/src/callisto/mst2_publication.rs +++ b/src/callisto/mst2_publication.rs @@ -16,7 +16,7 @@ use serde::{Deserialize, Serialize}; pub struct Model { #[sea_orm(primary_key, auto_increment = true)] pub id: i64, - /// Deterministic writer-side operation id (e.g. push old→new). + /// Server-assigned writer identity (e.g. `mst2:trunk-queue:`). pub operation_id: String, /// Namespace path this publication advances (e.g. "/"). pub namespace: String, @@ -29,6 +29,12 @@ pub struct Model { pub writer_epoch: i64, /// `trunk_push` | `web_edit` | `import` | … (spec 09 §5 writer matrix). pub writer_kind: String, + /// Missing on legacy receipts; never reconstructed from current refs. + #[sea_orm(column_type = "Text", nullable)] + pub request_digest: Option, + pub request_digest_version: Option, + /// NULL preserves historical path-only receipts; version 1 requires its certificate. + pub native_certificate_version: Option, pub created_at: DateTimeWithTimeZone, } diff --git a/src/callisto/mst2_queue_noop_receipt.rs b/src/callisto/mst2_queue_noop_receipt.rs new file mode 100644 index 00000000..15c3b137 --- /dev/null +++ b/src/callisto/mst2_queue_noop_receipt.rs @@ -0,0 +1,34 @@ +//! Committed trunk-queue no-op results, separate from visible publications. + +use sea_orm::entity::prelude::*; +use serde::{Deserialize, Serialize}; + +#[derive(Clone, Debug, PartialEq, DeriveEntityModel, Eq, Serialize, Deserialize)] +#[sea_orm(table_name = "mst2_queue_noop_receipt")] +pub struct Model { + #[sea_orm(primary_key, auto_increment = true)] + pub id: i64, + #[sea_orm(column_type = "Text")] + pub operation_id: String, + #[sea_orm(column_type = "Text")] + pub namespace: String, + #[sea_orm(column_type = "Text")] + pub request_digest: String, + pub request_digest_version: i32, + pub writer_epoch: i64, + #[sea_orm(column_type = "Text")] + pub writer_kind: String, + pub observed_sequence: i64, + #[sea_orm(column_type = "Text")] + pub observed_root_commit: String, + #[sea_orm(column_type = "Text")] + pub observed_root_tree: String, + #[sea_orm(column_type = "Text")] + pub landed_commit_id: String, + pub created_at: DateTimeWithTimeZone, +} + +#[derive(Copy, Clone, Debug, EnumIter, DeriveRelation)] +pub enum Relation {} + +impl ActiveModelBehavior for ActiveModel {} diff --git a/src/callisto/mst2_retention_edge.rs b/src/callisto/mst2_retention_edge.rs new file mode 100644 index 00000000..2c861bdf --- /dev/null +++ b/src/callisto/mst2_retention_edge.rs @@ -0,0 +1,19 @@ +//! MST/2 retention graph edges (spec 10 §6). + +use sea_orm::entity::prelude::*; +use serde::{Deserialize, Serialize}; + +#[derive(Clone, Debug, PartialEq, DeriveEntityModel, Eq, Serialize, Deserialize)] +#[sea_orm(table_name = "mst2_retention_edge")] +pub struct Model { + #[sea_orm(primary_key, auto_increment = true)] + pub id: i64, + pub parent_id: String, + pub child_id: String, + pub created_at: Option, +} + +#[derive(Copy, Clone, Debug, EnumIter, DeriveRelation)] +pub enum Relation {} + +impl ActiveModelBehavior for ActiveModel {} diff --git a/src/callisto/mst2_retention_gc_op.rs b/src/callisto/mst2_retention_gc_op.rs new file mode 100644 index 00000000..833c949d --- /dev/null +++ b/src/callisto/mst2_retention_gc_op.rs @@ -0,0 +1,26 @@ +//! Durable MST/2 GC operation log (spec 10 §6 crash replay). + +use sea_orm::entity::prelude::*; +use serde::{Deserialize, Serialize}; + +#[derive(Clone, Debug, PartialEq, DeriveEntityModel, Eq, Serialize, Deserialize)] +#[sea_orm(table_name = "mst2_retention_gc_op")] +pub struct Model { + /// Caller supplied idempotency key. A retry of one physical GC step must + /// address the same row, so replay cannot decrement an edge twice. + #[sea_orm(primary_key, auto_increment = false)] + pub operation_id: String, + pub node_id: String, + /// `MARK_DELETING` or `REMOVE`. + pub operation: String, + /// `PENDING`, `APPLIED`, or `FAILED`; workers may replay `PENDING`. + pub state: String, + pub attempts: i32, + pub created_at: DateTimeWithTimeZone, + pub completed_at: Option, +} + +#[derive(Copy, Clone, Debug, EnumIter, DeriveRelation)] +pub enum Relation {} + +impl ActiveModelBehavior for ActiveModel {} diff --git a/src/callisto/mst2_retention_node.rs b/src/callisto/mst2_retention_node.rs new file mode 100644 index 00000000..40ce2999 --- /dev/null +++ b/src/callisto/mst2_retention_node.rs @@ -0,0 +1,25 @@ +//! MST/2 retention graph nodes (spec 10 §6). + +use sea_orm::entity::prelude::*; +use serde::{Deserialize, Serialize}; + +#[derive(Clone, Debug, PartialEq, DeriveEntityModel, Eq, Serialize, Deserialize)] +#[sea_orm(table_name = "mst2_retention_node")] +pub struct Model { + #[sea_orm(primary_key, auto_increment = false)] + pub node_id: String, + pub kind: String, + pub state: String, + #[sea_orm(column_type = "BigInteger")] + pub bytes: i64, + /// Number of unique incoming retention edges. Root coverage is kept in + /// `mst2_retention_root` and checked in the same transaction. + #[sea_orm(column_type = "BigInteger")] + pub incoming_refs: i64, + pub created_at: Option, +} + +#[derive(Copy, Clone, Debug, EnumIter, DeriveRelation)] +pub enum Relation {} + +impl ActiveModelBehavior for ActiveModel {} diff --git a/src/callisto/mst2_retention_root.rs b/src/callisto/mst2_retention_root.rs new file mode 100644 index 00000000..9b60cb5a --- /dev/null +++ b/src/callisto/mst2_retention_root.rs @@ -0,0 +1,20 @@ +//! MST/2 retention roots (spec 10 §5/§6). + +use sea_orm::entity::prelude::*; +use serde::{Deserialize, Serialize}; + +#[derive(Clone, Debug, PartialEq, DeriveEntityModel, Eq, Serialize, Deserialize)] +#[sea_orm(table_name = "mst2_retention_root")] +pub struct Model { + #[sea_orm(primary_key, auto_increment = true)] + pub id: i64, + pub node_id: String, + pub root_key: String, + pub root_kind: String, + pub created_at: Option, +} + +#[derive(Copy, Clone, Debug, EnumIter, DeriveRelation)] +pub enum Relation {} + +impl ActiveModelBehavior for ActiveModel {} diff --git a/src/callisto/mst2_snapshot_context.rs b/src/callisto/mst2_snapshot_context.rs new file mode 100644 index 00000000..5e7acefc --- /dev/null +++ b/src/callisto/mst2_snapshot_context.rs @@ -0,0 +1,24 @@ +use sea_orm::entity::prelude::*; + +#[derive(Clone, Debug, PartialEq, DeriveEntityModel, Eq)] +#[sea_orm(table_name = "mst2_snapshot_context")] +pub struct Model { + #[sea_orm(primary_key, auto_increment = false)] + pub snapshot_id: String, + pub canonical_descriptor: Vec, + pub instance_id: String, + pub commit_oid: String, + pub root_tree_oid: String, + pub metadata_root: Vec, + pub prepare_id: String, + pub publication_sequence: i64, + pub writer_epoch: i64, + pub certificate_receipt_id: Option, + pub authorization_epoch: i64, + pub state: String, + pub created_at: DateTimeWithTimeZone, +} + +#[derive(Copy, Clone, Debug, EnumIter, DeriveRelation)] +pub enum Relation {} +impl ActiveModelBehavior for ActiveModel {} diff --git a/src/callisto/mst2_snapshot_lease.rs b/src/callisto/mst2_snapshot_lease.rs new file mode 100644 index 00000000..2d5cc887 --- /dev/null +++ b/src/callisto/mst2_snapshot_lease.rs @@ -0,0 +1,20 @@ +use sea_orm::entity::prelude::*; + +#[derive(Clone, Debug, PartialEq, DeriveEntityModel, Eq)] +#[sea_orm(table_name = "mst2_snapshot_lease")] +pub struct Model { + #[sea_orm(primary_key, auto_increment = false)] + pub lease_id: String, + pub snapshot_id: String, + pub authorization_epoch: i64, + pub publication_sequence: i64, + pub writer_epoch: i64, + pub certificate_receipt_id: i64, + pub expires_at_unix: i64, + pub state: String, + pub created_at: DateTimeWithTimeZone, +} + +#[derive(Copy, Clone, Debug, EnumIter, DeriveRelation)] +pub enum Relation {} +impl ActiveModelBehavior for ActiveModel {} diff --git a/src/callisto/prelude.rs b/src/callisto/prelude.rs index f85bd9cb..de2fd09c 100644 --- a/src/callisto/prelude.rs +++ b/src/callisto/prelude.rs @@ -25,6 +25,10 @@ pub use super::{ mega_view_root_chain_scan::Entity as MegaViewRootChainScan, mega_webhook::Entity as MegaWebhook, mega_webhook_delivery::Entity as MegaWebhookDelivery, mega_webhook_event_type::Entity as MegaWebhookEventType, + mst2_retention_edge::Entity as Mst2RetentionEdge, + mst2_retention_gc_op::Entity as Mst2RetentionGcOp, + mst2_retention_node::Entity as Mst2RetentionNode, + mst2_retention_root::Entity as Mst2RetentionRoot, notification_event_types::Entity as NotificationEventTypes, oci_blob_ref::Entity as OciBlobRef, oci_manifest::Entity as OciManifest, oci_tag::Entity as OciTag, oci_upload::Entity as OciUpload, path_check_configs::Entity as PathCheckConfigs, diff --git a/src/callisto/push_queue.rs b/src/callisto/push_queue.rs index 01638727..a8dea3d6 100644 --- a/src/callisto/push_queue.rs +++ b/src/callisto/push_queue.rs @@ -42,6 +42,9 @@ pub struct Model { pub expected_commit_hash: Option, #[sea_orm(column_type = "Text", nullable)] pub expected_tree_hash: Option, + pub expected_native_sequence: Option, + pub expected_native_epoch: Option, + pub expected_native_certificate: Option, pub pending_action: Option, pub enqueued_at: DateTimeWithTimeZone, pub started_at: Option, diff --git a/src/ceres/api_service/mod.rs b/src/ceres/api_service/mod.rs index 856087b2..7acbd06a 100644 --- a/src/ceres/api_service/mod.rs +++ b/src/ceres/api_service/mod.rs @@ -60,6 +60,13 @@ mod un25_freeze; pub trait ApiHandler: Send + Sync { fn get_context(&self) -> Storage; + /// Only native monorepo trees use the native snapshot memoization domain. + /// Import and custom handlers remain uncached unless they implement their + /// own source identity and verification contract. + fn native_snapshot_projection(&self) -> bool { + false + } + fn object_cache(&self) -> &GitObjectCache; fn strip_relative(&self, path: &Path) -> Result; @@ -171,6 +178,64 @@ pub trait ApiHandler: Send + Sync { } } + async fn get_raw_blob_stream_by_hash( + &self, + hash: &str, + ) -> Result { + let storage = self.get_context(); + match storage.git_service.get_object_stream(hash).await { + Ok(stream) => Ok(stream), + Err(error) => Err(storage + .classify_blob_objstorage_not_found(hash, error) + .await), + } + } + + /// Raw whole-file stream with the physical object's complete size. + async fn get_raw_blob_stream_with_meta( + &self, + hash: &str, + ) -> Result< + ( + crate::orbit_api::object_storage::ObjectByteStream, + crate::orbit_api::object_storage::ObjectMeta, + ), + MegaError, + > { + let storage = self.get_context(); + match storage.git_service.get_object_stream_with_meta(hash).await { + Ok(value) => Ok(value), + Err(error) => Err(storage + .classify_blob_objstorage_not_found(hash, error) + .await), + } + } + + async fn get_raw_blob_range_stream_exact( + &self, + hash: &str, + start: u64, + end: u64, + ) -> Result< + Option<( + crate::orbit_api::object_storage::ObjectByteStream, + crate::orbit_api::object_storage::ObjectMeta, + )>, + MegaError, + > { + let storage = self.get_context(); + match storage + .git_service + .get_object_range_stream_exact(hash, start, end) + .await + { + Ok(range) => Ok(range), + Err(error) => Err(storage + .classify_blob_objstorage_not_found(hash, error) + .await), + } + } + /// Preview unified diff for a single file change async fn preview_file_diff( &self, diff --git a/src/ceres/api_service/mono_api_service.rs b/src/ceres/api_service/mono_api_service.rs index f499b159..0a204c16 100644 --- a/src/ceres/api_service/mono_api_service.rs +++ b/src/ceres/api_service/mono_api_service.rs @@ -1177,6 +1177,10 @@ impl ApiHandler for MonoApiService { self.storage.clone() } + fn native_snapshot_projection(&self) -> bool { + true + } + fn object_cache(&self) -> &GitObjectCache { &self.git_object_cache } diff --git a/src/ceres/snapshot/chunk_map_gate.rs b/src/ceres/snapshot/chunk_map_gate.rs new file mode 100644 index 00000000..765e6c3b --- /dev/null +++ b/src/ceres/snapshot/chunk_map_gate.rs @@ -0,0 +1,185 @@ +//! Bounded exact-source install flights. Completed data is always re-read +//! from the requesting source's trusted receipt, never retained in a flight. + +use std::{ + collections::HashMap, + sync::{Arc, Mutex, OnceLock, Weak}, +}; + +use super::error::{SnapshotError, SnapshotErrorCode}; + +const MAX_INSTALL_FLIGHTS: usize = 128; +type SourceGate = tokio::sync::Mutex<()>; + +#[derive(Default)] +struct Registry { + entries: HashMap<[u8; 32], Weak>, +} + +pub(crate) struct InstallFlight { + key: [u8; 32], + gate: Option>, + registry: Arc>, +} + +impl InstallFlight { + pub(crate) fn acquire(key: [u8; 32]) -> Result { + static REGISTRY: OnceLock>> = OnceLock::new(); + Self::from_registry(REGISTRY.get_or_init(Arc::default).clone(), key) + } + + fn from_registry(registry: Arc>, key: [u8; 32]) -> Result { + let gate = { + let mut state = registry.lock().map_err(|_| internal())?; + if let Some(gate) = state.entries.get(&key).and_then(Weak::upgrade) { + gate + } else { + state.entries.remove(&key); + if state.entries.len() >= MAX_INSTALL_FLIGHTS { + return Err(SnapshotError::new( + SnapshotErrorCode::LimitExceeded, + "too many distinct source map installations in flight", + )); + } + let gate = Arc::new(SourceGate::new(())); + state.entries.insert(key, Arc::downgrade(&gate)); + gate + } + }; + Ok(Self { + key, + gate: Some(gate), + registry, + }) + } + + pub(crate) async fn lock(&self) -> Result, SnapshotError> { + Ok(self.gate.as_ref().ok_or_else(internal)?.lock().await) + } + + #[cfg(test)] + pub(crate) fn test_owner_count(&self) -> usize { + self.gate.as_ref().map_or(0, Arc::strong_count) + } +} + +impl Drop for InstallFlight { + fn drop(&mut self) { + let Some(gate) = self.gate.take() else { + return; + }; + if let Ok(mut state) = self.registry.lock() { + if Arc::strong_count(&gate) == 1 + && state + .entries + .get(&self.key) + .is_some_and(|entry| entry.ptr_eq(&Arc::downgrade(&gate))) + { + state.entries.remove(&self.key); + } + // Owners must release their strong reference while the registry + // is locked, so concurrent final drops cannot leave a stale slot. + drop(gate); + } + } +} + +fn internal() -> SnapshotError { + SnapshotError::new( + SnapshotErrorCode::Internal, + "source map flight registry is unavailable", + ) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[tokio::test] + async fn exact_source_gate_blocks_only_same_source_and_returns_capacity_after_last_owner() { + let registry = Arc::new(Mutex::new(Registry::default())); + let first = InstallFlight::from_registry(registry.clone(), [1; 32]).unwrap(); + let lock = first.lock().await.unwrap(); + let same = InstallFlight::from_registry(registry.clone(), [1; 32]).unwrap(); + assert!(same.gate.as_ref().unwrap().try_lock().is_err()); + let other = InstallFlight::from_registry(registry.clone(), [2; 32]).unwrap(); + assert!(other.gate.as_ref().unwrap().try_lock().is_ok()); + drop(lock); + drop(first); + assert!(same.gate.as_ref().unwrap().try_lock().is_ok()); + assert_eq!(registry.lock().unwrap().entries.len(), 2); + drop(same); + drop(other); + assert!(registry.lock().unwrap().entries.is_empty()); + let mut held = Vec::new(); + for i in 0..MAX_INSTALL_FLIGHTS { + let mut key = [0; 32]; + key[..8].copy_from_slice(&(i as u64).to_le_bytes()); + held.push(InstallFlight::from_registry(registry.clone(), key).unwrap()); + } + assert_eq!( + InstallFlight::from_registry(registry.clone(), [255; 32]) + .err() + .unwrap() + .code, + SnapshotErrorCode::LimitExceeded + ); + let joined_at_capacity = InstallFlight::from_registry(registry.clone(), [0; 32]).unwrap(); + assert_eq!(joined_at_capacity.test_owner_count(), 2); + assert_eq!(registry.lock().unwrap().entries.len(), MAX_INSTALL_FLIGHTS); + drop(joined_at_capacity); + drop(held); + assert!(registry.lock().unwrap().entries.is_empty()); + assert!(InstallFlight::from_registry(registry, [255; 32]).is_ok()); + } + + #[test] + fn concurrent_last_owners_release_registry_capacity() { + for _ in 0..64 { + let registry = Arc::new(Mutex::new(Registry::default())); + let first = InstallFlight::from_registry(registry.clone(), [1; 32]).unwrap(); + let second = InstallFlight::from_registry(registry.clone(), [1; 32]).unwrap(); + let barrier = Arc::new(std::sync::Barrier::new(2)); + std::thread::scope(|scope| { + let other = barrier.clone(); + scope.spawn(move || { + other.wait(); + drop(first); + }); + scope.spawn(move || { + barrier.wait(); + drop(second); + }); + }); + assert!(registry.lock().unwrap().entries.is_empty()); + } + } + + #[test] + fn dropping_an_old_claim_does_not_remove_a_replacement_gate() { + let registry = Arc::new(Mutex::new(Registry::default())); + let old = InstallFlight::from_registry(registry.clone(), [1; 32]).unwrap(); + let replacement = Arc::new(SourceGate::new(())); + registry + .lock() + .unwrap() + .entries + .insert([1; 32], Arc::downgrade(&replacement)); + drop(old); + assert_eq!(registry.lock().unwrap().entries.len(), 1); + let current = InstallFlight::from_registry(registry.clone(), [1; 32]).unwrap(); + assert!(Arc::ptr_eq(current.gate.as_ref().unwrap(), &replacement)); + drop(replacement); + drop(current); + assert!(registry.lock().unwrap().entries.is_empty()); + registry + .lock() + .unwrap() + .entries + .insert([1; 32], Weak::new()); + let reclaimed = InstallFlight::from_registry(registry.clone(), [1; 32]).unwrap(); + assert_eq!(registry.lock().unwrap().entries.len(), 1); + drop(reclaimed); + assert!(registry.lock().unwrap().entries.is_empty()); + } +} diff --git a/src/ceres/snapshot/chunk_map_index.rs b/src/ceres/snapshot/chunk_map_index.rs new file mode 100644 index 00000000..7e253f9c --- /dev/null +++ b/src/ceres/snapshot/chunk_map_index.rs @@ -0,0 +1,171 @@ +//! Canonical MCL2 Merkle subtrees addressed by their exact leaf interval. + +use mst2_codec::chunkmap::{ProofSide, ProofStep}; +use sha2::{Digest, Sha256}; + +use super::error::{SnapshotError, SnapshotErrorCode}; + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(crate) struct ChunkMapNode { + pub start: u64, + pub pages: u64, + pub digest: [u8; 32], +} + +pub(crate) fn indexed_nodes(hashes: &[[u8; 32]]) -> Result, SnapshotError> { + if hashes.is_empty() || hashes.len() > 32_768 { + return Err(integrity("chunk map is outside the indexed page profile")); + } + let mut nodes = Vec::new(); + nodes.try_reserve_exact(hashes.len() * 2 - 1).map_err(|_| { + SnapshotError::new( + SnapshotErrorCode::TemporaryUnavailable, + "chunk map index allocation failed", + ) + })?; + build(hashes, 0, &mut nodes); + Ok(nodes) +} + +fn build(hashes: &[[u8; 32]], start: u64, nodes: &mut Vec) -> [u8; 32] { + let pages = hashes.len() as u64; + let digest = if pages == 1 { + hashes[0] + } else { + let split = split(pages); + let left = build(&hashes[..split as usize], start, nodes); + let right = build(&hashes[split as usize..], start + split, nodes); + let mut hash = Sha256::new(); + hash.update(b"mega.mst2.chunkbranch\0"); + hash.update(split.to_le_bytes()); + hash.update(left); + hash.update((pages - split).to_le_bytes()); + hash.update(right); + hash.finalize().into() + }; + nodes.push(ChunkMapNode { + start, + pages, + digest, + }); + digest +} + +fn split(pages: u64) -> u64 { + 1 << (63 - (pages - 1).leading_zeros()) +} + +/// Only these sibling intervals may be fetched for a selected page. +pub(crate) fn proof_intervals( + pages: u64, + index: u64, +) -> Result, SnapshotError> { + if pages == 0 || pages > 32_768 || index >= pages { + return Err(SnapshotError::new( + SnapshotErrorCode::PathNotFound, + "chunk map page does not exist", + )); + } + let (mut start, mut count) = (0, pages); + let mut intervals = Vec::new(); + while count > 1 { + let left = split(count); + if index < start + left { + intervals.push((ProofSide::Right, start + left, count - left)); + count = left; + } else { + intervals.push((ProofSide::Left, start, left)); + start += left; + count -= left; + } + } + intervals.reverse(); + Ok(intervals) +} + +pub(crate) fn selected_proof( + intervals: &[(ProofSide, u64, u64)], + nodes: &[ChunkMapNode], +) -> Result, SnapshotError> { + if nodes.len() != intervals.len() { + return Err(integrity( + "persisted chunk map proof coverage is incomplete", + )); + } + intervals + .iter() + .map(|&(side, start, pages)| { + let mut matching = nodes + .iter() + .filter(|n| n.start == start && n.pages == pages); + let node = matching + .next() + .ok_or_else(|| integrity("persisted chunk map sibling is missing"))?; + if matching.next().is_some() { + return Err(integrity("persisted chunk map sibling is duplicated")); + } + Ok(ProofStep { + side, + sibling_pages: pages, + digest: node.digest, + }) + }) + .collect() +} + +fn integrity(message: &str) -> SnapshotError { + SnapshotError::new(SnapshotErrorCode::IntegrityError, message) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn indexed_selected_proofs_match_independent_codec_for_uneven_trees() { + for count in [1, 2, 3, 5, 9, 17, 257, 32_768] { + let hashes: Vec<[u8; 32]> = (0u64..count) + .map(|n| Sha256::digest(n.to_le_bytes()).into()) + .collect(); + let nodes = indexed_nodes(&hashes).unwrap(); + assert_eq!(nodes.len(), hashes.len() * 2 - 1); + let root = mst2_codec::chunkmap::merkle_root(&hashes).unwrap(); + assert_eq!(nodes.last().unwrap().digest, root); + for page in [0, count / 2, count - 1] { + let intervals = proof_intervals(count, page).unwrap(); + assert!(intervals.len() <= 15); + let selected: Vec<_> = nodes + .iter() + .filter(|n| { + intervals + .iter() + .any(|&(_, s, c)| n.start == s && n.pages == c) + }) + .copied() + .collect(); + let proof = selected_proof(&intervals, &selected).unwrap(); + assert_eq!( + proof, + mst2_codec::chunkmap::leaf_proof(&hashes, page).unwrap() + ); + mst2_codec::chunkmap::verify_leaf(count, page, hashes[page as usize], &proof, root) + .unwrap(); + if !selected.is_empty() { + assert!(selected_proof(&intervals, &selected[1..]).is_err()); + let mut bad = proof.clone(); + bad[0].digest[0] ^= 1; + assert!( + mst2_codec::chunkmap::verify_leaf( + count, + page, + hashes[page as usize], + &bad, + root + ) + .is_err() + ); + } + } + } + } +} diff --git a/src/ceres/snapshot/chunks.rs b/src/ceres/snapshot/chunks.rs index 18d9027b..8018c245 100644 --- a/src/ceres/snapshot/chunks.rs +++ b/src/ceres/snapshot/chunks.rs @@ -1,39 +1,151 @@ -//! On-the-fly, range-readable chunk projection for one file (spec 07). -//! -//! Persistent segment/locator storage is T04/T12 work; this identity slice -//! builds the MCM2 map and MCL2 leaves from a fixed view's verified blob and -//! stages the raw bytes in a bounded in-process cache, so a CHUNK request -//! slices its range out of an already-verified representation rather than -//! reconstructing the whole Git object per chunk (spec 07 §8). -//! -//! The cache is an optimization, never the authority: every projected file -//! is re-hashed against the `content_id` the fixed view advertised, and a -//! cache miss simply rebuilds from Git. Entries are addressed by content -//! digest, never by request path. +//! Cold full-stream chunk-map verifier. Actual content callers use immutable +//! source receipts and selected persisted pages, never a digest-only cache. -use std::{ - collections::{HashMap, VecDeque}, - sync::{Mutex, OnceLock}, -}; +use std::sync::Arc; use mst2_codec::chunkmap::{CHUNK_SIZE, CHUNKS_PER_PAGE, ChunkLeaf, ChunkMap}; use sha2::{Digest, Sha256}; +use super::content_budget::MemoryLease; use crate::ceres::snapshot::error::{SnapshotError, SnapshotErrorCode}; +#[path = "chunks_stream.rs"] +mod streaming; + +/// Full-stream admission for one exact source, independent of the digest cache. +/// This opaque value is the only production input to durable map installation. +pub(crate) struct VerifiedSourceChunkMap { + source: ChunkMapSource, + projection: ChunkProjection, +} + +#[derive(Debug, Clone, PartialEq, Eq)] +pub(crate) struct ChunkMapSource { + fact: crate::callisto::mst2_verified_object::Model, +} + +impl ChunkMapSource { + pub(crate) fn from_fact( + fact: crate::callisto::mst2_verified_object::Model, + oid: &str, + ) -> Result { + if fact.id <= 0 + || fact.storage_domain != "git" + || fact.object_kind != "blob" + || fact.git_oid != oid + || ![40, 64].contains(&oid.len()) + || !oid + .bytes() + .all(|b| b.is_ascii_hexdigit() && !b.is_ascii_uppercase()) + || fact.state != "VERIFIED" + || fact.verification_version + != crate::jupiter::storage::mono_storage::MST2_VERIFICATION_VERSION + || fact.raw_sha256.len() != 32 + || fact.size <= 0 + || fact.size as u64 > 8 * 1024 * 1024 * 1024 * 1024 + { + return Err(SnapshotError::new( + SnapshotErrorCode::IntegrityError, + "invalid exact chunk map source fact", + )); + } + Ok(Self { fact }) + } + + pub(crate) fn fact(&self) -> &crate::callisto::mst2_verified_object::Model { + &self.fact + } + + pub(crate) fn canonical_bytes(&self) -> Result, SnapshotError> { + let f = &self.fact; + serde_json::to_vec(&( + f.id, + &f.storage_domain, + &f.git_oid, + &f.object_kind, + &f.raw_sha256, + f.size, + f.verification_version, + &f.state, + f.created_at, + )) + .map_err(|_| internal("chunk map source encoding failed")) + } +} + +impl VerifiedSourceChunkMap { + pub(crate) async fn verify( + handler: &T, + source: ChunkMapSource, + budget: &Arc, + admission: &crate::jupiter::storage::native_chunk_map::retention::ChunkMapInstall, + ) -> Result { + let digest: [u8; 32] = source + .fact + .raw_sha256 + .as_slice() + .try_into() + .map_err(|_| internal("chunk map source digest shape changed"))?; + let projection = streaming::build_source_stream( + digest, + source.fact.size as u64, + || async { + handler + .get_raw_blob_stream_by_hash(&source.fact.git_oid) + .await + .map_err(|error| { + let code = match error { + crate::common::errors::MegaError::ObjStorageNotFound(_) => { + SnapshotErrorCode::ObjectUnavailable + } + crate::common::errors::MegaError::ObjStorageInconsistent(_) => { + SnapshotErrorCode::IntegrityError + } + _ => SnapshotErrorCode::Internal, + }; + SnapshotError::new(code, "exact chunk map source could not be read") + }) + }, + budget, + admission, + ) + .await?; + Ok(Self { source, projection }) + } + + pub(crate) fn source(&self) -> &ChunkMapSource { + &self.source + } + pub(crate) fn map(&self) -> &ChunkMap { + &self.projection.map + } + pub(crate) fn leaves(&self) -> &[ChunkLeaf] { + &self.projection.leaves + } + pub(crate) fn leaf_hashes(&self) -> &[[u8; 32]] { + &self.projection.leaf_hashes + } +} + +pub(crate) fn map_build_reservation_bytes(size: u64) -> Result { + streaming::reserved_map_bytes(size) +} + /// One file's verified range-readable projection. pub struct ChunkProjection { pub map: ChunkMap, pub map_id: [u8; 32], leaves: Vec, leaf_hashes: Vec<[u8; 32]>, - /// Full verified content. Indexed by fixed 1 MiB chunk boundaries. + /// Empty for range-only projections; empty files have no projection. raw: Vec, + memory: Option, } impl ChunkProjection { /// Build from bytes that the caller already resolved at a fixed path. /// `content_id` is the digest the fixed view advertises. + #[cfg(test)] pub fn build(content_id: [u8; 32], raw: Vec) -> Result { let mut hasher = Sha256::new(); hasher.update(&raw); @@ -95,14 +207,61 @@ impl ChunkProjection { leaves, leaf_hashes, raw, + memory: None, }) } + #[cfg(test)] + pub fn has_inline_bytes(&self) -> bool { + !self.raw.is_empty() + } + + fn retained_bytes(&self) -> usize { + self.memory + .as_ref() + .map_or_else(|| self.allocated_bytes(), |lease| lease.bytes) + } + + fn allocated_bytes(&self) -> usize { + std::mem::size_of::() + + self.raw.capacity() + + self.leaves.capacity() * std::mem::size_of::() + + self.leaf_hashes.capacity() * 32 + + self + .leaves + .iter() + .map(|leaf| leaf.chunk_sha256.capacity() * 32) + .sum::() + } + + #[cfg(test)] + pub fn verify_chunk(&self, index: u64, bytes: &[u8]) -> Result<(), SnapshotError> { + let want_len = self.map.chunk_len(index).map_err(codec_err)?; + if bytes.len() as u64 != want_len { + return Err(SnapshotError::new( + SnapshotErrorCode::IntegrityError, + "range length disagrees with the verified chunk map", + )); + } + let page = (index / CHUNKS_PER_PAGE as u64) as usize; + let slot = (index % CHUNKS_PER_PAGE as u64) as usize; + let got: [u8; 32] = Sha256::digest(bytes).into(); + if got != self.leaves[page].chunk_sha256[slot] { + return Err(SnapshotError::new( + SnapshotErrorCode::IntegrityError, + "range digest disagrees with the verified chunk map", + )); + } + Ok(()) + } + + #[cfg(test)] pub fn page_count(&self) -> u64 { self.map.page_count } /// One MCL2 leaf plus its bottom-up proof toward `pages_root`. + #[cfg(test)] pub fn leaf_and_proof( &self, page_index: u64, @@ -124,6 +283,7 @@ impl ChunkProjection { /// Raw bytes of chunk `index`, length-checked against the map (spec 07 /// §4: full 1 MiB chunks except the positive-length remainder). + #[cfg(test)] pub fn chunk_bytes(&self, index: u64) -> Result<&[u8], SnapshotError> { let want_len = self.map.chunk_len(index).map_err(codec_err)?; let start = (index as usize) @@ -167,78 +327,6 @@ fn internal(m: &str) -> SnapshotError { SnapshotError::new(SnapshotErrorCode::Internal, m) } -/// Bounded process-wide cache of verified projections. Eviction is pure -/// memory reclaim; the projection is reproducible from Git, so an evicted -/// entry is rebuilt, never an error to the client. -static STAGED: OnceLock> = OnceLock::new(); - -/// 512 MiB cap for staged file bytes in this identity slice. The real -/// persistent RAW_OBJECT locator layout replaces this (spec 10 §4). -const STAGED_CAP_BYTES: usize = 512 * 1024 * 1024; - -struct ProjectionCache { - entries: HashMap<[u8; 32], std::sync::Arc>, - /// Insertion/last-used order for FIFO reclaim. - order: VecDeque<[u8; 32]>, - total_bytes: usize, -} - -impl ProjectionCache { - fn new() -> Self { - ProjectionCache { - entries: HashMap::new(), - order: VecDeque::new(), - total_bytes: 0, - } - } - - fn get(&mut self, id: [u8; 32]) -> Option> { - self.entries.get(&id).cloned() - } - - fn put(&mut self, proj: std::sync::Arc) { - let id = proj.map.file_content_id; - if self.entries.contains_key(&id) { - return; - } - // Reclaim oldest entries until the new one fits. A file larger than - // the cap alone is not cached (rebuild per request) but still - // served correctly. - while self.total_bytes + proj.raw.len() > STAGED_CAP_BYTES - && let Some(victim) = self.order.pop_front() - { - if let Some(v) = self.entries.remove(&victim) { - self.total_bytes = self.total_bytes.saturating_sub(v.raw.len()); - } - } - if proj.raw.len() <= STAGED_CAP_BYTES { - self.total_bytes += proj.raw.len(); - self.order.push_back(id); - self.entries.insert(id, proj); - } - } -} - -/// Return the cached projection, or build one via `load` (which must resolve -/// the file in the fixed view and return its verified bytes). -pub async fn get_or_project( - content_id: [u8; 32], - load: F, -) -> Result, SnapshotError> -where - F: FnOnce() -> Fut, - Fut: std::future::Future, SnapshotError>>, -{ - let cache = STAGED.get_or_init(|| Mutex::new(ProjectionCache::new())); - if let Some(p) = cache.lock().unwrap().get(content_id) { - return Ok(p); - } - let raw = load().await?; - let proj = std::sync::Arc::new(ChunkProjection::build(content_id, raw)?); - cache.lock().unwrap().put(proj.clone()); - Ok(proj) -} - #[cfg(test)] mod tests { use super::*; diff --git a/src/ceres/snapshot/chunks_stream.rs b/src/ceres/snapshot/chunks_stream.rs new file mode 100644 index 00000000..35e33e97 --- /dev/null +++ b/src/ceres/snapshot/chunks_stream.rs @@ -0,0 +1,320 @@ +use std::sync::{Arc, OnceLock}; + +use futures::StreamExt; +use tokio::sync::Semaphore; + +use super::*; +use crate::orbit_api::object_storage::ObjectByteStream; + +const STREAM_ITEM_MAX_BYTES: usize = 8 * 1024 * 1024; +const MAX_FILE_BYTES: u64 = 8 * 1024 * 1024 * 1024 * 1024; +const MAX_BUILDERS: usize = 4; +#[cfg(test)] +const STAGED_CAP_BYTES: usize = 512 * 1024 * 1024; +const CONSTRUCTION_ALLOWANCE: usize = 64 * 1024; + +#[cfg(test)] +pub(super) fn reserved_bytes(size: u64) -> Result { + reserved_bytes_for(size, size <= STAGED_CAP_BYTES as u64) +} + +pub(super) fn reserved_map_bytes(size: u64) -> Result { + reserved_bytes_for(size, false) +} + +fn reserved_bytes_for(size: u64, inline: bool) -> Result { + if size == 0 || size > MAX_FILE_BYTES { + return Err(SnapshotError::new( + if size == 0 { + SnapshotErrorCode::ScopeInvalid + } else { + SnapshotErrorCode::LimitExceeded + }, + "file is outside the chunk projection profile", + )); + } + let chunks = size.div_ceil(CHUNK_SIZE as u64); + let pages = chunks.div_ceil(CHUNKS_PER_PAGE as u64); + let inline = if inline { size } else { 0 }; + let bytes = chunks + .checked_mul(32) + .and_then(|n| { + pages + .checked_mul((std::mem::size_of::() + 32) as u64) + .and_then(|pages| n.checked_add(pages)) + }) + .and_then(|n| n.checked_add(inline)) + .and_then(|n| n.checked_add(CONSTRUCTION_ALLOWANCE as u64)) + .and_then(|n| usize::try_from(n).ok()) + .ok_or_else(|| internal("chunk projection memory arithmetic overflow"))?; + Ok(bytes) +} + +#[cfg(test)] +async fn build_stream( + content_id: [u8; 32], + size: u64, + input: ObjectByteStream, + lease: MemoryLease, +) -> Result { + build_stream_with_inline( + content_id, + size, + input, + lease, + size <= STAGED_CAP_BYTES as u64, + None, + ) + .await +} + +pub(super) async fn build_source_stream( + content_id: [u8; 32], + size: u64, + open: F, + budget: &Arc, + admission: &crate::jupiter::storage::native_chunk_map::retention::ChunkMapInstall, +) -> Result +where + F: FnOnce() -> Fut, + Fut: std::future::Future>, +{ + static BUILDERS: OnceLock> = OnceLock::new(); + build_source_with_resources( + content_id, + size, + open, + budget, + BUILDERS.get_or_init(|| Arc::new(Semaphore::new(MAX_BUILDERS))), + Some(admission), + ) + .await +} + +async fn build_source_with_resources( + content_id: [u8; 32], + size: u64, + open: F, + budget: &Arc, + builders: &Arc, + admission: Option<&crate::jupiter::storage::native_chunk_map::retention::ChunkMapInstall>, +) -> Result +where + F: FnOnce() -> Fut, + Fut: std::future::Future>, +{ + let _builder = builders.clone().try_acquire_owned().map_err(|_| { + SnapshotError::new( + SnapshotErrorCode::TemporaryUnavailable, + "chunk map builders are occupied", + ) + })?; + let lease = budget.reserve(source_reservation_bytes(size)?)?; + let input = if let Some(admission) = admission { + admission.ensure_live().await?; + let result = admission.await_backend(open()).await?; + admission.ensure_live().await?; + result? + } else { + open().await? + }; + build_stream_with_inline(content_id, size, input, lease, false, admission).await +} + +pub(super) fn source_reservation_bytes(size: u64) -> Result { + let map_bytes = reserved_map_bytes(size)?; + let pages = size + .div_ceil(CHUNK_SIZE as u64) + .div_ceil(CHUNKS_PER_PAGE as u64); + let nodes = pages + .checked_mul(2) + .and_then(|n| n.checked_sub(1)) + .and_then(|n| { + n.checked_mul(std::mem::size_of::() as u64) + }) + .and_then(|n| usize::try_from(n).ok()) + .ok_or_else(|| internal("chunk map install reservation overflow"))?; + map_bytes + .checked_add(nodes) + .and_then(|n| n.checked_add(4 * 1024 * 1024)) + .ok_or_else(|| internal("chunk map install reservation overflow")) +} + +async fn build_stream_with_inline( + content_id: [u8; 32], + size: u64, + mut input: ObjectByteStream, + lease: MemoryLease, + inline: bool, + admission: Option<&crate::jupiter::storage::native_chunk_map::retention::ChunkMapInstall>, +) -> Result { + let chunk_count = size.div_ceil(CHUNK_SIZE as u64); + let page_count = chunk_count.div_ceil(CHUNKS_PER_PAGE as u64); + let mut raw = Vec::new(); + let mut leaves = Vec::new(); + let mut leaf_hashes = Vec::new(); + let mut current = Vec::new(); + if inline { + raw.try_reserve_exact(size as usize) + .map_err(allocation_error)?; + } + leaves + .try_reserve_exact(page_count as usize) + .map_err(allocation_error)?; + leaf_hashes + .try_reserve_exact(page_count as usize) + .map_err(allocation_error)?; + current + .try_reserve_exact(chunk_count.min(CHUNKS_PER_PAGE as u64) as usize) + .map_err(allocation_error)?; + let mut full_hash = Sha256::new(); + let mut chunk_hash = Sha256::new(); + let mut received = 0u64; + let mut chunk_bytes = 0usize; + let mut digests = 0u64; + let mut since_yield = 0usize; + let mut empty_parts = 0usize; + loop { + let next = if let Some(admission) = admission { + admission.ensure_live().await?; + let next = admission.await_backend(input.next()).await?; + admission.ensure_live().await?; + next + } else { + input.next().await + }; + let Some(part) = next else { break }; + let bytes = part.map_err(|error| { + tracing::warn!(error = %error, "fixed-view chunk projection stream failed"); + SnapshotError::new( + SnapshotErrorCode::ObjectUnavailable, + "content stream failed", + ) + })?; + if bytes.len() as u64 > size - received { + return Err(length_error()); + } + // A visible-item admission limit cannot bound a producer's backing + // allocation, buffers or work surviving cancellation. + if bytes.len() > STREAM_ITEM_MAX_BYTES { + return Err(SnapshotError::new( + SnapshotErrorCode::TemporaryUnavailable, + "content producer item exceeds the consumer processing limit", + )); + } + full_hash.update(&bytes); + if inline { + raw.extend_from_slice(&bytes); + } + received += bytes.len() as u64; + if let Some(admission) = admission { + if !bytes.is_empty() { + admission.record_progress(received).await?; + } else { + empty_parts += 1; + if empty_parts == 32 { + tokio::task::yield_now().await; + admission.ensure_live().await?; + empty_parts = 0; + } + } + } + let mut rest = bytes.as_ref(); + while !rest.is_empty() { + let len = rest.len().min(CHUNK_SIZE as usize - chunk_bytes); + chunk_hash.update(&rest[..len]); + chunk_bytes += len; + rest = &rest[len..]; + if chunk_bytes == CHUNK_SIZE as usize { + current.push(chunk_hash.finalize_reset().into()); + digests += 1; + chunk_bytes = 0; + if current.len() == CHUNKS_PER_PAGE { + append_leaf(&mut leaves, &mut leaf_hashes, std::mem::take(&mut current))?; + if digests < chunk_count { + current + .try_reserve_exact( + (chunk_count - digests).min(CHUNKS_PER_PAGE as u64) as usize + ) + .map_err(allocation_error)?; + } + } + } + } + since_yield += bytes.len(); + if since_yield >= STREAM_ITEM_MAX_BYTES { + // A ready stream still cooperates with cancellation and the + // executor while hashing a large file. No detached CPU worker. + tokio::task::yield_now().await; + since_yield = 0; + } + } + if received != size { + return Err(length_error()); + } + let got: [u8; 32] = full_hash.finalize().into(); + if got != content_id { + return Err(SnapshotError::new( + SnapshotErrorCode::DigestMismatch, + "chunk projection input does not hash to content_id", + )); + } + if chunk_bytes != 0 { + current.push(chunk_hash.finalize().into()); + digests += 1; + } + if !current.is_empty() { + append_leaf(&mut leaves, &mut leaf_hashes, current)?; + } + if digests != chunk_count || leaves.len() as u64 != page_count { + return Err(internal( + "streamed chunk boundaries disagree with file size", + )); + } + let pages_root = mst2_codec::chunkmap::merkle_root(&leaf_hashes).map_err(codec_err)?; + let map = ChunkMap::new(content_id, size, pages_root).map_err(codec_err)?; + let projection = ChunkProjection { + map_id: map.map_id(), + map, + leaves, + leaf_hashes, + raw, + memory: Some(lease), + }; + if projection.allocated_bytes() > projection.retained_bytes() { + return Err(internal("projection allocation exceeds its reservation")); + } + Ok(projection) +} + +fn append_leaf( + leaves: &mut Vec, + hashes: &mut Vec<[u8; 32]>, + chunk_sha256: Vec<[u8; 32]>, +) -> Result<(), SnapshotError> { + let leaf = ChunkLeaf { + page_index: leaves.len() as u64, + chunk_sha256, + }; + hashes.push(leaf.leaf_hash().map_err(codec_err)?); + leaves.push(leaf); + Ok(()) +} + +fn allocation_error(_: std::collections::TryReserveError) -> SnapshotError { + SnapshotError::new( + SnapshotErrorCode::TemporaryUnavailable, + "chunk projection allocation could not be admitted", + ) +} + +fn length_error() -> SnapshotError { + SnapshotError::new( + SnapshotErrorCode::IntegrityError, + "fixed blob length disagrees with its verified size fact", + ) +} + +#[cfg(test)] +#[path = "chunks_stream_tests.rs"] +mod tests; diff --git a/src/ceres/snapshot/chunks_stream_tests.rs b/src/ceres/snapshot/chunks_stream_tests.rs new file mode 100644 index 00000000..d6f7e042 --- /dev/null +++ b/src/ceres/snapshot/chunks_stream_tests.rs @@ -0,0 +1,341 @@ +use std::{ + io, + sync::atomic::{AtomicUsize, Ordering}, + time::Duration, +}; + +use bytes::Bytes; +use tokio::{sync::Notify, time::timeout}; + +use super::*; +use crate::ceres::snapshot::content_budget::MemoryBudget; + +fn stream(parts: Vec>) -> ObjectByteStream { + Box::pin(futures::stream::iter(parts)) +} + +#[tokio::test] +async fn exact_source_builder_admits_all_install_workspace_before_open_and_cancellation_releases_it() + { + let raw = Bytes::from_static(b"exact source body"); + let digest: [u8; 32] = Sha256::digest(&raw).into(); + let weight = source_reservation_bytes(raw.len() as u64).unwrap(); + assert!(weight > 4 * 1024 * 1024); + let budget = MemoryBudget::new(weight); + let builders = Arc::new(Semaphore::new(1)); + let opens = Arc::new(AtomicUsize::new(0)); + let held = budget.reserve(weight).unwrap(); + let error = build_source_with_resources( + digest, + raw.len() as u64, + || async { + opens.fetch_add(1, Ordering::SeqCst); + Ok(stream(vec![Ok(raw.clone())])) + }, + &budget, + &builders, + None, + ) + .await + .err() + .unwrap(); + assert_eq!(error.code, SnapshotErrorCode::TemporaryUnavailable); + assert_eq!(opens.load(Ordering::SeqCst), 0); + assert_eq!(builders.available_permits(), 1); + drop(held); + let held_builder = builders.clone().acquire_owned().await.unwrap(); + assert_eq!( + build_source_with_resources( + digest, + raw.len() as u64, + || async { + opens.fetch_add(1, Ordering::SeqCst); + Ok(stream(vec![Ok(raw.clone())])) + }, + &budget, + &builders, + None + ) + .await + .err() + .unwrap() + .code, + SnapshotErrorCode::TemporaryUnavailable + ); + assert_eq!(opens.load(Ordering::SeqCst), 0); + assert_eq!(budget.used(), 0); + drop(held_builder); + let entered = Arc::new(Notify::new()); + let task_budget = budget.clone(); + let task_builders = builders.clone(); + let task_entered = entered.clone(); + let task_opens = opens.clone(); + let drops = Arc::new(AtomicUsize::new(0)); + let producer_owner = DropCount(drops.clone()); + let size = raw.len() as u64; + let task = tokio::spawn(async move { + build_source_with_resources( + digest, + size, + || async { + task_opens.fetch_add(1, Ordering::SeqCst); + task_entered.notify_one(); + Ok(Box::pin(futures::stream::unfold( + producer_owner, + |owner| async move { + futures::future::pending::<()>().await; + Some((Ok(Bytes::new()), owner)) + }, + )) as ObjectByteStream) + }, + &task_budget, + &task_builders, + None, + ) + .await + }); + timeout(Duration::from_secs(5), entered.notified()) + .await + .unwrap(); + assert_eq!(budget.used(), weight); + assert_eq!(builders.available_permits(), 0); + task.abort(); + assert!(task.await.err().unwrap().is_cancelled()); + assert_eq!(budget.used(), 0); + assert_eq!(builders.available_permits(), 1); + assert_eq!(drops.load(Ordering::SeqCst), 1); + let projection = build_source_with_resources( + digest, + size, + || async { + opens.fetch_add(1, Ordering::SeqCst); + Ok(stream(vec![Ok(raw.clone())])) + }, + &budget, + &builders, + None, + ) + .await + .unwrap(); + assert!(!projection.has_inline_bytes()); + assert_eq!(projection.map.file_content_id, digest); + assert_eq!(opens.load(Ordering::SeqCst), 2); + assert_eq!(builders.available_permits(), 1); + assert_eq!(budget.used(), weight); + projection.verify_chunk(0, &raw).unwrap(); + drop(projection); + assert_eq!(budget.used(), 0); +} + +async fn project(raw: &[u8], parts: Vec>) -> ChunkProjection { + let budget = MemoryBudget::new(reserved_bytes(raw.len() as u64).unwrap()); + let lease = budget + .reserve(reserved_bytes(raw.len() as u64).unwrap()) + .unwrap(); + build_stream( + Sha256::digest(raw).into(), + raw.len() as u64, + stream(parts), + lease, + ) + .await + .unwrap() +} + +#[tokio::test] +async fn arbitrary_fragmentation_preserves_canonical_map_leaves_and_full_final_chunk() { + for size in [ + 1, + CHUNK_SIZE as usize - 1, + CHUNK_SIZE as usize, + CHUNK_SIZE as usize + 7, + 2 * CHUNK_SIZE as usize, + ] { + let raw: Vec = (0..size).map(|i| (i % 251) as u8).collect(); + let parts = raw + .chunks(997) + .map(|part| Ok(Bytes::copy_from_slice(part))) + .collect(); + let projection = project(&raw, parts).await; + let oracle = ChunkProjection::build(Sha256::digest(&raw).into(), raw.clone()).unwrap(); + assert_eq!(projection.map, oracle.map); + assert_eq!(projection.map_id, oracle.map_id); + assert_eq!(projection.leaves, oracle.leaves); + assert_eq!(projection.leaf_hashes, oracle.leaf_hashes); + assert!(projection.has_inline_bytes()); + for index in 0..projection.map.chunk_count { + assert_eq!( + projection.chunk_bytes(index).unwrap(), + oracle.chunk_bytes(index).unwrap() + ); + let (leaf, proof) = projection + .leaf_and_proof(index / CHUNKS_PER_PAGE as u64) + .unwrap(); + mst2_codec::chunkmap::verify_leaf( + projection.page_count(), + leaf.page_index, + leaf.leaf_hash().unwrap(), + &proof, + projection.map.pages_root, + ) + .unwrap(); + } + } +} + +#[tokio::test] +async fn large_production_threshold_retains_only_map_and_verifies_every_page() { + // Reuse one producer allocation; never create the >512 MiB raw Vec. + let block = Bytes::from(vec![37; CHUNK_SIZE as usize]); + let size = STAGED_CAP_BYTES as u64 + 7; + let mut full = Sha256::new(); + for _ in 0..512 { + full.update(&block); + } + full.update(&block[..7]); + let content_id: [u8; 32] = full.finalize().into(); + let input = Box::pin(futures::stream::iter((0..513).map(move |index| { + Ok(if index == 512 { + block.slice(..7) + } else { + block.clone() + }) + }))); + let weight = reserved_bytes(size).unwrap(); + let budget = MemoryBudget::new(weight); + let lease = budget.reserve(weight).unwrap(); + let projection = build_stream(content_id, size, input, lease).await.unwrap(); + assert!(!projection.has_inline_bytes()); + assert_eq!(projection.raw.capacity(), 0); + assert_eq!(projection.map.chunk_count, 513); + assert_eq!(projection.page_count(), 3); + assert!(projection.allocated_bytes() < 128 * 1024); + assert_eq!(budget.used(), weight); + let full_chunk: [u8; 32] = Sha256::digest(vec![37; CHUNK_SIZE as usize]).into(); + let last_chunk: [u8; 32] = Sha256::digest([37; 7]).into(); + assert_eq!(projection.leaves[0].chunk_sha256, vec![full_chunk; 256]); + assert_eq!(projection.leaves[1].chunk_sha256, vec![full_chunk; 256]); + assert_eq!(projection.leaves[2].chunk_sha256, vec![last_chunk]); + for page in 0..3 { + let (leaf, proof) = projection.leaf_and_proof(page).unwrap(); + mst2_codec::chunkmap::verify_leaf( + 3, + page, + leaf.leaf_hash().unwrap(), + &proof, + projection.map.pages_root, + ) + .unwrap(); + } + projection.verify_chunk(512, &[37; 7]).unwrap(); + assert_eq!( + projection.verify_chunk(512, &[38; 7]).unwrap_err().code, + SnapshotErrorCode::IntegrityError + ); + drop(projection); + assert_eq!(budget.used(), 0); +} + +#[tokio::test] +async fn exact_eof_wrong_hash_growth_and_late_error_never_publish_a_projection() { + let cases = [ + ( + vec![Ok(Bytes::from_static(b"ab"))], + SnapshotErrorCode::IntegrityError, + ), + ( + vec![Ok(Bytes::from_static(b"abc")), Ok(Bytes::from_static(b"d"))], + SnapshotErrorCode::IntegrityError, + ), + ( + vec![Ok(Bytes::from_static(b"abd"))], + SnapshotErrorCode::DigestMismatch, + ), + ( + vec![ + Ok(Bytes::from_static(b"abc")), + Err(io::Error::other("late")), + ], + SnapshotErrorCode::ObjectUnavailable, + ), + ]; + for (parts, code) in cases { + let budget = MemoryBudget::new(reserved_bytes(3).unwrap()); + let lease = budget.reserve(reserved_bytes(3).unwrap()).unwrap(); + let failure = build_stream(Sha256::digest(b"abc").into(), 3, stream(parts), lease) + .await + .err() + .unwrap(); + assert_eq!(failure.code, code); + assert_eq!(budget.used(), 0); + } +} + +#[tokio::test] +async fn overlong_producer_is_rejected_before_hash_copy_and_tail_poll() { + let tail = Arc::new(AtomicUsize::new(0)); + let input = Box::pin(futures::stream::unfold( + (false, tail.clone()), + |(sent, tail)| async move { + if sent { + tail.fetch_add(1, Ordering::SeqCst); + Some((Err(io::Error::other("must not poll tail")), (true, tail))) + } else { + Some((Ok(Bytes::from_static(b"abcd")), (true, tail))) + } + }, + )); + let budget = MemoryBudget::new(reserved_bytes(3).unwrap()); + let lease = budget.reserve(reserved_bytes(3).unwrap()).unwrap(); + assert_eq!( + build_stream(Sha256::digest(b"abc").into(), 3, input, lease) + .await + .err() + .unwrap() + .code, + SnapshotErrorCode::IntegrityError + ); + assert_eq!(tail.load(Ordering::SeqCst), 0); + assert_eq!(budget.used(), 0); +} + +#[tokio::test] +async fn visible_producer_item_limit_does_not_collect_a_large_item() { + let size = STREAM_ITEM_MAX_BYTES + 1; + let budget = MemoryBudget::new(reserved_bytes(size as u64).unwrap()); + let lease = budget + .reserve(reserved_bytes(size as u64).unwrap()) + .unwrap(); + let input = stream(vec![Ok(Bytes::from(vec![1; size]))]); + assert_eq!( + build_stream([0; 32], size as u64, input, lease) + .await + .err() + .unwrap() + .code, + SnapshotErrorCode::TemporaryUnavailable + ); + assert_eq!(budget.used(), 0); +} + +struct DropCount(Arc); +impl Drop for DropCount { + fn drop(&mut self) { + self.0.fetch_add(1, Ordering::SeqCst); + } +} + +#[test] +fn maximum_file_metadata_fits_and_protocol_or_quota_rejection_needs_no_source() { + assert!(reserved_bytes(MAX_FILE_BYTES).unwrap() < 300 * 1024 * 1024); + assert_eq!( + reserved_bytes(0).unwrap_err().code, + SnapshotErrorCode::ScopeInvalid + ); + assert_eq!( + reserved_bytes(MAX_FILE_BYTES + 1).unwrap_err().code, + SnapshotErrorCode::LimitExceeded + ); + assert!(reserved_bytes(STAGED_CAP_BYTES as u64).unwrap() > STAGED_CAP_BYTES); + assert!(reserved_bytes(STAGED_CAP_BYTES as u64 + 1).unwrap() < 128 * 1024); +} diff --git a/src/ceres/snapshot/content_budget.rs b/src/ceres/snapshot/content_budget.rs new file mode 100644 index 00000000..18e16a51 --- /dev/null +++ b/src/ceres/snapshot/content_budget.rs @@ -0,0 +1,147 @@ +//! Ownership-based consumer memory admission. Backend allocations and allocator +//! overhead are separate; these credits are not a process RSS guarantee. + +use std::sync::{ + Arc, OnceLock, + atomic::{AtomicUsize, Ordering}, +}; + +use super::error::{SnapshotError, SnapshotErrorCode}; + +pub(crate) const PROJECTION_LIVE_BYTES: usize = 1024 * 1024 * 1024; +const RESPONSE_LIVE_BYTES: usize = 512 * 1024 * 1024; +const RANGE_SCRATCH_BYTES: usize = 128 * 1024 * 1024; +pub(crate) const RANGE_WORK_BYTES: usize = 8 * 1024 * 1024; + +pub(crate) struct MemoryBudget { + limit: usize, + used: AtomicUsize, +} + +impl MemoryBudget { + pub(crate) fn new(limit: usize) -> Arc { + Arc::new(Self { + limit, + used: AtomicUsize::new(0), + }) + } + + pub(crate) fn reserve(self: &Arc, bytes: usize) -> Result { + if bytes > self.limit { + return Err(SnapshotError::new( + SnapshotErrorCode::LimitExceeded, + "content operation exceeds its consumer memory budget", + )); + } + let mut used = self.used.load(Ordering::Acquire); + loop { + if bytes > self.limit - used { + return Err(SnapshotError::new( + SnapshotErrorCode::TemporaryUnavailable, + "content consumer memory budget is occupied", + )); + } + match self.used.compare_exchange_weak( + used, + used + bytes, + Ordering::AcqRel, + Ordering::Acquire, + ) { + Ok(_) => { + return Ok(MemoryLease { + budget: self.clone(), + bytes, + }); + } + Err(actual) => used = actual, + } + } + } + + #[cfg(test)] + pub(crate) fn used(&self) -> usize { + self.used.load(Ordering::Acquire) + } +} + +pub(crate) struct MemoryLease { + budget: Arc, + pub(crate) bytes: usize, +} + +impl Drop for MemoryLease { + fn drop(&mut self) { + self.budget.used.fetch_sub(self.bytes, Ordering::AcqRel); + } +} + +pub(crate) fn projection_budget() -> &'static Arc { + static BUDGET: OnceLock> = OnceLock::new(); + BUDGET.get_or_init(|| MemoryBudget::new(PROJECTION_LIVE_BYTES)) +} + +pub(crate) fn reserve_response(bytes: usize) -> Result { + response_budget().reserve(bytes) +} + +pub(crate) fn response_budget() -> &'static Arc { + static BUDGET: OnceLock> = OnceLock::new(); + BUDGET.get_or_init(|| MemoryBudget::new(RESPONSE_LIVE_BYTES)) +} + +pub(crate) fn range_budget() -> &'static Arc { + static BUDGET: OnceLock> = OnceLock::new(); + BUDGET.get_or_init(|| MemoryBudget::new(RANGE_SCRATCH_BYTES)) +} + +/// Every encoded frame owns shared credit, including after the body passes a +/// frame to a transport that retains or clones its Bytes. +pub(crate) struct BudgetedFrame { + pub(crate) bytes: Vec, + pub(crate) lease: Arc, +} + +impl AsRef<[u8]> for BudgetedFrame { + fn as_ref(&self) -> &[u8] { + &self.bytes + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn frames_retain_credit_until_last_transport_clone_drops() { + let budget = MemoryBudget::new(64); + let lease = Arc::new(budget.reserve(64).unwrap()); + let frame = bytes::Bytes::from_owner(BudgetedFrame { + bytes: vec![7; 32], + lease: lease.clone(), + }); + let transport = frame.clone(); + drop(lease); + drop(frame); + assert_eq!(budget.used(), 64); + assert!(budget.reserve(1).is_err()); + drop(transport); + assert_eq!(budget.used(), 0); + assert!(budget.reserve(64).is_ok()); + } + + #[test] + fn quota_rejects_without_waiting_for_a_callers_own_credits() { + let budget = MemoryBudget::new(12); + let first = budget.reserve(8).unwrap(); + assert_eq!( + budget.reserve(5).err().unwrap().code, + SnapshotErrorCode::TemporaryUnavailable + ); + assert_eq!( + budget.reserve(13).err().unwrap().code, + SnapshotErrorCode::LimitExceeded + ); + drop(first); + assert_eq!(budget.used(), 0); + } +} diff --git a/src/ceres/snapshot/error.rs b/src/ceres/snapshot/error.rs index 4fe54274..4c427126 100644 --- a/src/ceres/snapshot/error.rs +++ b/src/ceres/snapshot/error.rs @@ -18,6 +18,10 @@ pub enum SnapshotErrorCode { ScopeForbidden, ViewNotFound, SnapshotNotReady, + /// Verified fixed-object size/digest facts are not yet available. + MetadataNotReady, + /// A bounded operation could not complete in time (spec 14 §5). + TemporaryUnavailable, /// The fixed view no longer exists: spec 14 §5 SNAPSHOT_GONE (410). SnapshotGone, PathNotFound, @@ -49,6 +53,8 @@ impl SnapshotErrorCode { SnapshotErrorCode::ScopeForbidden => "SCOPE_FORBIDDEN", SnapshotErrorCode::ViewNotFound => "VIEW_NOT_FOUND", SnapshotErrorCode::SnapshotNotReady => "SNAPSHOT_NOT_READY", + SnapshotErrorCode::MetadataNotReady => "METADATA_NOT_READY", + SnapshotErrorCode::TemporaryUnavailable => "TEMPORARY_UNAVAILABLE", SnapshotErrorCode::SnapshotGone => "SNAPSHOT_GONE", SnapshotErrorCode::PathNotFound => "PATH_NOT_FOUND", SnapshotErrorCode::NotDirectory => "NOT_DIRECTORY", @@ -82,7 +88,9 @@ impl SnapshotErrorCode { SnapshotErrorCode::ViewNotFound | SnapshotErrorCode::PathNotFound | SnapshotErrorCode::LeaseUnknown => 404, - SnapshotErrorCode::SnapshotNotReady => 503, + SnapshotErrorCode::SnapshotNotReady + | SnapshotErrorCode::MetadataNotReady + | SnapshotErrorCode::TemporaryUnavailable => 503, SnapshotErrorCode::ObjectUnavailable => 503, SnapshotErrorCode::NotDirectory | SnapshotErrorCode::Conflict diff --git a/src/ceres/snapshot/metadata_install.rs b/src/ceres/snapshot/metadata_install.rs new file mode 100644 index 00000000..ab946290 --- /dev/null +++ b/src/ceres/snapshot/metadata_install.rs @@ -0,0 +1,463 @@ +//! Storage-owned identity for a durable native metadata installation. +//! This plan grants no publication, lease or content-retention authority. + +use std::collections::{BTreeMap, BTreeSet}; + +use mst2_codec::metapage::{HEADER_LEN, PAGE_MAX_BYTES}; +use sha2::{Digest, Sha256}; + +use super::{ + error::{SnapshotError, SnapshotErrorCode}, + retention_dag::{MetadataDagLimits, MetadataPageId, ValidatedMetadataDag}, +}; + +const DOMAIN: &[u8] = b"mega.mst2.metadata-install.v1\0"; +pub(crate) const MAX_PLAN_BYTES: usize = 2 * 1024 * 1024; + +#[derive(Debug, Clone, PartialEq, Eq)] +pub(crate) struct MetadataInstallIdentity { + pub source_domain: String, + pub tagged_root_tree_oid: String, + pub scope: String, + pub schema_version: u16, + pub metadata_codec: u16, + pub materialization_policy: u16, + pub fs_semantics: u16, + pub access_projection: u16, + pub verification_revision: i32, + pub projection_revision: u16, +} + +#[derive(Debug, Clone, PartialEq, Eq)] +pub(crate) struct MetadataInstallPlan { + pub identity: MetadataInstallIdentity, + pub root: MetadataPageId, + pub pages: BTreeMap, + pub edges: BTreeSet<(MetadataPageId, MetadataPageId)>, + pub total_bytes: u64, +} + +impl MetadataInstallPlan { + pub(super) fn from_validated( + identity: MetadataInstallIdentity, + dag: &ValidatedMetadataDag, + ) -> Result { + dag.check_limits(MetadataDagLimits::default())?; + let edges = dag + .edges() + .iter() + .map(|edge| Ok((parse_node_id(&edge.parent)?, parse_node_id(&edge.child)?))) + .collect::>()?; + let plan = Self { + identity, + root: dag.root(), + pages: dag + .payloads() + .iter() + .map(|page| (page.id, page.size)) + .collect(), + edges, + total_bytes: dag.payload_bytes(), + }; + plan.validate()?; + Ok(plan) + } + + pub fn digest(&self) -> Result<[u8; 32], SnapshotError> { + Ok(Sha256::digest(self.encode()?).into()) + } + + pub fn encode(&self) -> Result, SnapshotError> { + self.validate()?; + let mut bytes = DOMAIN.to_vec(); + bytes.extend_from_slice(&1u16.to_be_bytes()); + for value in [ + &self.identity.source_domain, + &self.identity.tagged_root_tree_oid, + &self.identity.scope, + ] { + bytes.extend_from_slice(&(value.len() as u32).to_be_bytes()); + bytes.extend_from_slice(value.as_bytes()); + } + for value in [ + self.identity.schema_version, + self.identity.metadata_codec, + self.identity.materialization_policy, + self.identity.fs_semantics, + self.identity.access_projection, + ] { + bytes.extend_from_slice(&value.to_be_bytes()); + } + bytes.extend_from_slice(&self.identity.verification_revision.to_be_bytes()); + bytes.extend_from_slice(&self.identity.projection_revision.to_be_bytes()); + bytes.extend_from_slice(&self.root); + bytes.extend_from_slice(&(self.pages.len() as u32).to_be_bytes()); + for (id, size) in &self.pages { + bytes.extend_from_slice(id); + bytes.extend_from_slice(&size.to_be_bytes()); + } + bytes.extend_from_slice(&(self.edges.len() as u32).to_be_bytes()); + for (parent, child) in &self.edges { + bytes.extend_from_slice(parent); + bytes.extend_from_slice(child); + } + if bytes.len() > MAX_PLAN_BYTES { + return Err(limit("metadata installation plan exceeds its byte budget")); + } + Ok(bytes) + } + + pub fn decode(bytes: &[u8], expected_digest: &[u8; 32]) -> Result { + if bytes.len() > MAX_PLAN_BYTES { + return Err(limit( + "stored metadata installation plan exceeds its byte budget", + )); + } + if Sha256::digest(bytes).as_slice() != expected_digest { + return Err(integrity( + "stored metadata installation plan digest mismatch", + )); + } + let mut reader = PlanReader(bytes); + if reader.take(DOMAIN.len())? != DOMAIN || reader.u16()? != 1 { + return Err(integrity("unsupported metadata installation plan encoding")); + } + let identity = MetadataInstallIdentity { + source_domain: reader.string(64)?, + tagged_root_tree_oid: reader.string(128)?, + scope: reader.string(4096)?, + schema_version: reader.u16()?, + metadata_codec: reader.u16()?, + materialization_policy: reader.u16()?, + fs_semantics: reader.u16()?, + access_projection: reader.u16()?, + verification_revision: i32::from_be_bytes(reader.array()?), + projection_revision: reader.u16()?, + }; + let root = reader.array()?; + let count = reader.count(MetadataDagLimits::default().nodes)?; + let mut pages = BTreeMap::new(); + let mut total_bytes = 0u64; + let mut last = None; + for _ in 0..count { + let id = reader.array()?; + let size = reader.u64()?; + if last.is_some_and(|previous| previous >= id) { + return Err(integrity( + "metadata installation pages are not uniquely ordered", + )); + } + last = Some(id); + total_bytes = total_bytes + .checked_add(size) + .ok_or_else(|| limit("metadata installation byte count overflow"))?; + pages.insert(id, size); + } + let count = reader.count(MetadataDagLimits::default().edges)?; + let mut edges = BTreeSet::new(); + let mut last = None; + for _ in 0..count { + let edge = (reader.array()?, reader.array()?); + if last.is_some_and(|previous| previous >= edge) { + return Err(integrity( + "metadata installation edges are not uniquely ordered", + )); + } + last = Some(edge); + edges.insert(edge); + } + if !reader.0.is_empty() { + return Err(integrity("metadata installation plan has trailing bytes")); + } + let plan = Self { + identity, + root, + pages, + edges, + total_bytes, + }; + plan.validate()?; + if plan.encode()?.as_slice() != bytes { + return Err(integrity("metadata installation plan is not canonical")); + } + Ok(plan) + } + + fn validate(&self) -> Result<(), SnapshotError> { + let identity = &self.identity; + if identity.source_domain != "native-git" + || identity.schema_version != mst2_codec::descriptor::SCHEMA_VERSION + || identity.metadata_codec != mst2_codec::descriptor::METADATA_CODEC + || identity.materialization_policy + != mst2_codec::descriptor::MATERIALIZATION_POLICY_GIT_RAW_V1 + || identity.fs_semantics != mst2_codec::descriptor::FS_SEMANTICS_LINUX_CODE_V1 + || identity.access_projection != mst2_codec::descriptor::ACCESS_PROJECTION_EXACT_FULL + || identity.verification_revision + != crate::jupiter::storage::mono_storage::MST2_VERIFICATION_VERSION + || identity.projection_revision != 1 + { + return Err(integrity( + "unsupported stored native metadata installation profile", + )); + } + let tagged = identity.tagged_root_tree_oid.as_str(); + let (kind, hex) = tagged + .split_once(':') + .ok_or_else(|| integrity("untagged fixed root tree"))?; + let length = match kind { + "sha1" => 40, + "sha256" | "blake3" => 64, + _ => return Err(integrity("unsupported fixed root tree hash kind")), + }; + if hex.len() != length + || !hex + .bytes() + .all(|byte| byte.is_ascii_digit() || (b'a'..=b'f').contains(&byte)) + { + return Err(integrity("noncanonical fixed root tree identity")); + } + super::view::validate_scope_relative_path(&identity.scope) + .map_err(|_| integrity("noncanonical metadata installation scope"))?; + let limits = MetadataDagLimits::default(); + if self.pages.is_empty() + || self.pages.len() > limits.nodes + || self.edges.len() > limits.edges + || self.total_bytes > limits.payload_bytes + { + return Err(limit("metadata installation group exceeds its budget")); + } + if !self.pages.contains_key(&self.root) + || self + .pages + .values() + .any(|size| *size < HEADER_LEN as u64 || *size > PAGE_MAX_BYTES as u64) + || self + .pages + .values() + .try_fold(0u64, |total, size| total.checked_add(*size)) + != Some(self.total_bytes) + { + return Err(integrity( + "metadata installation root or page size is invalid", + )); + } + let mut incoming: BTreeMap<_, usize> = self.pages.keys().map(|id| (*id, 0)).collect(); + let mut children: BTreeMap<_, Vec<_>> = BTreeMap::new(); + for (parent, child) in &self.edges { + if parent == child || !self.pages.contains_key(parent) { + return Err(integrity("invalid metadata installation edge")); + } + *incoming + .get_mut(child) + .ok_or_else(|| integrity("metadata installation child is missing"))? += 1; + children.entry(*parent).or_default().push(*child); + } + let mut ready: Vec<_> = incoming + .iter() + .filter_map(|(id, count)| (*count == 0).then_some(*id)) + .collect(); + let mut visited = 0; + while let Some(parent) = ready.pop() { + visited += 1; + for child in children.get(&parent).into_iter().flatten() { + let count = incoming + .get_mut(child) + .ok_or_else(|| integrity("missing installation child"))?; + *count -= 1; + if *count == 0 { + ready.push(*child); + } + } + } + if visited != self.pages.len() { + return Err(integrity("cyclic metadata installation plan")); + } + let mut reachable = BTreeSet::new(); + let mut pending = vec![self.root]; + while let Some(id) = pending.pop() { + if reachable.insert(id) { + pending.extend(children.get(&id).into_iter().flatten()); + } + } + if reachable.len() != self.pages.len() { + return Err(integrity("unreachable metadata installation pages")); + } + Ok(()) + } +} + +struct PlanReader<'a>(&'a [u8]); + +impl<'a> PlanReader<'a> { + fn take(&mut self, count: usize) -> Result<&'a [u8], SnapshotError> { + if count > self.0.len() { + return Err(integrity("truncated metadata installation plan")); + } + let (bytes, rest) = self.0.split_at(count); + self.0 = rest; + Ok(bytes) + } + fn array(&mut self) -> Result<[u8; N], SnapshotError> { + self.take(N)? + .try_into() + .map_err(|_| integrity("invalid installation field")) + } + fn u16(&mut self) -> Result { + Ok(u16::from_be_bytes(self.array()?)) + } + fn u32(&mut self) -> Result { + Ok(u32::from_be_bytes(self.array()?)) + } + fn u64(&mut self) -> Result { + Ok(u64::from_be_bytes(self.array()?)) + } + fn count(&mut self, maximum: usize) -> Result { + let count = self.u32()? as usize; + if count > maximum { + return Err(limit("metadata installation count exceeds its budget")); + } + Ok(count) + } + fn string(&mut self, maximum: usize) -> Result { + let count = self.count(maximum)?; + String::from_utf8(self.take(count)?.to_vec()) + .map_err(|_| integrity("non-UTF8 installation identity")) + } +} + +fn parse_node_id(value: &str) -> Result { + let value = value + .strip_prefix("page:sha256:") + .ok_or_else(|| integrity("non-page installation node"))?; + hex::decode(value) + .map_err(|_| integrity("invalid installation page ID"))? + .try_into() + .map_err(|_| integrity("invalid installation page ID length")) +} + +fn integrity(message: &str) -> SnapshotError { + SnapshotError::new(SnapshotErrorCode::IntegrityError, message) +} +fn limit(message: &str) -> SnapshotError { + SnapshotError::new(SnapshotErrorCode::LimitExceeded, message) +} + +#[cfg(test)] +mod tests { + use std::sync::Arc; + + use mst2_codec::metapage::{Page, page_id}; + + use super::*; + use crate::ceres::snapshot::{ + pages::PreparedNativeMetadataRetention, retention_dag::MetadataDagBuilder, + }; + + fn plan() -> MetadataInstallPlan { + let bytes = Page::build(&[]).unwrap(); + let mut builder = MetadataDagBuilder::new(MetadataDagLimits::default()); + builder.add_directory(&bytes, &[]).unwrap(); + PreparedNativeMetadataRetention::test_installation( + Arc::new(builder.finish(page_id(&bytes)).unwrap()), + "/", + ) + .install_plan() + .unwrap() + } + + #[test] + fn native_installation_plan_round_trip_binds_every_source_and_profile_field() { + let plan = plan(); + let bytes = plan.encode().unwrap(); + let digest = plan.digest().unwrap(); + assert_eq!(MetadataInstallPlan::decode(&bytes, &digest).unwrap(), plan); + let mut changed = plan.clone(); + changed.identity.tagged_root_tree_oid = format!("sha1:{}", "b".repeat(40)); + assert_ne!(changed.digest().unwrap(), digest); + changed = plan.clone(); + changed.identity.scope = "/other".into(); + assert_ne!(changed.digest().unwrap(), digest); + changed = plan; + changed.identity.projection_revision += 1; + assert_eq!( + changed.encode().unwrap_err().code, + SnapshotErrorCode::IntegrityError + ); + } + + #[test] + fn native_installation_stored_plan_rejects_bad_digest_truncation_trailing_bytes_and_profile() { + let plan = plan(); + let bytes = plan.encode().unwrap(); + let digest = plan.digest().unwrap(); + let mut bad_digest = digest; + bad_digest[0] ^= 1; + assert!(MetadataInstallPlan::decode(&bytes, &bad_digest).is_err()); + for length in [0, DOMAIN.len(), bytes.len() - 1] { + let truncated = &bytes[..length]; + let digest = Sha256::digest(truncated).into(); + assert!(MetadataInstallPlan::decode(truncated, &digest).is_err()); + } + let mut trailing = bytes.clone(); + trailing.push(0); + assert!(MetadataInstallPlan::decode(&trailing, &Sha256::digest(&trailing).into()).is_err()); + let mut wrong_domain = bytes; + wrong_domain[0] ^= 1; + assert!( + MetadataInstallPlan::decode(&wrong_domain, &Sha256::digest(&wrong_domain).into()) + .is_err() + ); + } + + #[test] + fn native_installation_plan_rejects_oversized_stored_bytes_before_allocating_fields() { + let bytes = vec![0; MAX_PLAN_BYTES + 1]; + assert_eq!( + MetadataInstallPlan::decode(&bytes, &[0; 32]) + .unwrap_err() + .code, + SnapshotErrorCode::LimitExceeded + ); + } + + #[test] + fn native_installation_posix_scopes_preserve_utf8_backslash_control_and_path_boundaries() { + let original = plan(); + let at_byte_limit = format!("/{}", vec!["a".repeat(255); 16].join("/")); + assert_eq!(at_byte_limit.len(), 4096); + let at_component_limit = format!("/{}", vec!["a"; 256].join("/")); + for scope in [ + "/a\\b".to_owned(), + "/目录/é".to_owned(), + "/line\ncontrol\t".to_owned(), + format!("/{}", "a".repeat(255)), + at_byte_limit.clone(), + at_component_limit, + ] { + let mut value = original.clone(); + value.identity.scope = scope.clone(); + let encoded = value.encode().unwrap(); + assert_eq!( + MetadataInstallPlan::decode(&encoded, &value.digest().unwrap()) + .unwrap() + .identity + .scope, + scope + ); + } + for scope in [ + format!("/{}", "a".repeat(256)), + format!("/{}", vec!["a"; 257].join("/")), + format!("{at_byte_limit}/a"), + "/a\0b".to_owned(), + "/a/..".to_owned(), + ] { + let mut value = original.clone(); + value.identity.scope = scope; + assert_eq!( + value.encode().unwrap_err().code, + SnapshotErrorCode::IntegrityError + ); + } + } +} diff --git a/src/ceres/snapshot/mod.rs b/src/ceres/snapshot/mod.rs index a7c44a61..267a78bb 100644 --- a/src/ceres/snapshot/mod.rs +++ b/src/ceres/snapshot/mod.rs @@ -3,14 +3,23 @@ //! Serving remains native-only. The independent namespace index is an unwired //! composition seam; identity, attestation and publication integration remain //! separate gates. Fixed-source readers must never look up current refs. +pub(crate) mod chunk_map_gate; +pub(crate) mod chunk_map_index; pub mod chunks; +pub(crate) mod content_budget; pub mod descriptor; pub mod error; pub mod frame_stream; +pub(crate) mod metadata_install; pub mod namespace; pub mod pages; +pub(crate) mod projection_observation; +pub(crate) mod projection_writer; pub mod publication; pub mod resolver; pub mod retention; +pub mod retention_dag; +pub(crate) mod rooted_metadata_install; +pub(crate) mod rooted_metadata_projection; pub mod runtime; pub mod view; diff --git a/src/ceres/snapshot/native_projection_tests.rs b/src/ceres/snapshot/native_projection_tests.rs new file mode 100644 index 00000000..0cdddbd5 --- /dev/null +++ b/src/ceres/snapshot/native_projection_tests.rs @@ -0,0 +1,646 @@ +//! Real persisted commits and object storage, with an independent raw-content +//! oracle. This exercises on-demand memoization, not durable publication. + +use std::{collections::BTreeMap, sync::Arc}; + +use git_internal::{ + hash::{HashKind, ObjectHash}, + internal::object::{ + blob::Blob, + commit::Commit, + tree::{Tree, TreeItem, TreeItemMode}, + types::ObjectType, + }, +}; +use sha2::{Digest, Sha256}; + +use super::{FsKind, ProjectionWork, build_directory_page, build_directory_page_with_work}; +use crate::{ + ceres::api_service::{ApiHandler, cache::GitObjectCache, mono_api_service::MonoApiService}, + config::testing::isolated_config, + jupiter::{ + service::git_service::GitService, + storage::object_storage::build_object_storage, + tests::{test_redis_manager, test_storage_with_config}, + }, +}; + +type Files = BTreeMap>; + +#[tokio::test] +async fn mst2_native_retention_prepares_full_shared_radix_closure_once() { + use crate::ceres::snapshot::retention_dag::MetadataDagLimits; + + let temp = tempfile::tempdir().unwrap(); + let handler = handler(temp.path()).await; + let mut files = Files::new(); + for prefix in ["/project/alpha", "/project/beta"] { + for index in 0..129 { + files.insert( + format!("{prefix}/f{index:03}"), + format!("raw-{index}").into_bytes(), + ); + } + files.insert(format!("{prefix}/nested/file"), b"shared nested".to_vec()); + } + let (_, root) = persist_commit(&handler, &files, vec![], "retention closure").await; + let scope = build_directory_page(&handler, &root, "/project") + .await + .unwrap(); + let prepared = super::prepare_native_metadata_retention( + &handler, + &root, + "/project", + MetadataDagLimits::default(), + ) + .await + .unwrap(); + assert_eq!(prepared.fixed_root_tree_oid(), root.id.to_tagged_string()); + assert_eq!(prepared.scope(), "/project"); + assert_eq!(prepared.metadata_codec(), 1); + assert_eq!(prepared.schema_version(), 2); + assert_eq!(prepared.dag().root(), scope.page_id); + assert!(prepared.dag().payloads().iter().any(|payload| { + matches!( + mst2_codec::metapage::Page::decode(&payload.bytes) + .unwrap() + .0, + mst2_codec::metapage::Page::Branch { .. } + ) + })); + assert!(prepared.dag().payloads().len() > 3); + let root_node = format!("page:sha256:{}", super::hex(&scope.page_id)); + assert_eq!( + prepared + .dag() + .edges() + .iter() + .filter(|edge| edge.parent == root_node) + .count(), + 1 + ); + + // Remove all directory memo entries: the prepared hit must return its Arc + // without rebuilding or walking even this fixed view's child Git trees. + { + let mut state = handler + .storage + .native_projection_cache + .state + .lock() + .unwrap(); + state.pages.clear(); + state.retained_payload_bytes = 0; + } + let again = super::prepare_native_metadata_retention( + &handler, + &root, + "/project", + MetadataDagLimits::default(), + ) + .await + .unwrap(); + assert!(Arc::ptr_eq(prepared.dag(), again.dag())); + { + let state = handler + .storage + .native_projection_cache + .state + .lock() + .unwrap(); + assert!( + state.pages.is_empty(), + "prepare hit must not rebuild directory projections" + ); + } + let tightened = super::prepare_native_metadata_retention( + &handler, + &root, + "/project", + MetadataDagLimits { + nodes: 1, + ..MetadataDagLimits::default() + }, + ) + .await + .unwrap_err(); + assert_eq!( + tightened.code, + crate::ceres::snapshot::error::SnapshotErrorCode::LimitExceeded + ); +} + +#[tokio::test] +async fn mst2_native_retention_checks_source_budgets_before_unknown_child_fetch() { + use crate::ceres::snapshot::{error::SnapshotErrorCode, retention_dag::MetadataDagLimits}; + + let temp = tempfile::tempdir().unwrap(); + let handler = handler(temp.path()).await; + let unknown = ObjectHash::from_hex_for_kind(HashKind::Sha1, &"a".repeat(40)).unwrap(); + let root = Tree::from_tree_items_with_kind( + HashKind::Sha1, + vec![TreeItem::new( + TreeItemMode::Tree, + unknown, + "unknown-child".into(), + )], + ) + .unwrap(); + for limits in [ + MetadataDagLimits { + nodes: 1, + ..MetadataDagLimits::default() + }, + MetadataDagLimits { + edges: 0, + ..MetadataDagLimits::default() + }, + MetadataDagLimits { + entries: 0, + ..MetadataDagLimits::default() + }, + MetadataDagLimits { + payload_bytes: 19, + ..MetadataDagLimits::default() + }, + ] { + let error = super::prepare_native_metadata_retention(&handler, &root, "/", limits) + .await + .unwrap_err(); + assert_eq!( + error.code, + SnapshotErrorCode::LimitExceeded, + "budget must reject before querying the nonexistent tree: {error}" + ); + } + let state = handler + .storage + .native_projection_cache + .state + .lock() + .unwrap(); + assert!(state.retention_dags.is_empty()); +} + +async fn handler(temp: &std::path::Path) -> MonoApiService { + let config = isolated_config(temp.join("config")); + let object_storage = build_object_storage(&config.object_storage).await.unwrap(); + let mut storage = test_storage_with_config(temp, config).await; + storage.git_service = GitService { + obj_storage: object_storage, + }; + MonoApiService { + storage, + git_object_cache: Arc::new(GitObjectCache { + connection: test_redis_manager().await, + prefix: String::new(), + }), + } +} + +#[derive(Default)] +struct Directory { + children: BTreeMap, + files: BTreeMap, +} + +fn build_trees(directory: Directory, trees: &mut Vec) -> Tree { + let mut items: BTreeMap = directory + .files + .into_iter() + .map(|(name, oid)| { + let item = TreeItem::new(TreeItemMode::Blob, oid, name.clone()); + (name, item) + }) + .collect(); + for (name, child) in directory.children { + let child = build_trees(child, trees); + items.insert( + name.clone(), + TreeItem::new(TreeItemMode::Tree, child.id, name), + ); + } + let tree = + Tree::from_tree_items_with_kind(HashKind::Sha1, items.into_values().collect()).unwrap(); + trees.push(tree.clone()); + tree +} + +async fn persist_commit( + handler: &MonoApiService, + files: &Files, + parents: Vec, + message: &str, +) -> (Commit, Tree) { + let mut directory = Directory::default(); + for (path, raw) in files { + let blob = Blob::from_content_bytes_with_kind(HashKind::Sha1, raw.clone()).unwrap(); + handler + .storage + .git_service + .save_object_from_model(raw.clone(), &blob.id.to_string()) + .await + .unwrap(); + let components: Vec<_> = path.trim_start_matches('/').split('/').collect(); + let mut parent = &mut directory; + for component in &components[..components.len() - 1] { + parent = parent.children.entry((*component).to_string()).or_default(); + } + parent + .files + .insert(components.last().unwrap().to_string(), blob.id); + } + let mut trees = Vec::new(); + let root = build_trees(directory, &mut trees); + persist_tree_commit(handler, root, trees, parents, message).await +} + +async fn persist_tree_commit( + handler: &MonoApiService, + root: Tree, + trees: Vec, + parents: Vec, + message: &str, +) -> (Commit, Tree) { + let commit = Commit::from_tree_id_with_kind(HashKind::Sha1, root.id, parents, message).unwrap(); + let mono = handler.storage.mono_storage(); + mono.save_mega_trees(trees, commit.id, None).await.unwrap(); + mono.save_mega_commits(vec![commit.clone()], None) + .await + .unwrap(); + // Read back the real persisted commit and its fixed tree; no fake blob OID + // is used as a commit identity and no live ref is required by the builder. + let persisted = handler + .get_commit_by_hash(&commit.id.to_string()) + .await + .unwrap(); + assert_eq!(persisted.tree_id, root.id); + let root = handler + .get_tree_by_hash(&persisted.tree_id.to_string()) + .await + .unwrap(); + (persisted, root) +} + +fn fixture(modules: usize, buckets: usize, files_per_bucket: usize, bytes: usize) -> Files { + let mut files = Files::new(); + for module in 0..modules { + for bucket in 0..buckets { + for file in 0..files_per_bucket { + let path = format!("/project/m{module:03}/d{bucket:02}/f{file:03}"); + let mut raw = vec![b'x'; bytes.max(path.len())]; + raw[..path.len()].copy_from_slice(path.as_bytes()); + files.insert(path, raw); + } + } + } + files +} + +async fn assert_raw_oracle(handler: &MonoApiService, root: &Tree, scope: &str, files: &Files) { + let mut paths = vec![scope.to_string()]; + let mut projected = BTreeMap::new(); + while let Some(path) = paths.pop() { + let directory = build_directory_page(handler, root, &path).await.unwrap(); + for entry in &directory.entries { + let child = format!("{}/{name}", path.trim_end_matches('/'), name = entry.name); + if entry.fs_kind == FsKind::Directory { + paths.push(child); + } else { + assert_eq!(entry.fs_kind, FsKind::Regular); + projected.insert(child, (entry.size.unwrap(), entry.content_digest.unwrap())); + } + } + } + let expected: BTreeMap<_, _> = files + .iter() + .map(|(path, raw)| { + ( + path.clone(), + (raw.len() as u64, <[u8; 32]>::from(Sha256::digest(raw))), + ) + }) + .collect(); + assert_eq!(projected, expected); +} + +async fn leaf_update_case( + modules: usize, + buckets: usize, + files_per_bucket: usize, + bytes: usize, +) -> ProjectionWork { + let temp = tempfile::tempdir().unwrap(); + let handler = handler(temp.path()).await; + let mut files = fixture(modules, buckets, files_per_bucket, bytes); + let original = files.clone(); + let (v1, root1) = persist_commit(&handler, &files, vec![], "native projection V1").await; + let (page1, cold) = build_directory_page_with_work(&handler, &root1, "/project") + .await + .unwrap(); + assert_eq!( + cold.directories_rebuilt, + (1 + modules + modules * buckets) as u64 + ); + assert_eq!(cold.reused_subtree_roots, 0); + assert_eq!(cold.directory_root_pages_built, cold.directories_rebuilt); + assert_eq!(cold.verified_blob_misses, files.len() as u64); + + files.get_mut("/project/m000/d00/f000").unwrap().push(b'2'); + let (v2, root2) = persist_commit(&handler, &files, vec![v1.id], "native projection V2").await; + let (page2, update) = build_directory_page_with_work(&handler, &root2, "/project") + .await + .unwrap(); + assert_ne!(page1.page_id, page2.page_id); + assert_eq!(update.directories_rebuilt, 3); + assert_eq!(update.directory_root_pages_built, 3); + assert_eq!( + update.tree_fetches, 3, + "scope + two changed child trees only" + ); + assert_eq!( + update.reused_subtree_roots, + (modules - 1 + buckets - 1) as u64 + ); + assert_eq!( + update.directory_entries_scanned, + (modules + buckets + files_per_bucket) as u64 + ); + assert_eq!(update.verified_blob_hits, (files_per_bucket - 1) as u64); + assert_eq!(update.verified_blob_misses, 1); + assert_eq!( + update.raw_bytes_fetched, + files["/project/m000/d00/f000"].len() as u64 + ); + assert_eq!(update.raw_bytes_hashed, update.raw_bytes_fetched); + let scope_tree = + super::fetch_tree_with_work(&handler, &root2, "/project", &mut ProjectionWork::default()) + .await + .unwrap(); + let mut full_work = ProjectionWork::default(); + let full = super::build_subtree( + &handler, + &handler.storage, + &scope_tree, + "/project", + false, + &mut full_work, + ) + .await + .unwrap(); + assert_eq!( + page2.page_id, full.page_id, + "cache reuse equals a fresh full projection" + ); + assert_eq!(full_work.directories_rebuilt, cold.directories_rebuilt); + assert_eq!(full_work.reused_subtree_roots, 0); + assert_raw_oracle(&handler, &root2, "/project", &files).await; + + // A third real commit repeats the operation; this cannot be a special + // first-update shortcut. Reading older fixed roots remains correct. + let second_files = files.clone(); + files.get_mut("/project/m000/d00/f000").unwrap().push(b'3'); + let (_, root3) = persist_commit(&handler, &files, vec![v2.id], "native projection V3").await; + let (_, third) = build_directory_page_with_work(&handler, &root3, "/project") + .await + .unwrap(); + assert_eq!(third.directories_rebuilt, update.directories_rebuilt); + assert_eq!( + third.directory_entries_scanned, + update.directory_entries_scanned + ); + assert_eq!(third.reused_subtree_roots, update.reused_subtree_roots); + assert_eq!(third.verified_blob_misses, 1); + assert_raw_oracle(&handler, &root3, "/project", &files).await; + assert_raw_oracle(&handler, &root2, "/project", &second_files).await; + + let (old, old_work) = build_directory_page_with_work(&handler, &root1, "/project") + .await + .unwrap(); + assert_eq!(old.page_id, page1.page_id); + assert_eq!(old_work.directories_rebuilt, 0); + assert_eq!(old_work.directory_entries_scanned, 0); + assert_eq!(old_work.reused_subtree_roots, 1); + assert_raw_oracle(&handler, &root1, "/project", &original).await; + let old_file = super::resolve_abs(&handler, &root1, "/project/m000/d00/f000") + .await + .unwrap(); + assert!( + matches!(old_file, super::WalkOutcome::FoundFile { raw, .. } if raw == original["/project/m000/d00/f000"]) + ); + update +} + +#[tokio::test] +async fn mst2_native_projection_real_commit_leaf_update_only_rebuilds_ancestors() { + leaf_update_case(4, 3, 4, 64).await; +} + +#[tokio::test] +#[ignore = "opt-in 16,384-file/256MiB real object-storage projection workload"] +async fn mst2_native_projection_medium_real_commit_leaf_update() { + let update = leaf_update_case(64, 8, 32, 16 * 1024).await; + assert_eq!(update.directory_entries_scanned, 104); + assert_eq!(update.reused_subtree_roots, 70); + eprintln!("native projection memoization work: {update:?}"); +} + +#[tokio::test] +async fn mst2_native_projection_same_oid_isolated_storage_does_not_share_cache() { + let temp_a = tempfile::tempdir().unwrap(); + let temp_b = tempfile::tempdir().unwrap(); + let a = handler(temp_a.path()).await; + let b = handler(temp_b.path()).await; + let files = fixture(2, 2, 2, 64); + let (_, root) = persist_commit(&a, &files, vec![], "independent storage A").await; + let (cached, _) = build_directory_page_with_work(&a, &root, "/") + .await + .unwrap(); + assert!( + build_directory_page_with_work(&b, &root, "/") + .await + .is_err(), + "B must fetch its own missing child tree" + ); + let (_, own_root) = persist_commit(&b, &files, vec![], "independent storage B").await; + assert_eq!(root.id, own_root.id); + let (rebuilt, work) = build_directory_page_with_work(&b, &own_root, "/") + .await + .unwrap(); + assert_eq!(cached.page_id, rebuilt.page_id); + assert_eq!(work.directories_rebuilt, 8); + assert_eq!(work.reused_subtree_roots, 0); + let clone = a.clone(); + let (_, clone_work) = build_directory_page_with_work(&clone, &root, "/") + .await + .unwrap(); + assert_eq!(clone_work.directories_rebuilt, 0); + assert_eq!(clone_work.tree_fetches, 0); +} + +#[tokio::test] +async fn mst2_native_projection_real_commit_directory_move_reuses_identical_tree() { + let temp = tempfile::tempdir().unwrap(); + let handler = handler(temp.path()).await; + let files = Files::from([("/project/old/nested/file".into(), b"move me".to_vec())]); + let (v1, root1) = persist_commit(&handler, &files, vec![], "before move").await; + let (before, _) = build_directory_page_with_work(&handler, &root1, "/project") + .await + .unwrap(); + let moved = Files::from([("/project/new/nested/file".into(), b"move me".to_vec())]); + let (_, root2) = persist_commit(&handler, &moved, vec![v1.id], "after move").await; + let (after, work) = build_directory_page_with_work(&handler, &root2, "/project") + .await + .unwrap(); + assert_eq!( + before.entries[0].directory_root, + after.entries[0].directory_root + ); + assert_eq!(work.directories_rebuilt, 1); + assert_eq!(work.reused_subtree_roots, 1); + assert_eq!( + work.tree_fetches, 1, + "only scope; moved root is reused by OID" + ); + assert_eq!(work.verified_blob_hits + work.verified_blob_misses, 0); + assert_raw_oracle(&handler, &root2, "/project", &moved).await; +} + +fn byte_prefix(bytes: usize) -> String { + let mut path = "/project".to_string(); + while path.len() < bytes { + let length = (bytes - path.len() - 1).min(255); + assert_ne!(length, 0); + path.push('/'); + path.extend(std::iter::repeat_n('a', length)); + } + path +} + +fn wrap_tree_at(prefix: &str, mut child: Tree, trees: &mut Vec) -> Tree { + for name in prefix[1..].split('/').rev() { + child = Tree::from_tree_items_with_kind( + HashKind::Sha1, + vec![TreeItem::new( + TreeItemMode::Tree, + child.id, + name.to_string(), + )], + ) + .unwrap(); + trees.push(child.clone()); + } + child +} + +#[tokio::test] +async fn mst2_native_projection_cached_empty_child_move_rechecks_path_budgets() { + let temp = tempfile::tempdir().unwrap(); + let handler = handler(temp.path()).await; + let empty = Tree { + id: ObjectHash::from_type_and_data_for_kind(HashKind::Sha1, ObjectType::Tree, &[]).unwrap(), + tree_items: vec![], + }; + let source = Tree::from_tree_items_with_kind( + HashKind::Sha1, + vec![TreeItem::new( + TreeItemMode::Tree, + empty.id, + "empty".to_string(), + )], + ) + .unwrap(); + let mut trees = vec![empty.clone(), source.clone()]; + let root = wrap_tree_at("/project/source", source.clone(), &mut trees); + let (mut commit, root) = + persist_tree_commit(&handler, root, trees, vec![], "empty source").await; + build_directory_page_with_work(&handler, &root, "/project/source") + .await + .unwrap(); + for (index, (prefix, allowed)) in [ + (byte_prefix(4090), true), + (byte_prefix(4091), false), + (format!("/project/{}", vec!["a"; 254].join("/")), true), + (format!("/project/{}", vec!["a"; 255].join("/")), false), + ] + .into_iter() + .enumerate() + { + let mut trees = vec![empty.clone(), source.clone()]; + let root = wrap_tree_at(&prefix, source.clone(), &mut trees); + let (next, root) = persist_tree_commit( + &handler, + root, + trees, + vec![commit.id], + &format!("empty move {index}"), + ) + .await; + commit = next; + let result = build_directory_page_with_work(&handler, &root, &prefix).await; + if allowed { + let (directory, work) = result.unwrap(); + assert_eq!(directory.entries[0].name, "empty"); + assert_eq!(work.directories_rebuilt, 0); + assert_eq!(work.reused_subtree_roots, 1); + } else { + assert!( + matches!(result, Err(error) if error.code == super::SnapshotErrorCode::ScopeInvalid) + ); + } + } +} + +#[test] +fn mst2_native_projection_cache_separates_equal_hex_different_hash_kinds() { + let cache = super::NativeProjectionCache::default(); + let sha = ObjectHash::from_bytes_for_kind(HashKind::Sha256, &[7; 32]).unwrap(); + let blake = ObjectHash::from_bytes_for_kind(HashKind::Blake3, &[7; 32]).unwrap(); + assert_eq!(sha.to_string(), blake.to_string()); + let page_bytes = mst2_codec::metapage::Page::build(&[]).unwrap(); + let directory = Arc::new(super::BuiltDirectory { + page_id: mst2_codec::metapage::page_id(&page_bytes), + page_bytes, + entries: vec![], + codec_entries: vec![], + path_budget: super::DescendantPathBudget::default(), + }); + cache.insert(sha, Arc::clone(&directory)); + assert!(cache.get(blake).is_none()); + assert!(Arc::ptr_eq(&cache.get(sha).unwrap(), &directory)); +} + +#[tokio::test] +async fn mst2_native_projection_cached_move_rechecks_full_scope_path_budgets() { + let temp = tempfile::tempdir().unwrap(); + let handler = handler(temp.path()).await; + let raw = b"multibyte basename".to_vec(); + let files = Files::from([("/project/source/é".into(), raw.clone())]); + let (mut commit, root) = persist_commit(&handler, &files, vec![], "warm source").await; + build_directory_page_with_work(&handler, &root, "/project/source") + .await + .unwrap(); + let prefixes = [ + (byte_prefix(4093), true), + (byte_prefix(4094), false), + (format!("/project/{}", vec!["a"; 254].join("/")), true), + (format!("/project/{}", vec!["a"; 255].join("/")), false), + ]; + for (index, (prefix, allowed)) in prefixes.into_iter().enumerate() { + let files = Files::from([(format!("{prefix}/é"), raw.clone())]); + let (next, root) = + persist_commit(&handler, &files, vec![commit.id], &format!("move {index}")).await; + commit = next; + let result = build_directory_page_with_work(&handler, &root, &prefix).await; + if allowed { + let (_, work) = result.unwrap(); + assert_eq!(work.directories_rebuilt, 0); + assert_eq!(work.reused_subtree_roots, 1); + assert_eq!(work.directory_entries_scanned, 0); + } else { + assert!( + matches!(result, Err(error) if error.code == super::SnapshotErrorCode::ScopeInvalid) + ); + } + } +} diff --git a/src/ceres/snapshot/pages.rs b/src/ceres/snapshot/pages.rs index 555ed913..05e4e67f 100644 --- a/src/ceres/snapshot/pages.rs +++ b/src/ceres/snapshot/pages.rs @@ -6,25 +6,36 @@ //! yields identical page_ids across requests. use std::{ - collections::HashMap, - sync::{Arc, Mutex, OnceLock}, + borrow::Cow, + collections::{HashMap, HashSet}, + sync::{Arc, Mutex}, }; use base64::Engine; -use mst2_codec::metapage::{Entry, EntryKind, Page, page_id}; +use git_internal::{hash::ObjectHash, internal::object::tree::Tree}; +use mst2_codec::{ + descriptor::{ + ACCESS_PROJECTION_EXACT_FULL, FS_SEMANTICS_LINUX_CODE_V1, + MATERIALIZATION_POLICY_GIT_RAW_V1, METADATA_CODEC, SCHEMA_VERSION, + }, + metapage::{Entry, EntryKind, Page, page_id}, +}; use sea_orm::ActiveValue::Set; use sha2::{Digest, Sha256}; +pub use crate::ceres::snapshot::projection_observation::ProjectionWork; use crate::{ ceres::{ api_service::ApiHandler, snapshot::{ error::{SnapshotError, SnapshotErrorCode}, + projection_observation::NATIVE_PROJECTION_REVISION, resolver::FsKind, + retention_dag::{MetadataDagBuilder, MetadataDagLimits, ValidatedMetadataDag}, view::hex, }, }, - jupiter::storage::mono_storage::MST2_VERIFICATION_VERSION, + jupiter::storage::{Storage, mono_storage::MST2_VERIFICATION_VERSION}, }; /// A directory page plus everything the JSON layer needs. @@ -38,6 +49,189 @@ pub struct BuiltDirectory { /// The codec entries the page was built from, so callers can walk a route /// through the same canonical tree (`Page::pages_along_route`). pub codec_entries: Vec, + /// Bounds proven while validating this subtree; rechecked at every new + /// prefix so an identical tree cannot bypass full-path limits after a move. + path_budget: DescendantPathBudget, +} + +#[derive(Debug, Clone, Copy, Default)] +struct DescendantPathBudget { + /// Includes the slash before each descendant name. + suffix_bytes: usize, + components: usize, +} + +impl DescendantPathBudget { + fn include(&mut self, name: &str, child: Self) { + self.suffix_bytes = self.suffix_bytes.max(1 + name.len() + child.suffix_bytes); + self.components = self.components.max(1 + child.components); + } + + fn validate_at(self, prefix: &str) -> Result<(), SnapshotError> { + crate::ceres::snapshot::view::validate_scope_relative_path(prefix)?; + let (bytes, components) = if prefix == "/" { + (0, 0) + } else { + (prefix.len(), prefix[1..].split('/').count()) + }; + if bytes + self.suffix_bytes > 4096 || components + self.components > 256 { + return Err(SnapshotError::new( + SnapshotErrorCode::ScopeInvalid, + "subtree exceeds full-path budget at this prefix", + )); + } + Ok(()) + } +} + +#[derive(Debug, Clone, PartialEq, Eq, Hash)] +struct NativeProjectionKey { + source_domain: &'static str, + /// Tagged OID preserves SHA-256 vs BLAKE3 even though both are 64 hex. + tree_oid: String, + schema_version: u16, + metadata_codec: u16, + materialization_policy: u16, + fs_semantics: u16, + access_projection: u16, + verification_revision: i32, + projection_revision: u16, +} + +impl NativeProjectionKey { + fn new(oid: ObjectHash) -> Self { + Self { + source_domain: "native-git", + tree_oid: oid.to_tagged_string(), + schema_version: SCHEMA_VERSION, + metadata_codec: METADATA_CODEC, + materialization_policy: MATERIALIZATION_POLICY_GIT_RAW_V1, + fs_semantics: FS_SEMANTICS_LINUX_CODE_V1, + access_projection: ACCESS_PROJECTION_EXACT_FULL, + verification_revision: MST2_VERIFICATION_VERSION, + projection_revision: NATIVE_PROJECTION_REVISION, + } + } +} + +/// Storage-owned, disposable derived facts. This cache grants no authorization +/// and provides no durable retention evidence. Clearing it only costs a rebuild. +#[derive(Default)] +pub(crate) struct NativeProjectionCache { + state: Mutex, +} + +#[derive(Default)] +struct NativeProjectionCacheState { + pages: HashMap>, + retained_payload_bytes: usize, + retention_dags: HashMap>, + retained_dag_bytes: usize, +} + +#[derive(Debug, Clone, PartialEq, Eq, Hash)] +struct NativeRetentionKey { + projection: NativeProjectionKey, + scope: String, +} + +impl NativeProjectionCache { + fn retention_dag(&self, key: &NativeRetentionKey) -> Option> { + self.state.lock().ok()?.retention_dags.get(key).cloned() + } + + fn insert_retention_dag(&self, key: NativeRetentionKey, dag: Arc) { + // The memo has one aggregate residency bound, independent of the page + // cache. Caller Arcs may outlive eviction; this is no request admission. + const MAX_DAG_RESIDENT_BYTES: usize = 64 * 1024 * 1024; + const MAX_DAG_ENTRIES: usize = 16; + self.insert_retention_dag_with_budget(key, dag, MAX_DAG_RESIDENT_BYTES, MAX_DAG_ENTRIES); + } + + fn insert_retention_dag_with_budget( + &self, + key: NativeRetentionKey, + dag: Arc, + maximum_bytes: usize, + maximum_entries: usize, + ) { + let Some(bytes) = dag + .residency_bytes() + .and_then(|bytes| { + bytes.checked_add( + std::mem::size_of::() + + std::mem::size_of::>(), + ) + }) + .and_then(|bytes| bytes.checked_add(key.scope.capacity())) + .and_then(|bytes| bytes.checked_add(key.projection.tree_oid.capacity())) + else { + return; + }; + if bytes > maximum_bytes || maximum_entries == 0 { + return; + } + if let Ok(mut state) = self.state.lock() { + if state.retention_dags.contains_key(&key) { + return; + } + if state.retention_dags.len() >= maximum_entries + || state.retained_dag_bytes > maximum_bytes - bytes + { + state.retention_dags.clear(); + state.retained_dag_bytes = 0; + } + state.retained_dag_bytes += bytes; + state.retention_dags.insert(key, dag); + } + } + + fn get(&self, oid: ObjectHash) -> Option> { + // A poisoned optimization cache is a miss, not a serving failure. + self.state + .lock() + .ok()? + .pages + .get(&NativeProjectionKey::new(oid)) + .cloned() + } + + fn insert(&self, oid: ObjectHash, directory: Arc) { + // Bound retained payload as well as directory count: a wide directory + // retains all entries and codec names even though its root is <=16 KiB. + const MAX_PAYLOAD_BYTES: usize = 64 * 1024 * 1024; + let payload_bytes = std::mem::size_of::() + + directory.page_bytes.capacity() + + directory.entries.capacity() * std::mem::size_of::() + + directory.codec_entries.capacity() * std::mem::size_of::() + + directory + .entries + .iter() + .map(|entry| entry.name.capacity() + entry.oid.capacity()) + .sum::() + + directory + .codec_entries + .iter() + .map(|entry| entry.name.capacity()) + .sum::(); + if payload_bytes > MAX_PAYLOAD_BYTES { + return; + } + if let Ok(mut state) = self.state.lock() { + let key = NativeProjectionKey::new(oid); + if state.pages.contains_key(&key) { + return; + } + if state.pages.len() >= 8192 + || state.retained_payload_bytes + payload_bytes > MAX_PAYLOAD_BYTES + { + state.pages.clear(); + state.retained_payload_bytes = 0; + } + state.retained_payload_bytes += payload_bytes; + state.pages.insert(key, directory); + } + } } #[derive(Debug, Clone)] @@ -52,54 +246,375 @@ pub struct DirEntry { pub content_digest: Option<[u8; 32]>, /// Child directory page id (directories only). pub directory_root: Option<[u8; 32]>, + /// Preserve the source hash kind when preparing a cached child closure. + directory_tree_oid: Option, +} + +/// Fixed source/profile identity for a prepared metadata closure. This grants +/// no authorization, durable retention or coverage of file/chunk payloads. +#[derive(Debug)] +pub struct PreparedNativeMetadataRetention { + key: NativeRetentionKey, + dag: Arc, +} + +impl PreparedNativeMetadataRetention { + #[cfg(test)] + pub(crate) fn test_installation(dag: Arc, scope: &str) -> Self { + let tree_oid = + ObjectHash::from_hex_for_kind(git_internal::hash::HashKind::Sha1, &"a".repeat(40)) + .unwrap(); + Self { + key: NativeRetentionKey { + projection: NativeProjectionKey::new(tree_oid), + scope: scope.to_owned(), + }, + dag, + } + } + pub(crate) fn install_plan( + &self, + ) -> Result { + use super::metadata_install::{MetadataInstallIdentity, MetadataInstallPlan}; + let key = &self.key.projection; + MetadataInstallPlan::from_validated( + MetadataInstallIdentity { + source_domain: key.source_domain.to_owned(), + tagged_root_tree_oid: key.tree_oid.clone(), + scope: self.key.scope.clone(), + schema_version: key.schema_version, + metadata_codec: key.metadata_codec, + materialization_policy: key.materialization_policy, + fs_semantics: key.fs_semantics, + access_projection: key.access_projection, + verification_revision: key.verification_revision, + projection_revision: key.projection_revision, + }, + &self.dag, + ) + } + pub fn fixed_root_tree_oid(&self) -> &str { + &self.key.projection.tree_oid + } + pub fn scope(&self) -> &str { + &self.key.scope + } + pub fn metadata_codec(&self) -> u16 { + self.key.projection.metadata_codec + } + pub fn schema_version(&self) -> u16 { + self.key.projection.schema_version + } + pub fn dag(&self) -> &Arc { + &self.dag + } +} + +/// Explicit preparation, separate from resolve/lease success. A hit returns +/// the immutable Arc before traversing any child Git tree. A miss collects +/// from existing native codec entries, rebuilding an evicted child as needed. +pub async fn prepare_native_metadata_retention( + handler: &T, + root_tree: &Tree, + scope: &str, + limits: MetadataDagLimits, +) -> Result { + crate::ceres::snapshot::view::validate_scope_relative_path(scope)?; + if !handler.native_snapshot_projection() { + return Err(SnapshotError::new( + SnapshotErrorCode::ScopeInvalid, + "retention preparation requires native Git projection", + )); + } + let storage = handler.get_context(); + let limits = limits.effective(); + let key = NativeRetentionKey { + projection: NativeProjectionKey::new(root_tree.id), + scope: scope.to_owned(), + }; + if let Some(dag) = storage.native_projection_cache.retention_dag(&key) { + dag.check_limits(limits)?; + return Ok(PreparedNativeMetadataRetention { key, dag }); + } + let mut projection_budget = MetadataProjectionBudget::new(limits); + let mut work = ProjectionWork::default(); + let root = if scope == "/" { + build_subtree_inner( + handler, + &storage, + root_tree, + scope, + true, + &mut work, + Some(&mut projection_budget), + ) + .await? + } else { + let tree = fetch_tree_with_work(handler, root_tree, scope, &mut work).await?; + build_subtree_inner( + handler, + &storage, + &tree, + scope, + true, + &mut work, + Some(&mut projection_budget), + ) + .await? + }; + let root_id = root.page_id; + let mut builder = MetadataDagBuilder::new(limits); + let mut scheduled = HashSet::from([root_id]); + let mut pending = vec![(root, scope.to_owned())]; + while let Some((directory, path)) = pending.pop() { + builder.add_directory(&directory.page_bytes, &directory.codec_entries)?; + for entry in directory + .entries + .iter() + .filter(|entry| entry.fs_kind == FsKind::Directory) + { + let child_id = entry.directory_root.ok_or_else(|| { + SnapshotError::new( + SnapshotErrorCode::IntegrityError, + "directory lacks its metadata root", + ) + })?; + if !scheduled.insert(child_id) { + continue; + } + let child_oid = entry.directory_tree_oid.ok_or_else(|| { + SnapshotError::new( + SnapshotErrorCode::IntegrityError, + "directory lacks its tagged source identity", + ) + })?; + let child_path = if path == "/" { + format!("/{}", entry.name) + } else { + format!("{path}/{}", entry.name) + }; + let child = if let Some(hit) = + cached_subtree(&storage, child_oid, &child_path, true, &mut work)? + { + hit + } else { + let tree = handler + .get_tree_by_hash(&entry.oid) + .await + .map_err(|error| { + SnapshotError::new(SnapshotErrorCode::Internal, error.to_string()) + })?; + if tree.id != child_oid { + return Err(SnapshotError::new( + SnapshotErrorCode::IntegrityError, + "retention child tree identity mismatch", + )); + } + build_subtree_inner( + handler, + &storage, + &tree, + &child_path, + true, + &mut work, + Some(&mut projection_budget), + ) + .await? + }; + if child.page_id != child_id { + return Err(SnapshotError::new( + SnapshotErrorCode::DigestMismatch, + "retention child page disagrees with fixed parent", + )); + } + pending.push((child, child_path)); + } + } + let dag = Arc::new(builder.finish(root_id)?); + storage + .native_projection_cache + .insert_retention_dag(key.clone(), Arc::clone(&dag)); + Ok(PreparedNativeMetadataRetention { key, dag }) } /// Build the MTP2 page for one directory of the fixed view. `rel_path` is -/// scope-relative ("/" = the root directory); `root_tree` is the fixed view's -/// root tree — never a current-ref read. Every entry is represented; +/// an absolute path within the fixed global view, including the descriptor +/// scope prefix ("/" = the global root); `root_tree` is that view's global +/// root tree — never a rebased scope tree or current-ref read. Every entry is represented; /// unsupported entries reject the whole projection (spec: no silent drops). /// -/// Results are memoized per (root tree, path): a page is a pure function of -/// the pinned root tree, and the recursive build otherwise re-walks the same -/// subtree once per ancestor and once per request (a sync over N directories -/// would rebuild the tree N times). Blob sizes/digests are persisted in -/// `mst2_verified_object`, so a cache miss after eviction is a bounded, -/// correctness-identical recomputation. -/// Page memoization table: (root tree id, scope-relative path) → built page. -type PageCache = Mutex>>; - +/// Native results are memoized by storage assembly, tagged subtree tree OID +/// and verification/profile revision. A new global root does not invalidate +/// unchanged subtrees. The fixed root is resolved to `rel_path` once; recursive +/// descent then follows direct child OIDs rather than re-walking from the root. pub async fn build_directory_page( handler: &T, root_tree: &git_internal::internal::object::tree::Tree, rel_path: &str, ) -> Result, SnapshotError> { - static PAGE_CACHE: OnceLock = OnceLock::new(); - // Pages are small (≤16 KiB + entries); 200k pages is far beyond any real - // view. On overflow the cache clears wholesale — a miss only costs a - // rebuild, never correctness. - const PAGE_CACHE_MAX: usize = 200_000; - - let key = (root_tree.id.to_string(), rel_path.to_string()); - let cache = PAGE_CACHE.get_or_init(|| Mutex::new(HashMap::new())); - if let Some(hit) = cache.lock().unwrap().get(&key) { - return Ok(Arc::clone(hit)); - } - let built = Arc::new(build_directory_page_uncached(handler, root_tree, rel_path).await?); - let mut cache = cache.lock().unwrap(); - if cache.len() >= PAGE_CACHE_MAX { - cache.clear(); - } - cache.insert(key, Arc::clone(&built)); - Ok(built) + let (directory, work) = build_directory_page_with_work(handler, root_tree, rel_path).await?; + tracing::debug!( + ?work, + path = rel_path, + "native snapshot projection memoization work" + ); + Ok(directory) } -async fn build_directory_page_uncached( +/// Build one directory, returning operation-local counters. Authorization and +/// the fixed-view source selection remain the caller's responsibility. +pub async fn build_directory_page_with_work( handler: &T, - root_tree: &git_internal::internal::object::tree::Tree, + root_tree: &Tree, rel_path: &str, -) -> Result { - let tree = fetch_tree(handler, root_tree, rel_path).await?; - let dirents = crate::ceres::snapshot::resolver::direct_entries(&tree)?; +) -> Result<(Arc, ProjectionWork), SnapshotError> { + let storage = handler.get_context(); + let mut work = ProjectionWork::default(); + let native = handler.native_snapshot_projection(); + if rel_path == "/" { + // Even cloning a wide fixed root would copy all its entries on a hit. + let directory = + build_subtree(handler, &storage, root_tree, rel_path, native, &mut work).await?; + return Ok((directory, work)); + } + let tree = fetch_tree_with_work(handler, root_tree, rel_path, &mut work).await?; + let directory = build_subtree(handler, &storage, &tree, rel_path, native, &mut work).await?; + Ok((directory, work)) +} + +fn cached_subtree( + storage: &Storage, + oid: ObjectHash, + rel_path: &str, + native: bool, + work: &mut ProjectionWork, +) -> Result>, SnapshotError> { + if native && let Some(hit) = storage.native_projection_cache.get(oid) { + hit.path_budget.validate_at(rel_path)?; + work.reused_subtree_roots += 1; + work.directory_root_pages_reused += 1; + work.directory_root_page_bytes_reused += hit.page_bytes.len() as u64; + return Ok(Some(hit)); + } + Ok(None) +} + +async fn build_subtree( + handler: &T, + storage: &Storage, + tree: &Tree, + rel_path: &str, + native: bool, + work: &mut ProjectionWork, +) -> Result, SnapshotError> { + build_subtree_inner(handler, storage, tree, rel_path, native, work, None).await +} + +struct MetadataProjectionBudget { + limits: MetadataDagLimits, + seen: HashSet, + required: HashSet, + edges: HashSet<(ObjectHash, ObjectHash)>, + entries: usize, + encoded_entry_bytes: u64, +} + +impl MetadataProjectionBudget { + fn new(limits: MetadataDagLimits) -> Self { + Self { + limits, + seen: HashSet::new(), + required: HashSet::new(), + edges: HashSet::new(), + entries: 0, + encoded_entry_bytes: 0, + } + } + + fn tree(&mut self, tree: &Tree) -> Result<(), SnapshotError> { + let failed = || { + SnapshotError::new( + SnapshotErrorCode::LimitExceeded, + "native metadata source projection budget exceeded", + ) + }; + if self.seen.contains(&tree.id) { + return Ok(()); + } + self.require_node(tree.id)?; + self.entries = self + .entries + .checked_add(tree.tree_items.len()) + .filter(|count| *count <= self.limits.entries) + .ok_or_else(failed)?; + self.encoded_entry_bytes = self + .encoded_entry_bytes + .checked_add(mst2_codec::metapage::HEADER_LEN as u64) + .filter(|bytes| *bytes <= self.limits.payload_bytes) + .ok_or_else(failed)?; + for item in &tree.tree_items { + let kind = FsKind::from_git_mode(item.mode).ok_or_else(|| { + SnapshotError::new( + SnapshotErrorCode::UnsupportedEntry, + "unsupported entry in native retention projection", + ) + })?; + let value_bytes = if kind == FsKind::Directory { 32 } else { 40 }; + self.encoded_entry_bytes = self + .encoded_entry_bytes + .checked_add((3 + item.name.len() + value_bytes) as u64) + .filter(|bytes| *bytes <= self.limits.payload_bytes) + .ok_or_else(failed)?; + if kind == FsKind::Directory { + self.require_node(item.id)?; + if !self.edges.contains(&(tree.id, item.id)) { + if self.edges.len() >= self.limits.edges { + return Err(failed()); + } + self.edges.insert((tree.id, item.id)); + } + } + } + if self.required.len() > self.limits.nodes { + return Err(failed()); + } + self.seen.insert(tree.id); + Ok(()) + } + + fn require_node(&mut self, id: ObjectHash) -> Result<(), SnapshotError> { + if !self.required.contains(&id) { + if self.required.len() >= self.limits.nodes { + return Err(SnapshotError::new( + SnapshotErrorCode::LimitExceeded, + "native metadata source node budget exceeded", + )); + } + self.required.insert(id); + } + Ok(()) + } +} + +async fn build_subtree_inner( + handler: &T, + storage: &Storage, + tree: &Tree, + rel_path: &str, + native: bool, + work: &mut ProjectionWork, + mut budget: Option<&mut MetadataProjectionBudget>, +) -> Result, SnapshotError> { + crate::ceres::snapshot::view::validate_scope_relative_path(rel_path)?; + if let Some(budget) = budget.as_deref_mut() { + budget.tree(tree)?; + } + if let Some(hit) = cached_subtree(storage, tree.id, rel_path, native, work)? { + return Ok(hit); + } + let dirents = crate::ceres::snapshot::resolver::direct_entries(tree)?; + work.directories_rebuilt += 1; + work.directory_entries_scanned += dirents.len() as u64; // T03 write-through verification: consult verified records first, then // fetch + hash the misses and persist them. Invalid records and lookup @@ -109,8 +624,7 @@ async fn build_directory_page_uncached( .filter(|(_, k, _)| *k != FsKind::Directory) .map(|(_, _, oid)| oid.clone()) .collect(); - let verified = handler - .get_context() + let verified = storage .mono_storage() .get_verified_blobs(blob_oids) .await @@ -128,6 +642,7 @@ async fn build_directory_page_uncached( })?; let mut new_verified = HashMap::new(); + let mut path_budget = DescendantPathBudget::default(); let mut entries = Vec::with_capacity(dirents.len()); let mut codec_entries = Vec::with_capacity(dirents.len()); for (name, fs_kind, oid) in dirents { @@ -139,8 +654,37 @@ async fn build_directory_page_uncached( crate::ceres::snapshot::view::validate_scope_relative_path(&child_rel)?; match fs_kind { FsKind::Directory => { - let child_page = - Box::pin(build_directory_page(handler, root_tree, &child_rel)).await?; + let child_oid = + ObjectHash::from_hex_for_kind(tree.id.kind(), &oid).map_err(|e| { + SnapshotError::new(SnapshotErrorCode::IntegrityError, e.to_string()) + })?; + let child_page = if let Some(hit) = + cached_subtree(storage, child_oid, &child_rel, native, work)? + { + hit + } else { + work.tree_fetches += 1; + let child_tree = handler.get_tree_by_hash(&oid).await.map_err(|e| { + SnapshotError::new(SnapshotErrorCode::Internal, e.to_string()) + })?; + if child_tree.id != child_oid { + return Err(SnapshotError::new( + SnapshotErrorCode::IntegrityError, + "fetched child tree identity mismatch", + )); + } + Box::pin(build_subtree_inner( + handler, + storage, + &child_tree, + &child_rel, + native, + work, + budget.as_deref_mut(), + )) + .await? + }; + path_budget.include(&name, child_page.path_budget); codec_entries.push(Entry::dir(name.as_bytes(), child_page.page_id)); entries.push(DirEntry { name, @@ -149,10 +693,13 @@ async fn build_directory_page_uncached( size: None, content_digest: None, directory_root: Some(child_page.page_id), + directory_tree_oid: Some(child_oid), }); } FsKind::Regular | FsKind::Executable | FsKind::Symlink => { + path_budget.include(&name, DescendantPathBudget::default()); let (size, digest) = if let Some(v) = verified.get(&oid) { + work.verified_blob_hits += 1; // Verified record: 64-bit size + raw digest, no content read. let d: [u8; 32] = v.raw_sha256.as_slice().try_into().map_err(|_| { SnapshotError::new( @@ -168,9 +715,12 @@ async fn build_directory_page_uncached( })?; (size, d) } else { + work.verified_blob_misses += 1; let raw = fetch_raw_blob(handler, &oid).await?; + work.raw_bytes_fetched += raw.len() as u64; let mut h = Sha256::new(); h.update(&raw); + work.raw_bytes_hashed += raw.len() as u64; let digest: [u8; 32] = h.finalize().into(); new_verified.entry(oid.clone()).or_insert( crate::callisto::mst2_verified_object::ActiveModel { @@ -201,6 +751,7 @@ async fn build_directory_page_uncached( size: Some(size), content_digest: Some(digest), directory_root: None, + directory_tree_oid: None, }); } } @@ -208,8 +759,7 @@ async fn build_directory_page_uncached( if !new_verified.is_empty() { // Records only ever describe already-fetched content; a persist // failure loses an optimization, never correctness. - if let Err(e) = handler - .get_context() + if let Err(e) = storage .mono_storage() .insert_verified_blobs(new_verified.into_values().collect()) .await @@ -224,13 +774,22 @@ async fn build_directory_page_uncached( format!("MTP2 build failed for {rel_path}: {e}"), ) })?; + work.directory_root_pages_built += 1; + work.directory_root_page_bytes_built += page_bytes.len() as u64; let pid = page_id(&page_bytes); - Ok(BuiltDirectory { + let built = Arc::new(BuiltDirectory { page_bytes, page_id: pid, entries, codec_entries, - }) + path_budget, + }); + if native { + storage + .native_projection_cache + .insert(tree.id, Arc::clone(&built)); + } + Ok(built) } /// Proof pages from the scope root down to (and including) `rel_path`, @@ -259,23 +818,24 @@ pub fn base64_of(data: &[u8]) -> String { base64::engine::general_purpose::STANDARD.encode(data) } -async fn fetch_tree( +async fn fetch_tree_with_work( handler: &T, root_tree: &git_internal::internal::object::tree::Tree, rel_path: &str, + work: &mut ProjectionWork, ) -> Result { crate::ceres::snapshot::view::validate_scope_relative_path(rel_path)?; if rel_path == "/" { return Ok(root_tree.clone()); } let comps: Vec<&str> = rel_path[1..].split('/').collect(); - let mut current = root_tree.clone(); - for (i, comp) in comps.iter().enumerate() { - let last = i == comps.len() - 1; - let item = current - .tree_items - .iter() - .find(|x| x.name == *comp) + let mut current = Cow::Borrowed(root_tree); + for comp in comps { + let position = current.tree_items.iter().position(|x| x.name == comp); + work.scope_path_entries_examined += + position.map_or(current.tree_items.len(), |index| index + 1) as u64; + let item = position + .map(|index| ¤t.tree_items[index]) .ok_or_else(|| { SnapshotError::new( SnapshotErrorCode::PathNotFound, @@ -288,30 +848,15 @@ async fn fetch_tree( "gitlink entries are not supported in this profile", ) })?; - if last { - if kind != FsKind::Directory { - return Err(SnapshotError::new( - SnapshotErrorCode::NotDirectory, - format!("{rel_path} is not a directory"), - )); - } - return handler - .get_tree_by_hash(&item.id.to_string()) - .await - .map_err(|e| { - SnapshotError::new( - SnapshotErrorCode::Internal, - format!("tree fetch failed for {rel_path}: {e}"), - ) - }); - } if kind != FsKind::Directory { return Err(SnapshotError::new( SnapshotErrorCode::NotDirectory, - "intermediate component is not a directory", + format!("{rel_path} is not a directory"), )); } - current = handler + let expected_oid = item.id; + work.tree_fetches += 1; + let fetched = handler .get_tree_by_hash(&item.id.to_string()) .await .map_err(|e| { @@ -320,8 +865,15 @@ async fn fetch_tree( format!("tree fetch failed for component '{comp}': {e}"), ) })?; + if fetched.id != expected_oid { + return Err(SnapshotError::new( + SnapshotErrorCode::IntegrityError, + "fetched scope tree identity mismatch", + )); + } + current = Cow::Owned(fetched); } - unreachable!("loop returns on the last component") + Ok(current.into_owned()) } /// Outcome of resolving one absolute view path (spec 04 §7 statuses). @@ -352,16 +904,51 @@ pub async fn resolve_abs( root_tree: &git_internal::internal::object::tree::Tree, abs_path: &str, ) -> Result { + Ok( + match resolve_abs_metadata(handler, root_tree, abs_path).await? { + MetadataWalkOutcome::FoundDir => WalkOutcome::FoundDir, + MetadataWalkOutcome::Absent => WalkOutcome::Absent, + MetadataWalkOutcome::NotDirectory { symlink } => WalkOutcome::NotDirectory { symlink }, + MetadataWalkOutcome::FoundFile { fs_kind, oid } => { + let raw = fetch_raw_blob(handler, &oid).await?; + let mut h = Sha256::new(); + h.update(&raw); + WalkOutcome::FoundFile { + fs_kind, + oid, + size: raw.len() as u64, + digest: h.finalize().into(), + raw, + } + } + }, + ) +} + +/// Fixed Git path metadata, without reading or hashing file content. +#[derive(Debug)] +pub(crate) enum MetadataWalkOutcome { + FoundDir, + FoundFile { fs_kind: FsKind, oid: String }, + Absent, + NotDirectory { symlink: bool }, +} + +pub(crate) async fn resolve_abs_metadata( + handler: &T, + root_tree: &git_internal::internal::object::tree::Tree, + abs_path: &str, +) -> Result { crate::ceres::snapshot::view::validate_scope_relative_path(abs_path)?; if abs_path == "/" { - return Ok(WalkOutcome::FoundDir); + return Ok(MetadataWalkOutcome::FoundDir); } let comps: Vec<&str> = abs_path[1..].split('/').collect(); - let mut current = root_tree.clone(); + let mut current = Cow::Borrowed(root_tree); for (i, comp) in comps.iter().enumerate() { let last = i == comps.len() - 1; let Some(item) = current.tree_items.iter().find(|x| x.name == *comp) else { - return Ok(WalkOutcome::Absent); + return Ok(MetadataWalkOutcome::Absent); }; let Some(kind) = FsKind::from_git_mode(item.mode) else { return Err(SnapshotError::new( @@ -372,33 +959,31 @@ pub async fn resolve_abs( let oid = item.id.to_string(); if last { return match kind { - FsKind::Directory => Ok(WalkOutcome::FoundDir), + FsKind::Directory => Ok(MetadataWalkOutcome::FoundDir), FsKind::Regular | FsKind::Executable | FsKind::Symlink => { - let raw = fetch_raw_blob(handler, &oid).await?; - let mut h = Sha256::new(); - h.update(&raw); - let digest: [u8; 32] = h.finalize().into(); - Ok(WalkOutcome::FoundFile { - fs_kind: kind, - oid, - size: raw.len() as u64, - digest, - raw, - }) + Ok(MetadataWalkOutcome::FoundFile { fs_kind: kind, oid }) } }; } if kind != FsKind::Directory { - return Ok(WalkOutcome::NotDirectory { + return Ok(MetadataWalkOutcome::NotDirectory { symlink: kind == FsKind::Symlink, }); } - current = handler.get_tree_by_hash(&oid).await.map_err(|e| { + let expected_oid = item.id; + let fetched = handler.get_tree_by_hash(&oid).await.map_err(|e| { SnapshotError::new( SnapshotErrorCode::Internal, format!("tree fetch failed for component '{comp}': {e}"), ) })?; + if fetched.id != expected_oid { + return Err(SnapshotError::new( + SnapshotErrorCode::IntegrityError, + "fetched fixed tree identity mismatch", + )); + } + current = Cow::Owned(fetched); } unreachable!("loop returns on the last component") } @@ -446,6 +1031,14 @@ pub fn hex_of(id: &[u8; 32]) -> String { hex(id) } +#[cfg(test)] +#[path = "native_projection_tests.rs"] +mod native_projection_tests; + +#[cfg(test)] +#[path = "retention_prepare_tests.rs"] +mod retention_prepare_tests; + #[cfg(test)] mod tests { use std::sync::Arc; @@ -829,9 +1422,16 @@ mod tests { ]) .unwrap(); let handler = crate::ceres::api_service::mono_api_service::MonoApiService::from(&state); - let built = build_directory_page_uncached(&handler, &tree, "/") - .await - .unwrap(); + let built = build_subtree( + &handler, + &state.storage, + &tree, + "/", + false, + &mut ProjectionWork::default(), + ) + .await + .unwrap(); assert_eq!(built.entries.len(), 2); assert!(built.entries.iter().all(|entry| entry.size == Some(10) && entry.content_digest == Some(Sha256::digest(raw).into()))); diff --git a/src/ceres/snapshot/projection_observation.rs b/src/ceres/snapshot/projection_observation.rs new file mode 100644 index 00000000..532b3f18 --- /dev/null +++ b/src/ceres/snapshot/projection_observation.rs @@ -0,0 +1,720 @@ +//! Operation-local resolve diagnostics, not publication or retention receipts. + +use std::time::Duration; + +use git_internal::hash::ObjectHash; +use mst2_codec::descriptor::{ + ACCESS_PROJECTION_EXACT_FULL, FS_SEMANTICS_LINUX_CODE_V1, MATERIALIZATION_POLICY_GIT_RAW_V1, + METADATA_CODEC, SCHEMA_VERSION, ServingDescriptor, +}; +use serde::Serialize; +use sha2::{Digest, Sha256}; +use uuid::Uuid; + +use crate::{ + ceres::snapshot::error::{SnapshotError, SnapshotErrorCode}, + jupiter::storage::mono_storage::MST2_VERIFICATION_VERSION, +}; + +pub(crate) const NATIVE_PROJECTION_REVISION: u16 = 1; + +/// Work in one directory projection. Page counters cover returned directory +/// roots, excluding codec-internal radix encoding and later route traversal. +#[derive(Debug, Clone, Default, PartialEq, Eq, Serialize)] +pub struct ProjectionWork { + pub directories_rebuilt: u64, + /// Cache-hit boundary visits, not all descendants or unique source OIDs. + pub reused_subtree_roots: u64, + pub directory_root_pages_built: u64, + pub directory_root_page_bytes_built: u64, + pub directory_root_pages_reused: u64, + pub directory_root_page_bytes_reused: u64, + pub directory_entries_scanned: u64, + pub scope_path_entries_examined: u64, + /// Backend tree requests, excluding the caller-supplied fixed root tree. + pub tree_fetches: u64, + /// Verification-record outcomes per file entry, not unique body OIDs. + pub verified_blob_hits: u64, + pub verified_blob_misses: u64, + /// Returned raw-body lengths, not physical disk/network traffic. + pub raw_bytes_fetched: u64, + /// Bytes passed to this builder's explicit raw-file SHA-256. + pub raw_bytes_hashed: u64, +} + +/// Captured only from the validated native head, before resolve projects it. +pub(crate) struct NativeResolveSource { + instance: Uuid, + commit: ObjectHash, + tree: ObjectHash, + certificate_receipt_id: u64, + writer_epoch: u64, + publication_sequence: u64, +} + +pub(crate) struct ResolvedProjection<'a> { + pub descriptor: &'a ServingDescriptor, + pub snapshot_id: &'a str, + pub metadata_root: &'a str, + pub context_commit: &'a str, + pub context_root_tree: &'a str, + pub fixed_root_tree: ObjectHash, + pub requested_scope: &'a str, + pub request_id: &'a str, +} + +/// Immutable successful resolve observation. This grants no authorization, +/// retention, durable publication or proof of codec-internal construction work. +pub(crate) struct NativeProjectionObservation { + source: NativeResolveSource, + scope: String, + namespace_view_id: String, + snapshot_id: String, + metadata_root: String, + request_id: String, + projection_elapsed_micros: u64, + work: ProjectionWork, + rooted_work: Option, +} + +/// Closed wire fields, borrowed only from the already validated observation. +#[derive(Serialize)] +pub(crate) struct ProjectionWireRecord<'a> { + observation_revision: u16, + phase: &'static str, + source_domain: &'static str, + request_id: &'a str, + instance_id: String, + #[serde(serialize_with = "tagged_oid")] + root_commit_oid: ObjectHash, + #[serde(serialize_with = "tagged_oid")] + root_tree_oid: ObjectHash, + native_certificate_receipt_id: u64, + native_writer_epoch: u64, + native_publication_sequence: u64, + scope: &'a str, + schema_version: u16, + metadata_codec: u16, + materialization_policy: u16, + fs_semantics: u16, + access_projection: u16, + verification_revision: i32, + projection_revision: u16, + namespace_view_id: &'a str, + snapshot_id: &'a str, + metadata_root: &'a str, + projection_elapsed_micros: u64, + page_counter_scope: &'static str, + codec_radix_work: &'static str, + #[serde(flatten)] + work: Option<&'a ProjectionWork>, + #[serde(skip_serializing_if = "Option::is_none")] + rooted_work: Option<&'a super::rooted_metadata_projection::RootedProjectionWork>, + message: &'static str, +} + +fn tagged_oid(oid: &ObjectHash, serializer: S) -> Result { + serializer.serialize_str(&oid.to_tagged_string()) +} + +fn invalid() -> SnapshotError { + SnapshotError::new( + SnapshotErrorCode::IntegrityError, + "native resolve observation does not match its fixed source and context", + ) +} + +impl NativeResolveSource { + pub(crate) fn capture( + instance_id: &str, + commit: ObjectHash, + tree: ObjectHash, + certificate_receipt_id: Option, + writer_epoch: i64, + publication_sequence: i64, + ) -> Result { + let instance = Uuid::parse_str(instance_id).map_err(|_| invalid())?; + let positive = |value: i64| { + u64::try_from(value) + .ok() + .filter(|value| *value > 0) + .ok_or_else(invalid) + }; + if instance.to_string() != instance_id || commit.kind() != tree.kind() { + return Err(invalid()); + } + Ok(Self { + instance, + commit, + tree, + certificate_receipt_id: positive(certificate_receipt_id.ok_or_else(invalid)?)?, + writer_epoch: positive(writer_epoch)?, + publication_sequence: positive(publication_sequence)?, + }) + } + + pub(crate) fn observe( + self, + resolved: ResolvedProjection<'_>, + work: ProjectionWork, + projection_elapsed: Duration, + ) -> Result { + let mut view_hash = Sha256::new(); + view_hash.update(b"mega.mst2.namespaceview\0"); + view_hash.update(self.commit.to_string().as_bytes()); + let expected_view: [u8; 32] = view_hash.finalize().into(); + let descriptor = resolved.descriptor; + let digest_text = |bytes: &[u8]| format!("sha256:{}", hex::encode(bytes)); + let snapshot = descriptor.snapshot_id().map_err(|_| invalid())?; + let bounded_request_id = !resolved.request_id.is_empty() + && resolved.request_id.len() <= 128 + && resolved.request_id.bytes().all(|byte| { + matches!(byte, b'a'..=b'z' | b'A'..=b'Z' | b'0'..=b'9' | b'-' | b'_' | b'.' | b':' | b'/') + }); + if resolved.context_commit != self.commit.to_string() + || resolved.context_root_tree != self.tree.to_string() + || resolved.fixed_root_tree != self.tree + || descriptor.instance_uuid != *self.instance.as_bytes() + || descriptor.namespace_view_id != expected_view + || descriptor.scope != resolved.requested_scope + || resolved.snapshot_id != digest_text(&snapshot) + || resolved.metadata_root != digest_text(&descriptor.metadata_root) + || !bounded_request_id + { + return Err(invalid()); + } + Ok(NativeProjectionObservation { + source: self, + scope: descriptor.scope.clone(), + namespace_view_id: digest_text(&descriptor.namespace_view_id), + snapshot_id: resolved.snapshot_id.to_owned(), + metadata_root: resolved.metadata_root.to_owned(), + request_id: resolved.request_id.to_owned(), + projection_elapsed_micros: u64::try_from(projection_elapsed.as_micros()) + .map_err(|_| invalid())?, + work, + rooted_work: None, + }) + } + + pub(crate) fn observe_rooted( + self, + resolved: ResolvedProjection<'_>, + work: super::rooted_metadata_projection::RootedProjectionWork, + projection_elapsed: Duration, + ) -> Result { + let mut observation = + self.observe(resolved, ProjectionWork::default(), projection_elapsed)?; + observation.rooted_work = Some(work); + Ok(observation) + } +} + +impl NativeProjectionObservation { + pub(crate) fn wire_record(&self) -> ProjectionWireRecord<'_> { + ProjectionWireRecord { + observation_revision: if self.rooted_work.is_some() { 2 } else { 1 }, + phase: if self.rooted_work.is_some() { + "resolve_rooted_projection" + } else { + "resolve_directory_projection" + }, + source_domain: "native-git", + request_id: &self.request_id, + instance_id: self.source.instance.to_string(), + root_commit_oid: self.source.commit, + root_tree_oid: self.source.tree, + native_certificate_receipt_id: self.source.certificate_receipt_id, + native_writer_epoch: self.source.writer_epoch, + native_publication_sequence: self.source.publication_sequence, + scope: &self.scope, + schema_version: SCHEMA_VERSION, + metadata_codec: METADATA_CODEC, + materialization_policy: MATERIALIZATION_POLICY_GIT_RAW_V1, + fs_semantics: FS_SEMANTICS_LINUX_CODE_V1, + access_projection: ACCESS_PROJECTION_EXACT_FULL, + verification_revision: MST2_VERIFICATION_VERSION, + projection_revision: NATIVE_PROJECTION_REVISION, + namespace_view_id: &self.namespace_view_id, + snapshot_id: &self.snapshot_id, + metadata_root: &self.metadata_root, + projection_elapsed_micros: self.projection_elapsed_micros, + page_counter_scope: if self.rooted_work.is_some() { + "physical-delta-and-reuse-boundaries" + } else { + "returned-directory-root-pages" + }, + codec_radix_work: if self.rooted_work.is_some() { + "OPERATION_CALLS_ONLY_NOT_TOTAL_SQL" + } else { + "NOT_EXPOSED" + }, + work: self.rooted_work.is_none().then_some(&self.work), + rooted_work: self.rooted_work.as_ref(), + message: "native resolve directory projection succeeded", + } + } + + pub(crate) fn emit(self) { + if let Some(work) = &self.rooted_work { + tracing::debug!(target:"mst2::native_projection_observation", + observation_revision=2u16,phase="resolve_rooted_projection",source_domain="native-git", + request_id=%self.request_id,instance_id=%self.source.instance, + root_commit_oid=%self.source.commit.to_tagged_string(),root_tree_oid=%self.source.tree.to_tagged_string(), + native_certificate_receipt_id=self.source.certificate_receipt_id, + native_writer_epoch=self.source.writer_epoch,native_publication_sequence=self.source.publication_sequence, + scope=?self.scope,snapshot_id=%self.snapshot_id,metadata_root=%self.metadata_root, + projection_elapsed_micros=self.projection_elapsed_micros,rooted_work=?work, + "native rooted projection succeeded; counters exclude total SQL body and catalog work"); + #[cfg(test)] + let _ = OBSERVATIONS.try_with(|observations| observations.lock().unwrap().push(self)); + return; + } + tracing::debug!( + target: "mst2::native_projection_observation", + observation_revision = 1u16, + phase = "resolve_directory_projection", + source_domain = "native-git", + request_id = %self.request_id, + instance_id = %self.source.instance, + root_commit_oid = %self.source.commit.to_tagged_string(), + root_tree_oid = %self.source.tree.to_tagged_string(), + native_certificate_receipt_id = self.source.certificate_receipt_id, + native_writer_epoch = self.source.writer_epoch, + native_publication_sequence = self.source.publication_sequence, + scope = ?self.scope, + schema_version = SCHEMA_VERSION, + metadata_codec = METADATA_CODEC, + materialization_policy = MATERIALIZATION_POLICY_GIT_RAW_V1, + fs_semantics = FS_SEMANTICS_LINUX_CODE_V1, + access_projection = ACCESS_PROJECTION_EXACT_FULL, + verification_revision = MST2_VERIFICATION_VERSION, + projection_revision = NATIVE_PROJECTION_REVISION, + namespace_view_id = %self.namespace_view_id, + snapshot_id = %self.snapshot_id, + metadata_root = %self.metadata_root, + projection_elapsed_micros = self.projection_elapsed_micros, + page_counter_scope = "returned-directory-root-pages", + codec_radix_work = "NOT_EXPOSED", + directories_rebuilt = self.work.directories_rebuilt, + reused_subtree_roots = self.work.reused_subtree_roots, + directory_root_pages_built = self.work.directory_root_pages_built, + directory_root_page_bytes_built = self.work.directory_root_page_bytes_built, + directory_root_pages_reused = self.work.directory_root_pages_reused, + directory_root_page_bytes_reused = self.work.directory_root_page_bytes_reused, + directory_entries_scanned = self.work.directory_entries_scanned, + scope_path_entries_examined = self.work.scope_path_entries_examined, + tree_fetches = self.work.tree_fetches, + verified_blob_hits = self.work.verified_blob_hits, + verified_blob_misses = self.work.verified_blob_misses, + raw_bytes_fetched = self.work.raw_bytes_fetched, + raw_bytes_hashed = self.work.raw_bytes_hashed, + "native resolve directory projection succeeded" + ); + #[cfg(test)] + let _ = OBSERVATIONS.try_with(|observations| observations.lock().unwrap().push(self)); + } +} + +#[cfg(test)] +tokio::task_local! { + static OBSERVATIONS: std::sync::Arc>>; +} + +#[cfg(test)] +pub(crate) async fn with_observations( + future: F, +) -> (F::Output, Vec) { + let observations = std::sync::Arc::new(std::sync::Mutex::new(Vec::new())); + let result = OBSERVATIONS.scope(observations.clone(), future).await; + let records = std::mem::take(&mut *observations.lock().unwrap()); + (result, records) +} + +#[cfg(test)] +impl NativeProjectionObservation { + pub(crate) fn test_identity(&self) -> serde_json::Value { + serde_json::json!({ + "instance_id": self.source.instance.to_string(), + "root_commit_oid": self.source.commit.to_tagged_string(), + "root_tree_oid": self.source.tree.to_tagged_string(), + "native_certificate_receipt_id": self.source.certificate_receipt_id, + "native_writer_epoch": self.source.writer_epoch.to_string(), + "native_publication_sequence": self.source.publication_sequence.to_string(), + "scope": self.scope, + "namespace_view_id": self.namespace_view_id, + "snapshot_id": self.snapshot_id, + "metadata_root": self.metadata_root, + "request_id": self.request_id, + }) + } +} + +#[cfg(test)] +mod tests { + use std::{ + collections::BTreeMap, + sync::{Arc, Mutex}, + }; + + use git_internal::hash::HashKind; + use tracing::{ + Event, Metadata, Subscriber, + field::{Field, Visit}, + span::{Attributes, Id, Record}, + }; + + use super::*; + + const INSTANCE: &str = "11111111-2222-4333-8444-555555555559"; + + struct Fixture { + commit: ObjectHash, + tree: ObjectHash, + descriptor: ServingDescriptor, + snapshot_id: String, + metadata_root: String, + } + + impl Fixture { + fn new() -> Self { + let commit = ObjectHash::from_hex_for_kind(HashKind::Sha256, &"a".repeat(64)).unwrap(); + let tree = ObjectHash::from_hex_for_kind(HashKind::Sha256, &"b".repeat(64)).unwrap(); + let mut view_hash = Sha256::new(); + view_hash.update(b"mega.mst2.namespaceview\0"); + view_hash.update(commit.to_string().as_bytes()); + let descriptor = ServingDescriptor { + instance_uuid: *Uuid::parse_str(INSTANCE).unwrap().as_bytes(), + namespace_view_id: view_hash.finalize().into(), + scope: "/project".to_owned(), + metadata_root: [0xc; 32], + }; + Self { + commit, + tree, + snapshot_id: format!("sha256:{}", hex::encode(descriptor.snapshot_id().unwrap())), + metadata_root: format!("sha256:{}", hex::encode(descriptor.metadata_root)), + descriptor, + } + } + + fn source(&self) -> NativeResolveSource { + NativeResolveSource::capture(INSTANCE, self.commit, self.tree, Some(97), 3, 7).unwrap() + } + + fn resolved<'a>(&'a self, commit: &'a str, tree: &'a str) -> ResolvedProjection<'a> { + ResolvedProjection { + descriptor: &self.descriptor, + snapshot_id: &self.snapshot_id, + metadata_root: &self.metadata_root, + context_commit: commit, + context_root_tree: tree, + fixed_root_tree: self.tree, + requested_scope: "/project", + request_id: "resolve:request-1", + } + } + } + + #[test] + fn observation_binds_fixed_source_context_descriptor_and_operation_work() { + let fixture = Fixture::new(); + let commit = fixture.commit.to_string(); + let tree = fixture.tree.to_string(); + let work = ProjectionWork { + directories_rebuilt: 3, + reused_subtree_roots: 70, + directory_entries_scanned: 104, + verified_blob_misses: 1, + ..ProjectionWork::default() + }; + let observation = fixture + .source() + .observe( + fixture.resolved(&commit, &tree), + work.clone(), + Duration::from_micros(123), + ) + .unwrap(); + assert_eq!(observation.source.certificate_receipt_id, 97); + assert_eq!(observation.source.publication_sequence, 7); + assert_eq!(observation.source.writer_epoch, 3); + assert_eq!( + observation.source.tree.to_tagged_string(), + format!("sha256:{tree}") + ); + assert_eq!(observation.projection_elapsed_micros, 123); + assert_eq!(observation.work, work); + assert_eq!(observation.scope, "/project"); + assert_eq!(observation.snapshot_id, fixture.snapshot_id); + } + + #[test] + fn rooted_observation_exposes_delta_boundaries_without_legacy_or_total_sql_counters() { + let fixture = Fixture::new(); + let commit = fixture.commit.to_string(); + let tree = fixture.tree.to_string(); + let work = super::super::rooted_metadata_projection::RootedProjectionWork { + tree_fetches: 1, + delta_pages: 2, + reused_roots: 7, + ..Default::default() + }; + let observation = fixture + .source() + .observe_rooted( + fixture.resolved(&commit, &tree), + work, + Duration::from_micros(456), + ) + .unwrap(); + let wire = serde_json::to_value(observation.wire_record()).unwrap(); + assert_eq!(wire["observation_revision"], 2); + assert_eq!(wire["phase"], "resolve_rooted_projection"); + assert_eq!( + wire["page_counter_scope"], + "physical-delta-and-reuse-boundaries" + ); + assert_eq!( + wire["codec_radix_work"], + "OPERATION_CALLS_ONLY_NOT_TOTAL_SQL" + ); + assert_eq!(wire["rooted_work"]["tree_fetches"], 1); + assert_eq!(wire["rooted_work"]["delta_pages"], 2); + assert_eq!(wire["rooted_work"]["reused_roots"], 7); + assert!(wire.get("directories_rebuilt").is_none()); + assert!(wire.get("directory_root_pages_returned").is_none()); + assert_eq!(wire["snapshot_id"], fixture.snapshot_id); + assert_eq!(wire["projection_elapsed_micros"], 456); + } + + #[test] + fn observation_rejects_each_changed_binding_or_unbounded_trace_token() { + let fixture = Fixture::new(); + let commit = fixture.commit.to_string(); + let tree = fixture.tree.to_string(); + for mode in 0..8 { + let mut descriptor = fixture.descriptor.clone(); + let mut resolved = fixture.resolved(&commit, &tree); + match mode { + 0 => resolved.context_commit = &tree, + 1 => resolved.context_root_tree = &commit, + 2 => resolved.fixed_root_tree = fixture.commit, + 3 => descriptor.instance_uuid[0] ^= 1, + 4 => descriptor.namespace_view_id[0] ^= 1, + 5 => resolved.requested_scope = "/different", + 6 => resolved.snapshot_id = &fixture.metadata_root, + 7 => resolved.metadata_root = &fixture.snapshot_id, + _ => unreachable!(), + } + resolved.descriptor = &descriptor; + assert!( + fixture + .source() + .observe(resolved, ProjectionWork::default(), Duration::ZERO) + .is_err(), + "mode {mode}" + ); + } + for request_id in [ + "".to_owned(), + "x".repeat(129), + "bad\ntrace".to_owned(), + "actor secret".to_owned(), + ] { + let mut resolved = fixture.resolved(&commit, &tree); + resolved.request_id = &request_id; + assert!( + fixture + .source() + .observe(resolved, ProjectionWork::default(), Duration::ZERO) + .is_err() + ); + } + } + + #[test] + fn capture_requires_positive_native_certificate_and_preserves_hash_kind() { + let fixture = Fixture::new(); + for (certificate, epoch, sequence) in [ + (None, 1, 1), + (Some(0), 1, 1), + (Some(-1), 1, 1), + (Some(1), 0, 1), + (Some(1), 1, 0), + ] { + assert!( + NativeResolveSource::capture( + INSTANCE, + fixture.commit, + fixture.tree, + certificate, + epoch, + sequence + ) + .is_err() + ); + } + let blake_tree = + ObjectHash::from_hex_for_kind(HashKind::Blake3, &fixture.tree.to_string()).unwrap(); + assert!( + NativeResolveSource::capture(INSTANCE, fixture.commit, blake_tree, Some(1), 1, 1) + .is_err() + ); + } + + struct TraceCapture(Arc>>>); + + impl Subscriber for TraceCapture { + fn enabled(&self, _: &Metadata<'_>) -> bool { + true + } + fn new_span(&self, _: &Attributes<'_>) -> Id { + Id::from_u64(1) + } + fn record(&self, _: &Id, _: &Record<'_>) {} + fn record_follows_from(&self, _: &Id, _: &Id) {} + fn enter(&self, _: &Id) {} + fn exit(&self, _: &Id) {} + fn event(&self, event: &Event<'_>) { + if event.metadata().target() != "mst2::native_projection_observation" { + return; + } + struct Fields(BTreeMap); + impl Visit for Fields { + fn record_debug(&mut self, field: &Field, value: &dyn std::fmt::Debug) { + self.0.insert(field.name().to_owned(), format!("{value:?}")); + } + fn record_u64(&mut self, field: &Field, value: u64) { + self.0.insert(field.name().to_owned(), value.to_string()); + } + fn record_str(&mut self, field: &Field, value: &str) { + self.0.insert(field.name().to_owned(), value.to_owned()); + } + } + let mut fields = Fields(BTreeMap::new()); + event.record(&mut fields); + self.0.lock().unwrap().push(fields.0); + } + } + + #[test] + fn actual_trace_has_closed_identity_profile_counter_fields_and_no_body_or_credentials() { + let fixture = Fixture::new(); + let commit = fixture.commit.to_string(); + let tree = fixture.tree.to_string(); + let observation = fixture + .source() + .observe( + fixture.resolved(&commit, &tree), + ProjectionWork::default(), + Duration::from_micros(4), + ) + .unwrap(); + let wire = serde_json::to_value(observation.wire_record()).unwrap(); + let events = Arc::new(Mutex::new(Vec::new())); + tracing::subscriber::with_default(TraceCapture(events.clone()), || observation.emit()); + let events = events.lock().unwrap(); + assert_eq!(events.len(), 1); + let fields = &events[0]; + let expected = [ + "observation_revision", + "phase", + "source_domain", + "request_id", + "instance_id", + "root_commit_oid", + "root_tree_oid", + "native_certificate_receipt_id", + "native_writer_epoch", + "native_publication_sequence", + "scope", + "schema_version", + "metadata_codec", + "materialization_policy", + "fs_semantics", + "access_projection", + "verification_revision", + "projection_revision", + "namespace_view_id", + "snapshot_id", + "metadata_root", + "projection_elapsed_micros", + "page_counter_scope", + "codec_radix_work", + "directories_rebuilt", + "reused_subtree_roots", + "directory_root_pages_built", + "directory_root_page_bytes_built", + "directory_root_pages_reused", + "directory_root_page_bytes_reused", + "directory_entries_scanned", + "scope_path_entries_examined", + "tree_fetches", + "verified_blob_hits", + "verified_blob_misses", + "raw_bytes_fetched", + "raw_bytes_hashed", + "message", + ]; + assert_eq!(fields.len(), expected.len()); + assert!(expected.iter().all(|field| fields.contains_key(*field))); + let wire = wire.as_object().unwrap(); + assert_eq!(wire.len(), expected.len()); + for key in expected { + let typed = &wire[key]; + let expected = if key == "scope" { + typed.to_string() + } else if let Some(text) = typed.as_str() { + text.into() + } else { + typed.to_string() + }; + assert_eq!(fields[key], expected, "typed observation diverged at {key}"); + } + assert_eq!(fields["phase"], "resolve_directory_projection"); + assert_eq!( + fields["page_counter_scope"], + "returned-directory-root-pages" + ); + assert_eq!(fields["codec_radix_work"], "NOT_EXPOSED"); + assert_eq!(fields["projection_elapsed_micros"], "4"); + assert_eq!(fields["native_certificate_receipt_id"], "97"); + assert_eq!(fields["scope"], "\"/project\""); + assert_eq!(fields["root_tree_oid"], fixture.tree.to_tagged_string()); + } + + #[tokio::test] + async fn emitted_observations_are_operation_local_with_no_publication_aggregation() { + let collect = |request: &'static str, sequence: i64| async move { + let fixture = Fixture::new(); + let commit = fixture.commit.to_string(); + let tree = fixture.tree.to_string(); + let mut resolved = fixture.resolved(&commit, &tree); + resolved.request_id = request; + let source = NativeResolveSource::capture( + INSTANCE, + fixture.commit, + fixture.tree, + Some(97), + 3, + sequence, + ) + .unwrap(); + source + .observe(resolved, ProjectionWork::default(), Duration::ZERO) + .unwrap() + .emit(); + }; + let (left, right) = tokio::join!( + with_observations(collect("request-left", 7)), + with_observations(collect("request-right", 8)) + ); + assert_eq!(left.1.len(), 1); + assert_eq!(right.1.len(), 1); + assert_eq!(left.1[0].request_id, "request-left"); + assert_eq!(right.1[0].request_id, "request-right"); + assert_eq!(left.1[0].source.publication_sequence, 7); + assert_eq!(right.1[0].source.publication_sequence, 8); + } +} diff --git a/src/ceres/snapshot/projection_writer.rs b/src/ceres/snapshot/projection_writer.rs new file mode 100644 index 00000000..a3d51852 --- /dev/null +++ b/src/ceres/snapshot/projection_writer.rs @@ -0,0 +1,506 @@ +//! Default-off typed projection observations. Buffer capacity is reserved +//! before serialization; logger filters and HTTP response bodies are unchanged. + +use std::{ + fs::{self, File, OpenOptions}, + io::{self, Write}, + path::{Path, PathBuf}, + sync::{ + Arc, Mutex, + atomic::{AtomicU8, AtomicU64, Ordering}, + mpsc::{self, Receiver, SyncSender, TrySendError}, + }, + thread::{self, JoinHandle}, + time::{Duration, Instant}, +}; + +use serde::Serialize; +use sha2::{Digest, Sha256}; +use uuid::Uuid; + +use super::projection_observation::NativeProjectionObservation; + +pub(crate) const RECORD_BYTES: usize = 32 * 1024; +pub(crate) const RECORD_LIMIT: usize = 64; +const STATUS_BYTES: usize = 4096; + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +#[repr(u8)] +pub(crate) enum WriterFailure { + QueueSaturated = 1, + WriterUnavailable, + RecordTooLarge, + RecordLimitExceeded, + ObservationBindingRejected, + IoFailure, + WorkerPanic, + DrainTimeout, +} + +impl WriterFailure { + fn label(self) -> &'static str { + match self { + Self::QueueSaturated => "QUEUE_SATURATED", + Self::WriterUnavailable => "WRITER_UNAVAILABLE", + Self::RecordTooLarge => "RECORD_TOO_LARGE", + Self::RecordLimitExceeded => "RECORD_LIMIT_EXCEEDED", + Self::ObservationBindingRejected => "OBSERVATION_BINDING_REJECTED", + Self::IoFailure => "IO_FAILURE", + Self::WorkerPanic => "WORKER_PANIC", + Self::DrainTimeout => "DRAIN_TIMEOUT", + } + } +} + +struct Health { + first_error: AtomicU8, + accepted: AtomicU64, +} + +impl Health { + fn fail(&self, error: WriterFailure) { + if self + .first_error + .compare_exchange(0, error as u8, Ordering::SeqCst, Ordering::SeqCst) + .is_ok() + { + tracing::warn!(target: "mst2::projection_writer", failure_code = error.label(), "projection observation delivery failed"); + } + } +} + +// Exactly 64 preallocated payload slots exist, including the slot currently +// being serialized and the worker's slot. No serialize-to-Vec-then-check path. +struct RecordBuffer { + bytes: Box<[u8; CAPACITY]>, + len: usize, +} + +impl Write for RecordBuffer { + fn write(&mut self, bytes: &[u8]) -> io::Result { + let end = self + .len + .checked_add(bytes.len()) + .filter(|end| *end <= CAPACITY) + .ok_or_else(|| io::Error::other("projection record capacity exhausted"))?; + self.bytes[self.len..end].copy_from_slice(bytes); + self.len = end; + Ok(bytes.len()) + } + fn flush(&mut self) -> io::Result<()> { + Ok(()) + } +} + +struct QueuedRecord { + sequence: u64, + buffer: RecordBuffer, +} +type Pool = Arc>>; +struct Producer { + sender: Option>, +} + +pub(crate) struct ProjectionObservationSink { + id: Uuid, + producer: Mutex, + pool: Pool, + health: Arc, + worker: Mutex>>, +} + +impl ProjectionObservationSink { + pub(crate) fn start(cache: &Path) -> io::Result> { + let id = Uuid::new_v4(); + let root = prepare_directory(cache, id)?; + let records = create_private(&root.join("records.jsonl"))?; + let health = Arc::new(Health { + first_error: AtomicU8::new(0), + accepted: AtomicU64::new(0), + }); + let pool = Arc::new(Mutex::new( + (0..RECORD_LIMIT) + .map(|_| RecordBuffer { + bytes: Box::new([0; RECORD_BYTES]), + len: 0, + }) + .collect::>(), + )); + let mut writer = FileWriter::new(root, records, id, health.clone()); + writer.status(false)?; + let (sender, receiver) = mpsc::sync_channel(RECORD_LIMIT); + let worker_pool = pool.clone(); + let worker_health = health.clone(); + let worker = thread::Builder::new() + .name("mst2-projection-writer".into()) + .spawn(move || { + let outcome = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| { + writer.run(receiver, worker_pool); + })); + if outcome.is_err() { + worker_health.fail(WriterFailure::WorkerPanic); + let _ = writer.status(true); + } + })?; + tracing::info!(target: "mst2::projection_writer", writer_revision = 1u16, sink_instance = %id, + payload_slots = RECORD_LIMIT, payload_slot_bytes = RECORD_BYTES, + "typed projection writer started"); + Ok(Arc::new(Self { + id, + producer: Mutex::new(Producer { + sender: Some(sender), + }), + pool, + health, + worker: Mutex::new(Some(worker)), + })) + } + + pub(crate) fn reject_binding(&self) { + self.health.fail(WriterFailure::ObservationBindingRejected); + } + + pub(crate) fn enqueue( + &self, + observation: &NativeProjectionObservation, + ) -> Result<(), WriterFailure> { + let result = self.enqueue_inner(observation); + if let Err(error) = result { + self.health.fail(error); + } + result + } + + fn enqueue_inner( + &self, + observation: &NativeProjectionObservation, + ) -> Result<(), WriterFailure> { + if self.health.first_error.load(Ordering::SeqCst) != 0 { + return Err(WriterFailure::WriterUnavailable); + } + let producer = self + .producer + .try_lock() + .map_err(|_| WriterFailure::QueueSaturated)?; + let sender = producer + .sender + .as_ref() + .ok_or(WriterFailure::WriterUnavailable)?; + let sequence = self.health.accepted.load(Ordering::SeqCst) + 1; + if sequence > RECORD_LIMIT as u64 { + return Err(WriterFailure::RecordLimitExceeded); + } + let mut buffer = self + .pool + .try_lock() + .map_err(|_| WriterFailure::QueueSaturated)? + .pop() + .ok_or(WriterFailure::QueueSaturated)?; + buffer.len = 0; + let serialized = (|| -> io::Result<()> { + write!( + buffer, + "{{\"writer_revision\":1,\"sink_instance\":\"{}\",\"record_sequence\":{},\"payload\":", + self.id, sequence + )?; + let begin = buffer.len; + serde_json::to_writer(&mut buffer, &observation.wire_record()) + .map_err(io::Error::other)?; + let digest = Sha256::digest(&buffer.bytes[begin..buffer.len]); + writeln!( + buffer, + ",\"payload_sha256\":\"sha256:{}\"}}", + hex::encode(digest) + ) + })(); + if serialized.is_err() { + self.return_slot(buffer); + return Err(WriterFailure::RecordTooLarge); + } + self.health.accepted.store(sequence, Ordering::SeqCst); + match sender.try_send(QueuedRecord { sequence, buffer }) { + Ok(()) => Ok(()), + Err(error) => { + self.health.accepted.store(sequence - 1, Ordering::SeqCst); + let (error, record) = match error { + TrySendError::Full(record) => (WriterFailure::QueueSaturated, record), + TrySendError::Disconnected(record) => { + (WriterFailure::WriterUnavailable, record) + } + }; + self.return_slot(record.buffer); + Err(error) + } + } + } + + fn return_slot(&self, mut buffer: RecordBuffer) { + buffer.len = 0; + if let Ok(mut pool) = self.pool.lock() { + pool.push(buffer); + } + } + + pub(crate) async fn shutdown(&self, deadline: Instant) -> Result<(), WriterFailure> { + self.producer + .lock() + .map_err(|_| WriterFailure::WriterUnavailable)? + .sender + .take(); + loop { + let finished = self + .worker + .lock() + .map_err(|_| WriterFailure::WriterUnavailable)? + .as_ref() + .is_none_or(JoinHandle::is_finished); + if finished { + break; + } + if Instant::now() >= deadline { + self.health.fail(WriterFailure::DrainTimeout); + return Err(WriterFailure::DrainTimeout); + } + tokio::time::sleep( + Duration::from_millis(10).min(deadline.saturating_duration_since(Instant::now())), + ) + .await; + } + if let Some(worker) = self + .worker + .lock() + .map_err(|_| WriterFailure::WriterUnavailable)? + .take() + && worker.join().is_err() + { + self.health.fail(WriterFailure::WorkerPanic); + } + if self.health.first_error.load(Ordering::SeqCst) != 0 { + return Err(WriterFailure::WriterUnavailable); + } + if Instant::now() >= deadline { + self.health.fail(WriterFailure::DrainTimeout); + return Err(WriterFailure::DrainTimeout); + } + Ok(()) + } + + #[cfg(test)] + pub(crate) fn test_directory(&self, cache: &Path) -> PathBuf { + cache + .join("logs/mst2-native-projection") + .join(self.id.to_string()) + } +} + +#[derive(Serialize)] +struct WriterStatus { + writer_revision: u16, + sink_instance: String, + accepted_records: u64, + written_sequence: u64, + written_records: u64, + written_bytes: u64, + rolling_sha256: String, + first_error_code: u8, + closed: bool, +} + +struct FileWriter { + root: PathBuf, + records: File, + id: Uuid, + health: Arc, + sequence: u64, + bytes: u64, + digest: Sha256, +} +impl FileWriter { + fn new(root: PathBuf, records: File, id: Uuid, health: Arc) -> Self { + Self { + root, + records, + id, + health, + sequence: 0, + bytes: 0, + digest: Sha256::new(), + } + } + fn status(&self, closed: bool) -> io::Result<()> { + let status = WriterStatus { + writer_revision: 1, + sink_instance: self.id.to_string(), + accepted_records: self.health.accepted.load(Ordering::SeqCst), + written_sequence: self.sequence, + written_records: self.sequence, + written_bytes: self.bytes, + rolling_sha256: format!("sha256:{}", hex::encode(self.digest.clone().finalize())), + first_error_code: self.health.first_error.load(Ordering::SeqCst), + closed, + }; + let mut buffer = RecordBuffer:: { + bytes: Box::new([0; STATUS_BYTES]), + len: 0, + }; + serde_json::to_writer(&mut buffer, &status).map_err(io::Error::other)?; + let temporary = self.root.join("status.tmp"); + let mut file = create_private(&temporary)?; + file.write_all(&buffer.bytes[..buffer.len])?; + file.sync_all()?; + fs::rename(temporary, self.root.join("status.json"))?; + sync_directory(&self.root) + } + fn run(&mut self, receiver: Receiver, pool: Pool) { + let mut reported_error = 0; + loop { + match receiver.recv_timeout(Duration::from_millis(100)) { + Ok(mut record) => { + let result = (|| -> io::Result<()> { + if record.sequence != self.sequence + 1 + || record.sequence > RECORD_LIMIT as u64 + { + return Err(io::Error::other("projection writer sequence mismatch")); + } + self.records + .write_all(&record.buffer.bytes[..record.buffer.len])?; + self.records.flush()?; + self.records.sync_all()?; + self.digest + .update(&record.buffer.bytes[..record.buffer.len]); + self.bytes += record.buffer.len as u64; + self.sequence = record.sequence; + self.status(false) + })(); + record.buffer.len = 0; + if let Ok(mut slots) = pool.lock() { + slots.push(record.buffer); + } + if result.is_err() { + self.health.fail(WriterFailure::IoFailure); + break; + } + } + Err(mpsc::RecvTimeoutError::Timeout) => {} + Err(mpsc::RecvTimeoutError::Disconnected) => break, + } + let error = self.health.first_error.load(Ordering::SeqCst); + if error != 0 && error != reported_error { + if self.status(false).is_err() { + self.health.fail(WriterFailure::IoFailure); + break; + } + reported_error = error; + } + } + if self.status(true).is_err() { + self.health.fail(WriterFailure::IoFailure); + } + } +} + +fn create_private(path: &Path) -> io::Result { + let mut options = OpenOptions::new(); + options.write(true).create_new(true); + #[cfg(unix)] + { + use std::os::unix::fs::OpenOptionsExt; + options.mode(0o600); + } + options.open(path) +} + +fn prepare_directory(cache: &Path, id: Uuid) -> io::Result { + prepare_directory_with(cache, id, sync_directory) +} + +fn real_directory(path: &Path) -> io::Result<()> { + if !fs::symlink_metadata(path)?.is_dir() { + return Err(io::Error::other( + "projection storage must be a real directory", + )); + } + Ok(()) +} + +fn create_directory(path: &Path, private: bool) -> io::Result<()> { + #[cfg(unix)] + let mut directory = fs::DirBuilder::new(); + #[cfg(not(unix))] + let directory = fs::DirBuilder::new(); + #[cfg(unix)] + { + use std::os::unix::fs::DirBuilderExt; + directory.mode(if private { 0o700 } else { 0o755 }); + } + #[cfg(not(unix))] + let _ = private; + directory.create(path) +} + +fn private_directory(path: &Path) -> io::Result<()> { + real_directory(path)?; + #[cfg(unix)] + { + use std::os::unix::fs::{MetadataExt, PermissionsExt}; + let metadata = fs::symlink_metadata(path)?; + // SAFETY: geteuid reads the process identity and has no pointer arguments. + let owner = unsafe { libc::geteuid() }; + if metadata.uid() != owner || metadata.permissions().mode() & 0o7777 != 0o700 { + return Err(io::Error::other( + "projection directory must be privately owned", + )); + } + } + Ok(()) +} + +fn prepare_directory_with( + cache: &Path, + id: Uuid, + sync: impl Fn(&Path) -> io::Result<()>, +) -> io::Result { + // The actual configured cache is an existing anchor. No arbitrary output + // path or recursively created ancestor is accepted by this writer. + real_directory(cache)?; + let logs = cache.join("logs"); + match create_directory(&logs, false) { + Ok(()) => {} + Err(error) if error.kind() == io::ErrorKind::AlreadyExists => {} + Err(error) => return Err(error), + } + real_directory(&logs)?; + sync(cache)?; + let parent = logs.join("mst2-native-projection"); + match create_directory(&parent, true) { + Ok(()) => {} + Err(error) if error.kind() == io::ErrorKind::AlreadyExists => {} + Err(error) => return Err(error), + } + private_directory(&parent)?; + sync(&logs)?; + let root = parent.join(id.to_string()); + create_directory(&root, true)?; + private_directory(&root)?; + sync(&parent)?; + Ok(root) +} + +fn sync_directory(path: &Path) -> io::Result<()> { + #[cfg(unix)] + { + File::open(path)?.sync_all() + } + #[cfg(not(unix))] + { + let _ = path; + Err(io::Error::new( + io::ErrorKind::Unsupported, + "durable projection writer requires directory fsync", + )) + } +} + +#[cfg(test)] +#[path = "projection_writer_tests.rs"] +mod tests; diff --git a/src/ceres/snapshot/projection_writer_tests.rs b/src/ceres/snapshot/projection_writer_tests.rs new file mode 100644 index 00000000..2de737a2 --- /dev/null +++ b/src/ceres/snapshot/projection_writer_tests.rs @@ -0,0 +1,358 @@ +use git_internal::hash::{HashKind, ObjectHash}; +use mst2_codec::descriptor::ServingDescriptor; + +use super::*; +use crate::ceres::snapshot::projection_observation::{ + NativeResolveSource, ProjectionWork, ResolvedProjection, +}; + +fn observation(scope: &str) -> NativeProjectionObservation { + let commit = ObjectHash::from_hex_for_kind(HashKind::Sha256, &"a".repeat(64)).unwrap(); + let tree = ObjectHash::from_hex_for_kind(HashKind::Sha256, &"b".repeat(64)).unwrap(); + let instance = Uuid::new_v4(); + let mut namespace = Sha256::new(); + namespace.update(b"mega.mst2.namespaceview\0"); + namespace.update(commit.to_string().as_bytes()); + let descriptor = ServingDescriptor { + instance_uuid: *instance.as_bytes(), + namespace_view_id: namespace.finalize().into(), + scope: scope.into(), + metadata_root: [3; 32], + }; + let snapshot = format!("sha256:{}", hex::encode(descriptor.snapshot_id().unwrap())); + let root = format!("sha256:{}", hex::encode(descriptor.metadata_root)); + NativeResolveSource::capture(&instance.to_string(), commit, tree, Some(1), 2, 3) + .unwrap() + .observe( + ResolvedProjection { + descriptor: &descriptor, + snapshot_id: &snapshot, + metadata_root: &root, + context_commit: &commit.to_string(), + context_root_tree: &tree.to_string(), + fixed_root_tree: tree, + requested_scope: scope, + request_id: "writer:test:a1", + }, + ProjectionWork::default(), + Duration::from_micros(4), + ) + .unwrap() +} + +fn idle_sink() -> (ProjectionObservationSink, Receiver) { + let (sender, receiver) = mpsc::sync_channel(RECORD_LIMIT); + let sink = ProjectionObservationSink { + id: Uuid::new_v4(), + producer: Mutex::new(Producer { + sender: Some(sender), + }), + pool: Arc::new(Mutex::new( + (0..RECORD_LIMIT) + .map(|_| RecordBuffer { + bytes: Box::new([0; RECORD_BYTES]), + len: 0, + }) + .collect(), + )), + health: Arc::new(Health { + first_error: AtomicU8::new(0), + accepted: AtomicU64::new(0), + }), + worker: Mutex::new(None), + }; + (sink, receiver) +} + +#[test] +fn bounded_slots_precede_serialization_and_saturation_is_sticky_with_no_extra_allocation() { + let (sink, receiver) = idle_sink(); + let observation = observation("/project"); + for _ in 0..RECORD_LIMIT { + sink.enqueue(&observation).unwrap(); + } + assert_eq!(sink.pool.lock().unwrap().len(), 0); + assert_eq!( + sink.health.accepted.load(Ordering::SeqCst), + RECORD_LIMIT as u64 + ); + assert_eq!( + sink.enqueue(&observation), + Err(WriterFailure::RecordLimitExceeded) + ); + assert_eq!( + sink.health.first_error.load(Ordering::SeqCst), + WriterFailure::RecordLimitExceeded as u8 + ); + let record = receiver.try_recv().unwrap(); + sink.return_slot(record.buffer); + assert_eq!( + sink.enqueue(&observation), + Err(WriterFailure::WriterUnavailable) + ); + assert_eq!(sink.pool.lock().unwrap().len(), 1); +} + +#[test] +fn pool_saturation_and_disconnection_have_typed_failure_and_return_the_actual_slot() { + let (sink, receiver) = idle_sink(); + let observation = observation("/project"); + let held = sink.pool.lock().unwrap().drain(..).collect::>(); + assert_eq!( + sink.enqueue(&observation), + Err(WriterFailure::QueueSaturated) + ); + assert_eq!(held.len(), RECORD_LIMIT); + assert_eq!(sink.health.accepted.load(Ordering::SeqCst), 0); + let (sink, receiver2) = idle_sink(); + drop(receiver2); + assert_eq!( + sink.enqueue(&observation), + Err(WriterFailure::WriterUnavailable) + ); + assert_eq!(sink.pool.lock().unwrap().len(), RECORD_LIMIT); + assert_eq!(sink.health.accepted.load(Ordering::SeqCst), 0); + drop(receiver); +} + +#[test] +fn typed_wire_contract_has_exact_38_fields_and_rejects_oversize_without_growing_buffer() { + let (sink, receiver) = idle_sink(); + let captured = observation("/project\"quoted"); + sink.enqueue(&captured).unwrap(); + let record = receiver.try_recv().unwrap(); + assert_eq!(record.buffer.bytes.len(), RECORD_BYTES); + let envelope: serde_json::Value = + serde_json::from_slice(&record.buffer.bytes[..record.buffer.len]).unwrap(); + let payload = &envelope["payload"]; + assert_eq!(payload.as_object().unwrap().len(), 38); + assert_eq!(payload["scope"], "/project\"quoted"); + assert_eq!(payload["request_id"], "writer:test:a1"); + assert_eq!(payload["projection_elapsed_micros"], 4); + assert_eq!(payload["codec_radix_work"], "NOT_EXPOSED"); + let mut small = RecordBuffer::<64> { + bytes: Box::new([0; 64]), + len: 0, + }; + // Serialize a genuine validated observation into an insufficient reserved + // slot, rather than bypassing the descriptor's component length checks. + assert!(serde_json::to_writer(&mut small, &captured.wire_record()).is_err()); + assert!(small.len <= 64); + assert_eq!(small.bytes.len(), 64); + let len = small.len; + assert!(small.write_all(&[0; 65]).is_err()); + assert_eq!(small.len, len); + let wire = &record.buffer.bytes[..record.buffer.len]; + let prefix = b"\"payload\":"; + let begin = wire + .windows(prefix.len()) + .position(|part| part == prefix) + .unwrap() + + prefix.len(); + let suffix = b",\"payload_sha256\":"; + let end = wire + .windows(suffix.len()) + .position(|part| part == suffix) + .unwrap(); + assert_eq!( + envelope["payload_sha256"], + format!("sha256:{}", hex::encode(Sha256::digest(&wire[begin..end]))) + ); +} + +#[cfg(unix)] +#[tokio::test] +async fn actual_writer_fsync_ack_tracks_exact_bytes_and_closed_drain() { + let temp = tempfile::tempdir().unwrap(); + let sink = ProjectionObservationSink::start(temp.path()).unwrap(); + sink.enqueue(&observation("/project")).unwrap(); + sink.shutdown(Instant::now() + Duration::from_secs(5)) + .await + .unwrap(); + let root = temp + .path() + .join("logs/mst2-native-projection") + .join(sink.id.to_string()); + let wire = fs::read(root.join("records.jsonl")).unwrap(); + let status: serde_json::Value = + serde_json::from_slice(&fs::read(root.join("status.json")).unwrap()).unwrap(); + assert_eq!(status["written_bytes"], wire.len()); + assert_eq!(status["written_sequence"], 1); + assert_eq!(status["accepted_records"], 1); + assert_eq!(status["first_error_code"], 0); + assert_eq!(status["closed"], true); + assert_eq!( + status["rolling_sha256"], + format!("sha256:{}", hex::encode(Sha256::digest(&wire))) + ); + let record: serde_json::Value = serde_json::from_slice(&wire).unwrap(); + assert_eq!(record["payload"].as_object().unwrap().len(), 38); + assert_eq!(sink.pool.lock().unwrap().len(), RECORD_LIMIT); + use std::os::unix::fs::PermissionsExt; + assert_eq!( + fs::metadata(&root).unwrap().permissions().mode() & 0o777, + 0o700 + ); + assert_eq!( + fs::metadata(root.parent().unwrap()) + .unwrap() + .permissions() + .mode() + & 0o7777, + 0o700 + ); + for path in [root.join("records.jsonl"), root.join("status.json")] { + assert_eq!( + fs::metadata(path).unwrap().permissions().mode() & 0o777, + 0o600 + ); + } +} + +#[tokio::test] +async fn expired_drain_deadline_cannot_report_a_successful_capture() { + let (sink, _) = idle_sink(); + assert_eq!( + sink.shutdown(Instant::now()).await, + Err(WriterFailure::DrainTimeout) + ); + assert_eq!( + sink.health.first_error.load(Ordering::SeqCst), + WriterFailure::DrainTimeout as u8 + ); + assert_eq!( + sink.enqueue(&observation("/project")), + Err(WriterFailure::WriterUnavailable) + ); +} + +#[cfg(unix)] +#[tokio::test] +async fn actual_status_write_failure_and_binding_rejection_cannot_drain_successfully() { + let temp = tempfile::tempdir().unwrap(); + let sink = ProjectionObservationSink::start(temp.path()).unwrap(); + let root = temp + .path() + .join("logs/mst2-native-projection") + .join(sink.id.to_string()); + fs::write(root.join("status.tmp"), b"blocked exclusive temporary").unwrap(); + sink.enqueue(&observation("/project")).unwrap(); + assert!( + sink.shutdown(Instant::now() + Duration::from_secs(5)) + .await + .is_err() + ); + let status: serde_json::Value = + serde_json::from_slice(&fs::read(root.join("status.json")).unwrap()).unwrap(); + assert_eq!( + status["written_sequence"], 0, + "no durable sidecar acknowledgement for this record" + ); + assert_eq!( + sink.health.first_error.load(Ordering::SeqCst), + WriterFailure::IoFailure as u8 + ); + let temp = tempfile::tempdir().unwrap(); + let sink = ProjectionObservationSink::start(temp.path()).unwrap(); + sink.reject_binding(); + assert!( + sink.shutdown(Instant::now() + Duration::from_secs(5)) + .await + .is_err() + ); + let root = temp + .path() + .join("logs/mst2-native-projection") + .join(sink.id.to_string()); + let status: serde_json::Value = + serde_json::from_slice(&fs::read(root.join("status.json")).unwrap()).unwrap(); + assert_eq!( + status["first_error_code"], + WriterFailure::ObservationBindingRejected as u8 + ); +} + +#[cfg(unix)] +#[test] +fn actual_writer_rejects_a_symlinked_output_directory_before_creating_records() { + use std::os::unix::fs::symlink; + let temp = tempfile::tempdir().unwrap(); + let target = tempfile::tempdir().unwrap(); + fs::create_dir(temp.path().join("logs")).unwrap(); + symlink( + target.path(), + temp.path().join("logs/mst2-native-projection"), + ) + .unwrap(); + assert!(ProjectionObservationSink::start(temp.path()).is_err()); + assert_eq!(target.path().read_dir().unwrap().count(), 0); + let temp = tempfile::tempdir().unwrap(); + symlink(target.path(), temp.path().join("logs")).unwrap(); + assert!(ProjectionObservationSink::start(temp.path()).is_err()); + assert_eq!(target.path().read_dir().unwrap().count(), 0); +} + +#[cfg(unix)] +#[test] +fn directory_chain_is_synced_from_existing_anchor_before_any_record_can_be_created() { + use std::cell::RefCell; + let temp = tempfile::tempdir().unwrap(); + let calls = RefCell::new(Vec::new()); + let root = prepare_directory_with(temp.path(), Uuid::new_v4(), |path| { + sync_directory(path)?; + calls.borrow_mut().push(path.to_path_buf()); + Ok(()) + }) + .unwrap(); + assert_eq!( + *calls.borrow(), + [ + temp.path().to_path_buf(), + temp.path().join("logs"), + temp.path().join("logs/mst2-native-projection") + ] + ); + assert_eq!(root.read_dir().unwrap().count(), 0); + for fail_at in 0..3 { + let temp = tempfile::tempdir().unwrap(); + let id = Uuid::new_v4(); + let calls = RefCell::new(0); + assert!( + prepare_directory_with(temp.path(), id, |path| { + let current = *calls.borrow(); + *calls.borrow_mut() += 1; + if current == fail_at { + Err(io::Error::other("injected directory durability failure")) + } else { + sync_directory(path) + } + }) + .is_err() + ); + assert_eq!(*calls.borrow(), fail_at + 1); + let root = temp + .path() + .join("logs/mst2-native-projection") + .join(id.to_string()); + assert!(!root.join("records.jsonl").exists()); + assert!(!root.join("status.json").exists()); + } +} + +#[cfg(unix)] +#[test] +fn missing_or_symlink_anchor_and_nonprivate_existing_parent_fail_before_records() { + use std::os::unix::fs::{PermissionsExt, symlink}; + let temp = tempfile::tempdir().unwrap(); + assert!(ProjectionObservationSink::start(&temp.path().join("missing")).is_err()); + assert!(!temp.path().join("missing").exists()); + let alias = temp.path().join("alias"); + symlink(temp.path(), &alias).unwrap(); + assert!(ProjectionObservationSink::start(&alias).is_err()); + assert!(!temp.path().join("logs").exists()); + let parent = temp.path().join("logs/mst2-native-projection"); + fs::create_dir_all(&parent).unwrap(); + fs::set_permissions(&parent, fs::Permissions::from_mode(0o755)).unwrap(); + assert!(ProjectionObservationSink::start(temp.path()).is_err()); + assert_eq!(parent.read_dir().unwrap().count(), 0); +} diff --git a/src/ceres/snapshot/retention.rs b/src/ceres/snapshot/retention.rs index 54bba4da..fa9a12f6 100644 --- a/src/ceres/snapshot/retention.rs +++ b/src/ceres/snapshot/retention.rs @@ -105,7 +105,7 @@ pub enum ReapDecision { #[derive(Debug, Clone, PartialEq, Eq, Default)] pub struct CollectionReport { - /// Node ids that became unreachable this run. + /// Node ids atomically marked DELETING this run. pub unreachable: Vec, /// Bytes that would be / were reclaimed. pub reclaimed_bytes: u64, @@ -140,16 +140,30 @@ impl Reaper for NoopReaper { pub trait RetentionStore { fn node(&self, id: &str) -> Option; fn root_covers(&self, node_id: &str) -> bool; + /// Count incoming edges from parents not yet removed, including + /// DELETING parents whose physical reclaim has not succeeded. fn live_incoming(&self, node_id: &str) -> usize; - /// Idempotent: create the node if absent, upsert edges and roots in - /// one atomic step (spec §6 "no empty window"). + /// Idempotent: retain one LIVE node with its edges and root coverage. fn retain( &self, node: RetentionNode, edges: &[RetentionEdge], roots: &[RetentionRoot], + ) -> Result<(), SnapshotError> { + self.retain_group(std::slice::from_ref(&node), edges, roots) + } + /// Atomically retain the entire group, covering every supplied node + /// with `roots`. All nodes and edge endpoints must be LIVE; acquiring + /// a DELETING node is forbidden. Any error leaves the graph unchanged. + fn retain_group( + &self, + nodes: &[RetentionNode], + edges: &[RetentionEdge], + roots: &[RetentionRoot], ) -> Result<(), SnapshotError>; - /// Atomically CAS a node LIVE→DELETING; false if it is no longer LIVE. + /// Atomically check zero roots/incoming references and CAS LIVE→DELETING. + /// A parent still protects its children until the parent is removed, + /// including while the parent is DELETING and physical reclaim is pending. fn mark_deleting(&self, id: &str) -> bool; /// Remove a DELETING node after the reaper succeeds. fn remove(&self, id: &str); @@ -182,14 +196,6 @@ impl RetentionCoordinator { parents: &[String], roots: &[RetentionRoot], ) -> Result<(), SnapshotError> { - for p in parents { - if self.store.node(p).is_none() { - return Err(SnapshotError::new( - SnapshotErrorCode::Internal, - format!("retention edge to unknown parent {p}"), - )); - } - } let edges: Vec = parents .iter() .map(|p| RetentionEdge { @@ -200,38 +206,24 @@ impl RetentionCoordinator { self.store.retain(node, &edges, roots) } - /// Add coverage from a root to a node, atomically re-lifting it out of - /// a DELETING state (spec §10). The root is also materialized as a live - /// anchor node so reachability can start from it. The store's `retain` - /// is the upsert, so both the fresh and the already-known cases take - /// the same path. + /// Acquire root coverage for the complete group in one store operation. + /// The root is also materialized as a LIVE anchor. If any covered node + /// is DELETING, neither the anchor nor any partial coverage is retained. pub fn pin_root( &self, root: &RetentionRoot, covered: &[RetentionNode], ) -> Result<(), SnapshotError> { - let roots = std::slice::from_ref(root); - self.store.retain( - RetentionNode { - id: root.key(), - kind: RetainedKind::Frame, - state: NodeState::Live, - bytes: 0, - }, - &[], - roots, - )?; - for n in covered { - self.store.retain( - RetentionNode { - state: NodeState::Live, - ..n.clone() - }, - &[], - roots, - )?; - } - Ok(()) + let mut nodes = Vec::with_capacity(covered.len() + 1); + nodes.push(RetentionNode { + id: root.key(), + kind: RetainedKind::Frame, + state: NodeState::Live, + bytes: 0, + }); + nodes.extend_from_slice(covered); + self.store + .retain_group(&nodes, &[], std::slice::from_ref(root)) } pub fn release(&self, root: &RetentionRoot) { @@ -239,11 +231,10 @@ impl RetentionCoordinator { } /// One collection pass (spec 10 §6 steps 1–7): - /// mark unreachable LIVE nodes DELETING, then reap. Nodes marked - /// DELETING are skipped entirely on later passes if a new root/edge - /// raced in only after the CAS — here reachability is recomputed under - /// the store lock, so a retain concurrent with a pass either observes - /// the node (keeping it LIVE) or lands before the next pass. + /// mark unreachable LIVE nodes DELETING, then reap. The reachability + /// scan selects candidates; the store rechecks roots and incoming + /// references atomically with the CAS. Acquisition either wins before + /// that CAS or fails because the node is already DELETING. pub fn collect(&self, reaper: &R) -> Result { let live = self.store.all_live(); let reachable = self.reachable_set(); @@ -254,18 +245,11 @@ impl RetentionCoordinator { continue; } if !self.store.mark_deleting(&node.id) { - // Lost the CAS: a new reference made it LIVE again. + // Another collector or a newly acquired reference won. continue; } report.unreachable.push(node.id.clone()); report.reclaimed_bytes += node.bytes; - // Re-verify after the CAS: the store serializes retain against - // mark, so a zero incoming count here is stable for this pass. - if self.store.root_covers(&node.id) || self.store.live_incoming(&node.id) > 0 { - // A concurrent retain resurrected it; the CAS semantics of - // the store keep it LIVE, so do not reap. - continue; - } if reaper.physical() { reaper.reap(&node)?; self.store.remove(&node.id); @@ -311,6 +295,8 @@ pub mod mem { edges: HashSet, /// node id -> set of root keys covering it. roots: HashMap>, + #[cfg(test)] + fail_next_retain: bool, } /// Single-Mutex store; the lock is the serialization point that makes @@ -327,6 +313,14 @@ pub mod mem { } } + #[cfg(test)] + impl InMemoryRetentionStore { + /// Inject a backend failure after validation and before committing. + pub(crate) fn fail_next_retain_for_test(&self) { + self.inner.lock().unwrap().fail_next_retain = true; + } + } + impl RetentionStore for InMemoryRetentionStore { fn node(&self, id: &str) -> Option { self.inner.lock().unwrap().nodes.get(id).cloned() @@ -350,41 +344,82 @@ pub mod mem { .count() } - fn retain( + fn retain_group( &self, - node: RetentionNode, + nodes: &[RetentionNode], edges: &[RetentionEdge], roots: &[RetentionRoot], ) -> Result<(), SnapshotError> { let mut g = self.inner.lock().unwrap(); - // Atomic re-lift: never downgrade a LIVE node to DELETING. - g.nodes - .entry(node.id.clone()) - .and_modify(|existing| { - existing.bytes = node.bytes; - existing.kind = node.kind; - existing.state = NodeState::Live; - }) - .or_insert_with(|| node.clone()); - // Defensive: do not insert an edge to a missing child node. - for e in edges { - if !g.nodes.contains_key(&e.child) || !g.nodes.contains_key(&e.parent) { + // Validate the entire transaction before mutating any graph data. + let mut staged = HashMap::new(); + for node in nodes { + if node.state != NodeState::Live + || g.nodes + .get(&node.id) + .is_some_and(|existing| existing.state != NodeState::Live) + { + return Err(SnapshotError::new( + SnapshotErrorCode::ObjectUnavailable, + format!("retention acquire requires LIVE node {}", node.id), + )); + } + if staged + .insert(node.id.as_str(), node) + .is_some_and(|previous| previous != node) + { return Err(SnapshotError::new( SnapshotErrorCode::Internal, - "retention edge references missing node", + "retention group has conflicting node definitions", )); } - g.edges.insert(e.clone()); } - let entry = g.roots.entry(node.id.clone()).or_default(); - for r in roots { - entry.insert(r.key()); + for e in edges { + for id in [&e.parent, &e.child] { + match staged.get(id.as_str()).copied().or_else(|| g.nodes.get(id)) { + Some(node) if node.state == NodeState::Live => {} + Some(_) => { + return Err(SnapshotError::new( + SnapshotErrorCode::ObjectUnavailable, + format!("retention edge requires LIVE endpoint {id}"), + )); + } + None => { + return Err(SnapshotError::new( + SnapshotErrorCode::Internal, + format!("retention edge references missing node {id}"), + )); + } + } + } + } + #[cfg(test)] + if std::mem::take(&mut g.fail_next_retain) { + return Err(SnapshotError::new( + SnapshotErrorCode::Internal, + "injected retention commit failure", + )); + } + for node in nodes { + g.nodes.insert(node.id.clone(), node.clone()); + let entry = g.roots.entry(node.id.clone()).or_default(); + for root in roots { + entry.insert(root.key()); + } } + g.edges.extend(edges.iter().cloned()); Ok(()) } fn mark_deleting(&self, id: &str) -> bool { let mut g = self.inner.lock().unwrap(); + if g.roots.get(id).is_some_and(|roots| !roots.is_empty()) + || g.edges + .iter() + .any(|edge| edge.child == id && g.nodes.contains_key(&edge.parent)) + { + return false; + } match g.nodes.get_mut(id) { Some(n) if n.state == NodeState::Live => { n.state = NodeState::Deleting; @@ -396,6 +431,13 @@ pub mod mem { fn remove(&self, id: &str) { let mut g = self.inner.lock().unwrap(); + if !g + .nodes + .get(id) + .is_some_and(|node| node.state == NodeState::Deleting) + { + return; + } g.nodes.remove(id); g.edges.retain(|e| e.parent != id && e.child != id); g.roots.remove(id); @@ -435,6 +477,15 @@ pub mod mem { #[cfg(test)] mod tests { + use std::{ + sync::{ + Arc, + mpsc::{Receiver, SyncSender, sync_channel}, + }, + thread, + time::Duration, + }; + use super::{mem::InMemoryRetentionStore, *}; fn node(id: &str, bytes: u64) -> RetentionNode { @@ -450,6 +501,243 @@ mod tests { RetentionCoordinator::new(InMemoryRetentionStore::default()) } + /// Pause a collection after its scan, immediately before the store CAS. + /// Timeouts make both sides of the forced interleaving bounded. + struct PausedMarkStore { + inner: InMemoryRetentionStore, + ready: SyncSender<()>, + resume: Mutex>, + } + + impl RetentionStore for PausedMarkStore { + fn node(&self, id: &str) -> Option { + self.inner.node(id) + } + + fn root_covers(&self, node_id: &str) -> bool { + self.inner.root_covers(node_id) + } + + fn live_incoming(&self, node_id: &str) -> usize { + self.inner.live_incoming(node_id) + } + + fn retain_group( + &self, + nodes: &[RetentionNode], + edges: &[RetentionEdge], + roots: &[RetentionRoot], + ) -> Result<(), SnapshotError> { + self.inner.retain_group(nodes, edges, roots) + } + + fn mark_deleting(&self, id: &str) -> bool { + self.ready.send(()).unwrap(); + self.resume + .lock() + .unwrap() + .recv_timeout(Duration::from_secs(5)) + .expect("test did not resume deletion"); + self.inner.mark_deleting(id) + } + + fn remove(&self, id: &str) { + self.inner.remove(id); + } + + fn release_root(&self, root: &RetentionRoot) { + self.inner.release_root(root); + } + + fn all_live(&self) -> Vec { + self.inner.all_live() + } + + fn children(&self, parent: &str) -> Vec { + self.inner.children(parent) + } + } + + #[test] + fn failed_pin_group_leaves_no_anchor_or_partial_coverage() { + let c = coord(); + let original = node("existing", 10); + c.store().retain(original.clone(), &[], &[]).unwrap(); + c.store().retain(node("deleting", 7), &[], &[]).unwrap(); + assert!(c.store().mark_deleting("deleting")); + let root = RetentionRoot::Lease("new".into()); + + // The last node fails after earlier entries would have been written + // by the old per-node pin loop, including an update to existing. + let err = c + .pin_root( + &root, + &[node("fresh", 1), node("existing", 99), node("deleting", 7)], + ) + .unwrap_err(); + assert_eq!(err.code, SnapshotErrorCode::ObjectUnavailable); + assert!(c.store().node(&root.key()).is_none()); + assert!(c.store().node("fresh").is_none()); + assert_eq!(c.store().node("existing"), Some(original)); + assert!(!c.store().root_covers("existing")); + assert!(!c.store().root_covers("deleting")); + assert_eq!( + c.store().node("deleting").unwrap().state, + NodeState::Deleting + ); + } + + #[test] + fn supplied_deleting_node_cannot_be_created_or_pinned() { + let c = coord(); + let root = RetentionRoot::Pin("invalid".into()); + let mut deleting = node("absent", 1); + deleting.state = NodeState::Deleting; + assert_eq!( + c.pin_root(&root, &[deleting]).unwrap_err().code, + SnapshotErrorCode::ObjectUnavailable + ); + assert!(c.store().all_live().is_empty()); + assert!(c.store().node("absent").is_none()); + assert!(c.store().node(&root.key()).is_none()); + } + + #[test] + fn late_invalid_edge_rolls_back_nodes_edges_and_roots() { + let c = coord(); + c.store().retain(node("parent", 0), &[], &[]).unwrap(); + let root = RetentionRoot::Lease("new".into()); + let edges = [ + RetentionEdge { + parent: "parent".into(), + child: "fresh".into(), + }, + RetentionEdge { + parent: "missing".into(), + child: "fresh".into(), + }, + ]; + let err = c + .store() + .retain(node("fresh", 2), &edges, &[root]) + .unwrap_err(); + assert_eq!(err.code, SnapshotErrorCode::Internal); + assert!(c.store().node("fresh").is_none()); + assert!(c.store().children("parent").is_empty()); + assert!(!c.store().root_covers("fresh")); + assert_eq!(c.store().live_incoming("fresh"), 0); + } + + #[test] + fn injected_commit_failure_preserves_existing_root_and_can_retry() { + let c = coord(); + let old = RetentionRoot::Pin("old".into()); + let new = RetentionRoot::Lease("new".into()); + c.store() + .retain(node("existing", 1), &[], std::slice::from_ref(&old)) + .unwrap(); + c.store().fail_next_retain_for_test(); + assert_eq!( + c.pin_root(&new, &[node("existing", 1), node("fresh", 2)]) + .unwrap_err() + .code, + SnapshotErrorCode::Internal + ); + assert!(c.store().node(&new.key()).is_none()); + assert!(c.store().node("fresh").is_none()); + assert!(c.store().root_covers("existing")); + c.release(&old); + assert!( + !c.store().root_covers("existing"), + "failed pin added no new root" + ); + + c.pin_root(&new, &[node("existing", 1), node("fresh", 2)]) + .unwrap(); + assert!(c.is_retained("existing")); + assert!(c.is_retained("fresh")); + } + + #[test] + fn collection_rechecks_roots_and_edges_acquired_after_its_scan() { + for via_edge in [false, true] { + let (ready_tx, ready_rx) = sync_channel(1); + let (resume_tx, resume_rx) = sync_channel(1); + let c = Arc::new(RetentionCoordinator::new(PausedMarkStore { + inner: InMemoryRetentionStore::default(), + ready: ready_tx, + resume: Mutex::new(resume_rx), + })); + c.store().retain(node("candidate", 1), &[], &[]).unwrap(); + let root = RetentionRoot::Lease("late".into()); + if via_edge { + c.store() + .retain(node("parent", 0), &[], std::slice::from_ref(&root)) + .unwrap(); + } + let collector = { + let c = Arc::clone(&c); + thread::spawn(move || c.collect(&NoopReaper)) + }; + ready_rx + .recv_timeout(Duration::from_secs(5)) + .expect("collector did not reach its deletion CAS"); + // Both references land after candidate selection, before CAS. + if via_edge { + c.retain(node("candidate", 1), &["parent".into()], &[]) + .unwrap(); + } else { + c.pin_root(&root, &[node("candidate", 1)]).unwrap(); + } + resume_tx.send(()).unwrap(); + let report = collector.join().unwrap().unwrap(); + assert!(report.unreachable.is_empty()); + assert_eq!(c.store().node("candidate").unwrap().state, NodeState::Live); + assert!(c.is_retained("candidate")); + } + } + + #[test] + fn deleting_parent_rejects_new_edges_and_protects_existing_children_until_removed() { + let c = coord(); + c.store().retain(node("parent", 0), &[], &[]).unwrap(); + c.retain(node("child", 1), &["parent".into()], &[]).unwrap(); + assert!(c.store().mark_deleting("parent")); + assert!(!c.store().mark_deleting("child")); + assert_eq!(c.store().live_incoming("child"), 1); + assert_eq!( + c.retain(node("new-child", 2), &["parent".into()], &[]) + .unwrap_err() + .code, + SnapshotErrorCode::ObjectUnavailable + ); + assert!(c.store().node("new-child").is_none()); + c.store().remove("parent"); + assert_eq!(c.store().live_incoming("child"), 0); + assert!(c.store().mark_deleting("child")); + } + + #[test] + fn retain_cannot_add_an_edge_to_a_deleting_child() { + let c = coord(); + c.store().retain(node("child", 1), &[], &[]).unwrap(); + assert!(c.store().mark_deleting("child")); + let edge = RetentionEdge { + parent: "new-parent".into(), + child: "child".into(), + }; + assert_eq!( + c.store() + .retain(node("new-parent", 0), &[edge], &[]) + .unwrap_err() + .code, + SnapshotErrorCode::ObjectUnavailable + ); + assert!(c.store().node("new-parent").is_none()); + assert_eq!(c.store().node("child").unwrap().state, NodeState::Deleting); + assert_eq!(c.store().live_incoming("child"), 0); + } + #[test] fn leaf_dies_when_root_released_but_other_root_keeps_it() { // root A and B both cover a shared leaf; releasing A must not @@ -486,8 +774,8 @@ mod tests { let r = c.collect(&NoopReaper).unwrap(); assert_eq!(r.unreachable, vec!["orphan".to_string()]); assert_eq!(r.reclaimed_bytes, 9); - // Releasing the root cascades unreachability to a and b only after - // the anchor itself is collected (the anchor node is root-covered). + // Releasing the root makes the chain unreachable, but children stay + // LIVE while their parents await physical removal under NoopReaper. c.release(&root); let _ = c.collect(&NoopReaper).unwrap(); assert!(!c.is_retained("b")); @@ -521,7 +809,7 @@ mod tests { ) .unwrap(); assert_eq!(c.store().live_incoming("shared"), 2); - // Remove p1's coverage and p1 itself; shared still has p2. + // Drop p1's coverage; its pending deletion must not affect p2's view. c.release(&RetentionRoot::Pin("r1".into())); c.collect(&NoopReaper).unwrap(); assert!(c.store().node("shared").is_some()); @@ -562,12 +850,12 @@ mod tests { let c = coord(); c.store().retain(node("x", 3), &[], &[]).unwrap(); assert!(c.collect(&Fail).is_err()); - // Node remains (DELETING) so a retry can replay; it is not lost. - assert!(c.store().node("x").is_some()); + // Node remains DELETING; physical retry/recovery is a later slice. + assert_eq!(c.store().node("x").unwrap().state, NodeState::Deleting); } #[test] - fn new_root_revives_a_node_before_it_is_collected() { + fn new_root_protects_a_live_node_before_it_is_collected() { let c = coord(); c.store().retain(node("x", 4), &[], &[]).unwrap(); // Before any collect, a lease covers it. @@ -644,7 +932,12 @@ mod tests { c.retain(node("ch", 2), &["p".to_string()], &[]).unwrap(); c.release(&RetentionRoot::Lease("l".into())); c.collect(&Delete).unwrap(); + // A child selected before its parent was removed waits for the + // next pass; iteration order must not affect the final result. + c.collect(&Delete).unwrap(); // Both nodes and their edge are gone; no stale incoming edge. + assert!(c.store().node("p").is_none()); + assert!(c.store().node("ch").is_none()); assert_eq!(c.store().live_incoming("ch"), 0); } } diff --git a/src/ceres/snapshot/retention_dag.rs b/src/ceres/snapshot/retention_dag.rs new file mode 100644 index 00000000..b9f3cb4d --- /dev/null +++ b/src/ceres/snapshot/retention_dag.rs @@ -0,0 +1,554 @@ +//! Bounded canonical native metadata closure, ready for future durable storage. +//! This module retains no leases, persists no bytes and starts no collector. + +use std::collections::{BTreeMap, BTreeSet, VecDeque}; + +use mst2_codec::{ + descriptor::METADATA_CODEC, + metapage::{Entry, Page, page_id}, +}; + +use crate::ceres::snapshot::{ + error::{SnapshotError, SnapshotErrorCode}, + retention::{NodeState, RetainedKind, RetentionEdge, RetentionNode}, +}; + +pub type MetadataPageId = [u8; 32]; + +#[derive(Debug, Clone, Copy)] +pub struct MetadataDagLimits { + pub nodes: usize, + pub edges: usize, + pub payload_bytes: u64, + pub entries: usize, + pub prepare_entry_visits: usize, +} + +impl Default for MetadataDagLimits { + fn default() -> Self { + Self { + nodes: 4096, + edges: 16_384, + payload_bytes: 64 * 1024 * 1024, + entries: 131_072, + prepare_entry_visits: 64 * 1024 * 1024, + } + } +} + +impl MetadataDagLimits { + /// Caller budgets may tighten, never widen, the absolute group ceilings. + pub(crate) fn effective(self) -> Self { + let hard = Self::default(); + Self { + nodes: self.nodes.min(hard.nodes), + edges: self.edges.min(hard.edges), + payload_bytes: self.payload_bytes.min(hard.payload_bytes), + entries: self.entries.min(hard.entries), + prepare_entry_visits: self.prepare_entry_visits.min(hard.prepare_entry_visits), + } + } +} + +#[derive(Debug, Clone)] +pub struct MetadataPagePayload { + pub id: MetadataPageId, + pub size: u64, + pub bytes: Vec, +} + +#[derive(Debug, Clone)] +pub struct MetadataDagCandidate { + pub metadata_codec: u16, + pub root: MetadataPageId, + pub pages: Vec, + pub edges: Vec<(MetadataPageId, MetadataPageId)>, +} + +/// Immutable validated output. Supply these slices to PostgreSQL retain_group +/// only after the payloads have been durably stored. +#[derive(Debug)] +pub struct ValidatedMetadataDag { + root: MetadataPageId, + pages: Vec, + nodes: Vec, + edges: Vec, + payload_bytes: u64, + entry_count: usize, +} + +impl ValidatedMetadataDag { + pub fn validate( + candidate: MetadataDagCandidate, + limits: MetadataDagLimits, + ) -> Result { + let limits = limits.effective(); + if candidate.metadata_codec != METADATA_CODEC { + return Err(SnapshotError::new( + SnapshotErrorCode::ScopeInvalid, + "native metadata DAG requires metadata codec 1", + )); + } + check_count(candidate.pages.len(), limits.nodes, "metadata nodes")?; + check_count(candidate.edges.len(), limits.edges, "metadata edges")?; + let mut payload_bytes = 0u64; + let mut entry_count = 0usize; + let mut pages = BTreeMap::new(); + let mut decoded = BTreeMap::new(); + let mut expected_edges = BTreeSet::new(); + let mut directory_roots = BTreeSet::from([candidate.root]); + for payload in candidate.pages { + payload_bytes = payload_bytes + .checked_add(payload.bytes.len() as u64) + .filter(|total| *total <= limits.payload_bytes) + .ok_or_else(|| limit("metadata payload budget exceeded"))?; + if payload.size != payload.bytes.len() as u64 || page_id(&payload.bytes) != payload.id { + return Err(SnapshotError::new( + SnapshotErrorCode::DigestMismatch, + "metadata page digest or advertised size mismatch", + )); + } + if pages.contains_key(&payload.id) { + return Err(integrity("duplicate metadata page identity")); + } + let (page, subtree_entries) = Page::decode(&payload.bytes).map_err(codec_error)?; + let entries: &[Entry] = match &page { + Page::Leaf { entries } => entries, + Page::Branch { + terminal, children, .. + } => { + for child in children { + expected_edges.insert((payload.id, child.child_page_id)); + } + terminal.as_slice() + } + }; + entry_count = entry_count + .checked_add(entries.len()) + .filter(|count| *count <= limits.entries) + .ok_or_else(|| limit("metadata entry budget exceeded"))?; + for entry in entries.iter().filter(|entry| entry.is_dir()) { + expected_edges.insert((payload.id, entry.child_root)); + directory_roots.insert(entry.child_root); + } + check_count(expected_edges.len(), limits.edges, "metadata edges")?; + decoded.insert(payload.id, (page, subtree_entries)); + pages.insert(payload.id, payload); + } + let supplied_edges: BTreeSet<_> = candidate.edges.iter().copied().collect(); + if supplied_edges.len() != candidate.edges.len() { + return Err(integrity("duplicate metadata edge")); + } + validate_graph(candidate.root, &pages, &supplied_edges)?; + if supplied_edges != expected_edges { + return Err(integrity( + "metadata edges disagree with canonical page references", + )); + } + for (page, _) in decoded.values() { + if let Page::Branch { children, .. } = page { + for child in children { + if decoded.get(&child.child_page_id).map(|(_, count)| *count) + != Some(child.subtree_entries) + { + return Err(integrity("metadata branch subtree count mismatch")); + } + } + } + } + let mut validation_visits = 0usize; + for root in directory_roots { + let advertised_entries = decoded + .get(&root) + .map(|(_, count)| *count) + .ok_or_else(|| unavailable("metadata child is missing"))?; + if advertised_entries > limits.entries as u64 { + return Err(limit("metadata directory entry budget exceeded")); + } + let mut entries = Vec::new(); + let mut pending = vec![root]; + while let Some(id) = pending.pop() { + let (page, _) = decoded + .get(&id) + .ok_or_else(|| unavailable("metadata child payload is missing"))?; + let direct_entries = match page { + Page::Leaf { entries } => entries.as_slice(), + Page::Branch { + terminal, children, .. + } => { + pending.extend(children.iter().map(|child| child.child_page_id)); + terminal.as_slice() + } + }; + validation_visits = validation_visits + .checked_add(direct_entries.len().saturating_add(1)) + .filter(|count| *count <= limits.prepare_entry_visits) + .ok_or_else(|| limit("metadata canonical validation work budget exceeded"))?; + entries.extend_from_slice(direct_entries); + } + entries.sort_by(|left, right| left.name.cmp(&right.name)); + let canonical = Page::build(&entries).map_err(codec_error)?; + if pages.get(&root).map(|payload| payload.bytes.as_slice()) + != Some(canonical.as_slice()) + { + return Err(integrity( + "metadata directory is not its canonical Build(entries)", + )); + } + } + let nodes = pages + .values() + .map(|payload| RetentionNode { + id: node_id(&payload.id), + kind: RetainedKind::Page, + state: NodeState::Live, + bytes: payload.size, + }) + .collect(); + let edges = supplied_edges + .into_iter() + .map(|(parent, child)| RetentionEdge { + parent: node_id(&parent), + child: node_id(&child), + }) + .collect(); + Ok(Self { + root: candidate.root, + pages: pages.into_values().collect(), + nodes, + edges, + payload_bytes, + entry_count, + }) + } + + pub fn root(&self) -> MetadataPageId { + self.root + } + pub fn payloads(&self) -> &[MetadataPagePayload] { + &self.pages + } + pub fn nodes(&self) -> &[RetentionNode] { + &self.nodes + } + pub fn edges(&self) -> &[RetentionEdge] { + &self.edges + } + pub fn payload_bytes(&self) -> u64 { + self.payload_bytes + } + + pub fn check_limits(&self, limits: MetadataDagLimits) -> Result<(), SnapshotError> { + let limits = limits.effective(); + check_count(self.nodes.len(), limits.nodes, "metadata nodes")?; + check_count(self.edges.len(), limits.edges, "metadata edges")?; + check_count(self.entry_count, limits.entries, "metadata entries")?; + if self.payload_bytes > limits.payload_bytes { + return Err(limit("metadata payload budget exceeded")); + } + Ok(()) + } + + /// Cache residency only; outstanding caller Arcs are outside that bound. + pub(crate) fn residency_bytes(&self) -> Option { + let mut bytes = std::mem::size_of::() + .checked_add( + self.pages + .capacity() + .checked_mul(std::mem::size_of::())?, + )? + .checked_add( + self.nodes + .capacity() + .checked_mul(std::mem::size_of::())?, + )? + .checked_add( + self.edges + .capacity() + .checked_mul(std::mem::size_of::())?, + )?; + for payload in &self.pages { + bytes = bytes.checked_add(payload.bytes.capacity())?; + } + for node in &self.nodes { + bytes = bytes.checked_add(node.id.capacity())?; + } + for edge in &self.edges { + bytes = bytes + .checked_add(edge.parent.capacity())? + .checked_add(edge.child.capacity())?; + } + Some(bytes) + } +} + +pub(crate) struct MetadataDagBuilder { + limits: MetadataDagLimits, + pages: BTreeMap, + edges: BTreeSet<(MetadataPageId, MetadataPageId)>, + required_nodes: BTreeSet, + directories: BTreeSet, + payload_bytes: u64, + entries: usize, + entry_visits: usize, + failed: bool, +} + +impl MetadataDagBuilder { + pub(crate) fn new(limits: MetadataDagLimits) -> Self { + Self { + limits: limits.effective(), + pages: BTreeMap::new(), + edges: BTreeSet::new(), + required_nodes: BTreeSet::new(), + directories: BTreeSet::new(), + payload_bytes: 0, + entries: 0, + entry_visits: 0, + failed: false, + } + } + + pub(crate) fn contains_directory(&self, root: MetadataPageId) -> bool { + self.directories.contains(&root) + } + + pub(crate) fn add_directory( + &mut self, + root_bytes: &[u8], + entries: &[Entry], + ) -> Result<(), SnapshotError> { + if self.failed { + return Err(integrity("metadata preparation already failed")); + } + let result = self.add_directory_inner(root_bytes, entries); + if result.is_err() { + self.failed = true; + } + result + } + + fn add_directory_inner( + &mut self, + root_bytes: &[u8], + entries: &[Entry], + ) -> Result<(), SnapshotError> { + let root = page_id(root_bytes); + if self.contains_directory(root) { + return Ok(()); + } + self.entries = self + .entries + .checked_add(entries.len()) + .filter(|count| *count <= self.limits.entries) + .ok_or_else(|| limit("metadata entry budget exceeded"))?; + self.require_node(root)?; + if !self.pages.contains_key(&root) + && root_bytes.len() as u64 + > self.limits.payload_bytes.saturating_sub(self.payload_bytes) + { + return Err(limit("metadata payload budget exceeded")); + } + self.charge_work(entries.len())?; + let canonical = Page::build(entries).map_err(codec_error)?; + if canonical != root_bytes { + return Err(integrity("directory root disagrees with canonical entries")); + } + let mut routes = VecDeque::from([(Vec::::new(), root)]); + while let Some((route, expected_id)) = routes.pop_front() { + if self.pages.contains_key(&expected_id) { + continue; + } + self.charge_work(entries.len())?; + let pages = Page::pages_along_route(entries, &route).map_err(codec_error)?; + let bytes = pages + .into_iter() + .last() + .ok_or_else(|| integrity("canonical route returned no page"))?; + let id = page_id(&bytes); + if id != expected_id { + return Err(integrity("canonical route child identity mismatch")); + } + self.payload_bytes = self + .payload_bytes + .checked_add(bytes.len() as u64) + .filter(|total| *total <= self.limits.payload_bytes) + .ok_or_else(|| limit("metadata payload budget exceeded"))?; + let (page, _) = Page::decode(&bytes).map_err(codec_error)?; + match page { + Page::Leaf { entries } => { + for entry in entries.iter().filter(|entry| entry.is_dir()) { + self.add_edge(id, entry.child_root)?; + } + } + Page::Branch { + terminal, children, .. + } => { + if let Some(entry) = terminal.filter(Entry::is_dir) { + self.add_edge(id, entry.child_root)?; + } + for child in children { + self.add_edge(id, child.child_page_id)?; + let mut child_route = route.clone(); + child_route.push(child.label); + routes.push_back((child_route, child.child_page_id)); + } + } + } + self.pages.insert( + id, + MetadataPagePayload { + id, + size: bytes.len() as u64, + bytes, + }, + ); + } + self.directories.insert(root); + Ok(()) + } + + fn require_node(&mut self, id: MetadataPageId) -> Result<(), SnapshotError> { + if !self.required_nodes.contains(&id) { + check_count( + self.required_nodes.len().saturating_add(1), + self.limits.nodes, + "metadata nodes", + )?; + self.required_nodes.insert(id); + } + Ok(()) + } + + fn charge_work(&mut self, count: usize) -> Result<(), SnapshotError> { + self.entry_visits = self + .entry_visits + .checked_add(count) + .filter(|count| *count <= self.limits.prepare_entry_visits) + .ok_or_else(|| limit("metadata preparation work budget exceeded"))?; + Ok(()) + } + + fn add_edge( + &mut self, + parent: MetadataPageId, + child: MetadataPageId, + ) -> Result<(), SnapshotError> { + if !self.edges.contains(&(parent, child)) { + check_count( + self.edges.len().saturating_add(1), + self.limits.edges, + "metadata edges", + )?; + self.require_node(parent)?; + self.require_node(child)?; + self.edges.insert((parent, child)); + } + Ok(()) + } + + pub(crate) fn finish( + self, + root: MetadataPageId, + ) -> Result { + if self.failed { + return Err(integrity( + "metadata preparation failed; no group can be exposed", + )); + } + ValidatedMetadataDag::validate( + MetadataDagCandidate { + metadata_codec: METADATA_CODEC, + root, + pages: self.pages.into_values().collect(), + edges: self.edges.into_iter().collect(), + }, + self.limits, + ) + } +} + +fn validate_graph( + root: MetadataPageId, + pages: &BTreeMap, + edges: &BTreeSet<(MetadataPageId, MetadataPageId)>, +) -> Result<(), SnapshotError> { + if !pages.contains_key(&root) { + return Err(unavailable("metadata root payload is missing")); + } + let mut incoming: BTreeMap<_, usize> = pages.keys().map(|id| (*id, 0)).collect(); + let mut children: BTreeMap<_, Vec<_>> = BTreeMap::new(); + for &(parent, child) in edges { + if !pages.contains_key(&parent) || !pages.contains_key(&child) { + return Err(unavailable("metadata child payload is missing")); + } + *incoming + .get_mut(&child) + .ok_or_else(|| unavailable("metadata child is missing"))? += 1; + children.entry(parent).or_default().push(child); + } + let mut ready: VecDeque<_> = incoming + .iter() + .filter_map(|(id, count)| (*count == 0).then_some(*id)) + .collect(); + let mut visited = 0usize; + while let Some(id) = ready.pop_front() { + visited += 1; + for child in children.get(&id).into_iter().flatten() { + let count = incoming + .get_mut(child) + .ok_or_else(|| unavailable("metadata child is missing"))?; + *count -= 1; + if *count == 0 { + ready.push_back(*child); + } + } + } + if visited != pages.len() { + return Err(integrity("metadata graph contains a cycle")); + } + let mut reachable = BTreeSet::new(); + let mut pending = vec![root]; + while let Some(id) = pending.pop() { + if reachable.insert(id) { + pending.extend(children.get(&id).into_iter().flatten().copied()); + } + } + if reachable.len() != pages.len() { + return Err(integrity("metadata graph includes unreachable payloads")); + } + Ok(()) +} + +fn node_id(id: &MetadataPageId) -> String { + use std::fmt::Write; + let mut value = String::with_capacity(76); + value.push_str("page:sha256:"); + for byte in id { + let _ = write!(value, "{byte:02x}"); + } + value +} + +fn check_count(count: usize, maximum: usize, name: &str) -> Result<(), SnapshotError> { + if count > maximum { + return Err(limit(&format!("{name} budget exceeded"))); + } + Ok(()) +} +fn limit(message: &str) -> SnapshotError { + SnapshotError::new(SnapshotErrorCode::LimitExceeded, message) +} +fn integrity(message: &str) -> SnapshotError { + SnapshotError::new(SnapshotErrorCode::IntegrityError, message) +} +fn unavailable(message: &str) -> SnapshotError { + SnapshotError::new(SnapshotErrorCode::ObjectUnavailable, message) +} +fn codec_error(error: mst2_codec::CodecError) -> SnapshotError { + integrity(&format!("invalid metadata page: {error}")) +} + +#[cfg(test)] +#[path = "retention_dag_tests.rs"] +mod tests; diff --git a/src/ceres/snapshot/retention_dag_tests.rs b/src/ceres/snapshot/retention_dag_tests.rs new file mode 100644 index 00000000..3525a4b5 --- /dev/null +++ b/src/ceres/snapshot/retention_dag_tests.rs @@ -0,0 +1,345 @@ +use mst2_codec::metapage::{BranchChild, EntryKind}; + +use super::*; + +fn payload(entries: &[Entry]) -> MetadataPagePayload { + let bytes = Page::build(entries).unwrap(); + MetadataPagePayload { + id: page_id(&bytes), + size: bytes.len() as u64, + bytes, + } +} + +fn candidate(dag: &ValidatedMetadataDag) -> MetadataDagCandidate { + let ids: BTreeMap<_, _> = dag + .payloads() + .iter() + .map(|payload| (node_id(&payload.id), payload.id)) + .collect(); + MetadataDagCandidate { + metadata_codec: METADATA_CODEC, + root: dag.root(), + pages: dag.payloads().to_vec(), + edges: dag + .edges() + .iter() + .map(|edge| (ids[&edge.parent], ids[&edge.child])) + .collect(), + } +} + +fn nested_shared() -> ValidatedMetadataDag { + let empty = payload(&[]); + let mut wide: Vec<_> = (0..129) + .map(|index| { + Entry::file( + EntryKind::Regular, + format!("f{index:03}").as_bytes(), + index, + [index as u8; 32], + ) + }) + .collect(); + wide.push(Entry::dir(b"nested", empty.id)); + let shared = payload(&wide); + let root_entries = vec![ + Entry::dir(b"alpha", shared.id), + Entry::dir(b"beta", shared.id), + ]; + let root = payload(&root_entries); + let mut builder = MetadataDagBuilder::new(MetadataDagLimits::default()); + builder.add_directory(&root.bytes, &root_entries).unwrap(); + builder.add_directory(&shared.bytes, &wide).unwrap(); + builder.add_directory(&empty.bytes, &[]).unwrap(); + builder.add_directory(&shared.bytes, &wide).unwrap(); + builder.finish(root.id).unwrap() +} + +#[test] +fn complete_radix_nested_shared_closure_accounts_unique_payloads_and_edges() { + let dag = nested_shared(); + assert!( + dag.payloads() + .iter() + .any(|payload| matches!(Page::decode(&payload.bytes).unwrap().0, Page::Branch { .. })) + ); + assert!( + dag.payloads().len() > 3, + "internal radix pages must be retained too" + ); + assert_eq!( + dag.payload_bytes(), + dag.payloads() + .iter() + .map(|payload| payload.size) + .sum::() + ); + assert_eq!(dag.nodes().len(), dag.payloads().len()); + let mut ids = BTreeSet::new(); + for (node, payload) in dag.nodes().iter().zip(dag.payloads()) { + assert_eq!(node.kind, RetainedKind::Page); + assert_eq!(node.state, NodeState::Live); + assert_eq!(node.bytes, payload.bytes.len() as u64); + assert_eq!(node.id, node_id(&payload.id)); + assert!(ids.insert(node.id.clone())); + } + let edges: BTreeSet<_> = dag + .edges() + .iter() + .map(|edge| (&edge.parent, &edge.child)) + .collect(); + assert_eq!(edges.len(), dag.edges().len()); + assert!( + dag.edges() + .iter() + .all(|edge| ids.contains(&edge.parent) && ids.contains(&edge.child)) + ); + let root_id = node_id(&dag.root()); + assert_eq!( + dag.edges() + .iter() + .filter(|edge| edge.parent == root_id) + .count(), + 1, + "two names sharing one child acquire one edge" + ); +} + +#[test] +fn digest_and_advertised_size_are_verified_before_exposing_group() { + let dag = nested_shared(); + let mut wrong_id = candidate(&dag); + wrong_id.pages[0].id[0] ^= 1; + assert_eq!( + ValidatedMetadataDag::validate(wrong_id, MetadataDagLimits::default()) + .unwrap_err() + .code, + SnapshotErrorCode::DigestMismatch + ); + let mut wrong_size = candidate(&dag); + wrong_size.pages[0].size += 1; + assert_eq!( + ValidatedMetadataDag::validate(wrong_size, MetadataDagLimits::default()) + .unwrap_err() + .code, + SnapshotErrorCode::DigestMismatch + ); + let mut corrupt = candidate(&dag); + corrupt.pages[0].bytes[0] ^= 1; + assert_eq!( + ValidatedMetadataDag::validate(corrupt, MetadataDagLimits::default()) + .unwrap_err() + .code, + SnapshotErrorCode::DigestMismatch + ); +} + +#[test] +fn missing_child_and_missing_root_are_unavailable() { + let dag = nested_shared(); + let mut missing = candidate(&dag); + let index = missing + .pages + .iter() + .position(|payload| payload.id != missing.root) + .unwrap(); + missing.pages.remove(index); + assert_eq!( + ValidatedMetadataDag::validate(missing, MetadataDagLimits::default()) + .unwrap_err() + .code, + SnapshotErrorCode::ObjectUnavailable + ); + let mut missing_root = candidate(&dag); + missing_root.root = [255; 32]; + assert_eq!( + ValidatedMetadataDag::validate(missing_root, MetadataDagLimits::default()) + .unwrap_err() + .code, + SnapshotErrorCode::ObjectUnavailable + ); +} + +#[test] +fn duplicate_payloads_and_edges_are_rejected() { + let dag = nested_shared(); + let mut duplicate = candidate(&dag); + duplicate.pages.push(duplicate.pages[0].clone()); + assert_eq!( + ValidatedMetadataDag::validate(duplicate, MetadataDagLimits::default()) + .unwrap_err() + .code, + SnapshotErrorCode::IntegrityError + ); + let mut duplicate_edge = candidate(&dag); + duplicate_edge.edges.push(duplicate_edge.edges[0]); + assert_eq!( + ValidatedMetadataDag::validate(duplicate_edge, MetadataDagLimits::default()) + .unwrap_err() + .code, + SnapshotErrorCode::IntegrityError + ); +} + +#[test] +fn cycles_wrong_edges_and_unreachable_objects_are_rejected() { + let dag = nested_shared(); + let mut cyclic = candidate(&dag); + cyclic.edges.push((cyclic.root, cyclic.root)); + let error = ValidatedMetadataDag::validate(cyclic, MetadataDagLimits::default()).unwrap_err(); + assert_eq!(error.code, SnapshotErrorCode::IntegrityError); + assert!(error.message.contains("cycle")); + let mut missing_edge = candidate(&dag); + missing_edge.edges.pop(); + assert_eq!( + ValidatedMetadataDag::validate(missing_edge, MetadataDagLimits::default()) + .unwrap_err() + .code, + SnapshotErrorCode::IntegrityError + ); + let mut orphan = candidate(&dag); + orphan.pages.push(payload(&[Entry::file( + EntryKind::Regular, + b"orphan", + 1, + [91; 32], + )])); + assert_eq!( + ValidatedMetadataDag::validate(orphan, MetadataDagLimits::default()) + .unwrap_err() + .code, + SnapshotErrorCode::IntegrityError + ); +} + +#[test] +fn correct_hashes_do_not_make_a_noncanonical_directory_valid() { + let left_entry = Entry::file(EntryKind::Regular, b"a", 1, [1; 32]); + let right_entry = Entry::file(EntryKind::Regular, b"b", 1, [2; 32]); + let left = payload(&[left_entry]); + let right = payload(&[right_entry]); + let bytes = Page::Branch { + prefix: Vec::new(), + terminal: None, + children: vec![ + BranchChild { + label: b'a', + subtree_entries: 1, + child_page_id: left.id, + }, + BranchChild { + label: b'b', + subtree_entries: 1, + child_page_id: right.id, + }, + ], + } + .encode() + .unwrap(); + let root = page_id(&bytes); + let group = MetadataDagCandidate { + metadata_codec: METADATA_CODEC, + root, + edges: vec![(root, left.id), (root, right.id)], + pages: vec![ + MetadataPagePayload { + id: root, + size: bytes.len() as u64, + bytes, + }, + left, + right, + ], + }; + let error = ValidatedMetadataDag::validate(group, MetadataDagLimits::default()).unwrap_err(); + assert_eq!(error.code, SnapshotErrorCode::IntegrityError); + assert!(error.message.contains("canonical")); +} + +#[test] +fn node_edge_payload_entry_and_work_budgets_reject_complete_groups() { + let dag = nested_shared(); + for limits in [ + MetadataDagLimits { + nodes: dag.nodes().len() - 1, + ..MetadataDagLimits::default() + }, + MetadataDagLimits { + edges: dag.edges().len() - 1, + ..MetadataDagLimits::default() + }, + MetadataDagLimits { + payload_bytes: dag.payload_bytes() - 1, + ..MetadataDagLimits::default() + }, + MetadataDagLimits { + entries: 1, + ..MetadataDagLimits::default() + }, + MetadataDagLimits { + prepare_entry_visits: 0, + ..MetadataDagLimits::default() + }, + ] { + assert_eq!( + ValidatedMetadataDag::validate(candidate(&dag), limits) + .unwrap_err() + .code, + SnapshotErrorCode::LimitExceeded + ); + } +} + +#[test] +fn failed_preparation_cannot_expose_a_partial_group() { + let empty = payload(&[]); + let root_entries = vec![Entry::dir(b"child", empty.id)]; + let root = payload(&root_entries); + let mut builder = MetadataDagBuilder::new(MetadataDagLimits { + nodes: 1, + ..MetadataDagLimits::default() + }); + assert_eq!( + builder + .add_directory(&root.bytes, &root_entries) + .unwrap_err() + .code, + SnapshotErrorCode::LimitExceeded + ); + assert_eq!( + builder.finish(root.id).unwrap_err().code, + SnapshotErrorCode::IntegrityError + ); + let mut builder = MetadataDagBuilder::new(MetadataDagLimits::default()); + builder.add_directory(&root.bytes, &root_entries).unwrap(); + assert_eq!( + builder.finish(root.id).unwrap_err().code, + SnapshotErrorCode::ObjectUnavailable + ); +} + +#[test] +fn codec_identity_and_canonical_entry_source_are_fixed() { + let dag = nested_shared(); + let mut wrong_codec = candidate(&dag); + wrong_codec.metadata_codec += 1; + assert_eq!( + ValidatedMetadataDag::validate(wrong_codec, MetadataDagLimits::default()) + .unwrap_err() + .code, + SnapshotErrorCode::ScopeInvalid + ); + let empty = payload(&[]); + let mut builder = MetadataDagBuilder::new(MetadataDagLimits::default()); + assert_eq!( + builder + .add_directory( + &empty.bytes, + &[Entry::file(EntryKind::Regular, b"different", 1, [1; 32])] + ) + .unwrap_err() + .code, + SnapshotErrorCode::IntegrityError + ); +} diff --git a/src/ceres/snapshot/retention_prepare_tests.rs b/src/ceres/snapshot/retention_prepare_tests.rs new file mode 100644 index 00000000..2a427795 --- /dev/null +++ b/src/ceres/snapshot/retention_prepare_tests.rs @@ -0,0 +1,92 @@ +use git_internal::hash::HashKind; + +use super::*; +use crate::ceres::snapshot::retention_dag::{MetadataDagBuilder, MetadataDagLimits}; + +fn dag() -> Arc { + let bytes = Page::build(&[]).unwrap(); + let mut builder = MetadataDagBuilder::new(MetadataDagLimits::default()); + builder.add_directory(&bytes, &[]).unwrap(); + Arc::new(builder.finish(page_id(&bytes)).unwrap()) +} + +fn key(kind: HashKind, scope: &str) -> NativeRetentionKey { + NativeRetentionKey { + projection: NativeProjectionKey::new( + ObjectHash::from_hex_for_kind(kind, &"a".repeat(64)).unwrap(), + ), + scope: scope.to_owned(), + } +} + +#[test] +fn retention_memo_preserves_tagged_source_scope_and_profile_identity() { + let cache = NativeProjectionCache::default(); + let original = key(HashKind::Sha256, "/scope"); + let prepared = dag(); + cache.insert_retention_dag(original.clone(), Arc::clone(&prepared)); + assert!(Arc::ptr_eq( + &prepared, + &cache.retention_dag(&original).unwrap() + )); + assert!( + cache + .retention_dag(&key(HashKind::Blake3, "/scope")) + .is_none() + ); + assert!( + cache + .retention_dag(&key(HashKind::Sha256, "/different")) + .is_none() + ); + let mut different = original; + different.projection.projection_revision += 1; + assert!(cache.retention_dag(&different).is_none()); +} + +#[test] +fn retention_memo_limits_total_residency_and_eviction_keeps_active_arc() { + let cache = NativeProjectionCache::default(); + let first_key = key(HashKind::Sha256, "/first"); + let first = dag(); + let overhead = std::mem::size_of::() + + std::mem::size_of::>() + + first_key.scope.capacity() + + first_key.projection.tree_oid.capacity(); + let one_entry_bytes = first.residency_bytes().unwrap() + overhead; + cache.insert_retention_dag_with_budget( + first_key.clone(), + Arc::clone(&first), + one_entry_bytes, + 16, + ); + assert!(cache.retention_dag(&first_key).is_some()); + let next_key = key(HashKind::Sha256, "/other"); + let next = dag(); + cache.insert_retention_dag_with_budget( + next_key.clone(), + Arc::clone(&next), + one_entry_bytes, + 16, + ); + assert!(cache.retention_dag(&first_key).is_none()); + assert!(Arc::ptr_eq(&next, &cache.retention_dag(&next_key).unwrap())); + assert_eq!(first.payloads()[0].bytes, Page::build(&[]).unwrap()); + assert!(cache.state.lock().unwrap().retained_dag_bytes <= one_entry_bytes); + let skipped_key = key(HashKind::Sha256, "/too-large"); + cache.insert_retention_dag_with_budget(skipped_key.clone(), dag(), 1, 16); + assert!(cache.retention_dag(&skipped_key).is_none()); + assert!(cache.retention_dag(&next_key).is_some()); +} + +#[test] +fn retention_memo_entry_cap_bounds_distinct_roots() { + let cache = NativeProjectionCache::default(); + let first = key(HashKind::Sha256, "/one"); + let second = key(HashKind::Sha256, "/two"); + cache.insert_retention_dag_with_budget(first.clone(), dag(), 64 * 1024 * 1024, 1); + cache.insert_retention_dag_with_budget(second.clone(), dag(), 64 * 1024 * 1024, 1); + assert!(cache.retention_dag(&first).is_none()); + assert!(cache.retention_dag(&second).is_some()); + assert_eq!(cache.state.lock().unwrap().retention_dags.len(), 1); +} diff --git a/src/ceres/snapshot/rooted_metadata_install.rs b/src/ceres/snapshot/rooted_metadata_install.rs new file mode 100644 index 00000000..5086bfa5 --- /dev/null +++ b/src/ceres/snapshot/rooted_metadata_install.rs @@ -0,0 +1,848 @@ +//! Bounded native metadata delta and certified reused-root installation identity. +//! Reused roots are opaque boundaries; their database proofs grant no authority here. + +use std::collections::{BTreeMap, BTreeSet}; + +use mst2_codec::metapage::{HEADER_LEN, PAGE_MAX_BYTES}; +use sha2::{Digest, Sha256}; +use uuid::Uuid; + +use super::{ + error::{SnapshotError, SnapshotErrorCode}, + metadata_install::{MAX_PLAN_BYTES, MetadataInstallIdentity}, + projection_observation::NATIVE_PROJECTION_REVISION, + retention_dag::{MetadataDagLimits, MetadataPageId}, +}; + +const DOMAIN: &[u8] = b"mega.mst2.rooted-install.v1\0"; +const ENCODING_VERSION: u16 = 1; + +#[derive(Debug, Clone, PartialEq, Eq)] +pub(crate) struct RootedReuseRoot { + pub(crate) generation: i64, + pub(crate) attestation_id: Uuid, + pub(crate) attestation_digest: [u8; 32], + pub(crate) certificate_digest: [u8; 32], +} + +#[derive(Debug, Clone, PartialEq, Eq)] +pub(crate) struct RootedMetadataInstallPlan { + pub(crate) identity: MetadataInstallIdentity, + pub(crate) root: MetadataPageId, + pub(crate) delta: BTreeMap, + pub(crate) edges: BTreeSet<(MetadataPageId, MetadataPageId)>, + pub(crate) reused: BTreeMap, + pub(crate) source_roots: BTreeMap, +} + +impl RootedMetadataInstallPlan { + pub(crate) fn new( + identity: MetadataInstallIdentity, + root: MetadataPageId, + delta: BTreeMap, + edges: BTreeSet<(MetadataPageId, MetadataPageId)>, + reused: BTreeMap, + source_roots: BTreeMap, + ) -> Result { + let plan = Self { + identity, + root, + delta, + edges, + reused, + source_roots, + }; + plan.validate()?; + Ok(plan) + } + + pub(crate) fn digest(&self) -> Result<[u8; 32], SnapshotError> { + Ok(Sha256::digest(self.encode()?).into()) + } + + pub(crate) fn delta_bytes(&self) -> Result { + self.delta + .values() + .try_fold(0u64, |total, size| total.checked_add(*size)) + .ok_or_else(|| limit("rooted metadata delta byte count overflow")) + } + + pub(crate) fn encode(&self) -> Result, SnapshotError> { + self.validate()?; + let mut bytes = Vec::with_capacity(self.encoded_len()); + bytes.extend_from_slice(DOMAIN); + bytes.extend_from_slice(&ENCODING_VERSION.to_be_bytes()); + for value in [ + &self.identity.source_domain, + &self.identity.tagged_root_tree_oid, + &self.identity.scope, + ] { + write_string(&mut bytes, value); + } + for value in [ + self.identity.schema_version, + self.identity.metadata_codec, + self.identity.materialization_policy, + self.identity.fs_semantics, + self.identity.access_projection, + ] { + bytes.extend_from_slice(&value.to_be_bytes()); + } + bytes.extend_from_slice(&self.identity.verification_revision.to_be_bytes()); + bytes.extend_from_slice(&self.identity.projection_revision.to_be_bytes()); + bytes.extend_from_slice(&self.root); + bytes.extend_from_slice(&(self.delta.len() as u32).to_be_bytes()); + for (page, size) in &self.delta { + bytes.extend_from_slice(page); + bytes.extend_from_slice(&size.to_be_bytes()); + } + bytes.extend_from_slice(&(self.edges.len() as u32).to_be_bytes()); + for (parent, child) in &self.edges { + bytes.extend_from_slice(parent); + bytes.extend_from_slice(child); + } + bytes.extend_from_slice(&(self.reused.len() as u32).to_be_bytes()); + for (page, reused) in &self.reused { + bytes.extend_from_slice(page); + bytes.extend_from_slice(&reused.generation.to_be_bytes()); + bytes.extend_from_slice(reused.attestation_id.as_bytes()); + bytes.extend_from_slice(&reused.attestation_digest); + bytes.extend_from_slice(&reused.certificate_digest); + } + bytes.extend_from_slice(&(self.source_roots.len() as u32).to_be_bytes()); + for (tree_oid, page) in &self.source_roots { + write_string(&mut bytes, tree_oid); + bytes.extend_from_slice(page); + } + Ok(bytes) + } + + pub(crate) fn decode(bytes: &[u8], expected_digest: &[u8; 32]) -> Result { + if bytes.len() > MAX_PLAN_BYTES { + return Err(limit("stored rooted metadata plan exceeds its byte budget")); + } + let digest: [u8; 32] = Sha256::digest(bytes).into(); + if &digest != expected_digest { + return Err(integrity("stored rooted metadata plan digest mismatch")); + } + let mut reader = PlanReader(bytes); + if reader.take(DOMAIN.len())? != DOMAIN || reader.u16()? != ENCODING_VERSION { + return Err(integrity("unsupported rooted metadata plan encoding")); + } + let identity = MetadataInstallIdentity { + source_domain: reader.string(64)?, + tagged_root_tree_oid: reader.string(128)?, + scope: reader.string(4096)?, + schema_version: reader.u16()?, + metadata_codec: reader.u16()?, + materialization_policy: reader.u16()?, + fs_semantics: reader.u16()?, + access_projection: reader.u16()?, + verification_revision: i32::from_be_bytes(reader.array()?), + projection_revision: reader.u16()?, + }; + let root = reader.array()?; + let limits = MetadataDagLimits::default(); + let count = reader.count(limits.nodes)?; + let mut delta = BTreeMap::new(); + let mut last = None; + for _ in 0..count { + let page = reader.array()?; + let size = reader.u64()?; + if last.is_some_and(|previous| previous >= page) { + return Err(integrity("rooted metadata delta is not uniquely ordered")); + } + last = Some(page); + delta.insert(page, size); + } + let count = reader.count(limits.edges)?; + let mut edges = BTreeSet::new(); + let mut last = None; + for _ in 0..count { + let edge = (reader.array()?, reader.array()?); + if last.is_some_and(|previous| previous >= edge) { + return Err(integrity("rooted metadata edges are not uniquely ordered")); + } + last = Some(edge); + edges.insert(edge); + } + let count = reader.count(limits.nodes - delta.len())?; + let mut reused = BTreeMap::new(); + let mut last = None; + for _ in 0..count { + let page = reader.array()?; + let proof = RootedReuseRoot { + generation: i64::from_be_bytes(reader.array()?), + attestation_id: Uuid::from_bytes(reader.array()?), + attestation_digest: reader.array()?, + certificate_digest: reader.array()?, + }; + if last.is_some_and(|previous| previous >= page) { + return Err(integrity( + "rooted metadata reused roots are not uniquely ordered", + )); + } + last = Some(page); + reused.insert(page, proof); + } + let count = reader.count(limits.nodes)?; + let mut source_roots: BTreeMap = BTreeMap::new(); + for _ in 0..count { + let tree_oid = reader.string(128)?; + let page = reader.array()?; + if source_roots + .last_key_value() + .is_some_and(|(previous, _)| previous >= &tree_oid) + { + return Err(integrity( + "rooted metadata source roots are not uniquely ordered", + )); + } + source_roots.insert(tree_oid, page); + } + if !reader.0.is_empty() { + return Err(integrity("rooted metadata plan has trailing bytes")); + } + Self::new(identity, root, delta, edges, reused, source_roots) + } + + pub(crate) fn validate(&self) -> Result<(), SnapshotError> { + self.validate_and_order().map(|_| ()) + } + + /// Child-first delta order. Reused roots have no descendants in this plan. + pub(crate) fn child_first_delta(&self) -> Result, SnapshotError> { + self.validate_and_order() + } + + fn validate_and_order(&self) -> Result, SnapshotError> { + let identity = &self.identity; + if identity.source_domain != "native-git" + || identity.schema_version != mst2_codec::descriptor::SCHEMA_VERSION + || identity.metadata_codec != mst2_codec::descriptor::METADATA_CODEC + || identity.materialization_policy + != mst2_codec::descriptor::MATERIALIZATION_POLICY_GIT_RAW_V1 + || identity.fs_semantics != mst2_codec::descriptor::FS_SEMANTICS_LINUX_CODE_V1 + || identity.access_projection != mst2_codec::descriptor::ACCESS_PROJECTION_EXACT_FULL + || identity.verification_revision + != crate::jupiter::storage::mono_storage::MST2_VERIFICATION_VERSION + || identity.projection_revision != NATIVE_PROJECTION_REVISION + { + return Err(integrity("unsupported rooted native metadata profile")); + } + let hash_kind = tagged_hash_kind(&identity.tagged_root_tree_oid)?; + super::view::validate_scope_relative_path(&identity.scope) + .map_err(|_| integrity("noncanonical rooted metadata scope"))?; + let limits = MetadataDagLimits::default(); + if self.delta.len() > limits.nodes + || self.reused.len() > limits.nodes - self.delta.len() + || self.delta.is_empty() && self.reused.is_empty() + || self.edges.len() > limits.edges + || self.source_roots.is_empty() + || self.source_roots.len() > limits.nodes + { + return Err(limit("rooted metadata plan exceeds its group budget")); + } + if self.delta_bytes()? > limits.payload_bytes { + return Err(limit("rooted metadata delta exceeds its payload budget")); + } + if !self.contains(&self.root) + || self.delta.iter().any(|(page, size)| { + self.reused.contains_key(page) + || !(HEADER_LEN as u64..=PAGE_MAX_BYTES as u64).contains(size) + }) + || self.reused.values().any(|proof| proof.generation <= 0) + || self.delta.is_empty() + && (!self.reused.contains_key(&self.root) || !self.edges.is_empty()) + { + return Err(integrity( + "rooted metadata root, delta or reuse boundary is invalid", + )); + } + for (tree_oid, page) in &self.source_roots { + if tagged_hash_kind(tree_oid)? != hash_kind || !self.contains(page) { + return Err(integrity( + "rooted metadata source binding crossed its profile or boundary", + )); + } + } + if !self.source_roots.values().any(|page| page == &self.root) { + return Err(integrity( + "rooted metadata root lacks a source tree binding", + )); + } + if self.encoded_len() > MAX_PLAN_BYTES { + return Err(limit("rooted metadata plan exceeds its byte budget")); + } + + let mut remaining: BTreeMap<_, usize> = self.delta.keys().map(|page| (*page, 0)).collect(); + let mut parents: BTreeMap<_, Vec<_>> = BTreeMap::new(); + let mut children: BTreeMap<_, Vec<_>> = BTreeMap::new(); + for &(parent, child) in &self.edges { + if parent == child || !self.delta.contains_key(&parent) || !self.contains(&child) { + return Err(integrity( + "rooted metadata edge crossed its delta or reused boundary", + )); + } + children.entry(parent).or_default().push(child); + if self.delta.contains_key(&child) { + *remaining + .get_mut(&parent) + .ok_or_else(|| integrity("rooted metadata delta parent is missing"))? += 1; + parents.entry(child).or_default().push(parent); + } + } + let mut ready: BTreeSet<_> = remaining + .iter() + .filter_map(|(page, count)| (*count == 0).then_some(*page)) + .collect(); + let mut ordered = Vec::with_capacity(self.delta.len()); + while let Some(page) = ready.pop_first() { + ordered.push(page); + for parent in parents.get(&page).into_iter().flatten() { + let count = remaining + .get_mut(parent) + .ok_or_else(|| integrity("rooted metadata delta ancestor is missing"))?; + *count = count + .checked_sub(1) + .ok_or_else(|| integrity("rooted metadata delta edge accounting underflow"))?; + if *count == 0 { + ready.insert(*parent); + } + } + } + if ordered.len() != self.delta.len() { + return Err(integrity("rooted metadata delta contains a cycle")); + } + let mut reachable = BTreeSet::new(); + let mut pending = vec![self.root]; + while let Some(page) = pending.pop() { + if reachable.insert(page) { + pending.extend(children.get(&page).into_iter().flatten()); + } + } + if reachable.len() != self.delta.len() + self.reused.len() { + return Err(integrity( + "rooted metadata plan contains unreachable boundaries", + )); + } + Ok(ordered) + } + + fn contains(&self, page: &MetadataPageId) -> bool { + self.delta.contains_key(page) || self.reused.contains_key(page) + } + + fn encoded_len(&self) -> usize { + DOMAIN.len() + + 2 + + 3 * 4 + + self.identity.source_domain.len() + + self.identity.tagged_root_tree_oid.len() + + self.identity.scope.len() + + 5 * 2 + + 4 + + 2 + + 32 + + 4 * 4 + + self.delta.len() * 40 + + self.edges.len() * 64 + + self.reused.len() * 120 + + self + .source_roots + .keys() + .map(|tree_oid| 4 + tree_oid.len() + 32) + .sum::() + } +} + +fn tagged_hash_kind(oid: &str) -> Result<&str, SnapshotError> { + let (kind, hex) = oid + .split_once(':') + .ok_or_else(|| integrity("untagged rooted metadata source tree"))?; + let length = match kind { + "sha1" => 40, + "sha256" | "blake3" => 64, + _ => return Err(integrity("unsupported rooted metadata source hash kind")), + }; + if hex.len() != length + || !hex + .bytes() + .all(|byte| byte.is_ascii_digit() || (b'a'..=b'f').contains(&byte)) + { + return Err(integrity( + "noncanonical rooted metadata source tree identity", + )); + } + Ok(kind) +} + +fn write_string(bytes: &mut Vec, value: &str) { + bytes.extend_from_slice(&(value.len() as u32).to_be_bytes()); + bytes.extend_from_slice(value.as_bytes()); +} + +struct PlanReader<'a>(&'a [u8]); + +impl<'a> PlanReader<'a> { + fn take(&mut self, count: usize) -> Result<&'a [u8], SnapshotError> { + if count > self.0.len() { + return Err(integrity("truncated rooted metadata plan")); + } + let (bytes, rest) = self.0.split_at(count); + self.0 = rest; + Ok(bytes) + } + + fn array(&mut self) -> Result<[u8; N], SnapshotError> { + self.take(N)? + .try_into() + .map_err(|_| integrity("invalid rooted metadata plan field")) + } + + fn u16(&mut self) -> Result { + Ok(u16::from_be_bytes(self.array()?)) + } + + fn u32(&mut self) -> Result { + Ok(u32::from_be_bytes(self.array()?)) + } + + fn u64(&mut self) -> Result { + Ok(u64::from_be_bytes(self.array()?)) + } + + fn count(&mut self, maximum: usize) -> Result { + let count = self.u32()? as usize; + if count > maximum { + return Err(limit("rooted metadata plan count exceeds its budget")); + } + Ok(count) + } + + fn string(&mut self, maximum: usize) -> Result { + let count = self.count(maximum)?; + String::from_utf8(self.take(count)?.to_vec()) + .map_err(|_| integrity("non-UTF8 rooted metadata plan identity")) + } +} + +fn integrity(message: &str) -> SnapshotError { + SnapshotError::new(SnapshotErrorCode::IntegrityError, message) +} + +fn limit(message: &str) -> SnapshotError { + SnapshotError::new(SnapshotErrorCode::LimitExceeded, message) +} + +#[cfg(test)] +mod tests { + use super::*; + + fn page(value: u32) -> MetadataPageId { + let mut id = [0; 32]; + id[28..].copy_from_slice(&value.to_be_bytes()); + id + } + + fn identity() -> MetadataInstallIdentity { + MetadataInstallIdentity { + source_domain: "native-git".into(), + tagged_root_tree_oid: format!("sha1:{}", "a".repeat(40)), + scope: "/".into(), + schema_version: mst2_codec::descriptor::SCHEMA_VERSION, + metadata_codec: mst2_codec::descriptor::METADATA_CODEC, + materialization_policy: mst2_codec::descriptor::MATERIALIZATION_POLICY_GIT_RAW_V1, + fs_semantics: mst2_codec::descriptor::FS_SEMANTICS_LINUX_CODE_V1, + access_projection: mst2_codec::descriptor::ACCESS_PROJECTION_EXACT_FULL, + verification_revision: crate::jupiter::storage::mono_storage::MST2_VERIFICATION_VERSION, + projection_revision: NATIVE_PROJECTION_REVISION, + } + } + + fn reuse(value: u128) -> RootedReuseRoot { + RootedReuseRoot { + generation: 7, + attestation_id: Uuid::from_u128(value), + attestation_digest: [11; 32], + certificate_digest: [12; 32], + } + } + + fn plan() -> RootedMetadataInstallPlan { + let identity = identity(); + RootedMetadataInstallPlan::new( + identity.clone(), + page(1), + BTreeMap::from([(page(1), 20), (page(2), 57), (page(3), 16384)]), + BTreeSet::from([ + (page(1), page(2)), + (page(1), page(3)), + (page(1), page(5)), + (page(2), page(4)), + (page(3), page(4)), + ]), + BTreeMap::from([(page(4), reuse(1)), (page(5), reuse(2))]), + BTreeMap::from([ + (identity.tagged_root_tree_oid, page(1)), + (format!("sha1:{}", "b".repeat(40)), page(4)), + ]), + ) + .unwrap() + } + + fn decode(bytes: &[u8]) -> Result { + RootedMetadataInstallPlan::decode(bytes, &Sha256::digest(bytes).into()) + } + + fn records(bytes: &[u8]) -> [Vec>; 4] { + let mut reader = PlanReader(bytes); + reader.take(DOMAIN.len() + 2).unwrap(); + for _ in 0..3 { + reader.string(4096).unwrap(); + } + reader.take(16 + 32).unwrap(); + let mut sections: [Vec>; 4] = std::array::from_fn(|_| Vec::new()); + for (section, width) in sections[..3].iter_mut().zip([40, 64, 120]) { + let count = reader.u32().unwrap(); + for _ in 0..count { + let start = bytes.len() - reader.0.len(); + reader.take(width).unwrap(); + section.push(start..start + width); + } + } + let count = reader.u32().unwrap(); + for _ in 0..count { + let start = bytes.len() - reader.0.len(); + reader.string(128).unwrap(); + reader.take(32).unwrap(); + sections[3].push(start..bytes.len() - reader.0.len()); + } + assert!(reader.0.is_empty()); + sections + } + + #[test] + fn rooted_plan_roundtrip_and_shared_reuse_need_only_delta_order() { + let plan = plan(); + let bytes = plan.encode().unwrap(); + assert_eq!(bytes.len(), plan.encoded_len()); + assert_eq!(bytes.len(), 1004); + assert_eq!( + hex::encode(plan.digest().unwrap()), + "e74e98e8737cfcbdfb264d9f12e15ba2126e96b1e5f35cb0cfd89fb24dde808b" + ); + assert_eq!(decode(&bytes).unwrap(), plan); + assert_eq!( + plan.child_first_delta().unwrap(), + [page(2), page(3), page(1)] + ); + assert_eq!(plan.delta_bytes().unwrap(), 20 + 57 + 16384); + for range in &records(&bytes)[2] { + assert_eq!(range.len(), 120); + } + let original = plan.digest().unwrap(); + for field in 0..4 { + let mut changed = plan.clone(); + let proof = changed.reused.get_mut(&page(4)).unwrap(); + match field { + 0 => proof.generation += 1, + 1 => proof.attestation_id = Uuid::from_u128(9), + 2 => proof.attestation_digest[0] ^= 1, + _ => proof.certificate_digest[0] ^= 1, + } + assert_ne!(changed.digest().unwrap(), original); + } + let mut changed = plan.clone(); + changed.identity.scope = "/目录/a\\b".into(); + assert_ne!(changed.digest().unwrap(), original); + assert_eq!(decode(&changed.encode().unwrap()).unwrap(), changed); + changed = plan; + changed + .source_roots + .insert(format!("sha1:{}", "c".repeat(40)), page(3)); + assert_ne!(changed.digest().unwrap(), original); + } + + #[test] + fn rooted_plan_stored_order_duplicates_digest_domain_version_truncation_and_eof_reject() { + let plan = plan(); + let bytes = plan.encode().unwrap(); + let mut wrong_digest = plan.digest().unwrap(); + wrong_digest[0] ^= 1; + assert!(RootedMetadataInstallPlan::decode(&bytes, &wrong_digest).is_err()); + for section in records(&bytes) { + let mut changed = bytes.clone(); + let a = §ion[0]; + let b = §ion[1]; + assert_eq!(a.len(), b.len()); + changed[a.clone()].copy_from_slice(&bytes[b.clone()]); + changed[b.clone()].copy_from_slice(&bytes[a.clone()]); + assert!(decode(&changed).is_err()); + changed[b.clone()].copy_from_slice(&bytes[b.clone()]); + assert!(decode(&changed).is_err()); + } + for length in [0, DOMAIN.len(), bytes.len() - 1] { + assert!(decode(&bytes[..length]).is_err()); + } + let mut changed = bytes.clone(); + changed.push(0); + assert!(decode(&changed).is_err()); + changed = bytes.clone(); + changed[0] ^= 1; + assert!(decode(&changed).is_err()); + changed = bytes; + changed[DOMAIN.len() + 1] = 2; + assert!(decode(&changed).is_err()); + let oversized = vec![0; MAX_PLAN_BYTES + 1]; + assert_eq!( + decode(&oversized).unwrap_err().code, + SnapshotErrorCode::LimitExceeded + ); + } + + #[test] + fn rooted_plan_decode_bounds_declared_counts_strings_and_reused_generations() { + let bytes = plan().encode().unwrap(); + let sections = records(&bytes); + for section in §ions { + let mut changed = bytes.clone(); + let count_at = section[0].start - 4; + changed[count_at..count_at + 4].copy_from_slice(&u32::MAX.to_be_bytes()); + assert_eq!( + decode(&changed).unwrap_err().code, + SnapshotErrorCode::LimitExceeded + ); + } + let mut changed = bytes.clone(); + let source_at = sections[3][0].start; + changed[source_at..source_at + 4].copy_from_slice(&u32::MAX.to_be_bytes()); + assert_eq!( + decode(&changed).unwrap_err().code, + SnapshotErrorCode::LimitExceeded + ); + changed = bytes.clone(); + changed[DOMAIN.len() + 2 + 4] = 0xff; + assert_eq!( + decode(&changed).unwrap_err().code, + SnapshotErrorCode::IntegrityError + ); + changed = bytes; + let generation_at = sections[2][0].start + 32; + changed[generation_at..generation_at + 8].copy_from_slice(&i64::MIN.to_be_bytes()); + assert_eq!( + decode(&changed).unwrap_err().code, + SnapshotErrorCode::IntegrityError + ); + } + + #[test] + fn rooted_plan_rejects_cycles_unknown_endpoints_reused_parents_and_unreachable_members() { + let original = plan(); + for invalid in 0..7 { + let mut changed = original.clone(); + match invalid { + 0 => { + changed.edges.insert((page(2), page(1))); + } + 1 => { + changed.edges.insert((page(4), page(2))); + } + 2 => { + changed.edges.insert((page(2), page(99))); + } + 3 => { + changed.edges.insert((page(99), page(2))); + } + 4 => { + changed.delta.insert(page(6), 20); + } + 5 => { + changed.reused.insert(page(6), reuse(6)); + } + _ => { + changed.reused.insert(page(2), reuse(2)); + } + } + assert_eq!( + changed.encode().unwrap_err().code, + SnapshotErrorCode::IntegrityError + ); + } + } + + #[test] + fn rooted_plan_zero_delta_requires_its_single_reused_root_and_source_binding() { + let identity = identity(); + let plan = RootedMetadataInstallPlan::new( + identity.clone(), + page(4), + BTreeMap::new(), + BTreeSet::new(), + BTreeMap::from([(page(4), reuse(4))]), + BTreeMap::from([(identity.tagged_root_tree_oid, page(4))]), + ) + .unwrap(); + assert!(plan.child_first_delta().unwrap().is_empty()); + assert_eq!(plan.delta_bytes().unwrap(), 0); + assert_eq!(plan.encode().unwrap().len(), 363); + assert_eq!( + hex::encode(plan.digest().unwrap()), + "81b4cbdaaacd7efe8687477d05bd4db190b974f5aa99ce7e8101326857d0e7b1" + ); + assert_eq!(decode(&plan.encode().unwrap()).unwrap(), plan); + for invalid in 0..5 { + let mut changed = plan.clone(); + match invalid { + 0 => { + changed.root = page(9); + } + 1 => { + changed.edges.insert((page(4), page(4))); + } + 2 => { + changed.reused.insert(page(5), reuse(5)); + } + 3 => { + changed.delta.insert(page(5), 20); + } + _ => { + changed.source_roots.clear(); + } + } + assert!(changed.encode().is_err()); + } + } + + #[test] + fn rooted_plan_rejects_source_profile_scope_size_and_generation_mismatches() { + let original = plan(); + for invalid in 0..18 { + let mut changed = original.clone(); + match invalid { + 0 => changed.identity.metadata_codec += 1, + 1 => changed.identity.projection_revision += 1, + 2 => changed.identity.verification_revision += 1, + 3 => changed.identity.scope = "/a/..".into(), + 4 => changed.identity.scope = format!("/{}", "a".repeat(256)), + 5 => changed.identity.tagged_root_tree_oid = format!("sha1:{}", "A".repeat(40)), + 6 => { + changed.delta.insert(page(2), HEADER_LEN as u64 - 1); + } + 7 => { + changed.delta.insert(page(2), PAGE_MAX_BYTES as u64 + 1); + } + 8 => { + changed.reused.get_mut(&page(4)).unwrap().generation = 0; + } + 9 => { + changed + .source_roots + .insert(format!("sha256:{}", "a".repeat(64)), page(3)); + } + 10 => { + changed + .source_roots + .insert(format!("sha1:{}", "d".repeat(40)), page(99)); + } + 11 => { + changed + .source_roots + .retain(|_, page_id| *page_id != page(1)); + } + 12 => { + changed.source_roots.insert("sha1:bad".into(), page(3)); + } + 13 => changed.identity.source_domain = "other".into(), + 14 => changed.identity.schema_version += 1, + 15 => changed.identity.materialization_policy += 1, + 16 => changed.identity.fs_semantics += 1, + _ => changed.identity.access_projection += 1, + } + assert!(changed.encode().is_err()); + } + for kind in ["sha256", "blake3"] { + let mut changed = original.clone(); + changed.identity.tagged_root_tree_oid = format!("{kind}:{}", "a".repeat(64)); + changed.source_roots = + BTreeMap::from([(changed.identity.tagged_root_tree_oid.clone(), page(1))]); + assert_eq!(decode(&changed.encode().unwrap()).unwrap(), changed); + } + } + + #[test] + fn rooted_plan_admits_exact_group_payload_and_source_count_boundaries() { + let identity = identity(); + let limits = MetadataDagLimits::default(); + let delta: BTreeMap<_, _> = (1..=limits.nodes as u32) + .map(|value| (page(value), PAGE_MAX_BYTES as u64)) + .collect(); + let edges = (2..=limits.nodes as u32) + .map(|value| (page(1), page(value))) + .collect(); + let full_plan = RootedMetadataInstallPlan::new( + identity.clone(), + page(1), + delta, + edges, + BTreeMap::new(), + BTreeMap::from([(identity.tagged_root_tree_oid, page(1))]), + ) + .unwrap(); + assert_eq!(full_plan.delta_bytes().unwrap(), limits.payload_bytes); + assert_eq!(decode(&full_plan.encode().unwrap()).unwrap(), full_plan); + let mut changed = full_plan.clone(); + changed + .delta + .insert(page(limits.nodes as u32 + 1), HEADER_LEN as u64); + assert_eq!( + changed.encode().unwrap_err().code, + SnapshotErrorCode::LimitExceeded + ); + changed = full_plan.clone(); + changed + .reused + .insert(page(limits.nodes as u32 + 1), reuse(9)); + assert_eq!( + changed.encode().unwrap_err().code, + SnapshotErrorCode::LimitExceeded + ); + + let mut source_plan = plan(); + source_plan.source_roots = (0..limits.nodes) + .map(|value| (format!("sha1:{value:040x}"), page(1))) + .collect(); + assert_eq!(decode(&source_plan.encode().unwrap()).unwrap(), source_plan); + source_plan + .source_roots + .insert(format!("sha1:{:040x}", limits.nodes), page(1)); + assert_eq!( + source_plan.encode().unwrap_err().code, + SnapshotErrorCode::LimitExceeded + ); + } + + #[test] + fn rooted_plan_admits_exact_edge_boundary_then_rejects_one_more_edge() { + let mut value = plan(); + value.delta = (1..=4096).map(|id| (page(id), HEADER_LEN as u64)).collect(); + value.reused.clear(); + value.source_roots.retain(|_, root| *root == page(1)); + value.edges = (2..=4096).map(|id| (page(1), page(id))).collect(); + 'fill: for parent in 2..=4096 { + for child in parent + 1..=4096 { + if value.edges.len() == MetadataDagLimits::default().edges { + break 'fill; + } + value.edges.insert((page(parent), page(child))); + } + } + assert_eq!(value.edges.len(), MetadataDagLimits::default().edges); + assert_eq!(decode(&value.encode().unwrap()).unwrap(), value); + assert!(value.edges.insert((page(4095), page(4096)))); + assert_eq!( + value.encode().unwrap_err().code, + SnapshotErrorCode::LimitExceeded + ); + } +} diff --git a/src/ceres/snapshot/rooted_metadata_projection.rs b/src/ceres/snapshot/rooted_metadata_projection.rs new file mode 100644 index 00000000..b4b1a517 --- /dev/null +++ b/src/ceres/snapshot/rooted_metadata_projection.rs @@ -0,0 +1,1477 @@ +//! Native source projection that stops at certified reused-directory boundaries. +//! Hints and operation-local memoization are not database installation authority. + +use std::{ + borrow::Cow, + collections::{BTreeMap, BTreeSet, VecDeque}, + sync::Arc, + time::Duration, +}; + +use async_trait::async_trait; +use futures::StreamExt; +use git_internal::{hash::ObjectHash, internal::object::tree::Tree}; +use mst2_codec::{ + descriptor::{ + ACCESS_PROJECTION_EXACT_FULL, FS_SEMANTICS_LINUX_CODE_V1, + MATERIALIZATION_POLICY_GIT_RAW_V1, METADATA_CODEC, SCHEMA_VERSION, + }, + metapage::{Entry, EntryKind, HEADER_LEN, PAGE_MAX_BYTES, Page, page_id}, +}; +use sea_orm::ActiveValue::Set; +use sha2::{Digest, Sha256}; + +use super::{ + content_budget::{MemoryBudget, RANGE_WORK_BYTES, projection_budget}, + error::{SnapshotError, SnapshotErrorCode}, + metadata_install::MetadataInstallIdentity, + projection_observation::NATIVE_PROJECTION_REVISION, + resolver::{FsKind, direct_entries}, + retention_dag::{MetadataDagLimits, MetadataPageId, MetadataPagePayload}, + rooted_metadata_install::{RootedMetadataInstallPlan, RootedReuseRoot}, + view::validate_scope_relative_path, +}; +use crate::{ + ceres::api_service::ApiHandler, + common::errors::MegaError, + jupiter::storage::{Storage, mono_storage::MST2_VERIFICATION_VERSION}, + orbit_api::object_storage::{ObjectByteStream, ObjectMeta}, +}; + +const MAX_VERIFIED_FILE_BYTES: u64 = 8_796_093_022_208; +const SOURCE_PROGRESS_TIMEOUT: Duration = Duration::from_secs(30); + +#[async_trait] +pub(crate) trait RootedReuseLookup: Send + Sync { + async fn lookup_reuse( + &self, + tree_oid: &str, + identity: &MetadataInstallIdentity, + ) -> Result, SnapshotError>; +} + +#[derive(Debug, Clone, PartialEq, Eq)] +pub(crate) struct CertifiedReusableDirectory { + pub(crate) page_id: MetadataPageId, + pub(crate) proof: RootedReuseRoot, + pub(crate) relative_path_bytes: usize, + pub(crate) relative_components: usize, + pub(crate) closure_nodes_upper: usize, + pub(crate) closure_edges_upper: usize, + pub(crate) closure_bytes_upper: u64, + pub(crate) closure_entries_upper: usize, +} + +/// API calls and materialized outputs only, not total database or codec work. +#[derive(Debug, Clone, Default, PartialEq, Eq, serde::Serialize)] +pub(crate) struct RootedProjectionWork { + pub(crate) input_tree_entries: u64, + pub(crate) scope_path_entries_examined: u64, + pub(crate) tree_fetches: u64, + pub(crate) reuse_lookups: u64, + pub(crate) directory_memo_hits: u64, + pub(crate) directories_scanned: u64, + pub(crate) direct_entries_scanned: u64, + pub(crate) verified_blob_queries: u64, + pub(crate) verified_blob_facts_loaded: u64, + pub(crate) verified_blob_persistence_batches: u64, + pub(crate) blob_fetches: u64, + pub(crate) raw_bytes_fetched: u64, + pub(crate) raw_bytes_hashed: u64, + pub(crate) directory_root_builds: u64, + pub(crate) radix_route_builds: u64, + pub(crate) codec_input_entries: u64, + pub(crate) delta_pages: u64, + pub(crate) delta_bytes: u64, + pub(crate) reused_roots: u64, + pub(crate) reused_closure_nodes_upper: u64, + pub(crate) reused_closure_edges_upper: u64, + pub(crate) reused_closure_bytes_upper: u64, + pub(crate) reused_closure_entries_upper: u64, + pub(crate) plan_bytes: u64, +} + +#[derive(Debug)] +pub(crate) struct PreparedRootedNativeMetadata { + pub(crate) plan: RootedMetadataInstallPlan, + pub(crate) payloads: Vec, + pub(crate) work: RootedProjectionWork, +} + +pub(crate) async fn prepare_rooted_native_metadata< + T: ApiHandler + ?Sized, + R: RootedReuseLookup + ?Sized, +>( + handler: &T, + root_tree: &Tree, + scope: &str, + reuse: &R, +) -> Result { + validate_scope_relative_path(scope)?; + if !handler.native_snapshot_projection() { + return Err(SnapshotError::new( + SnapshotErrorCode::ScopeInvalid, + "rooted metadata projection requires native Git source semantics", + )); + } + let identity = MetadataInstallIdentity { + source_domain: "native-git".into(), + tagged_root_tree_oid: root_tree.id.to_tagged_string(), + scope: scope.to_owned(), + schema_version: SCHEMA_VERSION, + metadata_codec: METADATA_CODEC, + materialization_policy: MATERIALIZATION_POLICY_GIT_RAW_V1, + fs_semantics: FS_SEMANTICS_LINUX_CODE_V1, + access_projection: ACCESS_PROJECTION_EXACT_FULL, + verification_revision: MST2_VERIFICATION_VERSION, + projection_revision: NATIVE_PROJECTION_REVISION, + }; + let storage = handler.get_context(); + let mut state = ProjectionState::new(identity); + state.work.input_tree_entries = root_tree.tree_items.len() as u64; + let scoped_oid = resolve_scope_oid(handler, root_tree, scope, &mut state.work).await?; + let loaded = (scope == "/").then_some(root_tree); + let root = project_directory( + handler, reuse, &storage, scoped_oid, loaded, scope, &mut state, + ) + .await?; + state.finish(root.page_id) +} + +#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)] +struct PathBounds { + bytes: usize, + components: usize, +} + +impl PathBounds { + fn include(&mut self, name: &str, child: Self) -> Result<(), SnapshotError> { + let bytes = name + .len() + .checked_add(1) + .and_then(|value| value.checked_add(child.bytes)) + .filter(|value| *value <= 4096) + .ok_or_else(path_limit)?; + let components = child + .components + .checked_add(1) + .filter(|value| *value <= 256) + .ok_or_else(path_limit)?; + self.bytes = self.bytes.max(bytes); + self.components = self.components.max(components); + Ok(()) + } + + fn validate_at(self, prefix: &str) -> Result<(), SnapshotError> { + validate_scope_relative_path(prefix)?; + let (bytes, components) = if prefix == "/" { + (0, 0) + } else { + (prefix.len(), prefix[1..].split('/').count()) + }; + bytes + .checked_add(self.bytes) + .filter(|value| *value <= 4096) + .ok_or_else(path_limit)?; + components + .checked_add(self.components) + .filter(|value| *value <= 256) + .ok_or_else(path_limit)?; + Ok(()) + } +} + +#[derive(Debug, Clone, Copy)] +struct DirectorySummary { + page_id: MetadataPageId, + bounds: PathBounds, +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +struct BlobFact { + size: u64, + digest: [u8; 32], +} + +#[derive(Default)] +struct ClosureUpper { + nodes: usize, + edges: usize, + bytes: u64, + entries: usize, +} + +impl ClosureUpper { + fn add(&mut self, directory: &CertifiedReusableDirectory) -> Result<(), SnapshotError> { + self.nodes = self + .nodes + .checked_add(directory.closure_nodes_upper) + .ok_or_else(budget_limit)?; + self.edges = self + .edges + .checked_add(directory.closure_edges_upper) + .ok_or_else(budget_limit)?; + self.bytes = self + .bytes + .checked_add(directory.closure_bytes_upper) + .ok_or_else(budget_limit)?; + self.entries = self + .entries + .checked_add(directory.closure_entries_upper) + .ok_or_else(budget_limit)?; + Ok(()) + } +} + +struct ProjectionState { + identity: MetadataInstallIdentity, + required_trees: BTreeSet, + active_trees: BTreeSet, + directories: BTreeMap, + bounds: BTreeMap, + payloads: BTreeMap, + page_entries: BTreeMap, + edges: BTreeSet<(MetadataPageId, MetadataPageId)>, + reused: BTreeMap, + source_roots: BTreeMap, + blob_facts: BTreeMap, + source_entries: usize, + source_encoded_bytes: u64, + codec_entry_visits: usize, + payload_bytes: u64, + delta_entries: usize, + reuse_upper: ClosureUpper, + work: RootedProjectionWork, +} + +impl ProjectionState { + fn new(identity: MetadataInstallIdentity) -> Self { + Self { + identity, + required_trees: BTreeSet::new(), + active_trees: BTreeSet::new(), + directories: BTreeMap::new(), + bounds: BTreeMap::new(), + payloads: BTreeMap::new(), + page_entries: BTreeMap::new(), + edges: BTreeSet::new(), + reused: BTreeMap::new(), + source_roots: BTreeMap::new(), + blob_facts: BTreeMap::new(), + source_entries: 0, + source_encoded_bytes: 0, + codec_entry_visits: 0, + payload_bytes: 0, + delta_entries: 0, + reuse_upper: ClosureUpper::default(), + work: RootedProjectionWork::default(), + } + } + + fn require_tree(&mut self, tree_oid: &str) -> Result<(), SnapshotError> { + if !self.required_trees.contains(tree_oid) { + if self.required_trees.len() >= MetadataDagLimits::default().nodes { + return Err(budget_limit()); + } + self.required_trees.insert(tree_oid.to_owned()); + } + Ok(()) + } + + fn admit_entries(&mut self, entries: &[(String, FsKind, String)]) -> Result<(), SnapshotError> { + let limits = MetadataDagLimits::default(); + self.source_encoded_bytes = self + .source_encoded_bytes + .checked_add(HEADER_LEN as u64) + .filter(|value| *value <= limits.payload_bytes) + .ok_or_else(budget_limit)?; + let mut previous: Option<&str> = None; + for (name, kind, _) in entries { + if name.is_empty() + || name.len() > 255 + || name == "." + || name == ".." + || name.contains('/') + || name.contains('\0') + || previous.is_some_and(|value| value >= name.as_str()) + { + return Err(integrity( + "source tree has an invalid or duplicate UTF-8 name", + )); + } + previous = Some(name); + let value_bytes = if *kind == FsKind::Directory { 32 } else { 40 }; + self.source_encoded_bytes = self + .source_encoded_bytes + .checked_add((3 + name.len() + value_bytes) as u64) + .filter(|value| *value <= limits.payload_bytes) + .ok_or_else(budget_limit)?; + } + Ok(()) + } + + fn record_bounds( + &mut self, + page: MetadataPageId, + bounds: PathBounds, + ) -> Result<(), SnapshotError> { + if self + .bounds + .get(&page) + .is_some_and(|previous| previous != &bounds) + { + return Err(integrity( + "one canonical directory has inconsistent descendant path bounds", + )); + } + self.bounds.insert(page, bounds); + Ok(()) + } + + fn record_reuse( + &mut self, + tree_oid: &str, + directory: CertifiedReusableDirectory, + path: &str, + ) -> Result { + let limits = MetadataDagLimits::default(); + if directory.page_id == [0; 32] + || directory.proof.generation <= 0 + || directory.closure_nodes_upper == 0 + || directory.closure_nodes_upper > limits.nodes + || directory.closure_edges_upper > limits.edges + || !(HEADER_LEN as u64..=limits.payload_bytes).contains(&directory.closure_bytes_upper) + || directory.closure_entries_upper > limits.entries + || directory.relative_path_bytes > 4096 + || directory.relative_components > 256 + { + return Err(integrity( + "reused directory hint has an invalid certificate summary", + )); + } + let summary = DirectorySummary { + page_id: directory.page_id, + bounds: PathBounds { + bytes: directory.relative_path_bytes, + components: directory.relative_components, + }, + }; + summary.bounds.validate_at(path)?; + self.record_bounds(directory.page_id, summary.bounds)?; + if !self.payloads.contains_key(&directory.page_id) { + if let Some(previous) = self.reused.get(&directory.page_id) { + if previous.proof.generation != directory.proof.generation + || previous.proof.certificate_digest != directory.proof.certificate_digest + || previous.closure_nodes_upper != directory.closure_nodes_upper + || previous.closure_edges_upper != directory.closure_edges_upper + || previous.closure_bytes_upper != directory.closure_bytes_upper + || previous.closure_entries_upper != directory.closure_entries_upper + { + return Err(integrity( + "one reused boundary has inconsistent lifetime or certificate summaries", + )); + } + } else { + self.reuse_upper.add(&directory)?; + self.reused.insert(directory.page_id, directory); + self.check_budget()?; + } + } + self.source_roots + .insert(tree_oid.to_owned(), summary.page_id); + Ok(summary) + } + + fn charge_codec(&mut self, entries: usize, route_length: usize) -> Result<(), SnapshotError> { + let count = entries + .checked_mul(route_length.checked_add(1).ok_or_else(budget_limit)?) + .ok_or_else(budget_limit)?; + self.codec_entry_visits = self + .codec_entry_visits + .checked_add(count) + .filter(|value| *value <= MetadataDagLimits::default().prepare_entry_visits) + .ok_or_else(budget_limit)?; + self.work.codec_input_entries = self + .work + .codec_input_entries + .checked_add(entries as u64) + .ok_or_else(budget_limit)?; + Ok(()) + } + + fn add_edge( + &mut self, + parent: MetadataPageId, + child: MetadataPageId, + ) -> Result<(), SnapshotError> { + if self.edges.insert((parent, child)) { + self.check_budget()?; + } + Ok(()) + } + + fn collect_directory( + &mut self, + entries: &[Entry], + root_bytes: Vec, + ) -> Result { + let root = page_id(&root_bytes); + let mut routes = VecDeque::from([(Vec::::new(), root, Some(root_bytes))]); + while let Some((route, expected, provided)) = routes.pop_front() { + if self.payloads.contains_key(&expected) || self.reused.contains_key(&expected) { + continue; + } + let bytes = if let Some(bytes) = provided { + bytes + } else { + self.charge_codec(entries.len(), route.len())?; + self.work.radix_route_builds += 1; + Page::pages_along_route(entries, &route) + .map_err(codec_error)? + .pop() + .ok_or_else(|| integrity("canonical radix route returned no page"))? + }; + if page_id(&bytes) != expected || !(HEADER_LEN..=PAGE_MAX_BYTES).contains(&bytes.len()) + { + return Err(integrity( + "canonical radix page differs from its parent binding", + )); + } + let (page, _) = Page::decode(&bytes).map_err(codec_error)?; + let page_entries = match page { + Page::Leaf { entries } => { + for entry in entries.iter().filter(|entry| entry.is_dir()) { + self.add_edge(expected, entry.child_root)?; + } + entries.len() + } + Page::Branch { + terminal, children, .. + } => { + if let Some(entry) = terminal.as_ref().filter(|entry| entry.is_dir()) { + self.add_edge(expected, entry.child_root)?; + } + for child in children { + self.add_edge(expected, child.child_page_id)?; + let mut child_route = route.clone(); + child_route.push(child.label); + routes.push_back((child_route, child.child_page_id, None)); + } + usize::from(terminal.is_some()) + } + }; + self.payload_bytes = self + .payload_bytes + .checked_add(bytes.len() as u64) + .ok_or_else(budget_limit)?; + self.delta_entries = self + .delta_entries + .checked_add(page_entries) + .ok_or_else(budget_limit)?; + self.page_entries.insert(expected, page_entries); + self.payloads.insert( + expected, + MetadataPagePayload { + id: expected, + size: bytes.len() as u64, + bytes, + }, + ); + self.check_budget()?; + } + Ok(root) + } + + fn check_budget(&self) -> Result<(), SnapshotError> { + let limits = MetadataDagLimits::default(); + if self + .payloads + .len() + .checked_add(self.reuse_upper.nodes) + .is_none_or(|value| value > limits.nodes) + || self + .edges + .len() + .checked_add(self.reuse_upper.edges) + .is_none_or(|value| value > limits.edges) + || self + .payload_bytes + .checked_add(self.reuse_upper.bytes) + .is_none_or(|value| value > limits.payload_bytes) + || self + .delta_entries + .checked_add(self.reuse_upper.entries) + .is_none_or(|value| value > limits.entries) + { + return Err(budget_limit()); + } + Ok(()) + } + + fn finish( + mut self, + root: MetadataPageId, + ) -> Result { + let mut children: BTreeMap<_, Vec<_>> = BTreeMap::new(); + for &(parent, child) in &self.edges { + children.entry(parent).or_default().push(child); + } + let mut reachable = BTreeSet::new(); + let mut pending = vec![root]; + while let Some(page) = pending.pop() { + if reachable.insert(page) && !self.reused.contains_key(&page) { + pending.extend(children.get(&page).into_iter().flatten()); + } + } + self.payloads.retain(|page, _| reachable.contains(page)); + self.page_entries.retain(|page, _| reachable.contains(page)); + self.reused.retain(|page, _| reachable.contains(page)); + self.source_roots.retain(|_, page| reachable.contains(page)); + self.edges + .retain(|(parent, child)| reachable.contains(parent) && reachable.contains(child)); + self.payload_bytes = self + .payloads + .values() + .try_fold(0u64, |total, page| total.checked_add(page.size)) + .ok_or_else(budget_limit)?; + self.delta_entries = self + .page_entries + .values() + .try_fold(0usize, |total, count| total.checked_add(*count)) + .ok_or_else(budget_limit)?; + self.reuse_upper = ClosureUpper::default(); + for directory in self.reused.values() { + self.reuse_upper.add(directory)?; + } + self.check_budget()?; + let plan = RootedMetadataInstallPlan::new( + self.identity, + root, + self.payloads + .iter() + .map(|(page, payload)| (*page, payload.size)) + .collect(), + self.edges, + self.reused + .into_iter() + .map(|(page, directory)| (page, directory.proof)) + .collect(), + self.source_roots, + )?; + self.work.delta_pages = plan.delta.len() as u64; + self.work.delta_bytes = self.payload_bytes; + self.work.reused_roots = plan.reused.len() as u64; + self.work.reused_closure_nodes_upper = self.reuse_upper.nodes as u64; + self.work.reused_closure_edges_upper = self.reuse_upper.edges as u64; + self.work.reused_closure_bytes_upper = self.reuse_upper.bytes; + self.work.reused_closure_entries_upper = self.reuse_upper.entries as u64; + self.work.plan_bytes = plan.encode()?.len() as u64; + Ok(PreparedRootedNativeMetadata { + plan, + payloads: self.payloads.into_values().collect(), + work: self.work, + }) + } +} + +async fn resolve_scope_oid( + handler: &T, + root: &Tree, + scope: &str, + work: &mut RootedProjectionWork, +) -> Result { + if scope == "/" { + return Ok(root.id); + } + let mut current = Cow::Borrowed(root); + let mut components = scope[1..].split('/').peekable(); + while let Some(component) = components.next() { + let position = current + .tree_items + .iter() + .position(|item| item.name == component); + work.scope_path_entries_examined += + position.map_or(current.tree_items.len(), |index| index + 1) as u64; + let item = position + .map(|index| ¤t.tree_items[index]) + .ok_or_else(|| { + SnapshotError::new( + SnapshotErrorCode::PathNotFound, + "name absent in enumerated parent directory", + ) + })?; + let kind = FsKind::from_git_mode(item.mode).ok_or_else(|| { + SnapshotError::new( + SnapshotErrorCode::UnsupportedEntry, + "gitlink entries are not supported in this profile", + ) + })?; + if kind != FsKind::Directory { + return Err(SnapshotError::new( + SnapshotErrorCode::NotDirectory, + "rooted metadata scope is not a directory", + )); + } + let expected = item.id; + if expected.kind() != root.id.kind() { + return Err(integrity("scope tree crossed its source hash kind")); + } + if components.peek().is_none() { + return Ok(expected); + } + work.tree_fetches += 1; + let fetched = handler + .get_tree_by_hash(&expected.to_string()) + .await + .map_err(source_error)?; + if fetched.id != expected { + return Err(integrity("fetched scope ancestor identity mismatch")); + } + current = Cow::Owned(fetched); + } + Err(integrity("rooted metadata scope walk has no target")) +} + +async fn project_directory( + handler: &T, + reuse: &R, + storage: &Storage, + oid: ObjectHash, + loaded: Option<&Tree>, + path: &str, + state: &mut ProjectionState, +) -> Result { + validate_scope_relative_path(path)?; + let tagged = oid.to_tagged_string(); + if let Some(summary) = state.directories.get(&tagged).copied() { + summary.bounds.validate_at(path)?; + state.work.directory_memo_hits += 1; + return Ok(summary); + } + state.require_tree(&tagged)?; + if !state.active_trees.insert(tagged.clone()) { + return Err(integrity("native source trees contain a cycle")); + } + state.work.reuse_lookups += 1; + if let Some(hint) = reuse.lookup_reuse(&tagged, &state.identity).await? { + let summary = state.record_reuse(&tagged, hint, path)?; + state.active_trees.remove(&tagged); + state.directories.insert(tagged, summary); + return Ok(summary); + } + let fetched; + let tree = if let Some(tree) = loaded { + tree + } else { + state.work.tree_fetches += 1; + fetched = handler + .get_tree_by_hash(&oid.to_string()) + .await + .map_err(source_error)?; + &fetched + }; + if tree.id != oid + || tree + .tree_items + .iter() + .any(|item| item.id.kind() != oid.kind()) + { + return Err(integrity( + "fetched source directory identity or child hash kind mismatch", + )); + } + state.source_entries = state + .source_entries + .checked_add(tree.tree_items.len()) + .filter(|value| *value <= MetadataDagLimits::default().entries) + .ok_or_else(budget_limit)?; + let direct = direct_entries(tree)?; + state.admit_entries(&direct)?; + state.work.directories_scanned += 1; + state.work.direct_entries_scanned += direct.len() as u64; + let requested: Vec<_> = direct + .iter() + .filter(|(_, kind, oid)| *kind != FsKind::Directory && !state.blob_facts.contains_key(oid)) + .map(|(_, _, oid)| oid.clone()) + .collect::>() + .into_iter() + .collect(); + for batch in requested.chunks(64) { + state.work.verified_blob_queries += 1; + let facts = storage + .mono_storage() + .get_verified_blobs(batch.to_vec()) + .await + .map_err(source_error)?; + state.work.verified_blob_facts_loaded += facts.len() as u64; + for (oid, fact) in facts { + state.blob_facts.insert(oid, verified_fact(&fact)?); + } + } + let mut pending_facts = Vec::new(); + let mut bounds = PathBounds::default(); + let mut entries = Vec::with_capacity(direct.len()); + for (name, kind, raw_oid) in direct { + let child_path = if path == "/" { + format!("/{name}") + } else { + format!("{path}/{name}") + }; + validate_scope_relative_path(&child_path)?; + match kind { + FsKind::Directory => { + let child_oid = ObjectHash::from_hex_for_kind(oid.kind(), &raw_oid) + .map_err(|error| integrity(&error.to_string()))?; + let child = Box::pin(project_directory( + handler, + reuse, + storage, + child_oid, + None, + &child_path, + state, + )) + .await?; + bounds.include(&name, child.bounds)?; + entries.push(Entry::dir(name.as_bytes(), child.page_id)); + } + FsKind::Regular | FsKind::Executable | FsKind::Symlink => { + bounds.include(&name, PathBounds::default())?; + let fact = if let Some(fact) = state.blob_facts.get(&raw_oid).copied() { + fact + } else { + let fact = stream_blob_fact_with_resources( + || handler.get_raw_blob_stream_with_meta(&raw_oid), + projection_budget(), + &mut state.work, + ) + .await?; + state.blob_facts.insert(raw_oid.clone(), fact); + pending_facts.push((raw_oid.clone(), fact)); + if pending_facts.len() == 64 { + persist_facts(storage, &pending_facts, &mut state.work).await?; + pending_facts.clear(); + } + fact + }; + let entry_kind = match kind { + FsKind::Regular => EntryKind::Regular, + FsKind::Executable => EntryKind::Executable, + FsKind::Symlink => EntryKind::Symlink, + FsKind::Directory => { + return Err(integrity("file projection received a directory kind")); + } + }; + entries.push(Entry::file( + entry_kind, + name.as_bytes(), + fact.size, + fact.digest, + )); + } + } + } + if !pending_facts.is_empty() { + persist_facts(storage, &pending_facts, &mut state.work).await?; + } + bounds.validate_at(path)?; + state.charge_codec(entries.len(), 0)?; + state.work.directory_root_builds += 1; + let root_bytes = Page::build(&entries).map_err(codec_error)?; + let page_id = state.collect_directory(&entries, root_bytes)?; + state.record_bounds(page_id, bounds)?; + state.source_roots.insert(tagged.clone(), page_id); + let summary = DirectorySummary { page_id, bounds }; + state.active_trees.remove(&tagged); + state.directories.insert(tagged, summary); + Ok(summary) +} + +async fn stream_blob_fact_with_resources( + open: F, + budget: &Arc, + work: &mut RootedProjectionWork, +) -> Result +where + F: FnOnce() -> Fut, + Fut: std::future::Future>, +{ + // Credit bounds consumer-visible processing, not the backend's backing + // allocation or work surviving cancellation. No source bytes accumulate. + let _lease = budget.reserve(RANGE_WORK_BYTES)?; + work.blob_fetches = work.blob_fetches.checked_add(1).ok_or_else(budget_limit)?; + let (mut input, meta) = tokio::time::timeout(SOURCE_PROGRESS_TIMEOUT, open()) + .await + .map_err(|_| source_progress_timeout())? + .map_err(source_error)?; + let size = u64::try_from(meta.size) + .ok() + .filter(|size| *size <= MAX_VERIFIED_FILE_BYTES) + .ok_or_else(|| integrity("raw file exceeds the native verified-object size profile"))?; + let mut digest = Sha256::new(); + let mut received = 0u64; + let mut since_yield = 0usize; + let mut empty_parts = 0usize; + let mut progress_deadline = tokio::time::Instant::now() + SOURCE_PROGRESS_TIMEOUT; + loop { + if tokio::time::Instant::now() >= progress_deadline { + return Err(source_progress_timeout()); + } + let part = tokio::time::timeout_at(progress_deadline, input.next()) + .await + .map_err(|_| source_progress_timeout())?; + let Some(part) = part else { break }; + let bytes = part.map_err(|error| { + tracing::warn!(error = %error, "rooted metadata raw source stream failed"); + SnapshotError::new( + SnapshotErrorCode::ObjectUnavailable, + "raw source stream failed", + ) + })?; + if bytes.len() as u64 > size - received { + return Err(integrity( + "raw source length disagrees with its physical size", + )); + } + if bytes.len() > RANGE_WORK_BYTES { + return Err(SnapshotError::new( + SnapshotErrorCode::TemporaryUnavailable, + "raw source producer item exceeds the consumer processing limit", + )); + } + digest.update(&bytes); + received += bytes.len() as u64; + since_yield += bytes.len(); + if bytes.is_empty() { + empty_parts += 1; + } else { + empty_parts = 0; + progress_deadline = tokio::time::Instant::now() + SOURCE_PROGRESS_TIMEOUT; + } + if since_yield >= RANGE_WORK_BYTES || empty_parts == 32 { + tokio::task::yield_now().await; + since_yield = 0; + empty_parts = 0; + } + } + if received != size { + return Err(integrity( + "raw source length disagrees with its physical size", + )); + } + let fetched = work + .raw_bytes_fetched + .checked_add(size) + .ok_or_else(budget_limit)?; + let hashed = work + .raw_bytes_hashed + .checked_add(size) + .ok_or_else(budget_limit)?; + work.raw_bytes_fetched = fetched; + work.raw_bytes_hashed = hashed; + Ok(BlobFact { + size, + digest: digest.finalize().into(), + }) +} + +fn source_progress_timeout() -> SnapshotError { + SnapshotError::new( + SnapshotErrorCode::TemporaryUnavailable, + "rooted metadata raw source made no progress", + ) +} + +fn verified_fact( + fact: &crate::callisto::mst2_verified_object::Model, +) -> Result { + if fact.state != "VERIFIED" + || fact.verification_version != MST2_VERIFICATION_VERSION + || !(0..=MAX_VERIFIED_FILE_BYTES as i64).contains(&fact.size) + { + return Err(integrity( + "native projection received an invalid current verified blob fact", + )); + } + Ok(BlobFact { + size: u64::try_from(fact.size).map_err(|_| integrity("verified blob size is invalid"))?, + digest: fact + .raw_sha256 + .as_slice() + .try_into() + .map_err(|_| integrity("verified blob digest is invalid"))?, + }) +} + +async fn persist_facts( + storage: &Storage, + facts: &[(String, BlobFact)], + work: &mut RootedProjectionWork, +) -> Result<(), SnapshotError> { + let rows = facts + .iter() + .map( + |(oid, fact)| crate::callisto::mst2_verified_object::ActiveModel { + id: sea_orm::ActiveValue::NotSet, + storage_domain: Set("git".into()), + git_oid: Set(oid.clone()), + object_kind: Set("blob".into()), + raw_sha256: Set(fact.digest.to_vec()), + size: Set(fact.size as i64), + verification_version: Set(MST2_VERIFICATION_VERSION), + state: Set("VERIFIED".into()), + created_at: Set(chrono::Utc::now().fixed_offset()), + }, + ) + .collect(); + work.verified_blob_persistence_batches += 1; + storage + .mono_storage() + .insert_verified_blobs(rows) + .await + .map_err(source_error)?; + work.verified_blob_queries += 1; + let stored = storage + .mono_storage() + .get_verified_blobs(facts.iter().map(|(oid, _)| oid.clone()).collect()) + .await + .map_err(source_error)?; + work.verified_blob_facts_loaded += stored.len() as u64; + for (oid, expected) in facts { + let actual = stored + .get(oid) + .ok_or_else(|| integrity("verified blob persistence lost its current fact"))?; + if verified_fact(actual)? != *expected { + return Err(integrity( + "verified blob persistence conflicts with the hashed raw source", + )); + } + } + Ok(()) +} + +fn source_error(error: MegaError) -> SnapshotError { + let code = match error { + MegaError::ObjStorageNotFound(_) => SnapshotErrorCode::ObjectUnavailable, + MegaError::ObjStorageInconsistent(_) => SnapshotErrorCode::IntegrityError, + _ => SnapshotErrorCode::Internal, + }; + tracing::warn!(error = %error, "rooted native metadata source operation failed"); + SnapshotError::new(code, "rooted native metadata source operation failed") +} + +fn codec_error(error: mst2_codec::CodecError) -> SnapshotError { + integrity(&error.to_string()) +} +fn integrity(message: &str) -> SnapshotError { + SnapshotError::new(SnapshotErrorCode::IntegrityError, message) +} +fn budget_limit() -> SnapshotError { + SnapshotError::new( + SnapshotErrorCode::LimitExceeded, + "rooted metadata projection exceeds its fixed budget", + ) +} +fn path_limit() -> SnapshotError { + SnapshotError::new( + SnapshotErrorCode::ScopeInvalid, + "rooted metadata subtree exceeds its full-path budget", + ) +} + +#[cfg(test)] +mod tests { + use std::{ + pin::Pin, + sync::atomic::{AtomicUsize, Ordering}, + task::{Context, Poll}, + }; + + use bytes::Bytes; + use git_internal::hash::HashKind; + use tokio::sync::Notify; + use uuid::Uuid; + + use super::*; + + struct FactInput { + parts: VecDeque>, + polls: Arc, + drops: Arc, + hold_eof: Option>, + } + + impl futures::Stream for FactInput { + type Item = std::io::Result; + + fn poll_next(self: Pin<&mut Self>, _cx: &mut Context<'_>) -> Poll> { + let input = self.get_mut(); + input.polls.fetch_add(1, Ordering::SeqCst); + if let Some(part) = input.parts.pop_front() { + Poll::Ready(Some(part)) + } else if let Some(entered) = &input.hold_eof { + entered.notify_one(); + Poll::Pending + } else { + Poll::Ready(None) + } + } + } + + impl Drop for FactInput { + fn drop(&mut self) { + self.drops.fetch_add(1, Ordering::SeqCst); + } + } + + fn fact_input( + parts: Vec>, + hold_eof: Option>, + ) -> (ObjectByteStream, Arc, Arc) { + let polls = Arc::new(AtomicUsize::new(0)); + let drops = Arc::new(AtomicUsize::new(0)); + ( + Box::pin(FactInput { + parts: parts.into(), + polls: polls.clone(), + drops: drops.clone(), + hold_eof, + }), + polls, + drops, + ) + } + + #[tokio::test] + async fn streamed_cold_fact_keeps_large_memory_source_and_empty_raw_costs_exact() { + use crate::orbit_api::object_storage::{ObjectKey, ObjectNamespace}; + + for size in [0, RANGE_WORK_BYTES + 113] { + let backend = crate::jupiter::storage::object_storage::mock_object_storage(); + let key = ObjectKey { + namespace: ObjectNamespace::Git, + key: "ab".repeat(20), + }; + let mut raw = vec![0xff; size]; + if !raw.is_empty() { + raw[..11].copy_from_slice(b"blob 3\0abc\0"); + } + let expected: [u8; 32] = Sha256::digest(&raw).into(); + backend + .inner + .put_stream( + &key, + Box::pin(futures::stream::iter([Ok(Bytes::from(raw))])), + ObjectMeta::default(), + ) + .await + .unwrap(); + let budget = MemoryBudget::new(RANGE_WORK_BYTES); + let opens = AtomicUsize::new(0); + let mut work = RootedProjectionWork::default(); + let fact = stream_blob_fact_with_resources( + || async { + opens.fetch_add(1, Ordering::SeqCst); + backend + .inner + .get_stream(&key) + .await + .map_err(MegaError::from) + }, + &budget, + &mut work, + ) + .await + .unwrap(); + assert_eq!( + fact, + BlobFact { + size: size as u64, + digest: expected + } + ); + assert_eq!(opens.load(Ordering::SeqCst), 1); + assert_eq!(work.blob_fetches, 1); + assert_eq!(work.raw_bytes_fetched, size as u64); + assert_eq!(work.raw_bytes_hashed, size as u64); + assert_eq!(work.verified_blob_persistence_batches, 0); + assert_eq!(budget.used(), 0); + } + } + + #[tokio::test] + async fn streamed_cold_fact_rejects_physical_mismatch_and_late_error_before_fact_return() { + let raw = Bytes::from_static(b"raw"); + for (size, parts, code, polls_expected) in [ + ( + -1, + vec![Ok(raw.clone())], + SnapshotErrorCode::IntegrityError, + 0, + ), + ( + MAX_VERIFIED_FILE_BYTES as i64 + 1, + vec![Ok(raw.clone())], + SnapshotErrorCode::IntegrityError, + 0, + ), + ( + MAX_VERIFIED_FILE_BYTES as i64, + vec![], + SnapshotErrorCode::IntegrityError, + 1, + ), + ( + RANGE_WORK_BYTES as i64 + 1, + vec![Ok(Bytes::from(vec![7; RANGE_WORK_BYTES + 1]))], + SnapshotErrorCode::TemporaryUnavailable, + 1, + ), + ( + 2, + vec![Ok(raw.clone())], + SnapshotErrorCode::IntegrityError, + 1, + ), + ( + 4, + vec![Ok(raw.clone())], + SnapshotErrorCode::IntegrityError, + 2, + ), + ( + 3, + vec![Ok(raw.clone()), Ok(Bytes::from_static(b"x"))], + SnapshotErrorCode::IntegrityError, + 2, + ), + ( + 3, + vec![Ok(raw), Err(std::io::Error::other("late source error"))], + SnapshotErrorCode::ObjectUnavailable, + 2, + ), + ( + 0, + vec![Err(std::io::Error::other("empty source error"))], + SnapshotErrorCode::ObjectUnavailable, + 1, + ), + ] { + let (input, polls, drops) = fact_input(parts, None); + let budget = MemoryBudget::new(RANGE_WORK_BYTES); + let mut work = RootedProjectionWork::default(); + let error = stream_blob_fact_with_resources( + || async move { + Ok(( + input, + ObjectMeta { + size, + ..Default::default() + }, + )) + }, + &budget, + &mut work, + ) + .await + .unwrap_err(); + assert_eq!(error.code, code); + assert_eq!(polls.load(Ordering::SeqCst), polls_expected); + assert_eq!(drops.load(Ordering::SeqCst), 1); + assert_eq!(work.blob_fetches, 1); + assert_eq!(work.raw_bytes_fetched, 0); + assert_eq!(work.raw_bytes_hashed, 0); + assert_eq!(budget.used(), 0); + } + } + + #[tokio::test] + async fn streamed_cold_fact_cancellation_at_exact_size_drops_source_and_refunds_credit() { + let entered = Arc::new(Notify::new()); + let (input, polls, drops) = + fact_input(vec![Ok(Bytes::from_static(b"raw"))], Some(entered.clone())); + let budget = MemoryBudget::new(RANGE_WORK_BYTES); + let task_budget = budget.clone(); + let task = tokio::spawn(async move { + let mut work = RootedProjectionWork::default(); + stream_blob_fact_with_resources( + || async move { + Ok(( + input, + ObjectMeta { + size: 3, + ..Default::default() + }, + )) + }, + &task_budget, + &mut work, + ) + .await + }); + tokio::time::timeout(Duration::from_secs(2), entered.notified()) + .await + .unwrap(); + assert_eq!(polls.load(Ordering::SeqCst), 2); + assert_eq!(budget.used(), RANGE_WORK_BYTES); + assert_eq!(drops.load(Ordering::SeqCst), 0); + task.abort(); + assert!(task.await.unwrap_err().is_cancelled()); + assert_eq!(drops.load(Ordering::SeqCst), 1); + assert_eq!(budget.used(), 0); + } + + #[tokio::test] + async fn streamed_cold_fact_reserves_fixed_credit_before_open_and_preserves_open_errors() { + let budget = MemoryBudget::new(RANGE_WORK_BYTES); + let occupied = budget.reserve(RANGE_WORK_BYTES).unwrap(); + let mut work = RootedProjectionWork::default(); + let opens = AtomicUsize::new(0); + let error = stream_blob_fact_with_resources( + || async { + opens.fetch_add(1, Ordering::SeqCst); + Err(MegaError::Other("unexpected open".into())) + }, + &budget, + &mut work, + ) + .await + .unwrap_err(); + assert_eq!(error.code, SnapshotErrorCode::TemporaryUnavailable); + assert_eq!(opens.load(Ordering::SeqCst), 0); + assert_eq!(work, RootedProjectionWork::default()); + drop(occupied); + for (error, code) in [ + ( + MegaError::ObjStorageNotFound("missing".into()), + SnapshotErrorCode::ObjectUnavailable, + ), + ( + MegaError::ObjStorageInconsistent("inconsistent".into()), + SnapshotErrorCode::IntegrityError, + ), + ( + MegaError::Other("transport".into()), + SnapshotErrorCode::Internal, + ), + ] { + let mut work = RootedProjectionWork::default(); + let got = stream_blob_fact_with_resources(|| async { Err(error) }, &budget, &mut work) + .await + .unwrap_err(); + assert_eq!(got.code, code); + assert_eq!(work.blob_fetches, 1); + assert_eq!(work.raw_bytes_fetched, 0); + assert_eq!(work.raw_bytes_hashed, 0); + assert_eq!(budget.used(), 0); + } + } + + fn state() -> ProjectionState { + ProjectionState::new(MetadataInstallIdentity { + source_domain: "native-git".into(), + tagged_root_tree_oid: ObjectHash::from_hex_for_kind(HashKind::Sha1, &"a".repeat(40)) + .unwrap() + .to_tagged_string(), + scope: "/".into(), + schema_version: SCHEMA_VERSION, + metadata_codec: METADATA_CODEC, + materialization_policy: MATERIALIZATION_POLICY_GIT_RAW_V1, + fs_semantics: FS_SEMANTICS_LINUX_CODE_V1, + access_projection: ACCESS_PROJECTION_EXACT_FULL, + verification_revision: MST2_VERIFICATION_VERSION, + projection_revision: NATIVE_PROJECTION_REVISION, + }) + } + + fn reusable(page: MetadataPageId) -> CertifiedReusableDirectory { + CertifiedReusableDirectory { + page_id: page, + proof: RootedReuseRoot { + generation: 1, + attestation_id: Uuid::from_u128(1), + attestation_digest: [1; 32], + certificate_digest: [2; 32], + }, + relative_path_bytes: 0, + relative_components: 0, + closure_nodes_upper: 1, + closure_edges_upper: 0, + closure_bytes_upper: HEADER_LEN as u64, + closure_entries_upper: 0, + } + } + + #[test] + fn reused_path_bounds_recheck_moved_aliases_and_checked_overflow() { + let bounds = PathBounds { + bytes: 4093, + components: 255, + }; + assert!(bounds.validate_at("/").is_ok()); + assert!(bounds.validate_at("/a").is_ok()); + assert_eq!( + bounds.validate_at("/a/b").unwrap_err().code, + SnapshotErrorCode::ScopeInvalid + ); + assert!( + PathBounds { + bytes: usize::MAX, + components: 0 + } + .validate_at("/a") + .is_err() + ); + let mut parent = PathBounds::default(); + parent + .include( + "a", + PathBounds { + bytes: 3, + components: 1, + }, + ) + .unwrap(); + parent.include("longer", PathBounds::default()).unwrap(); + assert_eq!( + parent, + PathBounds { + bytes: 7, + components: 2 + } + ); + assert!( + parent + .include( + "a", + PathBounds { + bytes: usize::MAX, + components: 0 + } + ) + .is_err() + ); + } + + #[test] + fn canonical_radix_payloads_and_shared_reuse_edges_form_a_delta_only_plan() { + let empty = Page::build(&[]).unwrap(); + let reused_page = page_id(&empty); + let mut state = state(); + state + .record_reuse( + &format!("sha1:{}", "b".repeat(40)), + reusable(reused_page), + "/left", + ) + .unwrap(); + let mut entries: Vec<_> = (0..260) + .map(|index| { + Entry::file( + EntryKind::Regular, + format!("file-{index:03}").as_bytes(), + index, + [3; 32], + ) + }) + .collect(); + entries.push(Entry::dir(b"left", reused_page)); + entries.push(Entry::dir(b"right", reused_page)); + entries.sort_by(|left, right| left.name.cmp(&right.name)); + let bytes = Page::build(&entries).unwrap(); + let root = state.collect_directory(&entries, bytes).unwrap(); + state + .source_roots + .insert(state.identity.tagged_root_tree_oid.clone(), root); + let prepared = state.finish(root).unwrap(); + assert!(prepared.payloads.len() > 1); + assert_eq!(prepared.plan.reused.len(), 1); + assert!(!prepared.plan.delta.contains_key(&reused_page)); + assert_eq!(prepared.work.tree_fetches, 0); + assert_eq!(prepared.work.blob_fetches, 0); + assert_eq!( + prepared.plan.child_first_delta().unwrap().last(), + Some(&root) + ); + let mut decoded_edges = BTreeSet::new(); + for payload in &prepared.payloads { + assert_eq!(page_id(&payload.bytes), payload.id); + match Page::decode(&payload.bytes).unwrap().0 { + Page::Leaf { entries } => { + for entry in entries.iter().filter(|entry| entry.is_dir()) { + decoded_edges.insert((payload.id, entry.child_root)); + } + } + Page::Branch { + terminal, children, .. + } => { + if let Some(entry) = terminal.filter(Entry::is_dir) { + decoded_edges.insert((payload.id, entry.child_root)); + } + for child in children { + decoded_edges.insert((payload.id, child.child_page_id)); + } + } + } + } + assert_eq!(decoded_edges, prepared.plan.edges); + assert_eq!( + RootedMetadataInstallPlan::decode( + &prepared.plan.encode().unwrap(), + &prepared.plan.digest().unwrap() + ) + .unwrap(), + prepared.plan + ); + } + + #[test] + fn reuse_budget_is_deduplicated_conservative_and_never_expands_descendants() { + let empty = Page::build(&[]).unwrap(); + let page = page_id(&empty); + let hint = reusable(page); + let mut state = state(); + state + .record_reuse(&format!("sha1:{}", "b".repeat(40)), hint.clone(), "/one") + .unwrap(); + let mut alias = hint.clone(); + alias.proof.attestation_id = Uuid::from_u128(2); + alias.proof.attestation_digest = [9; 32]; + state + .record_reuse(&format!("sha1:{}", "c".repeat(40)), alias, "/two") + .unwrap(); + assert_eq!(state.reuse_upper.nodes, 1); + assert_eq!(state.reuse_upper.bytes, HEADER_LEN as u64); + assert_eq!(state.reused.get(&page).unwrap().proof, hint.proof); + assert!(state.payloads.is_empty()); + assert!(state.edges.is_empty()); + let mut different_lifetime = hint.clone(); + different_lifetime.proof.generation = 2; + assert!( + state + .record_reuse( + &format!("sha1:{}", "f".repeat(40)), + different_lifetime, + "/lifetime" + ) + .is_err() + ); + let mut bad = hint; + bad.relative_path_bytes = 1; + assert!( + state + .record_reuse(&format!("sha1:{}", "d".repeat(40)), bad, "/three") + .is_err() + ); + let mut huge = reusable([4; 32]); + huge.closure_nodes_upper = MetadataDagLimits::default().nodes; + assert_eq!( + state + .record_reuse(&format!("sha1:{}", "e".repeat(40)), huge, "/four") + .unwrap_err() + .code, + SnapshotErrorCode::LimitExceeded + ); + } +} diff --git a/src/ceres/snapshot/runtime.rs b/src/ceres/snapshot/runtime.rs index 770c4725..b8b4bce3 100644 --- a/src/ceres/snapshot/runtime.rs +++ b/src/ceres/snapshot/runtime.rs @@ -11,7 +11,7 @@ use std::{ collections::HashMap, - sync::{Mutex, OnceLock}, + sync::{Mutex, MutexGuard, OnceLock}, time::{Duration, SystemTime, UNIX_EPOCH}, }; @@ -101,16 +101,16 @@ impl Mst2Runtime { st.1 } - /// Insert a context and create its first lease. The lease also becomes - /// a retention root covering the snapshot's derived metadata root - /// (spec 10 §5: active leases pin what a fixed view needs). + /// Acquire the complete in-memory retention group before exposing a + /// context or lease. These placeholder anchors do not yet represent + /// the complete durable page/chunk graph required by T06. pub fn insert_context( &self, built: BuiltDescriptor, commit_oid: &str, root_tree_oid: &str, lease_seconds: u64, - ) -> SnapshotContext { + ) -> Result { let lease_id = Uuid::new_v4().to_string(); let expires = now_unix() + lease_seconds.clamp(1, 3600); let snapshot_id = built.snapshot_id.clone(); @@ -122,23 +122,9 @@ impl Mst2Runtime { lease_expires_at_unix: expires, authorization_epoch: 1, }; - self.contexts.lock().unwrap().insert( - snapshot_id.clone(), - ContextData { - built: built.clone(), - commit_oid: commit_oid.to_string(), - root_tree_oid: root_tree_oid.to_string(), - }, - ); - self.leases.lock().unwrap().insert( - lease_id.clone(), - LeaseRec { - snapshot_id: snapshot_id.clone(), - expires_at_unix: expires, - }, - ); - // Retention root for the lease: the metadata-root page node and the - // snapshot's chunk projections stay reachable while it is active. + // Serialize registration with release/expiry/renewal. No path holds + // contexts while acquiring leases, so this lock order cannot cycle. + let mut leases = self.leases.lock().unwrap(); let root = crate::ceres::snapshot::retention::RetentionRoot::Lease(lease_id.clone()); let page_node = crate::ceres::snapshot::retention::RetentionNode { id: format!("page:{}", built.metadata_root), @@ -152,15 +138,24 @@ impl Mst2Runtime { state: crate::ceres::snapshot::retention::NodeState::Live, bytes: 0, }; - if let Err(e) = self - .retention - .pin_root(&root, &[page_node, projection_node]) - { - // Coverage failure is surfaced, not swallowed: a lease that - // cannot pin its view must not pretend to. - tracing::warn!(error = %e, "retention pin for new lease failed"); - } - ctx + self.retention + .pin_root(&root, &[page_node, projection_node])?; + self.contexts.lock().unwrap().insert( + snapshot_id.clone(), + ContextData { + built, + commit_oid: commit_oid.to_string(), + root_tree_oid: root_tree_oid.to_string(), + }, + ); + leases.insert( + lease_id, + LeaseRec { + snapshot_id, + expires_at_unix: expires, + }, + ); + Ok(ctx) } /// Look up a snapshot context by snapshot_id. It is servable while at @@ -194,7 +189,7 @@ impl Mst2Runtime { /// as it goes. fn active_lease(&self, snapshot_id: &str) -> Result<(String, u64), SnapshotError> { let mut leases = self.leases.lock().unwrap(); - leases.retain(|_, r| r.expires_at_unix >= now_unix()); + self.prune_expired_leases(&mut leases, now_unix()); leases .iter() .find(|(_, r)| r.snapshot_id == snapshot_id) @@ -207,6 +202,22 @@ impl Mst2Runtime { }) } + /// The leases lock serializes expiry with renewal and registration; + /// releasing one expired lease never removes another lease's coverage. + /// The current in-memory store's release operation is infallible. + fn prune_expired_leases(&self, leases: &mut HashMap, now: u64) { + leases.retain(|lease_id, rec| { + if rec.expires_at_unix >= now { + return true; + } + self.retention + .release(&crate::ceres::snapshot::retention::RetentionRoot::Lease( + lease_id.clone(), + )); + false + }); + } + /// Validate the lease a request presented (spec 04 §1: identity lives in /// the `X-Mega-Snapshot-Lease` header, and a read path must check it — /// knowing the snapshot id alone is not a capability). Any lease that is @@ -216,8 +227,8 @@ impl Mst2Runtime { /// existed is answered identically: probing with lease ids must not /// distinguish released from forged. /// - /// Expired-row pruning happens in `active_lease`, not here: this runs on - /// every authenticated read and must stay a single key lookup. + /// Expired-row pruning happens in `active_lease` and collection: this + /// runs on every authenticated read and stays a single key lookup. pub fn validate_lease(&self, snapshot_id: &str, lease_id: &str) -> Result<(), SnapshotError> { let leases = self.leases.lock().unwrap(); match leases.get(lease_id) { @@ -237,8 +248,24 @@ impl Mst2Runtime { lease_id: &str, lease_seconds: u64, ) -> Result { - let now = now_unix(); - let mut leases = self.leases.lock().unwrap(); + Self::renew_lease_with_clock( + lease_id, + lease_seconds, + || self.leases.lock().unwrap(), + now_unix, + ) + } + + /// Sample the clock only after acquiring the lease lock. Separate + /// sources let tests force a deadline crossing during lock contention. + fn renew_lease_with_clock<'a>( + lease_id: &str, + lease_seconds: u64, + acquire_leases: impl FnOnce() -> MutexGuard<'a, HashMap>, + clock: impl FnOnce() -> u64, + ) -> Result { + let mut leases = acquire_leases(); + let now = clock(); let Some(rec) = leases.get_mut(lease_id) else { return Err(SnapshotError::new( SnapshotErrorCode::LeaseUnknown, @@ -268,7 +295,8 @@ impl Mst2Runtime { /// only drops the retention root so a later collection pass may /// reclaim derived pages no other lease/pin covers. pub fn release_lease(&self, lease_id: &str) -> bool { - let removed = self.leases.lock().unwrap().remove(lease_id).is_some(); + let mut leases = self.leases.lock().unwrap(); + let removed = leases.remove(lease_id).is_some(); self.retention .release(&crate::ceres::snapshot::retention::RetentionRoot::Lease( lease_id.to_string(), @@ -282,6 +310,7 @@ impl Mst2Runtime { pub fn collect_retention( &self, ) -> Result { + self.prune_expired_leases(&mut self.leases.lock().unwrap(), now_unix()); self.retention .collect(&crate::ceres::snapshot::retention::NoopReaper) } @@ -336,20 +365,24 @@ pub fn rfc3339(unix: u64) -> String { #[cfg(test)] mod tests { - use super::*; + use std::{ + sync::{ + Arc, TryLockError, + atomic::{AtomicU64, Ordering}, + mpsc::sync_channel, + }, + thread, + }; - #[test] - fn rfc3339_known_values() { - assert_eq!(rfc3339(0), "1970-01-01T00:00:00Z"); - assert_eq!(rfc3339(1_789_525_722), "2026-09-16T02:28:42Z"); - } + use mst2_codec::descriptor::ServingDescriptor; - #[tokio::test] - async fn lease_lifecycle_drives_retention() { - use crate::ceres::snapshot::retention::{ - NodeState, RetainedKind, RetentionRoot, RetentionStore as _, - }; - let r = Mst2Runtime { + use super::*; + use crate::ceres::snapshot::retention::{ + NodeState, RetainedKind, RetentionRoot, RetentionStore as _, + }; + + fn isolated_runtime() -> Mst2Runtime { + Mst2Runtime { hmac_key: blake3_key(), contexts: Mutex::new(HashMap::new()), leases: Mutex::new(HashMap::new()), @@ -357,9 +390,324 @@ mod tests { crate::ceres::snapshot::retention::mem::InMemoryRetentionStore::default(), ), tip_state: Mutex::new((String::new(), 0u64)), + } + } + + fn descriptor(root_byte: u8) -> BuiltDescriptor { + let descriptor = ServingDescriptor { + instance_uuid: [0x11; 16], + namespace_view_id: [0x22; 32], + scope: "/".into(), + metadata_root: [root_byte; 32], + }; + BuiltDescriptor { + instance_id: Uuid::from_bytes(descriptor.instance_uuid).to_string(), + snapshot_id: format!( + "sha256:{}", + crate::ceres::snapshot::view::hex(&descriptor.snapshot_id().unwrap()) + ), + metadata_root: format!( + "sha256:{}", + crate::ceres::snapshot::view::hex(&descriptor.metadata_root) + ), + descriptor, + } + } + + #[test] + fn pin_failure_exposes_no_context_lease_or_partial_graph_and_can_retry() { + let r = isolated_runtime(); + let built = descriptor(3); + r.retention.store().fail_next_retain_for_test(); + assert_eq!( + r.insert_context(built.clone(), "commit", "tree", 60) + .unwrap_err() + .code, + SnapshotErrorCode::Internal + ); + assert!(r.contexts.lock().unwrap().is_empty()); + assert!(r.leases.lock().unwrap().is_empty()); + assert!(r.retention.store().all_live().is_empty()); + assert_eq!( + r.context(&built.snapshot_id).unwrap_err().code, + SnapshotErrorCode::SnapshotGone + ); + + let ctx = r + .insert_context(built.clone(), "commit", "tree", 60) + .unwrap(); + r.validate_lease(&built.snapshot_id, &ctx.lease_id).unwrap(); + assert!( + r.retention + .is_retained(&format!("page:{}", built.metadata_root)) + ); + assert!(r.retention.is_retained("projection:commit")); + } + + #[test] + fn failed_resolve_preserves_an_existing_active_snapshot() { + let r = isolated_runtime(); + let old = descriptor(3); + let active = r + .insert_context(old.clone(), "old-commit", "old-tree", 60) + .unwrap(); + let new = descriptor(4); + r.retention.store().fail_next_retain_for_test(); + assert!( + r.insert_context(new.clone(), "new-commit", "new-tree", 60) + .is_err() + ); + assert_eq!(r.contexts.lock().unwrap().len(), 1); + assert_eq!(r.leases.lock().unwrap().len(), 1); + assert_eq!( + r.context(&old.snapshot_id).unwrap().lease_id, + active.lease_id + ); + assert_eq!( + r.context(&new.snapshot_id).unwrap_err().code, + SnapshotErrorCode::SnapshotGone + ); + assert!( + r.retention + .store() + .node(&format!("page:{}", new.metadata_root)) + .is_none() + ); + assert!(r.retention.store().node("projection:new-commit").is_none()); + } + + #[test] + fn resolve_cannot_create_a_lease_for_a_deleting_group() { + let r = isolated_runtime(); + let built = descriptor(3); + let ctx = r + .insert_context(built.clone(), "commit", "tree", 60) + .unwrap(); + assert!(r.release_lease(&ctx.lease_id)); + assert!(r.collect_retention().unwrap().reaped.is_empty()); + assert_eq!( + r.insert_context(built.clone(), "commit", "tree", 60) + .unwrap_err() + .code, + SnapshotErrorCode::ObjectUnavailable + ); + assert!(r.leases.lock().unwrap().is_empty()); + assert!( + r.retention.store().all_live().is_empty(), + "no new lease anchor leaked" + ); + assert_eq!( + r.context(&built.snapshot_id).unwrap_err().code, + SnapshotErrorCode::LeaseExpired + ); + assert_eq!( + r.retention + .store() + .node(&format!("page:{}", built.metadata_root)) + .unwrap() + .state, + NodeState::Deleting + ); + } + + #[test] + fn context_expiry_releases_only_the_expired_lease_on_a_shared_snapshot() { + let r = isolated_runtime(); + let built = descriptor(3); + let expired = r + .insert_context(built.clone(), "commit", "tree", 60) + .unwrap(); + let active = r + .insert_context(built.clone(), "commit", "tree", 60) + .unwrap(); + r.leases + .lock() + .unwrap() + .get_mut(&expired.lease_id) + .unwrap() + .expires_at_unix = 0; + + assert_eq!( + r.context(&built.snapshot_id).unwrap().lease_id, + active.lease_id + ); + assert!(!r.leases.lock().unwrap().contains_key(&expired.lease_id)); + assert!( + !r.retention + .store() + .root_covers(&format!("lease:{}", expired.lease_id)) + ); + assert!( + r.retention + .store() + .root_covers(&format!("lease:{}", active.lease_id)) + ); + assert!( + r.retention + .is_retained(&format!("page:{}", built.metadata_root)) + ); + r.validate_lease(&built.snapshot_id, &active.lease_id) + .unwrap(); + assert_eq!( + r.validate_lease(&built.snapshot_id, &expired.lease_id) + .unwrap_err() + .code, + SnapshotErrorCode::LeaseExpired + ); + assert_eq!( + r.collect_retention().unwrap().unreachable, + vec![format!("lease:{}", expired.lease_id)] + ); + assert_eq!( + r.context(&built.snapshot_id).unwrap().lease_id, + active.lease_id + ); + + assert!(r.release_lease(&active.lease_id)); + assert!( + !r.retention + .store() + .root_covers(&format!("page:{}", built.metadata_root)) + ); + assert!(!r.retention.store().root_covers("projection:commit")); + } + + #[test] + fn collection_prunes_expired_roots_without_a_context_read() { + let r = isolated_runtime(); + let built = descriptor(3); + let ctx = r + .insert_context(built.clone(), "commit", "tree", 60) + .unwrap(); + r.leases + .lock() + .unwrap() + .get_mut(&ctx.lease_id) + .unwrap() + .expires_at_unix = 0; + let report = r.collect_retention().unwrap(); + assert_eq!(report.unreachable.len(), 3, "anchor, page and projection"); + assert!(report.reaped.is_empty(), "NoopReaper stays fail-closed"); + assert!(r.leases.lock().unwrap().is_empty()); + for id in [ + format!("lease:{}", ctx.lease_id), + format!("page:{}", built.metadata_root), + "projection:commit".into(), + ] { + assert!(report.unreachable.contains(&id)); + assert!(!r.retention.store().root_covers(&id)); + assert_eq!( + r.retention.store().node(&id).unwrap().state, + NodeState::Deleting + ); + } + } + + #[test] + fn renewal_winning_before_expiry_keeps_its_root_and_expiry_winning_prevents_renewal() { + let r = isolated_runtime(); + let built = descriptor(3); + let ctx = r + .insert_context(built.clone(), "commit", "tree", 60) + .unwrap(); + let renewed = r.renew_lease(&ctx.lease_id, 60).unwrap(); + { + let mut leases = r.leases.lock().unwrap(); + // Advance past the old deadline, before the renewed deadline. + r.prune_expired_leases(&mut leases, ctx.lease_expires_at_unix + 1); + } + r.validate_lease(&built.snapshot_id, &ctx.lease_id).unwrap(); + assert!( + r.retention + .store() + .root_covers(&format!("lease:{}", ctx.lease_id)) + ); + { + let mut leases = r.leases.lock().unwrap(); + r.prune_expired_leases(&mut leases, renewed.expires_at_unix + 1); + } + assert!( + !r.retention + .store() + .root_covers(&format!("lease:{}", ctx.lease_id)) + ); + assert_eq!( + r.renew_lease(&ctx.lease_id, 60).unwrap_err().code, + SnapshotErrorCode::LeaseUnknown + ); + assert_eq!( + r.context(&built.snapshot_id).unwrap_err().code, + SnapshotErrorCode::LeaseExpired + ); + } + + #[test] + fn renewal_waiting_for_the_lease_lock_cannot_use_a_pre_expiry_clock_sample() { + let r = Arc::new(isolated_runtime()); + let ctx = r + .insert_context(descriptor(3), "commit", "tree", 60) + .unwrap(); + let clock = Arc::new(AtomicU64::new(9)); + let (blocked_tx, blocked_rx) = sync_channel(1); + let mut leases = r.leases.lock().unwrap(); + leases.get_mut(&ctx.lease_id).unwrap().expires_at_unix = 10; + let renewal = { + let r = Arc::clone(&r); + let clock = Arc::clone(&clock); + let lease_id = ctx.lease_id.clone(); + thread::spawn(move || { + Mst2Runtime::renew_lease_with_clock( + &lease_id, + 60, + || { + // Prove contention before allowing the clock to cross + // the deadline. A pre-lock sample would observe 9. + assert!(matches!(r.leases.try_lock(), Err(TryLockError::WouldBlock))); + blocked_tx.send(()).unwrap(); + r.leases.lock().unwrap() + }, + || clock.load(Ordering::SeqCst), + ) + }) }; - // Retain the snapshot's derived nodes directly, covered by the lease - // root that insert_context created. + blocked_rx + .recv_timeout(Duration::from_secs(5)) + .expect("renewal did not contend for the lease lock"); + clock.store(11, Ordering::SeqCst); + drop(leases); + + assert_eq!( + renewal.join().unwrap().unwrap_err().code, + SnapshotErrorCode::LeaseExpired + ); + assert_eq!( + r.leases + .lock() + .unwrap() + .get(&ctx.lease_id) + .unwrap() + .expires_at_unix, + 10 + ); + assert_eq!(r.leases.lock().unwrap().len(), 1, "no replacement lease"); + assert!(r.collect_retention().unwrap().reaped.is_empty()); + assert!( + !r.retention + .store() + .root_covers(&format!("lease:{}", ctx.lease_id)) + ); + } + + #[test] + fn rfc3339_known_values() { + assert_eq!(rfc3339(0), "1970-01-01T00:00:00Z"); + assert_eq!(rfc3339(1_789_525_722), "2026-09-16T02:28:42Z"); + } + + #[test] + fn lease_lifecycle_drives_retention() { + let r = isolated_runtime(); + // Cover a derived node directly with a manual test lease root. let lease_root = RetentionRoot::Lease("gc-lease-1".to_string()); let page = crate::ceres::snapshot::retention::RetentionNode { id: "page:sha256:test".to_string(), @@ -382,7 +730,7 @@ mod tests { assert_eq!(report.unreachable, vec!["page:sha256:test".to_string()]); assert_eq!(report.reclaimed_bytes, 100); assert!(report.reaped.is_empty(), "fail-closed reaper"); - // The node is DELETING, not gone: a re-resolve re-lifts it. + // The node stays DELETING; a new acquire must not resurrect it. assert_eq!( r.retention.store().node("page:sha256:test").unwrap().state, NodeState::Deleting diff --git a/src/commands/config.rs b/src/commands/config.rs index a4e2763a..f08a287d 100644 --- a/src/commands/config.rs +++ b/src/commands/config.rs @@ -1091,6 +1091,117 @@ mod tests { assert_eq!(load_mode(&matches), LoadMode::RawSources); } + #[test] + fn config_validate_accepts_mst2_file_source_and_checks_identity() { + use futures::FutureExt; + + let lock = env_lock(); + // The repository test environment still carries this removed setting. + // Keep the file-source regression isolated for both load and validate. + let _mail = crate::config::testing::EnvVarGuard::remove(&lock, "MEGA_MAIL__ENABLED"); + let _enabled = crate::config::testing::EnvVarGuard::remove(&lock, "MEGA_MST2__ENABLED"); + let _identity = + crate::config::testing::EnvVarGuard::remove(&lock, "MEGA_MST2__INSTANCE_UUID"); + let _publication = + crate::config::testing::EnvVarGuard::remove(&lock, "MEGA_MST2__PUBLICATION_ENABLED"); + let _token = crate::config::testing::EnvVarGuard::remove(&lock, "MEGA_MST2__AUTH_TOKEN"); + let dir = tempfile::tempdir().expect("temp dir"); + let config_path = dir.path().join("config.toml"); + let content = format!( + "{}\n[mst2]\nenabled = true\ninstance_uuid = \"12345678-1234-4234-9234-123456789abc\"\npublication_enabled = false\nauth_token = \"mst2-cli-test-token\"\n", + crate::config::template::config_init_template(dir.path()), + ); + fs::write(&config_path, &content).expect("write MST/2 config"); + let selected = + crate::config::loader::ConfigLoader::new(crate::config::loader::ConfigInput { + cli_path: Some(config_path.clone()), + ..Default::default() + }) + .load_readonly() + .expect("select explicit existing config"); + assert_eq!(selected.path, config_path); + let config = + load_config_for_validate(&selected.path, None).expect("CLI loader accepts MST/2 file"); + assert!(config.mst2.enabled); + assert_eq!( + config.mst2.auth_token.as_deref(), + Some("mst2-cli-test-token") + ); + let invalid_path = dir.path().join("invalid.toml"); + fs::write( + &invalid_path, + content.replace( + "12345678-1234-4234-9234-123456789abc", + "00000000-0000-0000-0000-000000000000", + ), + ) + .expect("write nil UUID config"); + let invalid_config = load_config_for_validate(&invalid_path, None) + .expect("syntactically valid MST/2 file loads before semantic validation"); + validate_config( + &config, + Some(&config_path), + None, + false, + false, + false, + "human", + ) + .now_or_never() + .expect("validation without secret resolution must complete synchronously") + .expect("ordinary CLI validation accepts MST/2"); + let message = validate_config( + &invalid_config, + Some(&invalid_path), + None, + false, + false, + false, + "human", + ) + .now_or_never() + .expect("validation without secret resolution must complete synchronously") + .expect_err("ordinary CLI validation must reject nil UUID") + .to_string(); + assert!(message.contains("mst2.instance_uuid"), "{message}"); + assert!(!message.contains("mst2-cli-test-token"), "{message}"); + } + + #[test] + fn config_validate_rejects_unknown_mst2_profile_field() { + let dir = tempfile::tempdir().expect("temp dir"); + let config_path = dir.path().join("config.toml"); + let profile_path = dir.path().join("config.test.toml"); + fs::write( + &config_path, + crate::config::template::config_init_template(dir.path()), + ) + .expect("write base config"); + fs::write( + &profile_path, + "[mst2]\nenabled = false\npublication_enabeld = true\n", + ) + .expect("write typo profile"); + let selected = + crate::config::loader::ConfigLoader::new(crate::config::loader::ConfigInput { + cli_path: Some(config_path), + cli_profile: Some("test".to_string()), + ..Default::default() + }) + .load_readonly() + .expect("select explicit existing profile"); + let message = load_config_for_validate( + &selected.path, + selected + .profile + .as_ref() + .map(|profile| profile.path.as_path()), + ) + .expect_err("MST/2 nested typo must fail in the CLI profile loader") + .to_string(); + assert!(message.contains("mst2.publication_enabeld"), "{message}"); + } + #[test] fn config_validate_accepts_deny_warnings_flag() { let matches = cli() diff --git a/src/commands/mod.rs b/src/commands/mod.rs index 92361218..4c7fc6d3 100644 --- a/src/commands/mod.rs +++ b/src/commands/mod.rs @@ -119,7 +119,7 @@ pub(crate) fn builtin_exec(cmd: &str) -> Option { pub(crate) fn load_mode(cmd: &str, args: &ArgMatches) -> Option { match cmd { "service" => match args.subcommand_name() { - Some("init") => Some(LoadMode::ParsedExistingConfig), + Some("init" | "native-publication-init") => Some(LoadMode::ParsedExistingConfig), _ => Some(LoadMode::FullAppContext), }, "debug" => Some(LoadMode::FullAppContext), diff --git a/src/commands/service/init.rs b/src/commands/service/init.rs index 4078d84b..69a7a712 100644 --- a/src/commands/service/init.rs +++ b/src/commands/service/init.rs @@ -1,6 +1,9 @@ use clap::{Arg, ArgAction, ArgMatches, Command}; -use crate::{common::errors::MegaResult, config::Config, context::bootstrap_monorepo}; +use crate::{ + common::errors::MegaResult, config::Config, context::bootstrap_monorepo_with_commit_time, + jupiter::utils::converter::BootstrapCommitTime, +}; pub fn cli() -> Command { Command::new("init") @@ -12,11 +15,19 @@ pub fn cli() -> Command { .required(true) .help("Confirm creation of the configured initial Monorepo graph"), ) + .arg( + Arg::new("commit-time") + .long("commit-time") + .value_name("UNIX_SECONDS") + .value_parser(clap::value_parser!(BootstrapCommitTime)) + .help("Set the initial commit's author and committer time (0..=4294967295)"), + ) } -pub(crate) async fn exec(config: Config, _args: &ArgMatches) -> MegaResult { +pub(crate) async fn exec(config: Config, args: &ArgMatches) -> MegaResult { let object_format = config.monorepo.object_format.as_str(); - bootstrap_monorepo(config).await?; + let commit_time = args.get_one::("commit-time").copied(); + bootstrap_monorepo_with_commit_time(config, commit_time).await?; tracing::info!( object_format, "Monorepo bootstrap completed; exiting without starting Git listeners" @@ -24,6 +35,10 @@ pub(crate) async fn exec(config: Config, _args: &ArgMatches) -> MegaResult { Ok(()) } +#[cfg(test)] +#[path = "init_commit_time_tests.rs"] +mod commit_time_tests; + #[cfg(test)] mod tests { use super::cli; diff --git a/src/commands/service/init_commit_time_tests.rs b/src/commands/service/init_commit_time_tests.rs new file mode 100644 index 00000000..543fca98 --- /dev/null +++ b/src/commands/service/init_commit_time_tests.rs @@ -0,0 +1,316 @@ +use std::{collections::BTreeMap, sync::Arc}; + +use git_internal::{ + hash::{HashKind, get_hash_kind, set_hash_kind_for_test}, + internal::object::{ObjectTrait, commit::Commit}, +}; +use sea_orm::{ActiveModelTrait, EntityTrait, IntoActiveModel, QueryOrder, Set}; +use serde_json::{Value, json}; + +use super::{cli, exec}; +use crate::{ + callisto::{mega_blob, mega_commit, mega_tree}, + ceres::pack::materialize::{lock_materialize_tests, materialize_path_refs}, + config::{Config, MonoConfig, MonoObjectFormat, testing::isolated_config}, + jupiter::{ + storage::{Storage, base_storage::StorageConnector, init::database_connection}, + tests::{TestSchemaGuard, test_db_config}, + utils::converter::{BootstrapCommitTime, FromMegaModel, MegaModelConverter}, + }, +}; + +#[test] +fn commit_time_cli_is_explicit_bounded_unsigned_unix_seconds() { + let default = cli().try_get_matches_from(["init", "--yes"]).unwrap(); + assert!( + default + .get_one::("commit-time") + .is_none() + ); + for value in ["0", "1790000000", "4294967295"] { + let args = cli() + .try_get_matches_from(["init", "--yes", "--commit-time", value]) + .unwrap(); + assert_eq!( + args.get_one::("commit-time"), + Some(&value.parse().unwrap()) + ); + } + for value in [ + "", + "-1", + "+1", + "1.0", + "1e3", + " 1", + "1 ", + "4294967296", + "18446744073709551616", + ] { + assert!( + cli() + .try_get_matches_from(["init", "--yes", &format!("--commit-time={value}")]) + .is_err(), + "invalid commit time accepted: {value:?}" + ); + assert!(value.parse::().is_err()); + } + assert!( + cli() + .try_get_matches_from(["init", "--commit-time", "1"]) + .is_err() + ); + assert!( + cli() + .try_get_matches_from(["init", "--yes", "--commit-time", "1", "--commit-time", "2"]) + .is_err() + ); +} + +#[test] +fn explicit_commit_time_reproduces_each_hash_format_and_restores_scope() { + let _scope = set_hash_kind_for_test(HashKind::Sha1); + for (format, kind) in [ + (MonoObjectFormat::Sha1, HashKind::Sha1), + (MonoObjectFormat::Sha256, HashKind::Sha256), + (MonoObjectFormat::Blake3, HashKind::Blake3), + ] { + let config = MonoConfig { + object_format: format, + ..Default::default() + }; + for seconds in ["0", "1790000000", "4294967295"] { + let time = seconds.parse().unwrap(); + let first = MegaModelConverter::init_with_commit_time(&config, Some(time)).unwrap(); + let second = MegaModelConverter::init_with_commit_time(&config, Some(time)).unwrap(); + assert_eq!(first.commit.id.kind(), kind); + assert_eq!(first.root_tree.id.kind(), kind); + assert_eq!(first.commit.id, second.commit.id); + assert_eq!( + first.commit.to_data().unwrap(), + second.commit.to_data().unwrap() + ); + assert_eq!( + first.root_tree.to_data().unwrap(), + second.root_tree.to_data().unwrap() + ); + assert_eq!(first.commit.author.timestamp.to_string(), seconds); + assert_eq!(first.commit.committer.timestamp.to_string(), seconds); + assert_eq!(get_hash_kind(), HashKind::Sha1); + } + } +} + +struct Fixture { + config: Config, + storage: Storage, + schema: TestSchemaGuard, + _directory: tempfile::TempDir, +} + +async fn fixture() -> Fixture { + let directory = tempfile::tempdir().unwrap(); + let (database, schema) = test_db_config(directory.path()).await; + let mut config = isolated_config(directory.path().join("config")); + config.database = database; + config.monorepo.root_dirs = vec!["bench".to_string(), "third-party".to_string()]; + config.redis.url = "redis://127.0.0.1:1".to_string(); + let args = cli() + .try_get_matches_from(["init", "--yes", "--commit-time", "1790000000"]) + .unwrap(); + exec(config.clone(), &args).await.unwrap(); + let connection = Arc::new(database_connection(&config.database).await.unwrap()); + let object_store = + crate::jupiter::storage::object_storage::build_object_storage(&config.object_storage) + .await + .unwrap(); + let storage = Storage::new_with_connection(Arc::new(config.clone()), connection, object_store) + .await + .unwrap(); + Fixture { + config, + storage, + schema, + _directory: directory, + } +} + +async fn graph_identity(storage: &Storage) -> Value { + let mono = storage.mono_storage(); + let db = mono.get_connection(); + let commits = mega_commit::Entity::find() + .order_by_asc(mega_commit::Column::CommitId) + .all(db) + .await + .unwrap(); + let trees = mega_tree::Entity::find().all(db).await.unwrap(); + let blobs = mega_blob::Entity::find().all(db).await.unwrap(); + let mut raw_blobs = BTreeMap::new(); + for blob in blobs { + raw_blobs.insert( + blob.blob_id.clone(), + storage + .git_service + .get_object_as_bytes(&blob.blob_id) + .await + .unwrap(), + ); + } + let refs = mono.get_all_refs("/", false).await.unwrap(); + let path_refs = mono.get_all_refs("/bench", false).await.unwrap(); + json!({ + "commits": commits.into_iter().map(|commit| json!({ + "id": commit.commit_id, "tree": commit.tree, "parents": commit.parents_id, + "author": commit.author, "committer": commit.committer, "message": commit.content, + })).collect::>(), + "trees": trees.into_iter().map(|tree| (tree.tree_id, tree.sub_trees)).collect::>(), + "blobs": raw_blobs, + "root_refs": refs.into_iter().map(|r| (r.ref_name, (r.ref_commit_hash, r.ref_tree_hash))).collect::>(), + "path_refs": path_refs.into_iter().map(|r| (r.ref_name, (r.ref_commit_hash, r.ref_tree_hash))).collect::>(), + }) +} + +#[tokio::test] +async fn cli_bootstrap_two_real_schemas_share_seed_root_tree_and_path_parent() { + let _lock = lock_materialize_tests().await; + let first = fixture().await; + let second = fixture().await; + assert_ne!(first.schema.schema(), second.schema.schema()); + let first_root = first + .storage + .mono_storage() + .get_main_ref("/") + .await + .unwrap() + .unwrap(); + let second_root = second + .storage + .mono_storage() + .get_main_ref("/") + .await + .unwrap() + .unwrap(); + assert_eq!(first_root.ref_commit_hash, second_root.ref_commit_hash); + assert_eq!(first_root.ref_tree_hash, second_root.ref_tree_hash); + let mut seeds = Vec::new(); + for fixture in [&first, &second] { + let refs = materialize_path_refs(&fixture.storage, "/bench") + .await + .unwrap(); + assert_eq!(refs.len(), 1); + let parent = fixture + .storage + .mono_storage() + .get_commit_by_hash(&refs[0].ref_commit_hash) + .await + .unwrap() + .unwrap(); + let root = fixture + .storage + .mono_storage() + .get_commit_by_hash(&first_root.ref_commit_hash) + .await + .unwrap() + .unwrap(); + assert_eq!(parent.author, root.author); + assert_eq!(parent.committer, root.committer); + assert_eq!(parent.parents_id, json!([])); + let parent = Commit::from_mega_model(parent); + let seed = Commit::new_with_kind( + HashKind::Sha1, + parent.author.clone(), + parent.committer.clone(), + parent.tree_id, + vec![parent.id], + "\npaired seed", + ) + .unwrap(); + assert_eq!(seed.parent_commit_ids, vec![parent.id]); + seeds.push(seed.to_data().unwrap()); + } + assert_eq!(seeds[0], seeds[1]); + assert_eq!( + graph_identity(&first.storage).await, + graph_identity(&second.storage).await + ); +} + +#[tokio::test] +async fn cli_bootstrap_existing_root_replays_exactly_and_rejects_changed_inputs_without_writes() { + let fixture = fixture().await; + let before = graph_identity(&fixture.storage).await; + let args = cli() + .try_get_matches_from(["init", "--yes", "--commit-time", "1790000000"]) + .unwrap(); + exec(fixture.config.clone(), &args).await.unwrap(); + assert_eq!(graph_identity(&fixture.storage).await, before); + let changed_time = cli() + .try_get_matches_from(["init", "--yes", "--commit-time", "1790000001"]) + .unwrap(); + let error = exec(fixture.config.clone(), &changed_time) + .await + .unwrap_err(); + assert!(error.to_string().contains("does not match")); + assert_eq!(graph_identity(&fixture.storage).await, before); + let mut changed_config = fixture.config.clone(); + changed_config + .monorepo + .root_dirs + .push("different".to_string()); + let error = exec(changed_config, &args).await.unwrap_err(); + assert!(error.to_string().contains("does not match")); + assert_eq!(graph_identity(&fixture.storage).await, before); + let default_args = cli().try_get_matches_from(["init", "--yes"]).unwrap(); + exec(fixture.config.clone(), &default_args).await.unwrap(); + assert_eq!(graph_identity(&fixture.storage).await, before); +} + +#[tokio::test] +async fn cli_bootstrap_checks_stored_root_objects_even_when_ref_ids_match() { + let fixture = fixture().await; + let mono = fixture.storage.mono_storage(); + let root = mono.get_main_ref("/").await.unwrap().unwrap(); + let args = cli() + .try_get_matches_from(["init", "--yes", "--commit-time", "1790000000"]) + .unwrap(); + let commit = mono + .get_commit_by_hash(&root.ref_commit_hash) + .await + .unwrap() + .unwrap(); + let original_message = commit.content.clone(); + let mut changed = commit.into_active_model(); + changed.content = Set(Some("\nchanged stored commit".to_string())); + let changed = changed.update(mono.get_connection()).await.unwrap(); + let corrupt_commit = graph_identity(&fixture.storage).await; + assert!( + exec(fixture.config.clone(), &args) + .await + .unwrap_err() + .to_string() + .contains("does not match") + ); + assert_eq!(graph_identity(&fixture.storage).await, corrupt_commit); + let mut restored = changed.into_active_model(); + restored.content = Set(original_message); + restored.update(mono.get_connection()).await.unwrap(); + let tree = mono + .get_tree_by_hash(&root.ref_tree_hash) + .await + .unwrap() + .unwrap(); + let mut bytes = tree.sub_trees.clone(); + bytes.push(b'x'); + let mut changed_tree = tree.into_active_model(); + changed_tree.sub_trees = Set(bytes); + changed_tree.update(mono.get_connection()).await.unwrap(); + let corrupt_tree = graph_identity(&fixture.storage).await; + assert!( + exec(fixture.config.clone(), &args) + .await + .unwrap_err() + .to_string() + .contains("does not match") + ); + assert_eq!(graph_identity(&fixture.storage).await, corrupt_tree); +} diff --git a/src/commands/service/mod.rs b/src/commands/service/mod.rs index 666dfb7c..204fc9ce 100644 --- a/src/commands/service/mod.rs +++ b/src/commands/service/mod.rs @@ -17,12 +17,19 @@ use crate::{ pub mod http; pub mod init; pub mod multi; +pub mod native_publication_init; pub mod ssh; const CONFIG_RELOAD_POLL_INTERVAL: Duration = Duration::from_secs(5); pub fn cli() -> Command { - let subcommands = vec![init::cli(), http::cli(), ssh::cli(), multi::cli()]; + let subcommands = vec![ + init::cli(), + native_publication_init::cli(), + http::cli(), + ssh::cli(), + multi::cli(), + ]; Command::new("service") .about("Start different kinds of server: for example https or ssh") .subcommands(subcommands) @@ -30,6 +37,9 @@ pub fn cli() -> Command { #[tokio::main] pub(crate) async fn exec(ctx: CommandContext, args: &ArgMatches) -> MegaResult { + if let Some(("native-publication-init", subcommand_args)) = args.subcommand() { + return native_publication_init::exec(ctx, subcommand_args).await; + } let config_path = ctx.config_path.clone(); let config_profile_path = ctx.config_profile_path.clone(); let config = require_config(ctx, "service")?; @@ -96,7 +106,7 @@ pub(crate) async fn exec(ctx: CommandContext, args: &ArgMatches) -> MegaResult { .await }; - match setup.await { + let result = match setup.await { Err(error) => cleanup_tail(Err(error), emitter, None, signal_forwarder).await, Ok(reload_watcher) => { let service = context.clone(); @@ -117,7 +127,16 @@ pub(crate) async fn exec(ctx: CommandContext, args: &ArgMatches) -> MegaResult { ) .await } + }; + if let Some(sink) = &context.storage.projection_observation_sink { + let deadline = std::time::Instant::now() + std::time::Duration::from_secs(5); + if sink.shutdown(deadline).await.is_err() && result.is_ok() { + return Err(MegaError::Other( + "typed projection writer did not drain successfully".into(), + )); + } } + result } /// Runs the service body under the unified cleanup tail (WH-13). @@ -256,7 +275,10 @@ mod tests { .map(|cmd| cmd.get_name().to_owned()) .collect::>(); - assert_eq!(names, vec!["init", "http", "ssh", "multi"]); + assert_eq!( + names, + vec!["init", "native-publication-init", "http", "ssh", "multi"] + ); } #[test] diff --git a/src/commands/service/native_publication_init.rs b/src/commands/service/native_publication_init.rs new file mode 100644 index 00000000..483f647c --- /dev/null +++ b/src/commands/service/native_publication_init.rs @@ -0,0 +1,198 @@ +use std::sync::Arc; + +use clap::{Arg, ArgAction, ArgMatches, Command}; +use sea_orm_migration::MigratorTrait; + +use crate::{ + commands::{CommandContext, require_config}, + common::errors::{MegaError, MegaResult}, + config::{DbConfig, loader::ConfigSource}, + jupiter::{ + migration::Migrator, + storage::{ + base_storage::{BaseStorage, StorageConnector}, + init::{postgres_connection, read_only_database_connection}, + mono_storage::MonoStorage, + native_publication_storage::NativeRoot, + }, + }, +}; + +#[path = "native_publication_init_preflight.rs"] +mod preflight; +use preflight::InitializationTarget; + +pub fn cli() -> Command { + let mut command = Command::new("native-publication-init") + .about("Prepare native publication in a stopped, paused and drained SHA-1 deployment"); + for (name, help) in [ + ("instance", "Deployment UUID; must match mst2.instance_uuid"), + ( + "expected-root-commit", + "Exact current native root commit object ID", + ), + ( + "expected-root-tree", + "Exact current native root tree object ID", + ), + ] { + command = command.arg(Arg::new(name).long(name).required(true).help(help)); + } + for (name, help) in [ + ("yes", "Confirm preparation of the native publication head"), + ( + "writers-stopped", + "Confirm all old writer processes have been stopped", + ), + ] { + command = command.arg( + Arg::new(name) + .long(name) + .action(ArgAction::SetTrue) + .required(true) + .help(help), + ); + } + command +} + +pub(crate) async fn exec(ctx: CommandContext, args: &ArgMatches) -> MegaResult { + if !ctx + .config_summary + .as_ref() + .is_some_and(|summary| matches!(summary.source, ConfigSource::Cli | ConfigSource::Env)) + { + return Err(MegaError::Other( + "MST2_NATIVE_INIT_CONFIG_REQUIRED: name the deployment with --config or MEGA_CONFIG" + .into(), + )); + } + if !args.get_flag("yes") || !args.get_flag("writers-stopped") { + return Err(MegaError::Other( + "MST2_NATIVE_INIT_CONFIRMATION_REQUIRED: --yes and --writers-stopped are required" + .into(), + )); + } + let config = require_config(ctx, "service native-publication-init")?; + config.validate()?; + if !config.mst2.enabled || !config.mst2.publication_enabled { + return Err(MegaError::Other( + "MST2_NATIVE_INIT_DISABLED: mst2.enabled and mst2.publication_enabled must be true" + .into(), + )); + } + let argument = |name| { + args.get_one::(name) + .map(String::as_str) + .ok_or_else(|| MegaError::Other(format!("--{name} is required"))) + }; + let target = InitializationTarget::parse( + config.monorepo.object_format.as_str(), + config.mst2.instance_uuid.as_deref(), + argument("instance")?, + argument("expected-root-commit")?, + argument("expected-root-tree")?, + ) + .map_err(|error| MegaError::Other(error.to_string()))?; + ensure_schema_current(&config.database).await?; + let db = Arc::new(postgres_connection(&config.database).await?); + let mono = MonoStorage { + base: BaseStorage::new(db.clone()), + }; + let result = mono + .initialize_native_publication_for_maintenance( + &target.instance, + &NativeRoot { + commit: target.commit, + tree: target.tree, + }, + ) + .await + .map_err(|error| MegaError::Other(error.to_string())); + let close = db.close_by_ref().await; + result?; + close?; + tracing::info!( + instance = %target.instance, + state = "INITIALIZING", + "Native publication initialization prepared; queue remains paused" + ); + Ok(()) +} + +async fn ensure_schema_current(config: &DbConfig) -> MegaResult { + let db = read_only_database_connection(config).await?; + let pending = Migrator::get_pending_migrations_read_only(&db).await; + let _ = db.close().await; + let pending = pending.map_err(|error| { + MegaError::Other(format!( + "MST2_NATIVE_INIT_SCHEMA_MISMATCH: cannot confirm the current schema: {error}" + )) + })?; + if !pending.is_empty() { + return Err(MegaError::Other(format!( + "MST2_NATIVE_INIT_SCHEMA_MISMATCH: {} migrations are pending; this command does not migrate", + pending.len() + ))); + } + Ok(()) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::commands::{LoadMode, load_mode}; + + #[test] + fn maintenance_cli_requires_each_confirmation_and_fixed_target() { + let arguments = [ + "native-publication-init", + "--yes", + "--writers-stopped", + "--instance", + "instance", + "--expected-root-commit", + "commit", + "--expected-root-tree", + "tree", + ]; + assert!(cli().try_get_matches_from(arguments).is_ok()); + for remove in [1, 2, 3, 5, 7] { + let count = if remove <= 2 { 1 } else { 2 }; + let mut incomplete = arguments.to_vec(); + incomplete.drain(remove..remove + count); + assert!(cli().try_get_matches_from(incomplete).is_err()); + } + let mut service_arguments = vec!["service"]; + service_arguments.extend(arguments); + let args = super::super::cli() + .try_get_matches_from(service_arguments) + .unwrap(); + assert_eq!( + load_mode("service", &args), + Some(LoadMode::ParsedExistingConfig) + ); + } + + #[tokio::test] + async fn maintenance_command_refuses_unnamed_config_before_any_database_io() { + let args = cli() + .try_get_matches_from([ + "native-publication-init", + "--yes", + "--writers-stopped", + "--instance", + "instance", + "--expected-root-commit", + "commit", + "--expected-root-tree", + "tree", + ]) + .unwrap(); + let error = exec(CommandContext::default(), &args).await.unwrap_err(); + assert!( + matches!(&error, MegaError::Other(message) if message.starts_with("MST2_NATIVE_INIT_CONFIG_REQUIRED")), + "unexpected error: {error}" + ); + } +} diff --git a/src/commands/service/native_publication_init_preflight.rs b/src/commands/service/native_publication_init_preflight.rs new file mode 100644 index 00000000..360751b6 --- /dev/null +++ b/src/commands/service/native_publication_init_preflight.rs @@ -0,0 +1,162 @@ +#[derive(Debug, thiserror::Error, PartialEq, Eq)] +pub(super) enum InitializationTargetError { + #[error( + "MST2_NATIVE_INIT_UNSUPPORTED_OBJECT_FORMAT: {0}; this maintenance entry supports sha1 only" + )] + UnsupportedObjectFormat(String), + #[error("MST2_NATIVE_INIT_INVALID_INSTANCE: both deployment instances must be non-nil UUIDs")] + InvalidInstance, + #[error("MST2_NATIVE_INIT_INSTANCE_MISMATCH: --instance differs from mst2.instance_uuid")] + InstanceMismatch, + #[error( + "MST2_NATIVE_INIT_INVALID_ROOT: expected commit and tree must be canonical SHA-1 object IDs" + )] + InvalidRoot, +} + +#[derive(Debug, PartialEq, Eq)] +pub(super) struct InitializationTarget { + pub(super) instance: String, + pub(super) commit: String, + pub(super) tree: String, +} + +impl InitializationTarget { + pub(super) fn parse( + object_format: &str, + configured_instance: Option<&str>, + requested_instance: &str, + commit: &str, + tree: &str, + ) -> Result { + if object_format != "sha1" { + return Err(InitializationTargetError::UnsupportedObjectFormat( + object_format.into(), + )); + } + let parse_instance = |value: &str| { + uuid::Uuid::parse_str(value) + .ok() + .filter(|id| !id.is_nil()) + .ok_or(InitializationTargetError::InvalidInstance) + }; + let configured = + parse_instance(configured_instance.ok_or(InitializationTargetError::InvalidInstance)?)?; + let requested = parse_instance(requested_instance)?; + if configured != requested { + return Err(InitializationTargetError::InstanceMismatch); + } + if [commit, tree].iter().any(|value| { + value.len() != 40 + || !value + .bytes() + .all(|byte| byte.is_ascii_digit() || (b'a'..=b'f').contains(&byte)) + }) { + return Err(InitializationTargetError::InvalidRoot); + } + Ok(Self { + instance: configured.to_string(), + commit: commit.into(), + tree: tree.into(), + }) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + const INSTANCE: &str = "6ab219b0-4275-45ba-9d7b-7b0b633018cd"; + + #[test] + fn target_is_canonical_and_bound_to_the_configured_instance() { + let target = InitializationTarget::parse( + "sha1", + Some(INSTANCE), + &INSTANCE.to_uppercase(), + &"a".repeat(40), + &"b".repeat(40), + ) + .unwrap(); + assert_eq!(target.instance, INSTANCE); + assert_eq!(target.commit, "a".repeat(40)); + assert_eq!(target.tree, "b".repeat(40)); + } + + #[test] + fn target_rejects_other_formats_without_inferring_from_equal_width() { + for format in ["sha256", "blake3", "unknown"] { + assert_eq!( + InitializationTarget::parse( + format, + Some(INSTANCE), + INSTANCE, + &"a".repeat(64), + &"b".repeat(64), + ), + Err(InitializationTargetError::UnsupportedObjectFormat( + format.into() + )) + ); + } + } + + #[test] + fn target_rejects_nil_missing_invalid_or_different_instances() { + for instance in [ + None, + Some("invalid"), + Some("00000000-0000-0000-0000-000000000000"), + ] { + assert_eq!( + InitializationTarget::parse( + "sha1", + instance, + INSTANCE, + &"a".repeat(40), + &"b".repeat(40) + ), + Err(InitializationTargetError::InvalidInstance) + ); + } + for instance in ["invalid", "00000000-0000-0000-0000-000000000000"] { + assert_eq!( + InitializationTarget::parse( + "sha1", + Some(INSTANCE), + instance, + &"a".repeat(40), + &"b".repeat(40) + ), + Err(InitializationTargetError::InvalidInstance) + ); + } + assert_eq!( + InitializationTarget::parse( + "sha1", + Some(INSTANCE), + "11111111-2222-4333-8444-555555555555", + &"a".repeat(40), + &"b".repeat(40) + ), + Err(InitializationTargetError::InstanceMismatch) + ); + } + + #[test] + fn target_rejects_noncanonical_or_wrong_width_object_ids() { + for value in [ + "a".repeat(39), + "a".repeat(64), + "A".repeat(40), + "z".repeat(40), + ] { + for (commit, tree) in [(&value, &"b".repeat(40)), (&"a".repeat(40), &value)] { + assert_eq!( + InitializationTarget::parse("sha1", Some(INSTANCE), INSTANCE, commit, tree), + Err(InitializationTargetError::InvalidRoot) + ); + } + } + } +} diff --git a/src/common/errors/mod.rs b/src/common/errors/mod.rs index d1872272..5d5ef9b1 100644 --- a/src/common/errors/mod.rs +++ b/src/common/errors/mod.rs @@ -184,6 +184,12 @@ impl From for MegaError { IoOrbitError::WriteManifestPreconditionFailed => { MegaError::Other("write manifest precondition failed".to_string()) } + error @ IoOrbitError::ChunkMapRetentionUnsupported => { + MegaError::ObjStorage(error.to_string()) + } + error @ IoOrbitError::ChunkMapRetentionCapacityExceeded => { + MegaError::ObjStorage(error.to_string()) + } IoOrbitError::Other(e) => MegaError::Other(e), } } diff --git a/src/config/model.rs b/src/config/model.rs index ec549858..db2c6220 100644 --- a/src/config/model.rs +++ b/src/config/model.rs @@ -957,6 +957,9 @@ impl Default for LFSSshConfig { /// MST/2 snapshot feature flags (default off; spec 00 §6 rollout). #[derive(Serialize, Deserialize, Debug, Clone, Default)] pub struct Mst2Config { + /// Startup-only typed projection diagnostics, independent of log.level. + #[serde(default)] + pub projection_observation_enabled: bool, /// Master switch for the `/api/v2/snapshots` surface. #[serde(default)] pub enabled: bool, diff --git a/src/config/reload.rs b/src/config/reload.rs index bbb6d47f..4674938d 100644 --- a/src/config/reload.rs +++ b/src/config/reload.rs @@ -669,6 +669,12 @@ fn collect_static_restart_fields( candidate: &Config, report: &mut ConfigReloadReport, ) { + if current.mst2.projection_observation_enabled != candidate.mst2.projection_observation_enabled + { + report + .restart_required_fields + .push("mst2.projection_observation_enabled"); + } if current.base_dir != candidate.base_dir { report.restart_required_fields.push("base_dir"); } @@ -1209,6 +1215,33 @@ mod tests { } } + #[test] + fn projection_writer_toggle_requires_restart_and_preserves_live_snapshot() { + let temp = tempfile::tempdir().unwrap(); + let mut current = isolated_config(temp.path().join("config")); + current.mst2.enabled = true; + current.mst2.publication_enabled = true; + current.mst2.instance_uuid = Some("12345678-1234-4234-9234-123456789abc".into()); + let handle = ConfigHandle::new(current); + let before = handle.snapshot().unwrap(); + let mut candidate = before.as_ref().clone(); + candidate.mst2.projection_observation_enabled = true; + let report = handle.reload(candidate).unwrap(); + assert_eq!( + report.restart_required_fields, + vec!["mst2.projection_observation_enabled"] + ); + assert!(!report.applied()); + assert!(Arc::ptr_eq(&before, &handle.snapshot().unwrap())); + assert!( + !handle + .snapshot() + .unwrap() + .mst2 + .projection_observation_enabled + ); + } + #[test] fn reload_applies_log_fields_and_preserves_restart_required_database_fields() { let temp_dir = tempfile::tempdir().expect("temp dir"); diff --git a/src/config/validate.rs b/src/config/validate.rs index 5d67fd35..bb14e9a6 100644 --- a/src/config/validate.rs +++ b/src/config/validate.rs @@ -9,9 +9,9 @@ use url::Url; use super::{ ArtifactGcConfig, BlameConfig, BuckConfig, CedarConfig, Config, DbConfig, GitConfig, - GithubSyncConfig, LFSConfig, LogConfig, MonoConfig, MonoObjectFormat, NotificationConfig, - OAuthConfig, PackConfig, PushAuth, PushPolicy, RedisConfig, VAULT_AUDIT_SINKS, VaultConfig, - normalize_token_path, + GithubSyncConfig, LFSConfig, LogConfig, MonoConfig, MonoObjectFormat, Mst2Config, + NotificationConfig, OAuthConfig, PackConfig, PushAuth, PushPolicy, RedisConfig, + VAULT_AUDIT_SINKS, VaultConfig, normalize_token_path, secret::{SecretRef, SecretResolver, is_secret_ref_value}, }; use crate::{ @@ -124,6 +124,7 @@ impl Config { validate_lfs_config(&self.lfs)?; validate_redis_config(&self.redis)?; validate_cedar_config(&self.cedar)?; + validate_mst2_config(&self.mst2)?; if let Some(buck_config) = &self.buck { validate_buck_config(buck_config)?; } @@ -150,6 +151,39 @@ impl Config { } } +fn validate_mst2_config(config: &Mst2Config) -> Result<(), MegaError> { + if config.projection_observation_enabled && (!config.enabled || !config.publication_enabled) { + return Err(MegaError::Other( + "mst2.projection_observation_enabled requires mst2.enabled and mst2.publication_enabled".into(), + )); + } + if !config.enabled { + return Ok(()); + } + let instance_uuid = config.instance_uuid.as_deref().ok_or_else(|| { + MegaError::Other("mst2.instance_uuid is required when mst2.enabled is true".to_string()) + })?; + let instance_uuid = uuid::Uuid::parse_str(instance_uuid).map_err(|_| { + MegaError::Other("mst2.instance_uuid must be a valid non-nil UUID".to_string()) + })?; + if instance_uuid.is_nil() { + return Err(MegaError::Other( + "mst2.instance_uuid must be a valid non-nil UUID".to_string(), + )); + } + if config + .auth_token + .as_deref() + .is_some_and(|token| token.trim().is_empty()) + { + return Err(MegaError::Other( + "mst2.auth_token must not be empty when configured and mst2.enabled is true" + .to_string(), + )); + } + Ok(()) +} + /// Validate `[oauth]` settings: website API base URL, session cookie names, and /// each CORS origin must be a browser Origin of the form `scheme://host[:port]` /// (http/https, no path/query/fragment) that also parses as an HTTP header value @@ -2093,6 +2127,7 @@ pub(crate) fn known_fields(path: &str) -> Option<&'static [&'static str]> { "object_storage", "oauth", "blame", + "mst2", "redis", "buck", "artifacts_gc", @@ -2157,6 +2192,13 @@ pub(crate) fn known_fields(path: &str) -> Option<&'static [&'static str]> { "enable_caching", ]), "redis" => Some(&["url"]), + "mst2" => Some(&[ + "enabled", + "instance_uuid", + "publication_enabled", + "projection_observation_enabled", + "auth_token", + ]), "buck" => Some(&[ "session_timeout", "max_file_size", @@ -2263,7 +2305,7 @@ mod tests { StorageEventsTargetConfig, ViewsConfig, secret::{SecretRef, SecretResolver}, template::config_init_template, - testing::{env_lock, isolated_config}, + testing::{EnvVarGuard, env_lock, isolated_config}, }; #[rustfmt::skip] use crate::orbit_api::factory::{GcsConfig, LocalConfig, ObjectStorageBackend, S3Config}; @@ -2279,6 +2321,148 @@ mod tests { assert!(err.to_string().contains(field), "{err}"); } + #[test] + fn typed_projection_writer_is_default_off_and_requires_native_publication() { + assert!(!Mst2Config::default().projection_observation_enabled); + let loaded: Mst2Config = toml::from_str("enabled = false").unwrap(); + assert!(!loaded.projection_observation_enabled); + let mut config = valid_config(); + config.mst2.projection_observation_enabled = true; + config.mst2.instance_uuid = Some("12345678-1234-4234-9234-123456789abc".into()); + for (enabled, publication) in [(false, false), (false, true), (true, false)] { + config.mst2.enabled = enabled; + config.mst2.publication_enabled = publication; + assert!( + config + .validate() + .unwrap_err() + .to_string() + .contains("projection_observation_enabled") + ); + } + config.mst2.enabled = true; + config.mst2.publication_enabled = true; + config.validate().unwrap(); + assert!(is_known_field_path("mst2.projection_observation_enabled")); + } + + #[test] + fn config_validate_mst2_checks_enabled_identity_and_configured_token() { + let mut config = valid_config(); + config.mst2.enabled = true; + for instance_uuid in [ + None, + Some(""), + Some("invalid-uuid"), + Some("00000000-0000-0000-0000-000000000000"), + ] { + config.mst2.instance_uuid = instance_uuid.map(str::to_owned); + config.mst2.auth_token = Some("mst2-private-token-must-not-appear".to_string()); + let message = config + .validate() + .expect_err("enabled MST/2 requires a non-nil UUID") + .to_string(); + assert!(message.contains("mst2.instance_uuid"), "{message}"); + assert!( + !message.contains("mst2-private-token-must-not-appear"), + "{message}" + ); + } + config.mst2.instance_uuid = Some("12345678-1234-4234-9234-123456789abc".to_string()); + for token in ["", " \t\n"] { + config.mst2.auth_token = Some(token.to_string()); + let message = config + .validate() + .expect_err("configured MST/2 token must not be blank") + .to_string(); + assert!(message.contains("mst2.auth_token"), "{message}"); + } + config.mst2.auth_token = None; + config + .validate() + .expect("existing unauthenticated lab mode remains valid"); + config.mst2.auth_token = Some("mst2-test-token".to_string()); + config.mst2.publication_enabled = true; + config + .validate() + .expect("authenticated publication config is valid"); + config.mst2.enabled = false; + config.mst2.publication_enabled = false; + config.mst2.instance_uuid = None; + config.mst2.auth_token = None; + config + .validate() + .expect("disabled MST/2 defaults remain valid"); + } + + #[test] + fn config_load_str_accepts_mst2_and_expands_file_token_without_reexpansion() { + let lock = env_lock(); + let _enabled = EnvVarGuard::remove(&lock, "MEGA_MST2__ENABLED"); + let _identity = EnvVarGuard::remove(&lock, "MEGA_MST2__INSTANCE_UUID"); + let _publication = EnvVarGuard::remove(&lock, "MEGA_MST2__PUBLICATION_ENABLED"); + let _token = EnvVarGuard::remove(&lock, "MEGA_MST2__AUTH_TOKEN"); + let dir = tempfile::tempdir().expect("temp dir"); + let token_file = dir.path().join("mst2-token"); + let token = "literal-${must_not_be_reexpanded}"; + std::fs::write(&token_file, format!("{token}\n")).expect("write private test token"); + let file_placeholder = format!("${{file:{}}}", token_file.display()); + let content = format!( + "{}\n[mst2]\nenabled = true\ninstance_uuid = \"12345678-1234-4234-9234-123456789abc\"\npublication_enabled = false\nauth_token = {}\n", + config_init_template(dir.path()), + toml::Value::String(file_placeholder), + ); + let loaded = + Config::load_str(&content).expect("ordinary config loader must recognize MST/2"); + assert!(loaded.mst2.enabled); + assert!(!loaded.mst2.publication_enabled); + assert_eq!(loaded.mst2.auth_token.as_deref(), Some(token)); + loaded + .validate() + .expect("expanded MST/2 config must validate"); + assert!( + known_unconsumed_fields(&toml::from_str(&content).expect("parse source")).is_empty() + ); + for field in [ + "enabled", + "instance_uuid", + "publication_enabled", + "auth_token", + ] { + assert!(is_known_field_path(&format!("mst2.{field}"))); + } + assert!(!is_known_field_path("mst2.auth_token.extra")); + assert!(!is_known_field_path("mst2.typo")); + std::fs::write(&token_file, "\n").expect("write empty private test token"); + let empty_token = + Config::load_str(&content).expect("file token still uses ordinary expansion"); + let message = empty_token + .validate() + .expect_err("expanded empty token must fail semantic validation") + .to_string(); + assert!(message.contains("mst2.auth_token"), "{message}"); + } + + #[test] + fn config_load_str_rejects_mst2_typo_and_unknown_nested_table() { + let content = r#" + [mst2] + enabled = false + publication_enabeld = true + [mst2.credentials] + auth_token = "private-value-must-not-appear" + "#; + let message = Config::load_str(content) + .expect_err("MST/2 unknown fields must fail closed") + .to_string(); + assert!(message.contains("mst2.publication_enabeld"), "{message}"); + assert!(message.contains("mst2.credentials"), "{message}"); + assert!( + !message.contains("private-value-must-not-appear"), + "{message}" + ); + } + #[test] fn config_validate_accepts_default_isolated_test_config() { valid_config() diff --git a/src/context/mod.rs b/src/context/mod.rs index dd97ed20..560cf6fa 100644 --- a/src/context/mod.rs +++ b/src/context/mod.rs @@ -31,6 +31,7 @@ use crate::{ mono_storage::MonoStorage, vault_storage::VaultStorage, }, + utils::converter::BootstrapCommitTime, }, }; @@ -75,6 +76,13 @@ pub struct AppContext { /// Redis, notification workers, and Git listeners are deliberately /// outside this one-shot path. pub(crate) async fn bootstrap_monorepo(config: crate::config::Config) -> Result<(), MegaError> { + bootstrap_monorepo_with_commit_time(config, None).await +} + +pub(crate) async fn bootstrap_monorepo_with_commit_time( + config: crate::config::Config, + commit_time: Option, +) -> Result<(), MegaError> { config.validate()?; config.monorepo.object_hash_kind()?; let config = Arc::new(config); @@ -111,7 +119,9 @@ pub(crate) async fn bootstrap_monorepo(config: crate::config::Config) -> Result< }, }; - mono_service.bootstrap_monorepo(&config.monorepo).await + mono_service + .bootstrap_monorepo_with_commit_time(&config.monorepo, commit_time) + .await } impl AppContext { @@ -295,6 +305,22 @@ impl AppContext { } }; + let mut storage = storage; + if config.mst2.projection_observation_enabled { + match crate::ceres::snapshot::projection_writer::ProjectionObservationSink::start( + &crate::config::mega_cache(), + ) { + Ok(sink) => storage.projection_observation_sink = Some(sink), + Err(_) => { + notification_shutdown.cancel(); + storage.storage_event_emitter.shutdown().await; + return Err(MegaError::Other( + "typed projection writer failed startup".into(), + )); + } + } + } + Ok(Self { storage, vault, diff --git a/src/context/un30_readonly.rs b/src/context/un30_readonly.rs index ad829a0f..d95a03b6 100644 --- a/src/context/un30_readonly.rs +++ b/src/context/un30_readonly.rs @@ -244,9 +244,9 @@ async fn un30_building_the_read_facade_writes_no_sidebar() { .await .expect("writable connection"), ); - let config = Arc::new(crate::config::testing::isolated_config( - temp.path().join("config"), - )); + let mut config = crate::config::testing::isolated_config(temp.path().join("config")); + config.database = db_config.clone(); + let config = Arc::new(config); assert!( !sidebar_table_exists(&writable).await, diff --git a/src/contract/git_protocol/mod.rs b/src/contract/git_protocol/mod.rs index e3b4dc7d..6a87531e 100644 --- a/src/contract/git_protocol/mod.rs +++ b/src/contract/git_protocol/mod.rs @@ -954,6 +954,7 @@ mod tests { .unwrap(); let bytes = Arc::new(Mutex::new(Vec::new())); let subscriber = tracing_subscriber::fmt() + .with_ansi(false) .with_writer(Writer(bytes.clone())) .finish(); let guard = tracing::subscriber::set_default(subscriber); diff --git a/src/contract/vault/pki.rs b/src/contract/vault/pki.rs index cc111aaa..2f262a2c 100644 --- a/src/contract/vault/pki.rs +++ b/src/contract/vault/pki.rs @@ -385,8 +385,16 @@ mod tests_raw { .unwrap() .clone(); + let issuance_started = SystemTime::now() + .duration_since(UNIX_EPOCH) + .unwrap() + .as_secs(); // issue cert let resp = test_write_api(core, "pki/issue/tls/test", true, Some(issue_data)).await; + let issuance_completed = SystemTime::now() + .duration_since(UNIX_EPOCH) + .unwrap() + .as_secs(); assert!(resp.is_ok()); let resp_body = resp.unwrap(); assert!(resp_body.is_some()); @@ -429,15 +437,10 @@ mod tests_raw { let ttl_compare = cert.not_after().compare(&expiration_time); assert!(ttl_compare.is_ok()); assert_eq!(ttl_compare.unwrap(), std::cmp::Ordering::Equal); - let now_timestamp = SystemTime::now() - .duration_since(UNIX_EPOCH) - .unwrap() - .as_secs(); let expiration_ttl = cert_data["expiration"].as_u64().unwrap(); - let ttl = expiration_ttl - now_timestamp; let expect_ttl = 10 * 24 * 60 * 60; - assert!(ttl <= expect_ttl); - assert!((ttl + 10) > expect_ttl); + assert!(expiration_ttl >= issuance_started + expect_ttl); + assert!(expiration_ttl <= issuance_completed + expect_ttl); let authority_key_id = cert.authority_key_id(); assert!(authority_key_id.is_some()); diff --git a/src/jupiter/migration/m20261005_000100_add_mst2_publication_request_digest.rs b/src/jupiter/migration/m20261005_000100_add_mst2_publication_request_digest.rs new file mode 100644 index 00000000..8186b7c4 --- /dev/null +++ b/src/jupiter/migration/m20261005_000100_add_mst2_publication_request_digest.rs @@ -0,0 +1,48 @@ +use sea_orm_migration::prelude::*; + +#[derive(DeriveMigrationName)] +pub struct Migration; + +#[async_trait::async_trait] +impl MigrationTrait for Migration { + async fn up(&self, manager: &SchemaManager) -> Result<(), DbErr> { + manager + .get_connection() + .execute_unprepared( + r#"ALTER TABLE mst2_publication + ADD COLUMN IF NOT EXISTS request_digest text, + ADD COLUMN IF NOT EXISTS request_digest_version integer; + DO $$ BEGIN + ALTER TABLE mst2_publication + ADD CONSTRAINT mst2_publication_request_digest_valid CHECK ( + (request_digest IS NULL AND request_digest_version IS NULL) OR + (request_digest IS NOT NULL AND request_digest_version IS NOT NULL + AND request_digest_version > 0 + AND request_digest ~ '^sha256:[0-9a-f]{64}$') + ); + EXCEPTION WHEN duplicate_object THEN NULL; + END $$; + CREATE TABLE IF NOT EXISTS mst2_queue_noop_receipt ( + id bigserial PRIMARY KEY, + operation_id text NOT NULL UNIQUE, + namespace text NOT NULL, + request_digest text NOT NULL CHECK (request_digest ~ '^sha256:[0-9a-f]{64}$'), + request_digest_version integer NOT NULL CHECK (request_digest_version > 0), + writer_epoch bigint NOT NULL, + writer_kind text NOT NULL, + observed_sequence bigint NOT NULL CHECK (observed_sequence >= 0), + observed_root_commit text NOT NULL, + observed_root_tree text NOT NULL, + landed_commit_id text NOT NULL, + created_at timestamptz NOT NULL + );"#, + ) + .await?; + Ok(()) + } + + async fn down(&self, _manager: &SchemaManager) -> Result<(), DbErr> { + // Receipts remain immutable, including the absence of legacy digests. + Ok(()) + } +} diff --git a/src/jupiter/migration/m20261005_000100_add_mst2_retention_durability.rs b/src/jupiter/migration/m20261005_000100_add_mst2_retention_durability.rs new file mode 100644 index 00000000..8f223af1 --- /dev/null +++ b/src/jupiter/migration/m20261005_000100_add_mst2_retention_durability.rs @@ -0,0 +1,119 @@ +//! T06-B: durable retention counters and replayable GC operations. +//! +//! The original T06 graph migration intentionally supplied only the graph +//! rows. This forward-only migration adds the counter used by the atomic +//! LIVE→DELETING CAS and an idempotent operation log for crash recovery. The +//! runtime is not switched to this schema by this migration alone. + +use sea_orm_migration::{prelude::*, schema::*}; + +#[derive(DeriveMigrationName)] +pub struct Migration; + +#[async_trait::async_trait] +impl MigrationTrait for Migration { + async fn up(&self, manager: &SchemaManager) -> Result<(), DbErr> { + manager + .alter_table( + Table::alter() + .table(Mst2RetentionNode::Table) + .add_column( + big_integer(Mst2RetentionNode::IncomingRefs) + .not_null() + .default(0), + ) + .to_owned(), + ) + .await?; + + manager + .create_index( + Index::create() + .if_not_exists() + .name("idx_mst2_retention_node_gc") + .table(Mst2RetentionNode::Table) + .col(Mst2RetentionNode::State) + .col(Mst2RetentionNode::IncomingRefs) + .to_owned(), + ) + .await?; + + manager + .create_table( + Table::create() + .table(Mst2RetentionGcOp::Table) + .if_not_exists() + .col( + string(Mst2RetentionGcOp::OperationId) + .not_null() + .primary_key(), + ) + .col(string(Mst2RetentionGcOp::NodeId).not_null()) + .col(string(Mst2RetentionGcOp::Operation).not_null()) + .col( + string(Mst2RetentionGcOp::State) + .not_null() + .default("PENDING"), + ) + .col(integer(Mst2RetentionGcOp::Attempts).not_null().default(0)) + .col( + timestamp_with_time_zone(Mst2RetentionGcOp::CreatedAt) + .not_null() + .default(Expr::current_timestamp()), + ) + .col(timestamp_with_time_zone_null( + Mst2RetentionGcOp::CompletedAt, + )) + .to_owned(), + ) + .await?; + + manager + .create_index( + Index::create() + .if_not_exists() + .name("idx_mst2_retention_gc_op_pending") + .table(Mst2RetentionGcOp::Table) + .col(Mst2RetentionGcOp::State) + .col(Mst2RetentionGcOp::CreatedAt) + .to_owned(), + ) + .await?; + + manager + .create_index( + Index::create() + .if_not_exists() + .name("idx_mst2_retention_gc_op_node") + .table(Mst2RetentionGcOp::Table) + .col(Mst2RetentionGcOp::NodeId) + .to_owned(), + ) + .await + } + + async fn down(&self, _manager: &SchemaManager) -> Result<(), DbErr> { + // Forward-only: retention state and replay evidence are never + // dropped automatically. + Ok(()) + } +} + +#[derive(DeriveIden)] +enum Mst2RetentionNode { + Table, + State, + IncomingRefs, +} + +#[derive(DeriveIden)] +enum Mst2RetentionGcOp { + Table, + OperationId, + NodeId, + Operation, + State, + Attempts, + CreatedAt, + CompletedAt, +} diff --git a/src/jupiter/migration/m20261005_000200_add_mst2_native_head.rs b/src/jupiter/migration/m20261005_000200_add_mst2_native_head.rs new file mode 100644 index 00000000..e29118cf --- /dev/null +++ b/src/jupiter/migration/m20261005_000200_add_mst2_native_head.rs @@ -0,0 +1,72 @@ +use sea_orm_migration::prelude::*; + +#[derive(DeriveMigrationName)] +pub struct Migration; + +#[async_trait::async_trait] +impl MigrationTrait for Migration { + async fn up(&self, manager: &SchemaManager) -> Result<(), DbErr> { + manager + .get_connection() + .execute_unprepared( + r#" + ALTER TABLE mst2_publication + ADD COLUMN IF NOT EXISTS native_certificate_version integer; + ALTER TABLE push_queue + ADD COLUMN IF NOT EXISTS expected_native_sequence bigint, + ADD COLUMN IF NOT EXISTS expected_native_epoch bigint, + ADD COLUMN IF NOT EXISTS expected_native_certificate bigint; + CREATE TABLE IF NOT EXISTS mst2_native_publication ( + receipt_id bigint PRIMARY KEY REFERENCES mst2_publication(id), + namespace text NOT NULL CHECK (namespace = '/'), + instance_id text NOT NULL, + sequence bigint NOT NULL CHECK (sequence > 0), + writer_epoch bigint NOT NULL CHECK (writer_epoch > 0), + old_root_commit text NOT NULL, + old_root_tree text NOT NULL, + root_commit text NOT NULL, + root_tree text NOT NULL, + origin_path text NOT NULL, + origin_ref text NOT NULL, + old_path_commit text, + old_path_tree text, + path_commit text NOT NULL, + path_tree text NOT NULL, + CHECK ((old_path_commit IS NULL) = (old_path_tree IS NULL)), + CHECK (old_path_commit IS DISTINCT FROM path_commit), + UNIQUE (namespace, sequence) + ); + CREATE TABLE IF NOT EXISTS mst2_native_head ( + namespace text PRIMARY KEY CHECK (namespace = '/'), + instance_id text NOT NULL, + sequence bigint NOT NULL CHECK (sequence >= 0), + writer_epoch bigint NOT NULL CHECK (writer_epoch > 0), + root_commit text NOT NULL, + root_tree text NOT NULL, + state text NOT NULL CHECK (state IN ('INITIALIZING', 'READY')), + certificate_receipt_id bigint REFERENCES mst2_native_publication(receipt_id), + CHECK ((state = 'READY') = (certificate_receipt_id IS NOT NULL)) + ); + DO $$ BEGIN + ALTER TABLE mst2_publication ADD CONSTRAINT mst2_native_certificate_version_valid + CHECK (native_certificate_version IS NULL OR native_certificate_version > 0); + EXCEPTION WHEN duplicate_object THEN NULL; END $$; + DO $$ BEGIN + ALTER TABLE push_queue ADD CONSTRAINT mst2_expected_native_token_valid CHECK ( + (expected_native_sequence IS NULL AND expected_native_epoch IS NULL + AND expected_native_certificate IS NULL) OR + (expected_native_sequence IS NOT NULL AND expected_native_sequence >= 0 + AND expected_native_epoch IS NOT NULL AND expected_native_epoch > 0) + ); + EXCEPTION WHEN duplicate_object THEN NULL; END $$; + "#, + ) + .await?; + Ok(()) + } + + async fn down(&self, _manager: &SchemaManager) -> Result<(), DbErr> { + // Historical native certificates and queue fences are never erased. + Ok(()) + } +} diff --git a/src/jupiter/migration/m20261005_000200_harden_mst2_retention_graph.rs b/src/jupiter/migration/m20261005_000200_harden_mst2_retention_graph.rs new file mode 100644 index 00000000..c42bcd28 --- /dev/null +++ b/src/jupiter/migration/m20261005_000200_harden_mst2_retention_graph.rs @@ -0,0 +1,76 @@ +//! T06-B: backfill graph counters and reject invalid durable graph state. +//! +//! Applies after the additive schema migration. Existing edges must form a +//! valid DAG and counters are derived from unique edges before any collector +//! can use them. Constraints keep failed writers from silently dangling a +//! node or underflowing a counter. Completed GC receipts intentionally do not +//! reference the removed node through a foreign key. + +use sea_orm_migration::prelude::*; + +#[derive(DeriveMigrationName)] +pub struct Migration; + +#[async_trait::async_trait] +impl MigrationTrait for Migration { + async fn up(&self, manager: &SchemaManager) -> Result<(), DbErr> { + let conn = manager.get_connection(); + conn.execute_unprepared( + "LOCK TABLE mst2_retention_node, mst2_retention_edge, mst2_retention_root, \ + mst2_retention_gc_op IN ACCESS EXCLUSIVE MODE", + ) + .await?; + let cycle = conn + .query_one_raw(sea_orm::Statement::from_string( + sea_orm::DbBackend::Postgres, + "WITH RECURSIVE walk(start_id, node_id) AS ( \ + SELECT parent_id, child_id FROM mst2_retention_edge \ + UNION SELECT w.start_id, e.child_id FROM walk w \ + JOIN mst2_retention_edge e ON e.parent_id = w.node_id \ + ) SELECT start_id FROM walk WHERE start_id = node_id LIMIT 1", + )) + .await?; + if cycle.is_some() { + return Err(DbErr::Migration( + "MST/2 retention graph contains a cycle".into(), + )); + } + conn.execute_unprepared( + "UPDATE mst2_retention_node n SET incoming_refs = \ + (SELECT count(*) FROM mst2_retention_edge e WHERE e.child_id = n.node_id)", + ) + .await?; + conn.execute_unprepared( + "ALTER TABLE mst2_retention_node \ + ADD CONSTRAINT mst2_retention_node_state_check CHECK (state IN ('LIVE', 'DELETING')), \ + ADD CONSTRAINT mst2_retention_node_kind_check \ + CHECK (kind IN ('page', 'chunk_map', 'frame', 'verified_object')), \ + ADD CONSTRAINT mst2_retention_node_nonnegative CHECK (bytes >= 0 AND incoming_refs >= 0); \ + ALTER TABLE mst2_retention_edge \ + ADD CONSTRAINT mst2_retention_edge_parent_fk FOREIGN KEY (parent_id) \ + REFERENCES mst2_retention_node(node_id), \ + ADD CONSTRAINT mst2_retention_edge_child_fk FOREIGN KEY (child_id) \ + REFERENCES mst2_retention_node(node_id), \ + ADD CONSTRAINT mst2_retention_edge_no_self CHECK (parent_id <> child_id); \ + ALTER TABLE mst2_retention_root \ + ADD CONSTRAINT mst2_retention_root_node_fk FOREIGN KEY (node_id) \ + REFERENCES mst2_retention_node(node_id), \ + ADD CONSTRAINT mst2_retention_root_kind_check CHECK (root_kind IN ('lease', 'pin', 'prepare')); \ + ALTER TABLE mst2_retention_gc_op \ + ADD CONSTRAINT mst2_retention_gc_op_kind_check CHECK (operation IN ('MARK_DELETING', 'REMOVE')), \ + ADD CONSTRAINT mst2_retention_gc_op_state_check CHECK (state IN ('PENDING', 'APPLIED', 'FAILED')), \ + ADD CONSTRAINT mst2_retention_gc_op_attempts_check CHECK (attempts >= 0), \ + ADD CONSTRAINT mst2_retention_gc_op_completion_check \ + CHECK ((state = 'APPLIED') = (completed_at IS NOT NULL)); \ + CREATE UNIQUE INDEX idx_mst2_retention_gc_op_pending_node \ + ON mst2_retention_gc_op(node_id) WHERE state = 'PENDING'", + ) + .await + .map(|_| ()) + } + + async fn down(&self, _manager: &SchemaManager) -> Result<(), DbErr> { + // Forward-only: preserve retained nodes and GC recovery evidence. + Ok(()) + } +} diff --git a/src/jupiter/migration/m20261005_000300_add_mst2_metadata_install.rs b/src/jupiter/migration/m20261005_000300_add_mst2_metadata_install.rs new file mode 100644 index 00000000..d0e8d168 --- /dev/null +++ b/src/jupiter/migration/m20261005_000300_add_mst2_metadata_install.rs @@ -0,0 +1,76 @@ +//! Additive, forward-only native metadata installation records. + +use mst2_codec::metapage::{HEADER_LEN, PAGE_MAX_BYTES}; +use sea_orm_migration::prelude::*; + +#[derive(DeriveMigrationName)] +pub struct Migration; + +#[async_trait::async_trait] +impl MigrationTrait for Migration { + async fn up(&self, manager: &SchemaManager) -> Result<(), DbErr> { + manager.get_connection().execute_unprepared(&format!( + "CREATE TABLE mst2_metadata_storage_scope ( + singleton smallint PRIMARY KEY CHECK (singleton = 1), + storage_uuid text NOT NULL UNIQUE + ); + CREATE TABLE mst2_metadata_payload ( + page_id bytea PRIMARY KEY CHECK (octet_length(page_id) = 32), + metadata_codec smallint NOT NULL CHECK (metadata_codec = 1), + byte_size integer NOT NULL CHECK (byte_size BETWEEN {HEADER_LEN} AND {PAGE_MAX_BYTES}), + payload bytea NOT NULL CHECK (octet_length(payload) = byte_size), + created_at timestamptz NOT NULL DEFAULT now() + ); + CREATE OR REPLACE FUNCTION mst2_metadata_payload_immutable() RETURNS trigger LANGUAGE plpgsql AS $$ + BEGIN RAISE EXCEPTION 'MST2 metadata payloads cannot be updated or deleted'; END $$; + CREATE TRIGGER mst2_metadata_payload_immutable BEFORE UPDATE OR DELETE + ON mst2_metadata_payload FOR EACH ROW EXECUTE FUNCTION mst2_metadata_payload_immutable(); + CREATE TRIGGER mst2_metadata_scope_immutable BEFORE UPDATE OR DELETE + ON mst2_metadata_storage_scope FOR EACH ROW EXECUTE FUNCTION mst2_metadata_payload_immutable(); + CREATE TABLE mst2_metadata_prepare ( + prepare_id text PRIMARY KEY CHECK (prepare_id ~ '^[0-9a-f]{{8}}-[0-9a-f]{{4}}-4[0-9a-f]{{3}}-[89ab][0-9a-f]{{3}}-[0-9a-f]{{12}}$'), + operation_id text NOT NULL UNIQUE CHECK (octet_length(operation_id) BETWEEN 1 AND 255), + manifest_digest bytea NOT NULL CHECK (octet_length(manifest_digest) = 32), + canonical_plan bytea NOT NULL CHECK (octet_length(canonical_plan) <= 2097152), + source_domain text NOT NULL CHECK (source_domain = 'native-git'), + tagged_root_tree_oid text NOT NULL, + scope text NOT NULL CHECK (octet_length(scope) <= 4096), + schema_version smallint NOT NULL, + metadata_codec smallint NOT NULL CHECK (metadata_codec = 1), + materialization_policy smallint NOT NULL, + fs_semantics smallint NOT NULL, + access_projection smallint NOT NULL, + verification_revision integer NOT NULL, + projection_revision smallint NOT NULL, + metadata_root bytea NOT NULL CHECK (octet_length(metadata_root) = 32), + node_count integer NOT NULL CHECK (node_count BETWEEN 1 AND 4096), + edge_count integer NOT NULL CHECK (edge_count BETWEEN 0 AND 16384), + total_bytes bigint NOT NULL CHECK (total_bytes BETWEEN 0 AND 67108864), + state text NOT NULL CHECK (state IN ('PREPARING', 'COMMITTED')), + created_at timestamptz NOT NULL DEFAULT now(), + committed_at timestamptz, + CHECK ((state = 'COMMITTED') = (committed_at IS NOT NULL)) + ); + CREATE TABLE mst2_metadata_prepare_page ( + prepare_id text NOT NULL REFERENCES mst2_metadata_prepare(prepare_id), + page_id bytea NOT NULL CHECK (octet_length(page_id) = 32), + expected_size integer NOT NULL CHECK (expected_size BETWEEN {HEADER_LEN} AND {PAGE_MAX_BYTES}), + PRIMARY KEY (prepare_id, page_id) + )" + )).await?; + manager + .get_connection() + .execute_raw(sea_orm::Statement::from_sql_and_values( + sea_orm::DbBackend::Postgres, + "INSERT INTO mst2_metadata_storage_scope(singleton,storage_uuid) VALUES(1,$1)", + [uuid::Uuid::new_v4().to_string().into()], + )) + .await + .map(|_| ()) + } + + async fn down(&self, _manager: &SchemaManager) -> Result<(), DbErr> { + // Preserve immutable payloads, prepare pins and recovery evidence. + Ok(()) + } +} diff --git a/src/jupiter/migration/m20261007_000100_add_mst2_snapshot_sessions.rs b/src/jupiter/migration/m20261007_000100_add_mst2_snapshot_sessions.rs new file mode 100644 index 00000000..be67efa5 --- /dev/null +++ b/src/jupiter/migration/m20261007_000100_add_mst2_snapshot_sessions.rs @@ -0,0 +1,72 @@ +use sea_orm_migration::prelude::*; + +#[derive(DeriveMigrationName)] +pub struct Migration; + +#[async_trait::async_trait] +impl MigrationTrait for Migration { + async fn up(&self, manager: &SchemaManager) -> Result<(), DbErr> { + manager.get_connection().execute_unprepared( + "CREATE TABLE mst2_snapshot_context ( + snapshot_id text PRIMARY KEY, + canonical_descriptor bytea NOT NULL, + instance_id text NOT NULL, + commit_oid text NOT NULL, + root_tree_oid text NOT NULL, + metadata_root bytea NOT NULL CHECK (octet_length(metadata_root)=32), + prepare_id text NOT NULL UNIQUE REFERENCES mst2_metadata_prepare(prepare_id), + publication_sequence bigint NOT NULL CHECK (publication_sequence>=0), + writer_epoch bigint NOT NULL CHECK (writer_epoch>0), + certificate_receipt_id bigint REFERENCES mst2_native_publication(receipt_id), + authorization_epoch bigint NOT NULL DEFAULT 1 CHECK (authorization_epoch>0), + state text NOT NULL CHECK (state IN ('READY','DISABLED')), + created_at timestamptz NOT NULL DEFAULT clock_timestamp() + ); + CREATE TABLE mst2_snapshot_lease ( + lease_id text PRIMARY KEY, + snapshot_id text NOT NULL REFERENCES mst2_snapshot_context(snapshot_id), + authorization_epoch bigint NOT NULL CHECK (authorization_epoch>0), + publication_sequence bigint NOT NULL CHECK (publication_sequence>0), + writer_epoch bigint NOT NULL CHECK (writer_epoch>0), + certificate_receipt_id bigint NOT NULL REFERENCES mst2_native_publication(receipt_id), + expires_at_unix bigint NOT NULL CHECK (expires_at_unix>=0), + state text NOT NULL CHECK (state IN ('ACTIVE','RELEASED','EXPIRED')), + created_at timestamptz NOT NULL DEFAULT clock_timestamp() + ); + CREATE FUNCTION mst2_snapshot_lease_identity_immutable() RETURNS trigger LANGUAGE plpgsql AS $$ + BEGIN + IF NEW.lease_id<>OLD.lease_id OR NEW.snapshot_id<>OLD.snapshot_id + OR NEW.authorization_epoch<>OLD.authorization_epoch OR NEW.publication_sequence<>OLD.publication_sequence + OR NEW.writer_epoch<>OLD.writer_epoch OR NEW.certificate_receipt_id<>OLD.certificate_receipt_id + OR (OLD.state<>'ACTIVE' AND NEW.state<>OLD.state) THEN + RAISE EXCEPTION 'MST2 lease source cannot change or be revived'; + END IF; + RETURN NEW; + END $$; + CREATE TRIGGER mst2_snapshot_lease_identity_immutable BEFORE UPDATE ON mst2_snapshot_lease + FOR EACH ROW EXECUTE FUNCTION mst2_snapshot_lease_identity_immutable(); + CREATE INDEX mst2_snapshot_lease_active ON mst2_snapshot_lease(snapshot_id,expires_at_unix) + WHERE state='ACTIVE'; + CREATE INDEX mst2_snapshot_lease_expiry ON mst2_snapshot_lease(expires_at_unix) + WHERE state='ACTIVE'; + CREATE FUNCTION mst2_snapshot_identity_immutable() RETURNS trigger LANGUAGE plpgsql AS $$ + BEGIN + IF NEW.snapshot_id<>OLD.snapshot_id OR NEW.canonical_descriptor<>OLD.canonical_descriptor + OR NEW.instance_id<>OLD.instance_id OR NEW.commit_oid<>OLD.commit_oid + OR NEW.root_tree_oid<>OLD.root_tree_oid OR NEW.metadata_root<>OLD.metadata_root + OR NEW.prepare_id<>OLD.prepare_id OR NEW.publication_sequence<>OLD.publication_sequence + OR NEW.writer_epoch<>OLD.writer_epoch + OR NEW.certificate_receipt_id IS DISTINCT FROM OLD.certificate_receipt_id THEN + RAISE EXCEPTION 'MST2 fixed snapshot identity cannot be changed'; + END IF; + RETURN NEW; + END $$; + CREATE TRIGGER mst2_snapshot_identity_immutable BEFORE UPDATE ON mst2_snapshot_context + FOR EACH ROW EXECUTE FUNCTION mst2_snapshot_identity_immutable();" + ).await.map(|_| ()) + } + + async fn down(&self, _manager: &SchemaManager) -> Result<(), DbErr> { + Ok(()) + } +} diff --git a/src/jupiter/migration/m20261007_000200_add_mst2_metadata_generations.rs b/src/jupiter/migration/m20261007_000200_add_mst2_metadata_generations.rs new file mode 100644 index 00000000..2c098c19 --- /dev/null +++ b/src/jupiter/migration/m20261007_000200_add_mst2_metadata_generations.rs @@ -0,0 +1,91 @@ +//! Fixed metadata lifetimes. Legacy rows remain explicitly unbound. + +use mst2_codec::metapage::{HEADER_LEN, PAGE_MAX_BYTES}; +use sea_orm_migration::prelude::*; + +#[derive(DeriveMigrationName)] +pub struct Migration; + +#[async_trait::async_trait] +impl MigrationTrait for Migration { + async fn up(&self, manager: &SchemaManager) -> Result<(), DbErr> { + manager.get_connection().execute_unprepared(&format!( + "CREATE TABLE mst2_metadata_lifetime ( + page_id bytea PRIMARY KEY CHECK (octet_length(page_id)=32), + node_id text NOT NULL UNIQUE CHECK (node_id='page:sha256:'||encode(page_id,'hex')), + generation bigint NOT NULL CHECK (generation>0), + state text NOT NULL CHECK (state IN ('RESERVED','LIVE','DELETING','REMOVED')), + metadata_codec smallint NOT NULL CHECK (metadata_codec=1), + expected_size integer NOT NULL CHECK (expected_size BETWEEN {HEADER_LEN} AND {PAGE_MAX_BYTES}), + UNIQUE(page_id,generation) + ); + CREATE INDEX idx_mst2_metadata_lifetime_state_page ON mst2_metadata_lifetime(state,page_id); + CREATE FUNCTION mst2_metadata_lifetime_guard() RETURNS trigger LANGUAGE plpgsql AS $$ + BEGIN + IF TG_OP='DELETE' THEN RAISE EXCEPTION 'metadata lifetime watermark cannot be deleted'; END IF; + IF NEW.page_id IS DISTINCT FROM OLD.page_id OR NEW.node_id IS DISTINCT FROM OLD.node_id + OR NEW.generation IS DISTINCT FROM OLD.generation + OR NEW.metadata_codec IS DISTINCT FROM OLD.metadata_codec + OR NEW.expected_size IS DISTINCT FROM OLD.expected_size THEN + RAISE EXCEPTION 'metadata lifetime identity is immutable'; + END IF; + IF NEW.state IS DISTINCT FROM OLD.state AND NOT (OLD.state='RESERVED' AND NEW.state='LIVE') THEN + RAISE EXCEPTION 'metadata lifetime transition requires generation collector'; + END IF; + RETURN NEW; + END $$; + CREATE TRIGGER mst2_metadata_lifetime_guard BEFORE UPDATE OR DELETE + ON mst2_metadata_lifetime FOR EACH ROW EXECUTE FUNCTION mst2_metadata_lifetime_guard(); + ALTER TABLE mst2_metadata_payload ADD COLUMN generation bigint CHECK (generation>0), + ADD CONSTRAINT mst2_metadata_payload_lifetime_fk FOREIGN KEY(page_id,generation) + REFERENCES mst2_metadata_lifetime(page_id,generation); + ALTER TABLE mst2_metadata_prepare_page ADD COLUMN generation bigint CHECK (generation>0), + ADD CONSTRAINT mst2_metadata_prepare_page_lifetime_fk FOREIGN KEY(page_id,generation) + REFERENCES mst2_metadata_lifetime(page_id,generation); + CREATE INDEX idx_mst2_metadata_prepare_page_lifetime + ON mst2_metadata_prepare_page(page_id,generation,prepare_id); + ALTER TABLE mst2_metadata_prepare + ADD COLUMN canonical_bindings bytea, + ADD COLUMN bindings_digest bytea CHECK (octet_length(bindings_digest)=32), + ADD COLUMN primary_scope bytea CHECK (octet_length(primary_scope) BETWEEN 1 AND 16384), + ADD COLUMN storage_seal bytea CHECK (octet_length(storage_seal)=32), + ADD CONSTRAINT mst2_metadata_prepare_generation_seal_check CHECK ( + (canonical_bindings IS NULL AND bindings_digest IS NULL AND primary_scope IS NULL AND storage_seal IS NULL) + OR (canonical_bindings IS NOT NULL AND octet_length(canonical_bindings) BETWEEN 60 AND 196620 + AND bindings_digest IS NOT NULL AND primary_scope IS NOT NULL AND storage_seal IS NOT NULL) + ); + CREATE FUNCTION mst2_metadata_generation_seal_guard() RETURNS trigger LANGUAGE plpgsql AS $$ + BEGIN + IF NEW.canonical_bindings IS DISTINCT FROM OLD.canonical_bindings + OR NEW.bindings_digest IS DISTINCT FROM OLD.bindings_digest + OR NEW.primary_scope IS DISTINCT FROM OLD.primary_scope + OR NEW.storage_seal IS DISTINCT FROM OLD.storage_seal THEN + RAISE EXCEPTION 'metadata generation seal cannot be rebound'; + END IF; + RETURN NEW; + END $$; + CREATE TRIGGER mst2_metadata_generation_seal_guard BEFORE UPDATE ON mst2_metadata_prepare + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_generation_seal_guard(); + CREATE FUNCTION mst2_metadata_generation_mapping_guard() RETURNS trigger LANGUAGE plpgsql AS $$ + DECLARE fixed boolean; preparing boolean; + BEGIN + SELECT storage_seal IS NOT NULL,state='PREPARING' INTO fixed,preparing + FROM mst2_metadata_prepare WHERE prepare_id=CASE WHEN TG_OP='DELETE' THEN OLD.prepare_id ELSE NEW.prepare_id END FOR UPDATE; + IF fixed AND (TG_OP<>'INSERT' OR NOT preparing) THEN + RAISE EXCEPTION 'metadata generation mappings are immutable'; + END IF; + IF TG_OP='UPDATE' AND OLD.prepare_id IS DISTINCT FROM NEW.prepare_id AND EXISTS ( + SELECT 1 FROM mst2_metadata_prepare WHERE prepare_id=OLD.prepare_id AND storage_seal IS NOT NULL + ) THEN RAISE EXCEPTION 'metadata generation mapping cannot leave its prepare'; END IF; + IF TG_OP='DELETE' THEN RETURN OLD; END IF; + RETURN NEW; + END $$; + CREATE TRIGGER mst2_metadata_generation_mapping_guard BEFORE INSERT OR UPDATE OR DELETE + ON mst2_metadata_prepare_page FOR EACH ROW EXECUTE FUNCTION mst2_metadata_generation_mapping_guard()" + )).await.map(|_| ()) + } + + async fn down(&self, _manager: &SchemaManager) -> Result<(), DbErr> { + Ok(()) + } +} diff --git a/src/jupiter/migration/m20261007_000300_add_mst2_metadata_lifetime_history.rs b/src/jupiter/migration/m20261007_000300_add_mst2_metadata_lifetime_history.rs new file mode 100644 index 00000000..51dcee8f --- /dev/null +++ b/src/jupiter/migration/m20261007_000300_add_mst2_metadata_lifetime_history.rs @@ -0,0 +1,67 @@ +//! Preserve historical incarnations and explicit preparation terminal states. + +use sea_orm_migration::prelude::*; + +#[derive(DeriveMigrationName)] +pub struct Migration; + +#[async_trait::async_trait] +impl MigrationTrait for Migration { + async fn up(&self, manager: &SchemaManager) -> Result<(), DbErr> { + manager.get_connection().execute_unprepared( + "ALTER TABLE mst2_metadata_lifetime + DROP CONSTRAINT mst2_metadata_lifetime_pkey, + DROP CONSTRAINT mst2_metadata_lifetime_node_id_key, + ADD PRIMARY KEY(page_id,generation); + CREATE TABLE mst2_metadata_current ( + page_id bytea PRIMARY KEY CHECK (octet_length(page_id)=32), + generation bigint NOT NULL CHECK (generation>0), + FOREIGN KEY(page_id,generation) REFERENCES mst2_metadata_lifetime(page_id,generation) + ); + INSERT INTO mst2_metadata_current(page_id,generation) + SELECT page_id,generation FROM mst2_metadata_lifetime; + CREATE FUNCTION mst2_metadata_current_guard() RETURNS trigger LANGUAGE plpgsql AS $$ + BEGIN RAISE EXCEPTION 'metadata current watermark requires generation-fenced collection'; END $$; + CREATE TRIGGER mst2_metadata_current_guard BEFORE UPDATE OR DELETE + ON mst2_metadata_current FOR EACH ROW EXECUTE FUNCTION mst2_metadata_current_guard(); + ALTER TABLE mst2_metadata_prepare + ADD COLUMN graph_domain text CHECK (graph_domain IN ('generic-v1','qualified-v1')), + ADD COLUMN aborted_at timestamptz, + ADD COLUMN coverage_retired_at timestamptz, + DROP CONSTRAINT mst2_metadata_prepare_state_check, + DROP CONSTRAINT mst2_metadata_prepare_check, + ADD CONSTRAINT mst2_metadata_prepare_state_check CHECK (state IN ('PREPARING','COMMITTED','ABORTED')), + ADD CONSTRAINT mst2_metadata_prepare_terminal_check CHECK ( + (state='COMMITTED')=(committed_at IS NOT NULL) + AND (state='ABORTED')=(aborted_at IS NOT NULL) + AND (coverage_retired_at IS NULL OR state='COMMITTED') + AND (state<>'ABORTED' OR storage_seal IS NOT NULL) + AND (graph_domain IS NULL OR storage_seal IS NOT NULL) + ); + CREATE OR REPLACE FUNCTION mst2_metadata_generation_seal_guard() RETURNS trigger LANGUAGE plpgsql AS $$ + BEGIN + IF NEW.canonical_bindings IS DISTINCT FROM OLD.canonical_bindings + OR NEW.bindings_digest IS DISTINCT FROM OLD.bindings_digest + OR NEW.primary_scope IS DISTINCT FROM OLD.primary_scope + OR NEW.storage_seal IS DISTINCT FROM OLD.storage_seal + OR NEW.graph_domain IS DISTINCT FROM OLD.graph_domain THEN + RAISE EXCEPTION 'metadata generation seal cannot be rebound'; + END IF; + IF OLD.state IN ('COMMITTED','ABORTED') AND NEW.state IS DISTINCT FROM OLD.state THEN + RAISE EXCEPTION 'metadata preparation terminal state cannot be revived'; + END IF; + IF OLD.committed_at IS NOT NULL AND NEW.committed_at IS DISTINCT FROM OLD.committed_at + OR OLD.aborted_at IS NOT NULL AND NEW.aborted_at IS DISTINCT FROM OLD.aborted_at + OR OLD.coverage_retired_at IS NOT NULL AND NEW.coverage_retired_at IS DISTINCT FROM OLD.coverage_retired_at THEN + RAISE EXCEPTION 'metadata preparation terminal receipt cannot be rewritten'; + END IF; + RETURN NEW; + END $$; + CREATE INDEX idx_mst2_metadata_lifetime_node_generation ON mst2_metadata_lifetime(node_id,generation)" + ).await.map(|_| ()) + } + + async fn down(&self, _manager: &SchemaManager) -> Result<(), DbErr> { + Ok(()) + } +} diff --git a/src/jupiter/migration/m20261007_000400_add_mst2_qualified_metadata_gc.rs b/src/jupiter/migration/m20261007_000400_add_mst2_qualified_metadata_gc.rs new file mode 100644 index 00000000..0faacb16 --- /dev/null +++ b/src/jupiter/migration/m20261007_000400_add_mst2_qualified_metadata_gc.rs @@ -0,0 +1,22 @@ +//! Exact-incarnation metadata graph and atomic, replayable payload collection. + +use mst2_codec::metapage::{HEADER_LEN, PAGE_MAX_BYTES}; +use sea_orm_migration::prelude::*; + +#[derive(DeriveMigrationName)] +pub struct Migration; + +#[async_trait::async_trait] +impl MigrationTrait for Migration { + async fn up(&self, manager: &SchemaManager) -> Result<(), DbErr> { + let sql = include_str!("m20261007_000400_qualified_metadata_gc.sql") + .replace("$HEADER_LEN$", &HEADER_LEN.to_string()) + .replace("$PAGE_MAX_BYTES$", &PAGE_MAX_BYTES.to_string()); + manager.get_connection().execute_unprepared(&sql).await?; + Ok(()) + } + + async fn down(&self, _manager: &SchemaManager) -> Result<(), DbErr> { + Ok(()) + } +} diff --git a/src/jupiter/migration/m20261007_000400_qualified_metadata_gc.sql b/src/jupiter/migration/m20261007_000400_qualified_metadata_gc.sql new file mode 100644 index 00000000..a2a84cf0 --- /dev/null +++ b/src/jupiter/migration/m20261007_000400_qualified_metadata_gc.sql @@ -0,0 +1,639 @@ +LOCK TABLE mst2_metadata_lifetime,mst2_metadata_current,mst2_metadata_payload, + mst2_metadata_prepare,mst2_metadata_prepare_page,mst2_retention_node, + mst2_retention_edge,mst2_retention_root,mst2_retention_gc_op, + mst2_snapshot_context,mst2_snapshot_lease IN ACCESS EXCLUSIVE MODE; + +ALTER TABLE mst2_metadata_lifetime ADD COLUMN graph_domain text NOT NULL DEFAULT 'generic-v1' + CHECK (graph_domain IN ('generic-v1','qualified-v1')); +ALTER TABLE mst2_metadata_prepare_page ADD UNIQUE(prepare_id,page_id,generation); +ALTER TABLE mst2_metadata_prepare ADD UNIQUE(prepare_id,storage_seal); +CREATE INDEX idx_mst2_snapshot_context_metadata_root ON mst2_snapshot_context(metadata_root,snapshot_id); +CREATE INDEX idx_mst2_metadata_lifetime_qualified_scan ON mst2_metadata_lifetime(page_id,generation) + WHERE graph_domain='qualified-v1' AND state IN ('RESERVED','LIVE'); + +CREATE TABLE mst2_metadata_graph_node ( + page_id bytea NOT NULL CHECK (octet_length(page_id)=32), + generation bigint NOT NULL CHECK (generation>0), + state text NOT NULL CHECK (state IN ('LIVE','DELETING')), + metadata_codec smallint NOT NULL CHECK (metadata_codec=1), + bytes bigint NOT NULL CHECK (bytes BETWEEN $HEADER_LEN$ AND $PAGE_MAX_BYTES$), + incoming_refs bigint NOT NULL DEFAULT 0 CHECK (incoming_refs>=0), + PRIMARY KEY(page_id,generation), + FOREIGN KEY(page_id,generation) REFERENCES mst2_metadata_lifetime(page_id,generation) +); +CREATE INDEX idx_mst2_metadata_graph_node_gc ON mst2_metadata_graph_node(state,incoming_refs,page_id,generation); +CREATE TABLE mst2_metadata_graph_edge ( + parent_page bytea NOT NULL, + parent_generation bigint NOT NULL, + child_page bytea NOT NULL, + child_generation bigint NOT NULL, + PRIMARY KEY(parent_page,parent_generation,child_page,child_generation), + FOREIGN KEY(parent_page,parent_generation) REFERENCES mst2_metadata_graph_node(page_id,generation), + FOREIGN KEY(child_page,child_generation) REFERENCES mst2_metadata_graph_node(page_id,generation), + CHECK (parent_page<>child_page OR parent_generation<>child_generation) +); +CREATE INDEX idx_mst2_metadata_graph_edge_child ON mst2_metadata_graph_edge(child_page,child_generation,parent_page,parent_generation); +CREATE TABLE mst2_metadata_graph_root ( + prepare_id text NOT NULL, + storage_seal bytea NOT NULL CHECK (octet_length(storage_seal)=32), + page_id bytea NOT NULL, + generation bigint NOT NULL, + PRIMARY KEY(prepare_id,page_id,generation), + FOREIGN KEY(prepare_id,storage_seal) REFERENCES mst2_metadata_prepare(prepare_id,storage_seal), + FOREIGN KEY(prepare_id,page_id,generation) REFERENCES mst2_metadata_prepare_page(prepare_id,page_id,generation), + FOREIGN KEY(page_id,generation) REFERENCES mst2_metadata_graph_node(page_id,generation) +); +CREATE INDEX idx_mst2_metadata_graph_root_page ON mst2_metadata_graph_root(page_id,generation,prepare_id); +CREATE TABLE mst2_metadata_gc_op ( + operation_id uuid PRIMARY KEY, + page_id bytea NOT NULL CHECK (octet_length(page_id)=32), + generation bigint NOT NULL CHECK (generation>0), + primary_scope bytea NOT NULL CHECK (octet_length(primary_scope) BETWEEN 1 AND 16384), + graph_domain text NOT NULL CHECK (graph_domain='qualified-v1'), + metadata_codec smallint NOT NULL CHECK (metadata_codec=1), + expected_size integer NOT NULL CHECK (expected_size BETWEEN $HEADER_LEN$ AND $PAGE_MAX_BYTES$), + graph_present boolean NOT NULL, + had_payload boolean NOT NULL, + payload_delete_xid bigint, + state text NOT NULL CHECK (state IN ('PENDING','APPLIED')), + created_at timestamptz NOT NULL DEFAULT clock_timestamp(), + completed_at timestamptz, + UNIQUE(page_id,generation), + FOREIGN KEY(page_id,generation) REFERENCES mst2_metadata_lifetime(page_id,generation), + CHECK ((state='APPLIED')=(completed_at IS NOT NULL)) +); +CREATE INDEX idx_mst2_metadata_gc_op_pending ON mst2_metadata_gc_op(created_at,operation_id) WHERE state='PENDING'; + +CREATE FUNCTION mst2_metadata_dml_barrier() RETURNS trigger LANGUAGE plpgsql VOLATILE AS $$ +BEGIN + IF pg_is_in_recovery() OR current_setting('transaction_isolation')<>'read committed' THEN + RAISE EXCEPTION 'metadata mutation requires primary READ COMMITTED'; + END IF; + PERFORM set_config('lock_timeout','5000ms',true); + PERFORM pg_advisory_xact_lock(1296717362,hashtext(current_schema())); + RETURN NULL; +END $$; +DO $$ DECLARE t text; BEGIN + FOREACH t IN ARRAY ARRAY['mst2_metadata_lifetime','mst2_metadata_current','mst2_metadata_payload', + 'mst2_metadata_prepare','mst2_metadata_prepare_page','mst2_metadata_graph_node', + 'mst2_metadata_graph_edge','mst2_metadata_graph_root','mst2_metadata_gc_op', + 'mst2_retention_node','mst2_retention_edge','mst2_retention_root','mst2_retention_gc_op', + 'mst2_snapshot_context','mst2_snapshot_lease'] LOOP + EXECUTE format('CREATE TRIGGER mst2_metadata_statement_barrier BEFORE INSERT OR UPDATE OR DELETE ON %I FOR EACH STATEMENT EXECUTE FUNCTION mst2_metadata_dml_barrier()',t); + END LOOP; +END $$; + +CREATE FUNCTION mst2_metadata_scope_matches(s bytea) RETURNS boolean LANGUAGE plpgsql VOLATILE AS $$ +DECLARE actual jsonb; +BEGIN + IF pg_is_in_recovery() OR current_setting('transaction_isolation')<>'read committed' THEN RETURN false; END IF; + SELECT jsonb_build_array(x.storage_uuid,current_database(),d.oid::bigint,current_schema(),n.oid::bigint, + inet_server_addr()::text,inet_server_port()) INTO actual + FROM mst2_metadata_storage_scope x JOIN pg_database d ON d.datname=current_database() + JOIN pg_namespace n ON n.nspname=current_schema() WHERE x.singleton=1; + RETURN actual IS NOT NULL AND convert_from(s,'UTF8')::jsonb=actual; +EXCEPTION WHEN OTHERS THEN RETURN false; +END $$; + +CREATE FUNCTION mst2_metadata_assert_uncovered(p bytea,g bigint) RETURNS void LANGUAGE plpgsql VOLATILE AS $$ +DECLARE nid text:='page:sha256:'||encode(p,'hex'); +BEGIN + IF EXISTS(SELECT 1 FROM mst2_metadata_graph_root WHERE page_id=p AND generation=g) + OR EXISTS(SELECT 1 FROM mst2_metadata_graph_edge WHERE child_page=p AND child_generation=g) + OR EXISTS(SELECT 1 FROM mst2_retention_node WHERE node_id=nid) + OR EXISTS(SELECT 1 FROM mst2_retention_edge WHERE parent_id=nid OR child_id=nid) + OR EXISTS(SELECT 1 FROM mst2_retention_root WHERE node_id=nid) + OR EXISTS(SELECT 1 FROM mst2_retention_gc_op WHERE node_id=nid) + OR EXISTS(SELECT 1 FROM mst2_metadata_prepare_page m JOIN mst2_metadata_prepare q USING(prepare_id) + WHERE m.page_id=p AND (q.state='PREPARING' OR (q.state='COMMITTED' AND q.coverage_retired_at IS NULL))) + OR EXISTS(SELECT 1 FROM mst2_snapshot_context WHERE metadata_root=p) + OR EXISTS(SELECT 1 FROM mst2_metadata_prepare_page m JOIN mst2_snapshot_context s USING(prepare_id) + WHERE m.page_id=p) THEN + RAISE EXCEPTION 'metadata lifetime has durable or ambiguous coverage'; + END IF; +END $$; + +CREATE FUNCTION mst2_metadata_gc_proof(p bytea,g bigint,stage text,graph_present boolean,payload_present boolean) +RETURNS void LANGUAGE plpgsql VOLATILE AS $$ +DECLARE l mst2_metadata_lifetime%ROWTYPE; n mst2_metadata_graph_node%ROWTYPE; b mst2_metadata_payload%ROWTYPE; +BEGIN + SELECT h.* INTO l FROM mst2_metadata_current c JOIN mst2_metadata_lifetime h USING(page_id,generation) + WHERE c.page_id=p AND c.generation=g FOR UPDATE OF c,h; + IF NOT FOUND OR l.graph_domain<>'qualified-v1' + OR (stage='CLAIM' AND l.state NOT IN ('RESERVED','LIVE')) + OR (stage='PENDING' AND l.state<>'DELETING') + OR (stage='REMOVED' AND l.state<>'REMOVED') + OR stage NOT IN ('CLAIM','PENDING','REMOVED') THEN + RAISE EXCEPTION 'metadata GC is not the exact current qualified lifetime'; + END IF; + PERFORM mst2_metadata_assert_uncovered(p,g); + SELECT * INTO n FROM mst2_metadata_graph_node WHERE page_id=p AND generation=g FOR UPDATE; + IF FOUND IS DISTINCT FROM graph_present THEN RAISE EXCEPTION 'metadata GC graph presence changed'; END IF; + IF graph_present AND (n.metadata_codec<>l.metadata_codec OR n.bytes<>l.expected_size OR n.incoming_refs<>0 + OR (stage='CLAIM' AND n.state<>'LIVE') OR (stage='PENDING' AND n.state<>'DELETING')) THEN + RAISE EXCEPTION 'metadata GC graph profile or counter changed'; + END IF; + IF graph_present AND (SELECT count(*) FROM (SELECT 1 FROM mst2_metadata_graph_edge + WHERE parent_page=p AND parent_generation=g LIMIT 16385) bounded)>16384 THEN + RAISE EXCEPTION 'metadata GC outgoing edge limit exceeded'; + END IF; + SELECT * INTO b FROM mst2_metadata_payload WHERE page_id=p FOR UPDATE; + IF FOUND IS DISTINCT FROM payload_present THEN RAISE EXCEPTION 'metadata GC payload presence changed'; END IF; + IF payload_present AND (b.generation IS DISTINCT FROM g OR b.metadata_codec<>l.metadata_codec OR b.byte_size<>l.expected_size) THEN + RAISE EXCEPTION 'metadata GC payload profile or generation changed'; + END IF; + IF stage='CLAIM' AND l.state='LIVE' AND NOT (graph_present AND payload_present) THEN + RAISE EXCEPTION 'LIVE metadata corruption is not collectable'; + END IF; + IF stage='CLAIM' AND l.state='RESERVED' AND (graph_present OR NOT EXISTS( + SELECT 1 FROM mst2_metadata_prepare_page m JOIN mst2_metadata_prepare q USING(prepare_id) + WHERE m.page_id=p AND m.generation=g AND q.graph_domain='qualified-v1' + AND q.state='ABORTED' AND q.storage_seal IS NOT NULL AND m.expected_size=l.expected_size) + OR EXISTS(SELECT 1 FROM mst2_metadata_prepare_page m JOIN mst2_metadata_prepare q USING(prepare_id) + WHERE m.page_id=p AND m.generation=g AND q.state<>'ABORTED')) THEN + RAISE EXCEPTION 'RESERVED metadata needs an explicitly aborted qualified origin'; + END IF; +END $$; + +CREATE FUNCTION mst2_metadata_gc_op_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE AS $$ +DECLARE l mst2_metadata_lifetime%ROWTYPE; +BEGIN + IF TG_OP='DELETE' THEN RAISE EXCEPTION 'metadata GC evidence cannot be deleted'; END IF; + IF NOT mst2_metadata_scope_matches(NEW.primary_scope) THEN RAISE EXCEPTION 'metadata GC wrong primary scope'; END IF; + IF TG_OP='INSERT' THEN + IF NEW.state<>'PENDING' OR NEW.completed_at IS NOT NULL OR NEW.payload_delete_xid IS NOT NULL THEN RAISE EXCEPTION 'metadata GC must start PENDING'; END IF; + PERFORM mst2_metadata_gc_proof(NEW.page_id,NEW.generation,'CLAIM',NEW.graph_present,NEW.had_payload); + SELECT * INTO l FROM mst2_metadata_lifetime WHERE page_id=NEW.page_id AND generation=NEW.generation; + IF NEW.graph_domain<>l.graph_domain OR NEW.metadata_codec<>l.metadata_codec OR NEW.expected_size<>l.expected_size THEN + RAISE EXCEPTION 'metadata GC immutable profile conflict'; + END IF; + NEW.created_at:=clock_timestamp(); + ELSE + IF ROW(NEW.operation_id,NEW.page_id,NEW.generation,NEW.primary_scope,NEW.graph_domain, + NEW.metadata_codec,NEW.expected_size,NEW.graph_present,NEW.had_payload,NEW.created_at) + IS DISTINCT FROM ROW(OLD.operation_id,OLD.page_id,OLD.generation,OLD.primary_scope,OLD.graph_domain, + OLD.metadata_codec,OLD.expected_size,OLD.graph_present,OLD.had_payload,OLD.created_at) THEN + RAISE EXCEPTION 'metadata GC identity cannot change'; + END IF; + IF OLD.state='APPLIED' AND (NEW.state<>OLD.state OR NEW.completed_at IS DISTINCT FROM OLD.completed_at + OR NEW.payload_delete_xid IS DISTINCT FROM OLD.payload_delete_xid) THEN + RAISE EXCEPTION 'metadata GC receipt cannot change'; + END IF; + IF OLD.state='PENDING' AND NEW.state='APPLIED' THEN + IF OLD.had_payload AND OLD.payload_delete_xid IS DISTINCT FROM txid_current() THEN + RAISE EXCEPTION 'metadata GC has no same-transaction payload delete proof'; + END IF; + PERFORM mst2_metadata_gc_proof(NEW.page_id,NEW.generation,'REMOVED',false,false); + NEW.completed_at:=clock_timestamp(); + ELSIF NEW.state IS DISTINCT FROM OLD.state OR NEW.completed_at IS DISTINCT FROM OLD.completed_at THEN + RAISE EXCEPTION 'metadata GC invalid receipt transition'; + END IF; + IF NEW.payload_delete_xid IS DISTINCT FROM OLD.payload_delete_xid THEN + IF OLD.state<>'PENDING' OR NEW.state<>'PENDING' OR NOT OLD.had_payload THEN + RAISE EXCEPTION 'invalid metadata delete transaction proof'; + END IF; + PERFORM mst2_metadata_gc_proof(OLD.page_id,OLD.generation,'PENDING',OLD.graph_present,true); + NEW.payload_delete_xid:=txid_current(); + END IF; + END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_metadata_gc_op_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_metadata_gc_op + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_gc_op_guard(); + +CREATE OR REPLACE FUNCTION mst2_metadata_lifetime_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE AS $$ +DECLARE op mst2_metadata_gc_op%ROWTYPE; +BEGIN + IF TG_OP='DELETE' THEN RAISE EXCEPTION 'metadata lifetime watermark cannot be deleted'; END IF; + IF ROW(NEW.page_id,NEW.node_id,NEW.generation,NEW.metadata_codec,NEW.expected_size,NEW.graph_domain) + IS DISTINCT FROM ROW(OLD.page_id,OLD.node_id,OLD.generation,OLD.metadata_codec,OLD.expected_size,OLD.graph_domain) THEN + RAISE EXCEPTION 'metadata lifetime identity is immutable'; + END IF; + IF NEW.state=OLD.state THEN RETURN NEW; END IF; + IF OLD.graph_domain='generic-v1' THEN + IF OLD.state='RESERVED' AND NEW.state='LIVE' THEN RETURN NEW; END IF; + RAISE EXCEPTION 'generic lifetime transition requires original collector'; + END IF; + IF OLD.state='RESERVED' AND NEW.state='LIVE' THEN + IF NOT EXISTS(SELECT 1 FROM mst2_metadata_graph_node n JOIN mst2_metadata_payload b USING(page_id,generation) + WHERE n.page_id=OLD.page_id AND n.generation=OLD.generation AND n.state='LIVE' + AND n.metadata_codec=OLD.metadata_codec AND n.bytes=OLD.expected_size + AND b.metadata_codec=OLD.metadata_codec AND b.byte_size=OLD.expected_size) + OR NOT EXISTS(SELECT 1 FROM mst2_metadata_graph_root r WHERE r.page_id=OLD.page_id AND r.generation=OLD.generation) THEN + RAISE EXCEPTION 'qualified LIVE transition needs its graph payload and prepare root'; + END IF; + RETURN NEW; + END IF; + SELECT * INTO op FROM mst2_metadata_gc_op WHERE page_id=OLD.page_id AND generation=OLD.generation AND state='PENDING'; + IF NOT FOUND OR NOT mst2_metadata_scope_matches(op.primary_scope) THEN RAISE EXCEPTION 'metadata transition has no exact GC op'; END IF; + IF OLD.state IN ('RESERVED','LIVE') AND NEW.state='DELETING' THEN + PERFORM mst2_metadata_assert_uncovered(OLD.page_id,OLD.generation); + RETURN NEW; + END IF; + IF OLD.state='DELETING' AND NEW.state='REMOVED' THEN + IF op.had_payload AND op.payload_delete_xid IS DISTINCT FROM txid_current() THEN + RAISE EXCEPTION 'metadata removal has no same-transaction payload delete proof'; + END IF; + PERFORM mst2_metadata_gc_proof(OLD.page_id,OLD.generation,'PENDING',false,false); + RETURN NEW; + END IF; + RAISE EXCEPTION 'metadata lifetime invalid state transition'; +END $$; + +CREATE FUNCTION mst2_metadata_lifetime_removed() RETURNS trigger LANGUAGE plpgsql VOLATILE AS $$ +BEGIN + IF OLD.state='DELETING' AND NEW.state='REMOVED' THEN + UPDATE mst2_metadata_gc_op SET state='APPLIED' + WHERE page_id=NEW.page_id AND generation=NEW.generation AND state='PENDING'; + IF NOT FOUND THEN RAISE EXCEPTION 'metadata removal lost its GC operation'; END IF; + END IF; + RETURN NULL; +END $$; +CREATE TRIGGER mst2_metadata_lifetime_removed AFTER UPDATE ON mst2_metadata_lifetime + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_lifetime_removed(); + +CREATE OR REPLACE FUNCTION mst2_metadata_current_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE AS $$ +DECLARE op mst2_metadata_gc_op%ROWTYPE; l mst2_metadata_lifetime%ROWTYPE; +BEGIN + IF TG_OP='DELETE' THEN RAISE EXCEPTION 'metadata current watermark cannot be deleted'; END IF; + IF TG_OP='INSERT' THEN + IF NEW.generation<>1 OR EXISTS(SELECT 1 FROM mst2_metadata_lifetime WHERE page_id=NEW.page_id AND generation<>1) THEN + RAISE EXCEPTION 'initial metadata current cannot reset history'; + END IF; + SELECT * INTO l FROM mst2_metadata_lifetime WHERE page_id=NEW.page_id AND generation=NEW.generation; + IF l.graph_domain='qualified-v1' THEN + PERFORM mst2_metadata_assert_uncovered(NEW.page_id,NEW.generation); + IF EXISTS(SELECT 1 FROM mst2_metadata_payload WHERE page_id=NEW.page_id) + OR EXISTS(SELECT 1 FROM mst2_metadata_graph_node WHERE page_id=NEW.page_id) THEN + RAISE EXCEPTION 'initial qualified current cannot adopt old bytes or graph'; + END IF; + END IF; + RETURN NEW; + END IF; + IF NEW.page_id<>OLD.page_id OR OLD.generation=9223372036854775807 OR NEW.generation<>OLD.generation+1 THEN + RAISE EXCEPTION 'metadata current requires exact next-generation CAS'; + END IF; + SELECT * INTO op FROM mst2_metadata_gc_op WHERE page_id=OLD.page_id AND generation=OLD.generation AND state='APPLIED'; + IF NOT FOUND OR NOT mst2_metadata_scope_matches(op.primary_scope) THEN RAISE EXCEPTION 'fresh metadata requires exact APPLIED proof'; END IF; + PERFORM mst2_metadata_gc_proof(OLD.page_id,OLD.generation,'REMOVED',false,false); + SELECT * INTO l FROM mst2_metadata_lifetime WHERE page_id=NEW.page_id AND generation=NEW.generation; + IF NOT FOUND OR l.state<>'RESERVED' OR l.graph_domain<>'qualified-v1' + OR l.metadata_codec<>op.metadata_codec OR l.expected_size<>op.expected_size THEN + RAISE EXCEPTION 'fresh metadata incarnation profile mismatch'; + END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_metadata_current_insert_guard BEFORE INSERT ON mst2_metadata_current + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_current_guard(); + +CREATE FUNCTION mst2_metadata_current_protected() RETURNS trigger LANGUAGE plpgsql VOLATILE AS $$ +BEGIN + IF EXISTS(SELECT 1 FROM mst2_metadata_lifetime WHERE page_id=NEW.page_id AND generation=NEW.generation AND graph_domain='qualified-v1') + AND NOT EXISTS(SELECT 1 FROM mst2_metadata_prepare_page m JOIN mst2_metadata_prepare q USING(prepare_id) + JOIN mst2_metadata_current c ON c.page_id=m.page_id AND c.generation=m.generation + WHERE m.page_id=NEW.page_id AND m.generation=NEW.generation AND q.graph_domain='qualified-v1' + AND q.storage_seal IS NOT NULL AND (q.state='PREPARING' OR (q.state='COMMITTED' AND q.coverage_retired_at IS NULL))) THEN + RAISE EXCEPTION 'qualified current change must commit with fresh prepare protection'; + END IF; + RETURN NULL; +END $$; +CREATE CONSTRAINT TRIGGER mst2_metadata_current_protected AFTER INSERT OR UPDATE ON mst2_metadata_current + DEFERRABLE INITIALLY DEFERRED FOR EACH ROW EXECUTE FUNCTION mst2_metadata_current_protected(); + +CREATE FUNCTION mst2_metadata_lifetime_insert_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE AS $$ +DECLARE c mst2_metadata_current%ROWTYPE; old_owner text; op mst2_metadata_gc_op%ROWTYPE; +BEGIN + SELECT * INTO c FROM mst2_metadata_current WHERE page_id=NEW.page_id; + IF FOUND THEN + SELECT graph_domain INTO old_owner FROM mst2_metadata_lifetime WHERE page_id=c.page_id AND generation=c.generation; + IF old_owner IS DISTINCT FROM NEW.graph_domain THEN RAISE EXCEPTION 'metadata incarnation domain cannot be adopted'; END IF; + IF NEW.graph_domain='qualified-v1' AND NEW.generation<>c.generation THEN + SELECT * INTO op FROM mst2_metadata_gc_op WHERE page_id=c.page_id AND generation=c.generation AND state='APPLIED'; + IF NOT FOUND OR NOT mst2_metadata_scope_matches(op.primary_scope) OR c.generation=9223372036854775807 + OR NEW.generation<>c.generation+1 OR NEW.state<>'RESERVED' + OR NEW.metadata_codec<>op.metadata_codec OR NEW.expected_size<>op.expected_size THEN + RAISE EXCEPTION 'fresh historical incarnation needs exact removal proof'; + END IF; + PERFORM mst2_metadata_gc_proof(c.page_id,c.generation,'REMOVED',false,false); + END IF; + ELSIF NEW.graph_domain='qualified-v1' AND (NEW.generation<>1 OR NEW.state<>'RESERVED' + OR EXISTS(SELECT 1 FROM mst2_metadata_lifetime WHERE page_id=NEW.page_id AND graph_domain<>NEW.graph_domain)) THEN + RAISE EXCEPTION 'initial qualified incarnation cannot adopt old history'; + END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_metadata_lifetime_insert_guard BEFORE INSERT ON mst2_metadata_lifetime + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_lifetime_insert_guard(); + +CREATE FUNCTION mst2_metadata_qualified_mapping_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE AS $$ +DECLARE owner text; domain text; st text; +BEGIN + IF TG_OP<>'INSERT' THEN RETURN NEW; END IF; + SELECT l.graph_domain INTO owner FROM mst2_metadata_current c JOIN mst2_metadata_lifetime l USING(page_id,generation) + WHERE c.page_id=NEW.page_id; + SELECT graph_domain INTO domain FROM mst2_metadata_prepare WHERE prepare_id=NEW.prepare_id; + IF owner='qualified-v1' OR domain='qualified-v1' THEN + SELECT l.state INTO st FROM mst2_metadata_current c JOIN mst2_metadata_lifetime l USING(page_id,generation) + WHERE c.page_id=NEW.page_id AND c.generation=NEW.generation AND l.graph_domain='qualified-v1'; + IF owner IS DISTINCT FROM domain OR st IS NULL OR st NOT IN ('RESERVED','LIVE') + OR EXISTS(SELECT 1 FROM mst2_metadata_gc_op WHERE page_id=NEW.page_id AND generation=NEW.generation) + OR NOT EXISTS(SELECT 1 FROM mst2_metadata_prepare q WHERE q.prepare_id=NEW.prepare_id + AND q.state='PREPARING' AND q.storage_seal IS NOT NULL AND mst2_metadata_scope_matches(q.primary_scope)) THEN + RAISE EXCEPTION 'qualified mapping cannot cross domain/current/state fence'; + END IF; + END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_metadata_qualified_mapping_guard BEFORE INSERT ON mst2_metadata_prepare_page + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_qualified_mapping_guard(); + +CREATE FUNCTION mst2_metadata_generic_domain_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE AS $$ +DECLARE record jsonb; ident text; key text; +BEGIN + FOREACH key IN ARRAY ARRAY['node_id','parent_id','child_id'] LOOP + IF TG_OP<>'INSERT' THEN + record:=to_jsonb(OLD); ident:=record->>key; + IF EXISTS(SELECT 1 FROM mst2_metadata_current c JOIN mst2_metadata_lifetime l USING(page_id,generation) + WHERE l.graph_domain='qualified-v1' AND l.node_id=ident) THEN RAISE EXCEPTION 'generic graph cannot touch qualified incarnation'; END IF; + END IF; + IF TG_OP<>'DELETE' THEN + record:=to_jsonb(NEW); ident:=record->>key; + IF EXISTS(SELECT 1 FROM mst2_metadata_current c JOIN mst2_metadata_lifetime l USING(page_id,generation) + WHERE l.graph_domain='qualified-v1' AND l.node_id=ident) THEN RAISE EXCEPTION 'generic graph cannot adopt qualified incarnation'; END IF; + END IF; + END LOOP; + IF TG_OP='DELETE' THEN RETURN OLD; END IF; + RETURN NEW; +END $$; +DO $$ DECLARE t text; BEGIN + FOREACH t IN ARRAY ARRAY['mst2_retention_node','mst2_retention_edge','mst2_retention_root','mst2_retention_gc_op'] LOOP + EXECUTE format('CREATE TRIGGER mst2_metadata_generic_domain_guard BEFORE INSERT OR UPDATE OR DELETE ON %I FOR EACH ROW EXECUTE FUNCTION mst2_metadata_generic_domain_guard()',t); + END LOOP; +END $$; + +CREATE FUNCTION mst2_metadata_session_domain_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE AS $$ +DECLARE sid text; pid text; +BEGIN + IF TG_OP='DELETE' THEN RETURN OLD; END IF; + IF TG_TABLE_NAME='mst2_snapshot_context' THEN pid:=NEW.prepare_id; + ELSE sid:=NEW.snapshot_id; SELECT prepare_id INTO pid FROM mst2_snapshot_context WHERE snapshot_id=sid; END IF; + IF EXISTS(SELECT 1 FROM mst2_metadata_prepare WHERE prepare_id=pid AND graph_domain='qualified-v1') THEN + RAISE EXCEPTION 'qualified production session adoption is closed'; + END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_metadata_session_domain_guard BEFORE INSERT OR UPDATE ON mst2_snapshot_context + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_session_domain_guard(); +CREATE TRIGGER mst2_metadata_session_domain_guard BEFORE INSERT OR UPDATE ON mst2_snapshot_lease + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_session_domain_guard(); + +CREATE FUNCTION mst2_metadata_graph_node_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE AS $$ +DECLARE l mst2_metadata_lifetime%ROWTYPE; op mst2_metadata_gc_op%ROWTYPE; +BEGIN + IF TG_OP='DELETE' THEN + SELECT * INTO op FROM mst2_metadata_gc_op WHERE page_id=OLD.page_id AND generation=OLD.generation AND state='PENDING'; + IF NOT FOUND OR NOT mst2_metadata_scope_matches(op.primary_scope) OR OLD.state<>'DELETING' THEN + RAISE EXCEPTION 'qualified node deletion needs exact pending operation'; + END IF; + IF op.had_payload AND op.payload_delete_xid IS DISTINCT FROM txid_current() THEN RAISE EXCEPTION 'node deletion has no payload delete proof'; END IF; + PERFORM mst2_metadata_gc_proof(OLD.page_id,OLD.generation,'PENDING',true,false); + IF EXISTS(SELECT 1 FROM mst2_metadata_graph_edge WHERE parent_page=OLD.page_id AND parent_generation=OLD.generation) THEN + RAISE EXCEPTION 'qualified node still has outgoing edges'; + END IF; + RETURN OLD; + END IF; + SELECT l0.* INTO l FROM mst2_metadata_current c JOIN mst2_metadata_lifetime l0 USING(page_id,generation) + WHERE c.page_id=NEW.page_id AND c.generation=NEW.generation; + IF NOT FOUND OR l.graph_domain<>'qualified-v1' OR l.metadata_codec<>NEW.metadata_codec OR l.expected_size<>NEW.bytes THEN + RAISE EXCEPTION 'qualified node does not match exact lifetime'; + END IF; + IF NEW.incoming_refs<>(SELECT count(*) FROM mst2_metadata_graph_edge WHERE child_page=NEW.page_id AND child_generation=NEW.generation) THEN + RAISE EXCEPTION 'qualified node counter differs from actual unique edges'; + END IF; + IF TG_OP='INSERT' THEN + IF NEW.state<>'LIVE' OR l.state NOT IN ('RESERVED','LIVE') OR NOT EXISTS( + SELECT 1 FROM mst2_metadata_prepare_page m JOIN mst2_metadata_prepare q USING(prepare_id) + WHERE m.page_id=NEW.page_id AND m.generation=NEW.generation AND q.graph_domain='qualified-v1' AND q.state='PREPARING') THEN + RAISE EXCEPTION 'qualified node creation needs active fixed preparation'; + END IF; + ELSE + IF ROW(NEW.page_id,NEW.generation,NEW.metadata_codec,NEW.bytes) IS DISTINCT FROM ROW(OLD.page_id,OLD.generation,OLD.metadata_codec,OLD.bytes) THEN + RAISE EXCEPTION 'qualified node identity is immutable'; + END IF; + IF NEW.state<>OLD.state AND NOT (OLD.state='LIVE' AND NEW.state='DELETING' AND EXISTS( + SELECT 1 FROM mst2_metadata_gc_op WHERE page_id=OLD.page_id AND generation=OLD.generation AND state='PENDING')) THEN + RAISE EXCEPTION 'qualified node invalid state transition'; + END IF; + END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_metadata_graph_node_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_metadata_graph_node + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_graph_node_guard(); + +CREATE FUNCTION mst2_metadata_graph_root_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE AS $$ +BEGIN + IF TG_OP='UPDATE' THEN RAISE EXCEPTION 'qualified roots cannot be retargeted'; END IF; + IF TG_OP='DELETE' THEN RETURN OLD; END IF; + IF NOT EXISTS(SELECT 1 FROM mst2_metadata_prepare q JOIN mst2_metadata_current c ON c.page_id=NEW.page_id AND c.generation=NEW.generation + JOIN mst2_metadata_lifetime l USING(page_id,generation) JOIN mst2_metadata_graph_node n USING(page_id,generation) + WHERE q.prepare_id=NEW.prepare_id AND q.storage_seal=NEW.storage_seal AND q.state='PREPARING' + AND q.graph_domain='qualified-v1' AND l.graph_domain='qualified-v1' AND l.state IN ('RESERVED','LIVE') + AND n.state='LIVE' AND mst2_metadata_scope_matches(q.primary_scope)) THEN + RAISE EXCEPTION 'qualified root requires its active exact sealed preparation'; + END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_metadata_graph_root_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_metadata_graph_root + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_graph_root_guard(); + +CREATE FUNCTION mst2_metadata_graph_edge_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE AS $$ +DECLARE op mst2_metadata_gc_op%ROWTYPE; +BEGIN + IF TG_OP='UPDATE' THEN RAISE EXCEPTION 'qualified edge identity is immutable'; END IF; + IF TG_OP='DELETE' THEN + SELECT * INTO op FROM mst2_metadata_gc_op WHERE page_id=OLD.parent_page AND generation=OLD.parent_generation AND state='PENDING'; + IF NOT FOUND OR NOT mst2_metadata_scope_matches(op.primary_scope) THEN RAISE EXCEPTION 'edge deletion has no exact operation'; END IF; + IF op.had_payload AND op.payload_delete_xid IS DISTINCT FROM txid_current() THEN RAISE EXCEPTION 'edge deletion has no payload delete proof'; END IF; + IF EXISTS(SELECT 1 FROM mst2_metadata_payload WHERE page_id=OLD.parent_page) + OR NOT EXISTS(SELECT 1 FROM mst2_metadata_graph_node WHERE page_id=OLD.parent_page + AND generation=OLD.parent_generation AND state='DELETING' AND incoming_refs=0) THEN + RAISE EXCEPTION 'edge deletion is outside its atomic byte-removal phase'; + END IF; + RETURN OLD; + END IF; + IF (SELECT count(*) FROM mst2_metadata_graph_node n JOIN mst2_metadata_current c USING(page_id,generation) + JOIN mst2_metadata_lifetime l USING(page_id,generation) + WHERE ((n.page_id=NEW.parent_page AND n.generation=NEW.parent_generation) + OR (n.page_id=NEW.child_page AND n.generation=NEW.child_generation)) + AND n.state='LIVE' AND l.state IN ('RESERVED','LIVE') AND l.graph_domain='qualified-v1')<>2 + OR NOT EXISTS(SELECT 1 FROM mst2_metadata_prepare_page a JOIN mst2_metadata_prepare_page b USING(prepare_id) + JOIN mst2_metadata_prepare q USING(prepare_id) WHERE a.page_id=NEW.parent_page AND a.generation=NEW.parent_generation + AND b.page_id=NEW.child_page AND b.generation=NEW.child_generation AND q.state='PREPARING' AND q.graph_domain='qualified-v1') THEN + RAISE EXCEPTION 'edge creation requires active exact qualified endpoints'; + END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_metadata_graph_edge_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_metadata_graph_edge + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_graph_edge_guard(); + +CREATE FUNCTION mst2_metadata_edges_added() RETURNS trigger LANGUAGE plpgsql VOLATILE AS $$ +DECLARE reachable text[]; links jsonb; adjacency jsonb; indegrees bigint[]; + queue integer[]:=ARRAY[]::integer[]; head integer:=1; parent_index integer; child_index integer; i integer; +BEGIN + IF (SELECT count(*) FROM added_edges)>16384 THEN RAISE EXCEPTION 'qualified edge batch exceeds limit'; END IF; + IF NOT EXISTS(SELECT 1 FROM added_edges) THEN RETURN NULL; END IF; + IF EXISTS(SELECT 1 FROM mst2_metadata_graph_node n JOIN ( + SELECT child_page,child_generation,count(*) AS delta FROM added_edges GROUP BY child_page,child_generation + ) d ON d.child_page=n.page_id AND d.child_generation=n.generation + WHERE n.incoming_refs+d.delta<>(SELECT count(*) FROM mst2_metadata_graph_edge x WHERE x.child_page=n.page_id AND x.child_generation=n.generation)) THEN + RAISE EXCEPTION 'qualified insertion found preexisting counter drift'; + END IF; + UPDATE mst2_metadata_graph_node n SET incoming_refs=n.incoming_refs+d.delta FROM ( + SELECT child_page,child_generation,count(*) AS delta FROM added_edges GROUP BY child_page,child_generation + ) d WHERE n.page_id=d.child_page AND n.generation=d.child_generation; + -- One bounded closure per statement; topological accounting below is local. + WITH RECURSIVE walk(page_id,generation) AS ( + SELECT page_id,generation FROM ( + SELECT parent_page AS page_id,parent_generation AS generation FROM added_edges + UNION SELECT child_page,child_generation FROM added_edges + ) starts UNION + SELECT x.child_page,x.child_generation FROM walk w JOIN mst2_metadata_graph_edge x + ON x.parent_page=w.page_id AND x.parent_generation=w.generation + ) SELECT array_agg(encode(page_id,'hex')||':'||generation ORDER BY page_id,generation) + INTO reachable FROM (SELECT * FROM walk LIMIT 4097) bounded; + IF cardinality(reachable)>4096 THEN RAISE EXCEPTION 'qualified graph node audit overflow'; END IF; + WITH nodes AS ( + SELECT ordinality::integer AS i,decode(split_part(key,':',1),'hex') AS page_id, + split_part(key,':',2)::bigint AS generation FROM unnest(reachable) WITH ORDINALITY n(key,ordinality) + ) SELECT coalesce(jsonb_agg(jsonb_build_array(p.i,c.i)),'[]'::jsonb) INTO links FROM ( + SELECT e.* FROM nodes p JOIN mst2_metadata_graph_edge e + ON e.parent_page=p.page_id AND e.parent_generation=p.generation LIMIT 16385 + ) e JOIN nodes p ON p.page_id=e.parent_page AND p.generation=e.parent_generation + JOIN nodes c ON c.page_id=e.child_page AND c.generation=e.child_generation; + IF jsonb_array_length(links)>16384 THEN RAISE EXCEPTION 'qualified graph edge audit overflow'; END IF; + SELECT coalesce(jsonb_object_agg(parent,children),'{}'::jsonb) INTO adjacency FROM ( + SELECT item->>0 AS parent,jsonb_agg((item->>1)::integer) AS children + FROM jsonb_array_elements(links) x(item) GROUP BY item->>0 + ) grouped; + SELECT array_agg(coalesce(counts.refs,0) ORDER BY n.i) INTO indegrees + FROM generate_series(1,cardinality(reachable)) n(i) LEFT JOIN ( + SELECT (item->>1)::integer AS child,count(*) AS refs FROM jsonb_array_elements(links) x(item) GROUP BY item->>1 + ) counts ON counts.child=n.i; + FOR i IN 1..cardinality(reachable) LOOP + IF indegrees[i]=0 THEN queue:=array_append(queue,i); END IF; + END LOOP; + WHILE head<=cardinality(queue) LOOP + parent_index:=queue[head]; head:=head+1; + FOR child_index IN SELECT value::text::integer FROM jsonb_array_elements(coalesce(adjacency->parent_index::text,'[]'::jsonb)) LOOP + indegrees[child_index]:=indegrees[child_index]-1; + IF indegrees[child_index]=0 THEN queue:=array_append(queue,child_index); END IF; + END LOOP; + END LOOP; + IF cardinality(queue)<>cardinality(reachable) THEN RAISE EXCEPTION 'qualified graph contains a cycle'; END IF; + RETURN NULL; +END $$; +CREATE TRIGGER mst2_metadata_edges_added AFTER INSERT ON mst2_metadata_graph_edge + REFERENCING NEW TABLE AS added_edges FOR EACH STATEMENT EXECUTE FUNCTION mst2_metadata_edges_added(); +CREATE FUNCTION mst2_metadata_edges_removed() RETURNS trigger LANGUAGE plpgsql VOLATILE AS $$ +BEGIN + IF EXISTS(SELECT 1 FROM mst2_metadata_graph_node n JOIN ( + SELECT child_page,child_generation,count(*) AS delta FROM removed_edges GROUP BY child_page,child_generation + ) d ON d.child_page=n.page_id AND d.child_generation=n.generation + WHERE n.incoming_refs(SELECT count(*) FROM mst2_metadata_graph_edge x WHERE x.child_page=n.page_id AND x.child_generation=n.generation)) THEN + RAISE EXCEPTION 'qualified subtraction found counter drift'; + END IF; + UPDATE mst2_metadata_graph_node n SET incoming_refs=n.incoming_refs-d.delta FROM ( + SELECT child_page,child_generation,count(*) AS delta FROM removed_edges GROUP BY child_page,child_generation + ) d WHERE n.page_id=d.child_page AND n.generation=d.child_generation; + RETURN NULL; +END $$; +CREATE TRIGGER mst2_metadata_edges_removed AFTER DELETE ON mst2_metadata_graph_edge + REFERENCING OLD TABLE AS removed_edges FOR EACH STATEMENT EXECUTE FUNCTION mst2_metadata_edges_removed(); + +CREATE FUNCTION mst2_metadata_gc_claimed() RETURNS trigger LANGUAGE plpgsql VOLATILE AS $$ +BEGIN + UPDATE mst2_metadata_graph_node SET state='DELETING' WHERE page_id=NEW.page_id AND generation=NEW.generation; + UPDATE mst2_metadata_lifetime SET state='DELETING' WHERE page_id=NEW.page_id AND generation=NEW.generation; + RETURN NULL; +END $$; +CREATE TRIGGER mst2_metadata_gc_claimed AFTER INSERT ON mst2_metadata_gc_op + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_gc_claimed(); + +CREATE FUNCTION mst2_metadata_gc_finish(id uuid) RETURNS void LANGUAGE plpgsql VOLATILE AS $$ +DECLARE op mst2_metadata_gc_op%ROWTYPE; +BEGIN + IF pg_is_in_recovery() OR current_setting('transaction_isolation')<>'read committed' THEN RAISE EXCEPTION 'GC finish needs primary READ COMMITTED'; END IF; + PERFORM set_config('lock_timeout','5000ms',true); + PERFORM pg_advisory_xact_lock(1296717362,hashtext(current_schema())); + SELECT * INTO op FROM mst2_metadata_gc_op WHERE operation_id=id FOR UPDATE; + IF NOT FOUND OR NOT mst2_metadata_scope_matches(op.primary_scope) THEN RAISE EXCEPTION 'GC finish wrong exact operation scope'; END IF; + IF op.state='APPLIED' THEN RETURN; END IF; + IF op.had_payload AND op.payload_delete_xid IS DISTINCT FROM txid_current() THEN + RAISE EXCEPTION 'GC finish has no same-transaction payload delete proof'; + END IF; + PERFORM mst2_metadata_gc_proof(op.page_id,op.generation,'PENDING',op.graph_present,false); + DELETE FROM mst2_metadata_graph_edge WHERE parent_page=op.page_id AND parent_generation=op.generation; + DELETE FROM mst2_metadata_graph_node WHERE page_id=op.page_id AND generation=op.generation; + UPDATE mst2_metadata_lifetime SET state='REMOVED' WHERE page_id=op.page_id AND generation=op.generation AND state='DELETING'; + IF NOT FOUND THEN RAISE EXCEPTION 'GC finish lost exact lifetime'; END IF; +END $$; + +CREATE FUNCTION mst2_metadata_payload_fenced() RETURNS trigger LANGUAGE plpgsql VOLATILE AS $$ +DECLARE l mst2_metadata_lifetime%ROWTYPE; op mst2_metadata_gc_op%ROWTYPE; selector text; +BEGIN + IF TG_OP='UPDATE' THEN RAISE EXCEPTION 'MST2 metadata payload UPDATE is forbidden'; END IF; + IF TG_OP='INSERT' THEN + SELECT h.* INTO l FROM mst2_metadata_current c JOIN mst2_metadata_lifetime h USING(page_id,generation) WHERE c.page_id=NEW.page_id; + IF FOUND AND l.graph_domain='qualified-v1' THEN + IF NEW.generation IS DISTINCT FROM l.generation OR l.state NOT IN ('RESERVED','LIVE') + OR NEW.metadata_codec<>l.metadata_codec OR NEW.byte_size<>l.expected_size + OR NOT EXISTS(SELECT 1 FROM mst2_metadata_prepare_page m JOIN mst2_metadata_prepare q USING(prepare_id) + WHERE m.page_id=NEW.page_id AND m.generation=NEW.generation AND q.state='PREPARING' + AND q.graph_domain='qualified-v1' AND q.storage_seal IS NOT NULL AND mst2_metadata_scope_matches(q.primary_scope)) THEN + RAISE EXCEPTION 'qualified payload INSERT crossed its exact active generation'; + END IF; + END IF; + RETURN NEW; + END IF; + selector:=current_setting('mega2.metadata_gc_operation',true); + IF OLD.generation IS NULL OR selector IS NULL OR selector='' THEN RAISE EXCEPTION 'payload DELETE has no qualified operation selector'; END IF; + SELECT * INTO op FROM mst2_metadata_gc_op WHERE operation_id=selector::uuid AND page_id=OLD.page_id AND generation=OLD.generation FOR UPDATE; + IF NOT FOUND OR op.state<>'PENDING' OR NOT op.had_payload OR NOT mst2_metadata_scope_matches(op.primary_scope) THEN + RAISE EXCEPTION 'payload DELETE requires exact persistent pending proof'; + END IF; + PERFORM mst2_metadata_gc_proof(OLD.page_id,OLD.generation,'PENDING',op.graph_present,true); + UPDATE mst2_metadata_gc_op SET payload_delete_xid=txid_current() WHERE operation_id=op.operation_id; + RETURN OLD; +END $$; +DROP TRIGGER mst2_metadata_payload_immutable ON mst2_metadata_payload; +CREATE TRIGGER mst2_metadata_payload_fenced BEFORE INSERT OR UPDATE OR DELETE ON mst2_metadata_payload + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_payload_fenced(); +CREATE FUNCTION mst2_metadata_payload_removed() RETURNS trigger LANGUAGE plpgsql VOLATILE AS $$ +DECLARE op uuid; +BEGIN + SELECT operation_id INTO op FROM mst2_metadata_gc_op WHERE page_id=OLD.page_id AND generation=OLD.generation AND state='PENDING'; + IF NOT FOUND THEN RAISE EXCEPTION 'deleted payload lost exact pending operation'; END IF; + PERFORM mst2_metadata_gc_finish(op); + RETURN NULL; +END $$; +CREATE TRIGGER mst2_metadata_payload_removed AFTER DELETE ON mst2_metadata_payload + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_payload_removed(); + +CREATE FUNCTION mst2_metadata_gc_apply(id uuid) RETURNS void LANGUAGE plpgsql VOLATILE AS $$ +DECLARE op mst2_metadata_gc_op%ROWTYPE; +BEGIN + IF pg_is_in_recovery() OR current_setting('transaction_isolation')<>'read committed' THEN RAISE EXCEPTION 'GC apply needs primary READ COMMITTED'; END IF; + PERFORM set_config('lock_timeout','5000ms',true); + PERFORM pg_advisory_xact_lock(1296717362,hashtext(current_schema())); + SELECT * INTO op FROM mst2_metadata_gc_op WHERE operation_id=id FOR UPDATE; + IF NOT FOUND OR NOT mst2_metadata_scope_matches(op.primary_scope) THEN RAISE EXCEPTION 'GC apply missing exact scope'; END IF; + IF op.state='APPLIED' THEN RETURN; END IF; + PERFORM mst2_metadata_gc_proof(op.page_id,op.generation,'PENDING',op.graph_present,op.had_payload); + IF op.had_payload THEN + PERFORM set_config('mega2.metadata_gc_operation',op.operation_id::text,true); + DELETE FROM mst2_metadata_payload WHERE page_id=op.page_id AND generation=op.generation; + IF NOT FOUND THEN RAISE EXCEPTION 'GC apply lost captured payload'; END IF; + ELSE + PERFORM mst2_metadata_gc_finish(id); + END IF; +END $$; diff --git a/src/jupiter/migration/m20261007_000500_add_mst2_install_capability.rs b/src/jupiter/migration/m20261007_000500_add_mst2_install_capability.rs new file mode 100644 index 00000000..95ddaca5 --- /dev/null +++ b/src/jupiter/migration/m20261007_000500_add_mst2_install_capability.rs @@ -0,0 +1,21 @@ +//! Freeze a fully validated legacy plan without granting generation authority. + +use sea_orm_migration::prelude::*; + +#[derive(DeriveMigrationName)] +pub struct Migration; + +#[async_trait::async_trait] +impl MigrationTrait for Migration { + async fn up(&self, manager: &SchemaManager) -> Result<(), DbErr> { + manager + .get_connection() + .execute_unprepared(include_str!("m20261007_000500_install_capability.sql")) + .await?; + Ok(()) + } + + async fn down(&self, _manager: &SchemaManager) -> Result<(), DbErr> { + Ok(()) + } +} diff --git a/src/jupiter/migration/m20261007_000500_install_capability.sql b/src/jupiter/migration/m20261007_000500_install_capability.sql new file mode 100644 index 00000000..b8114adf --- /dev/null +++ b/src/jupiter/migration/m20261007_000500_install_capability.sql @@ -0,0 +1,138 @@ +LOCK TABLE mst2_metadata_prepare,mst2_metadata_prepare_page IN ACCESS EXCLUSIVE MODE; + +CREATE TABLE mst2_metadata_install_seal ( + prepare_id text PRIMARY KEY REFERENCES mst2_metadata_prepare(prepare_id), + operation_id text NOT NULL CHECK (octet_length(operation_id) BETWEEN 1 AND 255), + manifest_digest bytea NOT NULL CHECK (octet_length(manifest_digest)=32), + members_digest bytea NOT NULL CHECK (octet_length(members_digest)=32), + primary_scope bytea NOT NULL CHECK (octet_length(primary_scope) BETWEEN 1 AND 16384), + install_seal bytea NOT NULL CHECK (octet_length(install_seal)=32), + created_at timestamptz NOT NULL DEFAULT clock_timestamp() +); + +-- The actual regulated table, rather than caller search_path, selects the lock. +CREATE FUNCTION mst2_install_capability_barrier() RETURNS trigger LANGUAGE plpgsql VOLATILE AS $$ +BEGIN + IF pg_catalog.current_schema() IS DISTINCT FROM TG_TABLE_SCHEMA + OR pg_catalog.pg_is_in_recovery() + OR pg_catalog.current_setting('transaction_isolation')<>'read committed' THEN + RAISE EXCEPTION 'install capability mutation requires its actual primary schema and READ COMMITTED'; + END IF; + PERFORM pg_catalog.set_config('search_path',pg_catalog.quote_ident(TG_TABLE_SCHEMA)||',pg_catalog,pg_temp',true); + PERFORM pg_catalog.set_config('lock_timeout','5000ms',true); + PERFORM pg_catalog.pg_advisory_xact_lock(1296717362,pg_catalog.hashtext(TG_TABLE_SCHEMA)); + RETURN NULL; +END $$; + +CREATE FUNCTION mst2_install_capability_register() RETURNS trigger LANGUAGE plpgsql VOLATILE AS $$ +DECLARE p record; members bytea; actual jsonb; member_count bigint; member_bytes bigint; has_root boolean; +BEGIN + IF TG_OP<>'INSERT' THEN RAISE EXCEPTION 'install capability registration is immutable'; END IF; + IF pg_catalog.current_schema() IS DISTINCT FROM TG_TABLE_SCHEMA THEN + RAISE EXCEPTION 'install capability registration is outside its actual schema'; + END IF; + EXECUTE pg_catalog.format('SELECT * FROM %I.mst2_metadata_prepare WHERE prepare_id=$1',TG_TABLE_SCHEMA) + INTO p USING NEW.prepare_id; + IF p.prepare_id IS NULL OR p.operation_id IS DISTINCT FROM NEW.operation_id + OR p.manifest_digest IS DISTINCT FROM NEW.manifest_digest + OR pg_catalog.sha256(p.canonical_plan) IS DISTINCT FROM NEW.manifest_digest + OR p.state NOT IN ('PREPARING','COMMITTED') OR p.coverage_retired_at IS NOT NULL + OR p.canonical_bindings IS NOT NULL OR p.bindings_digest IS NOT NULL + OR p.primary_scope IS NOT NULL OR p.storage_seal IS NOT NULL OR p.graph_domain IS NOT NULL THEN + RAISE EXCEPTION 'install capability requires a fixed unbound legacy preparation'; + END IF; + EXECUTE pg_catalog.format( + 'SELECT count(*),sum(expected_size),bool_or(page_id=$2), + sha256(string_agg(page_id||int4send(expected_size)|| + CASE WHEN generation IS NULL THEN decode(''00'',''hex'') + ELSE decode(''01'',''hex'')||int8send(generation) END,''''::bytea ORDER BY page_id)) + FROM %I.mst2_metadata_prepare_page WHERE prepare_id=$1',TG_TABLE_SCHEMA) + INTO member_count,member_bytes,has_root,members USING NEW.prepare_id,p.metadata_root; + IF member_count<>p.node_count OR member_bytes<>p.total_bytes OR has_root IS DISTINCT FROM true + OR members IS DISTINCT FROM NEW.members_digest THEN + RAISE EXCEPTION 'install capability complete membership proof disagrees'; + END IF; + EXECUTE pg_catalog.format( + 'SELECT EXISTS(SELECT 1 FROM %I.mst2_metadata_prepare_page WHERE prepare_id=$1 AND generation IS NOT NULL)', + TG_TABLE_SCHEMA) INTO has_root USING NEW.prepare_id; + IF has_root THEN RAISE EXCEPTION 'legacy install capability cannot adopt generation bindings'; END IF; + EXECUTE pg_catalog.format( + 'SELECT jsonb_build_array(s.storage_uuid,current_database(),d.oid::bigint,$1,n.oid::bigint, + inet_server_addr()::text,inet_server_port()) + FROM %I.mst2_metadata_storage_scope s + JOIN pg_catalog.pg_database d ON d.datname=current_database() + JOIN pg_catalog.pg_namespace n ON n.nspname=$1 WHERE s.singleton=1',TG_TABLE_SCHEMA) + INTO actual USING TG_TABLE_SCHEMA; + IF actual IS NULL OR pg_catalog.convert_from(NEW.primary_scope,'UTF8')::jsonb IS DISTINCT FROM actual + OR NEW.install_seal IS DISTINCT FROM pg_catalog.sha256( + pg_catalog.convert_to('MST2-LEGACY-INSTALL-CAPABILITY-1','UTF8')||pg_catalog.decode('00','hex')|| + pg_catalog.uuid_send(NEW.prepare_id::uuid)|| + pg_catalog.int4send(pg_catalog.octet_length(NEW.operation_id))||pg_catalog.convert_to(NEW.operation_id,'UTF8')|| + NEW.manifest_digest||NEW.members_digest|| + pg_catalog.int4send(pg_catalog.octet_length(NEW.primary_scope))||NEW.primary_scope) THEN + RAISE EXCEPTION 'install capability seal or actual primary scope disagrees'; + END IF; + RETURN NEW; +END $$; + +CREATE FUNCTION mst2_install_capability_prepare_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE AS $$ +DECLARE registered boolean; +BEGIN + IF pg_catalog.current_schema() IS DISTINCT FROM TG_TABLE_SCHEMA THEN + RAISE EXCEPTION 'install capability preparation is outside its actual schema'; + END IF; + EXECUTE pg_catalog.format('SELECT EXISTS(SELECT 1 FROM %I.mst2_metadata_install_seal WHERE prepare_id=$1)',TG_TABLE_SCHEMA) + INTO registered USING OLD.prepare_id; + IF registered AND (TG_OP='DELETE' OR + (pg_catalog.to_jsonb(NEW)-ARRAY['state','committed_at','aborted_at','coverage_retired_at']) IS DISTINCT FROM + (pg_catalog.to_jsonb(OLD)-ARRAY['state','committed_at','aborted_at','coverage_retired_at'])) THEN + RAISE EXCEPTION 'registered metadata preparation identity is immutable'; + END IF; + IF TG_OP='DELETE' THEN RETURN OLD; END IF; + RETURN NEW; +END $$; + +CREATE FUNCTION mst2_install_capability_mapping_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE AS $$ +DECLARE old_id text; new_id text; registered boolean; +BEGIN + IF pg_catalog.current_schema() IS DISTINCT FROM TG_TABLE_SCHEMA THEN + RAISE EXCEPTION 'install capability membership is outside its actual schema'; + END IF; + IF TG_OP<>'INSERT' THEN old_id:=OLD.prepare_id; END IF; + IF TG_OP<>'DELETE' THEN new_id:=NEW.prepare_id; END IF; + EXECUTE pg_catalog.format( + 'SELECT EXISTS(SELECT 1 FROM %I.mst2_metadata_install_seal WHERE prepare_id=$1 OR prepare_id=$2)',TG_TABLE_SCHEMA) + INTO registered USING old_id,new_id; + IF registered THEN RAISE EXCEPTION 'registered metadata membership is immutable'; END IF; + IF TG_OP='DELETE' THEN RETURN OLD; END IF; + RETURN NEW; +END $$; + +CREATE FUNCTION mst2_install_capability_truncate_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE AS $$ +DECLARE registered boolean; +BEGIN + IF pg_catalog.current_schema() IS DISTINCT FROM TG_TABLE_SCHEMA + OR pg_catalog.pg_is_in_recovery() + OR pg_catalog.current_setting('transaction_isolation')<>'read committed' THEN + RAISE EXCEPTION 'install capability truncation requires its actual primary schema and READ COMMITTED'; + END IF; + PERFORM pg_catalog.set_config('lock_timeout','5000ms',true); + PERFORM pg_catalog.pg_advisory_xact_lock(1296717362,pg_catalog.hashtext(TG_TABLE_SCHEMA)); + EXECUTE pg_catalog.format('SELECT EXISTS(SELECT 1 FROM %I.mst2_metadata_install_seal)',TG_TABLE_SCHEMA) + INTO registered; + IF registered THEN RAISE EXCEPTION 'registered metadata evidence cannot be truncated'; END IF; + RETURN NULL; +END $$; + +DO $$ DECLARE t text; BEGIN + FOREACH t IN ARRAY ARRAY['mst2_metadata_prepare','mst2_metadata_prepare_page','mst2_metadata_install_seal'] LOOP + EXECUTE pg_catalog.format('CREATE TRIGGER mst2_00_install_capability_barrier BEFORE INSERT OR UPDATE OR DELETE ON %I FOR EACH STATEMENT EXECUTE FUNCTION mst2_install_capability_barrier()',t); + EXECUTE pg_catalog.format('CREATE TRIGGER mst2_install_capability_truncate_guard BEFORE TRUNCATE ON %I FOR EACH STATEMENT EXECUTE FUNCTION mst2_install_capability_truncate_guard()',t); + END LOOP; +END $$; +CREATE TRIGGER mst2_install_capability_register BEFORE INSERT OR UPDATE OR DELETE + ON mst2_metadata_install_seal FOR EACH ROW EXECUTE FUNCTION mst2_install_capability_register(); +CREATE TRIGGER mst2_install_capability_prepare_guard BEFORE UPDATE OR DELETE + ON mst2_metadata_prepare FOR EACH ROW EXECUTE FUNCTION mst2_install_capability_prepare_guard(); +CREATE TRIGGER mst2_install_capability_mapping_guard BEFORE INSERT OR UPDATE OR DELETE + ON mst2_metadata_prepare_page FOR EACH ROW EXECUTE FUNCTION mst2_install_capability_mapping_guard(); diff --git a/src/jupiter/migration/m20261007_000600_add_mst2_storage_routes.rs b/src/jupiter/migration/m20261007_000600_add_mst2_storage_routes.rs new file mode 100644 index 00000000..541ce531 --- /dev/null +++ b/src/jupiter/migration/m20261007_000600_add_mst2_storage_routes.rs @@ -0,0 +1,42 @@ +//! Permanent semantic routes, with a separate existing generic session ledger. + +use sea_orm::{ConnectionTrait, DbBackend, Statement}; +use sea_orm_migration::prelude::*; + +#[derive(DeriveMigrationName)] +pub struct Migration; + +#[async_trait::async_trait] +impl MigrationTrait for Migration { + async fn up(&self, manager: &SchemaManager) -> Result<(), DbErr> { + let connection = manager.get_connection(); + let row = connection + .query_one_raw(Statement::from_string( + DbBackend::Postgres, + "SELECT current_schema() AS schema,n.oid::bigint AS oid + FROM pg_catalog.pg_namespace n WHERE n.nspname=current_schema()", + )) + .await? + .ok_or_else(|| DbErr::Custom("storage route core schema is missing".into()))?; + let schema: String = row.try_get("", "schema")?; + let oid: i64 = row.try_get("", "oid")?; + let quoted = format!("\"{}\"", schema.replace('"', "\"\"")); + let literal = format!("'{}'", schema.replace('\'', "''")); + #[cfg(test)] + let mono_key2 = format!("pg_catalog.hashtext({literal})"); + #[cfg(not(test))] + let mono_key2 = super::super::storage::push_queue_storage::MONO_WRITE_LOCK_KEY2.to_string(); + let sql = include_str!("m20261007_000600_storage_routes.sql") + .replace("$CORE_SCHEMA$", "ed) + .replace("$CORE_LITERAL$", &literal) + .replace("$CORE_OID$", &oid.to_string()) + .replace("$MONO_KEY2$", &mono_key2) + .replace("$NAMESPACE_UUID$", &uuid::Uuid::new_v4().to_string()); + connection.execute_unprepared(&sql).await?; + Ok(()) + } + + async fn down(&self, _manager: &SchemaManager) -> Result<(), DbErr> { + Ok(()) + } +} diff --git a/src/jupiter/migration/m20261007_000600_storage_routes.sql b/src/jupiter/migration/m20261007_000600_storage_routes.sql new file mode 100644 index 00000000..13e8b18c --- /dev/null +++ b/src/jupiter/migration/m20261007_000600_storage_routes.sql @@ -0,0 +1,329 @@ +DO $$ BEGIN + IF pg_catalog.pg_is_in_recovery() OR pg_catalog.current_setting('transaction_isolation')<>'read committed' + OR pg_catalog.current_schema()<>$CORE_LITERAL$ + OR NOT EXISTS(SELECT 1 FROM pg_catalog.pg_namespace WHERE oid=$CORE_OID$ AND nspname=$CORE_LITERAL$) THEN + RAISE EXCEPTION 'storage route migration requires its captured primary core schema'; + END IF; + IF EXISTS(SELECT 1 FROM pg_catalog.pg_locks WHERE locktype='advisory' AND pid=pg_catalog.pg_backend_pid() + AND classid=1296718001::oid AND objid=pg_catalog.hashtext($CORE_LITERAL$)::oid AND objsubid=2 AND granted) + AND NOT EXISTS(SELECT 1 FROM pg_catalog.pg_locks WHERE locktype='advisory' AND pid=pg_catalog.pg_backend_pid() + AND classid=1297043024::oid AND objid=($MONO_KEY2$)::oid AND objsubid=2 AND granted AND mode='ExclusiveLock') THEN + RAISE EXCEPTION 'storage route migration cannot acquire mono after route'; + END IF; + IF EXISTS(SELECT 1 FROM pg_catalog.pg_locks WHERE locktype='advisory' AND pid=pg_catalog.pg_backend_pid() + AND database=(SELECT oid FROM pg_catalog.pg_database WHERE datname=pg_catalog.current_database()) + AND classid=1296717362::oid AND objid=pg_catalog.hashtext($CORE_LITERAL$)::oid AND objsubid=2 AND granted) + AND NOT (EXISTS(SELECT 1 FROM pg_catalog.pg_locks WHERE locktype='advisory' AND pid=pg_catalog.pg_backend_pid() + AND classid=1297043024::oid AND objid=($MONO_KEY2$)::oid AND objsubid=2 AND granted AND mode='ExclusiveLock') + AND EXISTS(SELECT 1 FROM pg_catalog.pg_locks WHERE locktype='advisory' AND pid=pg_catalog.pg_backend_pid() + AND classid=1296718001::oid AND objid=pg_catalog.hashtext($CORE_LITERAL$)::oid + AND objsubid=2 AND granted AND mode='ExclusiveLock')) THEN + RAISE EXCEPTION 'storage route migration cannot acquire core locks after retention'; + END IF; +END $$; +SELECT pg_catalog.set_config('lock_timeout','5000ms',true); +SELECT pg_catalog.pg_advisory_xact_lock(1297043024,$MONO_KEY2$); +SELECT pg_catalog.pg_advisory_xact_lock(1296718001,pg_catalog.hashtext($CORE_LITERAL$)); +SELECT pg_catalog.pg_advisory_xact_lock(1296717362,pg_catalog.hashtext($CORE_LITERAL$)); +SET LOCAL search_path=$CORE_SCHEMA$,pg_catalog,pg_temp; +LOCK TABLE mst2_snapshot_context,mst2_snapshot_lease,mst2_metadata_prepare, + mst2_metadata_storage_scope IN ACCESS EXCLUSIVE MODE; + +CREATE TABLE mst2_metadata_namespace ( + singleton integer PRIMARY KEY CHECK (singleton=1), + namespace_uuid uuid NOT NULL UNIQUE, + core_schema text NOT NULL, + core_schema_oid oid NOT NULL, + database_name text NOT NULL, + database_oid oid NOT NULL, + storage_uuid text NOT NULL, + server_address text, + server_port integer, + mono_lock_key2 integer NOT NULL, + metadata_schema text NOT NULL, + metadata_schema_oid oid NOT NULL, + family_identity text NOT NULL CHECK (family_identity='v3-generic-session-1'), + graph_domain text NOT NULL CHECK (graph_domain='generic-v1'), + admission_state text NOT NULL CHECK (admission_state='G_ADMITTED_Q_CLOSED'), + collector_state text NOT NULL CHECK (collector_state='CLOSED'), + CHECK (core_schema=metadata_schema AND core_schema_oid=metadata_schema_oid) +); +INSERT INTO mst2_metadata_namespace +SELECT 1,'$NAMESPACE_UUID$'::uuid,$CORE_LITERAL$,$CORE_OID$,pg_catalog.current_database(),d.oid,s.storage_uuid, + pg_catalog.inet_server_addr()::text,pg_catalog.inet_server_port(),$MONO_KEY2$,$CORE_LITERAL$,$CORE_OID$, + 'v3-generic-session-1','generic-v1','G_ADMITTED_Q_CLOSED','CLOSED' +FROM mst2_metadata_storage_scope s JOIN pg_catalog.pg_database d ON d.datname=pg_catalog.current_database() +WHERE s.singleton=1; + +CREATE TABLE mst2_snapshot_storage_route ( + snapshot_id text PRIMARY KEY, + namespace_uuid uuid NOT NULL REFERENCES mst2_metadata_namespace(namespace_uuid), + canonical_descriptor bytea NOT NULL, + instance_id text NOT NULL, + commit_oid text NOT NULL, + root_tree_oid text NOT NULL, + metadata_root bytea NOT NULL CHECK (octet_length(metadata_root)=32), + source_profile jsonb NOT NULL, + UNIQUE(snapshot_id,namespace_uuid), + CHECK (snapshot_id='sha256:'||pg_catalog.encode(pg_catalog.sha256( + pg_catalog.convert_to('mega.mst2.descriptor','UTF8')||pg_catalog.decode('00','hex')||canonical_descriptor),'hex')) +); +ALTER TABLE mst2_snapshot_context ADD UNIQUE(snapshot_id,prepare_id,metadata_root); +CREATE TABLE mst2_generic_session_storage_binding ( + session_incarnation uuid PRIMARY KEY, + snapshot_id text NOT NULL UNIQUE, + namespace_uuid uuid NOT NULL, + prepare_id text NOT NULL, + metadata_root bytea NOT NULL, + FOREIGN KEY(snapshot_id,namespace_uuid) REFERENCES mst2_snapshot_storage_route(snapshot_id,namespace_uuid), + FOREIGN KEY(snapshot_id,prepare_id,metadata_root) REFERENCES mst2_snapshot_context(snapshot_id,prepare_id,metadata_root), + UNIQUE(snapshot_id,namespace_uuid,session_incarnation,prepare_id,metadata_root) +); +CREATE TABLE mst2_lease_storage_route ( + lease_id text PRIMARY KEY, + snapshot_id text NOT NULL, + namespace_uuid uuid NOT NULL, + session_incarnation uuid NOT NULL, + prepare_id text NOT NULL, + metadata_root bytea NOT NULL CHECK (octet_length(metadata_root)=32), + authorization_epoch bigint NOT NULL, + publication_sequence bigint NOT NULL, + writer_epoch bigint NOT NULL, + certificate_receipt_id bigint NOT NULL, + FOREIGN KEY(snapshot_id,namespace_uuid) REFERENCES mst2_snapshot_storage_route(snapshot_id,namespace_uuid) +); +CREATE INDEX mst2_lease_storage_route_incarnation ON mst2_lease_storage_route(namespace_uuid,session_incarnation,lease_id); + +CREATE FUNCTION mst2_route_scope_valid(caller_schema text) RETURNS boolean LANGUAGE sql VOLATILE +SET search_path=$CORE_SCHEMA$,pg_catalog,pg_temp AS $$ + SELECT NOT pg_catalog.pg_is_in_recovery() AND pg_catalog.current_setting('transaction_isolation')='read committed' + AND EXISTS(SELECT 1 FROM mst2_metadata_namespace n JOIN mst2_metadata_storage_scope s ON s.singleton=1 + JOIN pg_catalog.pg_database d ON d.datname=pg_catalog.current_database() + JOIN pg_catalog.pg_namespace c ON c.oid=n.core_schema_oid AND c.nspname=n.core_schema + WHERE n.singleton=1 AND caller_schema=n.core_schema AND n.core_schema=$CORE_LITERAL$ AND n.core_schema_oid=$CORE_OID$ + AND n.database_name=d.datname AND n.database_oid=d.oid AND n.storage_uuid=s.storage_uuid + AND n.server_address IS NOT DISTINCT FROM pg_catalog.inet_server_addr()::text + AND n.server_port IS NOT DISTINCT FROM pg_catalog.inet_server_port() + AND n.metadata_schema=n.core_schema AND n.metadata_schema_oid=n.core_schema_oid + AND n.graph_domain='generic-v1' AND n.family_identity='v3-generic-session-1' + AND n.admission_state='G_ADMITTED_Q_CLOSED' AND n.collector_state='CLOSED') +$$; + +CREATE FUNCTION mst2_route_lock_held(k1 integer,k2 integer,exclusive_only boolean DEFAULT true) RETURNS boolean LANGUAGE sql VOLATILE +SET search_path=$CORE_SCHEMA$,pg_catalog,pg_temp AS $$ + SELECT EXISTS(SELECT 1 FROM pg_catalog.pg_locks WHERE locktype='advisory' AND pid=pg_catalog.pg_backend_pid() + AND database=(SELECT oid FROM pg_catalog.pg_database WHERE datname=pg_catalog.current_database()) + AND classid=k1::oid AND objid=k2::oid AND objsubid=2 AND granted AND (NOT exclusive_only OR mode='ExclusiveLock')) +$$; + +CREATE FUNCTION mst2_route_enter(caller_schema text) RETURNS uuid LANGUAGE plpgsql VOLATILE +SET search_path=$CORE_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE n mst2_metadata_namespace%ROWTYPE; mono_held boolean; route_held boolean; +BEGIN + IF NOT mst2_route_scope_valid(caller_schema) THEN RAISE EXCEPTION 'storage route captured primary scope is unavailable'; END IF; + SELECT * INTO STRICT n FROM mst2_metadata_namespace WHERE singleton=1; + mono_held:=mst2_route_lock_held(1297043024,n.mono_lock_key2); + route_held:=mst2_route_lock_held(1296718001,pg_catalog.hashtext(n.core_schema)); + IF mst2_route_lock_held(1296718001,pg_catalog.hashtext(n.core_schema),false) AND NOT mono_held THEN + RAISE EXCEPTION 'storage route lock was acquired before core mono'; + END IF; + IF EXISTS(SELECT 1 FROM mst2_metadata_namespace x + WHERE mst2_route_lock_held(1296717362,pg_catalog.hashtext(x.metadata_schema),false)) + AND NOT (mono_held AND route_held) THEN + RAISE EXCEPTION 'storage route cannot acquire core locks after retention'; + END IF; + PERFORM pg_catalog.set_config('lock_timeout','5000ms',true); + PERFORM pg_catalog.pg_advisory_xact_lock(1297043024,n.mono_lock_key2); + PERFORM pg_catalog.pg_advisory_xact_lock(1296718001,pg_catalog.hashtext(n.core_schema)); + FOR n IN SELECT * FROM mst2_metadata_namespace ORDER BY namespace_uuid LOOP + PERFORM pg_catalog.pg_advisory_xact_lock(1296717362,pg_catalog.hashtext(n.metadata_schema)); + END LOOP; + RETURN n.namespace_uuid; +END $$; + +-- This wrapper observes the caller before entering a fixed trusted path. It +-- uses only pg_catalog and the explicitly captured core function/relation. +CREATE FUNCTION mst2_route_statement_barrier() RETURNS trigger LANGUAGE plpgsql VOLATILE AS $$ +BEGIN + IF TG_TABLE_SCHEMA<>$CORE_LITERAL$ OR NOT EXISTS(SELECT 1 FROM pg_catalog.pg_class + WHERE oid=TG_RELID AND relnamespace=$CORE_OID$) THEN + RAISE EXCEPTION 'storage route mutation is outside its captured core schema'; + END IF; + PERFORM $CORE_SCHEMA$.mst2_route_enter(pg_catalog.current_schema()); + RETURN NULL; +END $$; + +CREATE FUNCTION mst2_route_profile(source_domain text,tagged_root_tree_oid text,scope text,schema_version smallint, + metadata_codec smallint,materialization_policy smallint,fs_semantics smallint,access_projection smallint, + verification_revision integer,projection_revision smallint) RETURNS jsonb LANGUAGE sql IMMUTABLE STRICT +SET search_path=$CORE_SCHEMA$,pg_catalog,pg_temp AS $$ + SELECT pg_catalog.jsonb_build_object('source_domain',source_domain,'tagged_root_tree_oid',tagged_root_tree_oid, + 'scope',scope,'schema_version',schema_version,'metadata_codec',metadata_codec, + 'materialization_policy',materialization_policy,'fs_semantics',fs_semantics, + 'access_projection',access_projection,'verification_revision',verification_revision, + 'projection_revision',projection_revision) +$$; + +CREATE FUNCTION mst2_route_snapshot_proof(sid text) RETURNS boolean LANGUAGE sql VOLATILE +SET search_path=$CORE_SCHEMA$,pg_catalog,pg_temp AS $$ + SELECT EXISTS(SELECT 1 FROM mst2_snapshot_storage_route r JOIN mst2_snapshot_context s USING(snapshot_id) + JOIN mst2_generic_session_storage_binding b USING(snapshot_id,namespace_uuid) + JOIN mst2_metadata_namespace n USING(namespace_uuid) + JOIN mst2_metadata_prepare p ON p.prepare_id=s.prepare_id + WHERE r.snapshot_id=sid AND r.canonical_descriptor=s.canonical_descriptor + AND r.instance_id=s.instance_id AND r.commit_oid=s.commit_oid AND r.root_tree_oid=s.root_tree_oid + AND r.metadata_root=s.metadata_root AND r.source_profile=mst2_route_profile(p.source_domain,p.tagged_root_tree_oid, + p.scope,p.schema_version,p.metadata_codec,p.materialization_policy,p.fs_semantics,p.access_projection, + p.verification_revision,p.projection_revision) + AND b.prepare_id=s.prepare_id AND b.metadata_root=s.metadata_root + AND p.state='COMMITTED' AND p.metadata_root=s.metadata_root AND p.source_domain='native-git' + AND p.tagged_root_tree_oid IN ('sha1:'||s.root_tree_oid,'sha256:'||s.root_tree_oid) + AND n.graph_domain='generic-v1') +$$; + +CREATE FUNCTION mst2_route_lease_proof(lid text) RETURNS boolean LANGUAGE sql VOLATILE +SET search_path=$CORE_SCHEMA$,pg_catalog,pg_temp AS $$ + SELECT EXISTS(SELECT 1 FROM mst2_lease_storage_route r JOIN mst2_snapshot_lease l USING(lease_id) + JOIN mst2_generic_session_storage_binding b ON b.snapshot_id=r.snapshot_id AND b.namespace_uuid=r.namespace_uuid + AND b.session_incarnation=r.session_incarnation AND b.prepare_id=r.prepare_id AND b.metadata_root=r.metadata_root + WHERE r.lease_id=lid AND r.snapshot_id=l.snapshot_id AND r.authorization_epoch=l.authorization_epoch + AND r.publication_sequence=l.publication_sequence AND r.writer_epoch=l.writer_epoch + AND r.certificate_receipt_id=l.certificate_receipt_id AND mst2_route_snapshot_proof(r.snapshot_id)) +$$; + +CREATE FUNCTION mst2_route_select_snapshot(sid text,caller_schema text) +RETURNS TABLE(context_present boolean,route_present boolean,valid boolean) LANGUAGE plpgsql VOLATILE +SET search_path=$CORE_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + IF NOT mst2_route_scope_valid(caller_schema) THEN RAISE EXCEPTION 'storage route captured primary scope is unavailable'; END IF; + RETURN QUERY SELECT EXISTS(SELECT 1 FROM mst2_snapshot_context WHERE snapshot_id=sid), + EXISTS(SELECT 1 FROM mst2_snapshot_storage_route WHERE snapshot_id=sid),mst2_route_snapshot_proof(sid); +END $$; + +CREATE FUNCTION mst2_route_select_lease(lid text,caller_schema text) +RETURNS TABLE(actual_sid text,route_present boolean,valid boolean) LANGUAGE plpgsql VOLATILE +SET search_path=$CORE_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + IF NOT mst2_route_scope_valid(caller_schema) THEN RAISE EXCEPTION 'storage route captured primary scope is unavailable'; END IF; + RETURN QUERY SELECT (SELECT snapshot_id FROM mst2_snapshot_lease WHERE lease_id=lid), + EXISTS(SELECT 1 FROM mst2_lease_storage_route WHERE lease_id=lid),mst2_route_lease_proof(lid); +END $$; + +CREATE FUNCTION mst2_route_immutable() RETURNS trigger LANGUAGE plpgsql +SET search_path=$CORE_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN RAISE EXCEPTION 'storage route identity and historical ledger are immutable'; END $$; + +CREATE FUNCTION mst2_route_insert_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$CORE_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + IF TG_TABLE_NAME='mst2_snapshot_storage_route' THEN + IF NOT EXISTS(SELECT 1 FROM mst2_snapshot_context s JOIN mst2_metadata_prepare p ON p.prepare_id=s.prepare_id + JOIN mst2_metadata_namespace n ON n.namespace_uuid=NEW.namespace_uuid + WHERE s.snapshot_id=NEW.snapshot_id AND NEW.canonical_descriptor=s.canonical_descriptor + AND NEW.instance_id=s.instance_id AND NEW.commit_oid=s.commit_oid AND NEW.root_tree_oid=s.root_tree_oid + AND NEW.metadata_root=s.metadata_root AND NEW.source_profile=mst2_route_profile(p.source_domain,p.tagged_root_tree_oid, + p.scope,p.schema_version,p.metadata_codec,p.materialization_policy,p.fs_semantics,p.access_projection, + p.verification_revision,p.projection_revision) + AND p.state='COMMITTED' AND p.metadata_root=s.metadata_root AND p.source_domain='native-git' + AND (p.graph_domain IS NULL OR p.graph_domain='generic-v1') + AND p.tagged_root_tree_oid IN ('sha1:'||s.root_tree_oid,'sha256:'||s.root_tree_oid)) THEN + RAISE EXCEPTION 'storage route is not derived from its actual generic context'; + END IF; + ELSIF TG_TABLE_NAME='mst2_generic_session_storage_binding' THEN + IF NOT EXISTS(SELECT 1 FROM mst2_snapshot_context s JOIN mst2_snapshot_storage_route r USING(snapshot_id) + WHERE s.snapshot_id=NEW.snapshot_id AND r.namespace_uuid=NEW.namespace_uuid + AND s.prepare_id=NEW.prepare_id AND s.metadata_root=NEW.metadata_root) THEN + RAISE EXCEPTION 'storage route generic incarnation is not its actual context'; + END IF; + ELSIF TG_TABLE_NAME='mst2_lease_storage_route' THEN + IF NOT EXISTS(SELECT 1 FROM mst2_snapshot_lease l JOIN mst2_generic_session_storage_binding b USING(snapshot_id) + WHERE l.lease_id=NEW.lease_id AND l.snapshot_id=NEW.snapshot_id AND b.namespace_uuid=NEW.namespace_uuid + AND b.session_incarnation=NEW.session_incarnation AND b.prepare_id=NEW.prepare_id AND b.metadata_root=NEW.metadata_root + AND l.authorization_epoch=NEW.authorization_epoch AND l.publication_sequence=NEW.publication_sequence + AND l.writer_epoch=NEW.writer_epoch AND l.certificate_receipt_id=NEW.certificate_receipt_id + AND mst2_route_snapshot_proof(l.snapshot_id)) THEN + RAISE EXCEPTION 'storage route lease is not its exact generic incarnation and source'; + END IF; + ELSE RAISE EXCEPTION 'storage route insertion target is not registered'; END IF; + RETURN NEW; +END $$; + +CREATE FUNCTION mst2_route_context_insert() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$CORE_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + INSERT INTO mst2_snapshot_storage_route + SELECT s.snapshot_id,n.namespace_uuid,s.canonical_descriptor,s.instance_id,s.commit_oid,s.root_tree_oid, + s.metadata_root,mst2_route_profile(p.source_domain,p.tagged_root_tree_oid,p.scope,p.schema_version,p.metadata_codec, + p.materialization_policy,p.fs_semantics,p.access_projection,p.verification_revision,p.projection_revision) + FROM mst2_snapshot_context s JOIN mst2_metadata_prepare p ON p.prepare_id=s.prepare_id + CROSS JOIN mst2_metadata_namespace n WHERE s.snapshot_id=NEW.snapshot_id AND n.singleton=1 + ON CONFLICT(snapshot_id) DO NOTHING; + INSERT INTO mst2_generic_session_storage_binding + SELECT pg_catalog.gen_random_uuid(),s.snapshot_id,r.namespace_uuid,s.prepare_id,s.metadata_root + FROM mst2_snapshot_context s JOIN mst2_snapshot_storage_route r USING(snapshot_id) WHERE s.snapshot_id=NEW.snapshot_id + ON CONFLICT(snapshot_id) DO NOTHING; + RETURN NULL; +END $$; +CREATE FUNCTION mst2_route_lease_insert() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$CORE_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + INSERT INTO mst2_lease_storage_route + SELECT l.lease_id,l.snapshot_id,b.namespace_uuid,b.session_incarnation,b.prepare_id,b.metadata_root, + l.authorization_epoch,l.publication_sequence,l.writer_epoch,l.certificate_receipt_id + FROM mst2_snapshot_lease l JOIN mst2_generic_session_storage_binding b USING(snapshot_id) WHERE l.lease_id=NEW.lease_id + ON CONFLICT(lease_id) DO NOTHING; + RETURN NULL; +END $$; +CREATE FUNCTION mst2_route_complete() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$CORE_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + IF TG_TABLE_NAME='mst2_snapshot_context' THEN + IF NOT mst2_route_snapshot_proof(NEW.snapshot_id) THEN RAISE EXCEPTION 'storage route context committed without exact routing'; END IF; + ELSIF NOT mst2_route_lease_proof(NEW.lease_id) THEN RAISE EXCEPTION 'storage route lease committed without exact routing'; END IF; + RETURN NULL; +END $$; + +DO $$ DECLARE t text; BEGIN + FOREACH t IN ARRAY ARRAY['mst2_metadata_namespace','mst2_snapshot_storage_route', + 'mst2_generic_session_storage_binding','mst2_lease_storage_route'] LOOP + EXECUTE pg_catalog.format('CREATE TRIGGER mst2_00_route_statement_barrier BEFORE INSERT OR UPDATE OR DELETE ON %I FOR EACH STATEMENT EXECUTE FUNCTION mst2_route_statement_barrier()',t); + EXECUTE pg_catalog.format('CREATE TRIGGER mst2_route_immutable BEFORE UPDATE OR DELETE ON %I FOR EACH ROW EXECUTE FUNCTION mst2_route_immutable()',t); + EXECUTE pg_catalog.format('CREATE TRIGGER mst2_route_truncate_guard BEFORE TRUNCATE ON %I FOR EACH STATEMENT EXECUTE FUNCTION mst2_route_immutable()',t); + IF t<>'mst2_metadata_namespace' THEN + EXECUTE pg_catalog.format('CREATE TRIGGER mst2_route_insert_guard BEFORE INSERT ON %I FOR EACH ROW EXECUTE FUNCTION mst2_route_insert_guard()',t); + END IF; + END LOOP; + FOREACH t IN ARRAY ARRAY['mst2_snapshot_context','mst2_snapshot_lease'] LOOP + EXECUTE pg_catalog.format('CREATE TRIGGER mst2_00_route_statement_barrier BEFORE INSERT ON %I FOR EACH STATEMENT EXECUTE FUNCTION mst2_route_statement_barrier()',t); + EXECUTE pg_catalog.format('CREATE TRIGGER mst2_route_source_delete_guard BEFORE DELETE ON %I FOR EACH ROW EXECUTE FUNCTION mst2_route_immutable()',t); + EXECUTE pg_catalog.format('CREATE TRIGGER mst2_route_source_truncate_guard BEFORE TRUNCATE ON %I FOR EACH STATEMENT EXECUTE FUNCTION mst2_route_immutable()',t); + EXECUTE pg_catalog.format('CREATE CONSTRAINT TRIGGER mst2_route_complete AFTER INSERT ON %I DEFERRABLE INITIALLY DEFERRED FOR EACH ROW EXECUTE FUNCTION mst2_route_complete()',t); + END LOOP; +END $$; +CREATE TRIGGER mst2_route_namespace_registration_closed BEFORE INSERT ON mst2_metadata_namespace + FOR EACH ROW EXECUTE FUNCTION mst2_route_immutable(); +CREATE TRIGGER mst2_route_context_insert AFTER INSERT ON mst2_snapshot_context + FOR EACH ROW EXECUTE FUNCTION mst2_route_context_insert(); +CREATE TRIGGER mst2_route_lease_insert AFTER INSERT ON mst2_snapshot_lease + FOR EACH ROW EXECUTE FUNCTION mst2_route_lease_insert(); + +INSERT INTO mst2_snapshot_storage_route +SELECT s.snapshot_id,n.namespace_uuid,s.canonical_descriptor,s.instance_id,s.commit_oid,s.root_tree_oid, + s.metadata_root,mst2_route_profile(p.source_domain,p.tagged_root_tree_oid,p.scope,p.schema_version,p.metadata_codec, + p.materialization_policy,p.fs_semantics,p.access_projection,p.verification_revision,p.projection_revision) +FROM mst2_snapshot_context s JOIN mst2_metadata_prepare p ON p.prepare_id=s.prepare_id +CROSS JOIN mst2_metadata_namespace n; +INSERT INTO mst2_generic_session_storage_binding +SELECT pg_catalog.gen_random_uuid(),s.snapshot_id,r.namespace_uuid,s.prepare_id,s.metadata_root +FROM mst2_snapshot_context s JOIN mst2_snapshot_storage_route r USING(snapshot_id); +INSERT INTO mst2_lease_storage_route +SELECT l.lease_id,l.snapshot_id,b.namespace_uuid,b.session_incarnation,b.prepare_id,b.metadata_root, + l.authorization_epoch,l.publication_sequence,l.writer_epoch,l.certificate_receipt_id +FROM mst2_snapshot_lease l JOIN mst2_generic_session_storage_binding b USING(snapshot_id); +DO $$ BEGIN + IF NOT mst2_route_scope_valid($CORE_LITERAL$) + OR EXISTS(SELECT 1 FROM mst2_snapshot_context s WHERE NOT mst2_route_snapshot_proof(s.snapshot_id)) + OR EXISTS(SELECT 1 FROM mst2_snapshot_lease l WHERE NOT mst2_route_lease_proof(l.lease_id)) THEN + RAISE EXCEPTION 'storage route backfill did not cover the complete durable inventory'; + END IF; +END $$; diff --git a/src/jupiter/migration/m20261008_000100_add_mst2_chunk_maps.rs b/src/jupiter/migration/m20261008_000100_add_mst2_chunk_maps.rs new file mode 100644 index 00000000..cd597a16 --- /dev/null +++ b/src/jupiter/migration/m20261008_000100_add_mst2_chunk_maps.rs @@ -0,0 +1,28 @@ +//! Append-only, independently authenticated chunk-map read indexes. + +use sea_orm::{ConnectionTrait, DbBackend}; +use sea_orm_migration::prelude::*; + +#[derive(DeriveMigrationName)] +pub struct Migration; + +#[async_trait::async_trait] +impl MigrationTrait for Migration { + async fn up(&self, manager: &SchemaManager) -> Result<(), DbErr> { + if manager.get_database_backend() != DbBackend::Postgres { + return Err(DbErr::Custom( + "persisted chunk maps require primary PostgreSQL".into(), + )); + } + manager + .get_connection() + .execute_unprepared(include_str!("m20261008_000100_chunk_maps.sql")) + .await?; + Ok(()) + } + + async fn down(&self, _manager: &SchemaManager) -> Result<(), DbErr> { + // No collector or migration rollback may silently discard receipts. + Ok(()) + } +} diff --git a/src/jupiter/migration/m20261008_000100_chunk_maps.sql b/src/jupiter/migration/m20261008_000100_chunk_maps.sql new file mode 100644 index 00000000..75b452d0 --- /dev/null +++ b/src/jupiter/migration/m20261008_000100_chunk_maps.sql @@ -0,0 +1,90 @@ +CREATE TABLE mst2_chunk_map ( + map_id bytea PRIMARY KEY CHECK (pg_catalog.octet_length(map_id)=32), + descriptor bytea NOT NULL CHECK (pg_catalog.octet_length(descriptor)=100), + page_count integer NOT NULL CHECK (page_count BETWEEN 1 AND 32768), + pages_root bytea NOT NULL CHECK (pg_catalog.octet_length(pages_root)=32) +); +CREATE TABLE mst2_chunk_map_leaf ( + map_id bytea NOT NULL REFERENCES mst2_chunk_map(map_id), + page_index integer NOT NULL CHECK (page_index BETWEEN 0 AND 32767), + payload bytea NOT NULL CHECK (pg_catalog.octet_length(payload) BETWEEN 48 AND 8208), + PRIMARY KEY(map_id,page_index) +); +CREATE TABLE mst2_chunk_map_node ( + map_id bytea NOT NULL REFERENCES mst2_chunk_map(map_id), + first_page integer NOT NULL CHECK (first_page BETWEEN 0 AND 32767), + page_count integer NOT NULL CHECK (page_count BETWEEN 1 AND 32768), + digest bytea NOT NULL CHECK (pg_catalog.octet_length(digest)=32), + PRIMARY KEY(map_id,first_page,page_count), + CHECK (first_page+page_count<=32768) +); +CREATE TABLE mst2_chunk_map_source ( + storage_domain text NOT NULL CHECK (storage_domain='git'), + git_oid text NOT NULL CHECK (git_oid ~ '^([0-9a-f]{40}|[0-9a-f]{64})$'), + object_kind text NOT NULL CHECK (object_kind='blob'), + fact_id bigint NOT NULL CHECK (fact_id>0), + source_id bytea NOT NULL UNIQUE CHECK (pg_catalog.octet_length(source_id)=32), + source_bytes bytea NOT NULL CHECK (pg_catalog.octet_length(source_bytes) BETWEEN 1 AND 2048), + primary_scope bytea NOT NULL CHECK (pg_catalog.octet_length(primary_scope) BETWEEN 1 AND 1024), + map_id bytea NOT NULL REFERENCES mst2_chunk_map(map_id), + receipt_digest bytea NOT NULL CHECK (pg_catalog.octet_length(receipt_digest)=32), + PRIMARY KEY(storage_domain,git_oid,object_kind) +); + +-- These rows are indexes, not proof that any source body was consumed. The +-- trusted object writer alone publishes the independent immutable receipt. +CREATE FUNCTION mst2_chunk_map_immutable() RETURNS trigger LANGUAGE plpgsql AS $$ +BEGIN RAISE EXCEPTION 'chunk map indexes and source receipts are append-only'; END $$; + +CREATE FUNCTION mst2_chunk_map_primary() RETURNS trigger LANGUAGE plpgsql AS $$ +BEGIN + IF pg_catalog.current_schema() IS DISTINCT FROM TG_TABLE_SCHEMA + OR pg_catalog.pg_is_in_recovery() + OR pg_catalog.current_setting('transaction_isolation')<>'read committed' THEN + RAISE EXCEPTION 'chunk map insertion requires its actual primary schema and READ COMMITTED'; + END IF; + RETURN NULL; +END $$; + +CREATE FUNCTION mst2_chunk_map_complete() RETURNS trigger LANGUAGE plpgsql AS $$ +DECLARE m record; leaves bigint; nodes bigint; root bytea; first_leaf integer; last_leaf integer; +BEGIN + EXECUTE pg_catalog.format('SELECT * FROM %I.mst2_chunk_map WHERE map_id=$1',TG_TABLE_SCHEMA) + INTO m USING NEW.map_id; + EXECUTE pg_catalog.format('SELECT pg_catalog.count(*),pg_catalog.min(page_index),pg_catalog.max(page_index) FROM %I.mst2_chunk_map_leaf WHERE map_id=$1',TG_TABLE_SCHEMA) + INTO leaves,first_leaf,last_leaf USING NEW.map_id; + EXECUTE pg_catalog.format('SELECT pg_catalog.count(*) FROM %I.mst2_chunk_map_node WHERE map_id=$1',TG_TABLE_SCHEMA) + INTO nodes USING NEW.map_id; + EXECUTE pg_catalog.format('SELECT digest FROM %I.mst2_chunk_map_node WHERE map_id=$1 AND first_page=0 AND page_count=$2',TG_TABLE_SCHEMA) + INTO root USING NEW.map_id,m.page_count; + IF m.map_id IS NULL OR leaves<>m.page_count OR first_leaf<>0 OR last_leaf<>m.page_count-1 + OR nodes<>2*m.page_count-1 OR root IS DISTINCT FROM m.pages_root + OR pg_catalog.substring(m.descriptor,69,32) IS DISTINCT FROM m.pages_root + OR pg_catalog.sha256(pg_catalog.convert_to('mega.mst2.chunkmap','UTF8')||pg_catalog.decode('00','hex')||m.descriptor) IS DISTINCT FROM m.map_id THEN + RAISE EXCEPTION 'chunk map installation is incomplete or inconsistent'; + END IF; + RETURN NULL; +END $$; + +CREATE FUNCTION mst2_chunk_map_index_insert() RETURNS trigger LANGUAGE plpgsql AS $$ +DECLARE admitted boolean; +BEGIN + EXECUTE pg_catalog.format('SELECT EXISTS(SELECT 1 FROM %I.mst2_chunk_map_source WHERE map_id=$1)',TG_TABLE_SCHEMA) + INTO admitted USING NEW.map_id; + IF admitted THEN RAISE EXCEPTION 'admitted chunk map index cannot acquire more rows'; END IF; + RETURN NEW; +END $$; + +DO $$ DECLARE t text; BEGIN + FOREACH t IN ARRAY ARRAY['mst2_chunk_map','mst2_chunk_map_leaf','mst2_chunk_map_node','mst2_chunk_map_source'] LOOP + EXECUTE pg_catalog.format('CREATE TRIGGER mst2_chunk_map_primary BEFORE INSERT ON %I FOR EACH STATEMENT EXECUTE FUNCTION mst2_chunk_map_primary()',t); + EXECUTE pg_catalog.format('CREATE TRIGGER mst2_chunk_map_immutable BEFORE UPDATE OR DELETE ON %I FOR EACH ROW EXECUTE FUNCTION mst2_chunk_map_immutable()',t); + EXECUTE pg_catalog.format('CREATE TRIGGER mst2_chunk_map_no_truncate BEFORE TRUNCATE ON %I FOR EACH STATEMENT EXECUTE FUNCTION mst2_chunk_map_immutable()',t); + END LOOP; +END $$; +CREATE CONSTRAINT TRIGGER mst2_chunk_map_complete AFTER INSERT ON mst2_chunk_map_source + DEFERRABLE INITIALLY DEFERRED FOR EACH ROW EXECUTE FUNCTION mst2_chunk_map_complete(); +CREATE TRIGGER mst2_chunk_map_index_insert BEFORE INSERT ON mst2_chunk_map_leaf + FOR EACH ROW EXECUTE FUNCTION mst2_chunk_map_index_insert(); +CREATE TRIGGER mst2_chunk_map_index_insert BEFORE INSERT ON mst2_chunk_map_node + FOR EACH ROW EXECUTE FUNCTION mst2_chunk_map_index_insert(); diff --git a/src/jupiter/migration/m20261008_000200_add_mst2_rooted_qualified_family.rs b/src/jupiter/migration/m20261008_000200_add_mst2_rooted_qualified_family.rs new file mode 100644 index 00000000..fd613dce --- /dev/null +++ b/src/jupiter/migration/m20261008_000200_add_mst2_rooted_qualified_family.rs @@ -0,0 +1,125 @@ +//! A bounded namespace registry; physical Q provisioning occurs at bootstrap. + +use sea_orm::{ConnectionTrait, DbBackend, Statement}; +use sea_orm_migration::prelude::*; + +#[derive(DeriveMigrationName)] +pub struct Migration; + +#[async_trait::async_trait] +impl MigrationTrait for Migration { + async fn up(&self, manager: &SchemaManager) -> Result<(), DbErr> { + let connection = manager.get_connection(); + let row = connection.query_one_raw(Statement::from_string(DbBackend::Postgres, + "SELECT current_schema() AS schema,n.oid::bigint AS oid FROM pg_catalog.pg_namespace n WHERE n.nspname=current_schema()")) + .await?.ok_or_else(|| DbErr::Custom("qualified family core schema is missing".into()))?; + let schema: String = row.try_get("", "schema")?; + let oid: i64 = row.try_get("", "oid")?; + let quoted = format!("\"{}\"", schema.replace('"', "\"\"")); + let literal = format!("'{}'", schema.replace('\'', "''")); + let old = include_str!("m20261007_000600_storage_routes.sql"); + let start = old + .find("CREATE FUNCTION mst2_route_insert_guard()") + .ok_or_else(|| { + DbErr::Custom("generic route insertion guard source is missing".into()) + })?; + let end = old[start..] + .find("CREATE FUNCTION mst2_route_context_insert()") + .ok_or_else(|| { + DbErr::Custom("generic route insertion guard boundary is missing".into()) + })? + + start; + let guard = old[start..end].replace("CREATE FUNCTION", "CREATE OR REPLACE FUNCTION") + .replace("WHERE s.snapshot_id=NEW.snapshot_id AND NEW.canonical_descriptor=s.canonical_descriptor", + "WHERE n.singleton=1 AND n.graph_domain='generic-v1' AND s.snapshot_id=NEW.snapshot_id AND NEW.canonical_descriptor=s.canonical_descriptor"); + let catalog = include_str!("../storage/qualified_family_catalog.sql") + .replace("$CORE_OID$", "c_oid") + .replace("$Q_OID$", "q_oid") + .replace("$EXEMPT_Q_OID$", "exempt_q_oid"); + let shape = include_str!("../storage/qualified_family_shape.sql"); + let implementation = hex::encode( + super::super::storage::qualified_metadata_family::implementation_fingerprint(), + ); + let sql = include_str!("m20261008_000200_rooted_qualified_family.sql") + .replace("$CATALOG_SQL$", &catalog) + .replace("$SHAPE_SQL$", shape) + .replace("$GENERIC_INSERT_GUARD$", &guard) + .replace( + "$SOURCE_REVISION_SQL$", + include_str!("../storage/qualified_source_revision.sql"), + ) + .replace( + "$ROOTED_ROUTES_SQL$", + include_str!("m20261008_000200_rooted_routes.sql"), + ) + .replace("$CORE_SCHEMA$", "ed) + .replace("$CORE_LITERAL$", &literal) + .replace("$IMPLEMENTATION_SHA$", &implementation) + .replace("$CORE_OID$", &oid.to_string()); + connection.execute_unprepared(&sql).await?; + // Build the expected physical shape from trusted source exactly once, + // under this forward migration, without registering a namespace or + // preserving any template relations/history after the transaction. + let template_uuid = uuid::Uuid::new_v4().to_string(); + let template_schema = format!("mst2q_{}", template_uuid.replace('-', "")); + let template_quoted = format!("\"{template_schema}\""); + connection + .execute_unprepared(&format!("CREATE SCHEMA {template_quoted}")) + .await?; + let template_oid: i64 = connection + .query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT oid::bigint AS oid FROM pg_catalog.pg_namespace WHERE nspname=$1", + [template_schema.clone().into()], + )) + .await? + .ok_or_else(|| DbErr::Custom("qualified family template schema is missing".into()))? + .try_get("", "oid")?; + let template_storage = uuid::Uuid::new_v4().to_string(); + connection + .execute_unprepared( + &super::super::storage::qualified_metadata_family::render_family( + &schema, + oid, + &template_schema, + template_oid, + &template_uuid, + &template_storage, + ), + ) + .await?; + let expected_shape: Vec = connection.query_one_raw(Statement::from_sql_and_values(DbBackend::Postgres, + format!("SELECT {quoted}.mst2_route_family_shape($1::bigint::oid,$2::uuid,$3) AS fingerprint"), + [template_oid.into(),template_uuid.into(),template_storage.into()])).await? + .ok_or_else(||DbErr::Custom("qualified family template shape is missing".into()))?.try_get("","fingerprint")?; + connection.execute_unprepared(&format!("SET LOCAL search_path={quoted},pg_catalog,pg_temp; DROP SCHEMA {template_quoted} CASCADE")).await?; + let authority_catalog: Vec = connection + .query_one_raw(Statement::from_string( + DbBackend::Postgres, + format!("SELECT {quoted}.mst2_route_family_catalog({oid},0::oid) AS fingerprint"), + )) + .await? + .ok_or_else(|| { + DbErr::Custom("qualified family core authority catalog is missing".into()) + })? + .try_get("", "fingerprint")?; + connection + .execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + format!("INSERT INTO {quoted}.mst2_qualified_family_policy VALUES(1,$1,$2,$3)"), + [ + hex::decode(implementation) + .map_err(|e| DbErr::Custom(e.to_string()))? + .into(), + expected_shape.into(), + authority_catalog.into(), + ], + )) + .await?; + Ok(()) + } + + async fn down(&self, _manager: &SchemaManager) -> Result<(), DbErr> { + Ok(()) + } +} diff --git a/src/jupiter/migration/m20261008_000200_rooted_qualified_family.sql b/src/jupiter/migration/m20261008_000200_rooted_qualified_family.sql new file mode 100644 index 00000000..22038ef2 --- /dev/null +++ b/src/jupiter/migration/m20261008_000200_rooted_qualified_family.sql @@ -0,0 +1,161 @@ +SELECT mst2_route_enter(current_schema()); +SET LOCAL search_path=$CORE_SCHEMA$,pg_catalog,pg_temp; +LOCK TABLE mst2_metadata_namespace IN ACCESS EXCLUSIVE MODE; + +ALTER TABLE mst2_metadata_namespace + DROP CONSTRAINT mst2_metadata_namespace_pkey, + ALTER COLUMN singleton DROP NOT NULL, + DROP CONSTRAINT mst2_metadata_namespace_family_identity_check, + DROP CONSTRAINT mst2_metadata_namespace_graph_domain_check, + DROP CONSTRAINT mst2_metadata_namespace_admission_state_check, + DROP CONSTRAINT mst2_metadata_namespace_collector_state_check, + DROP CONSTRAINT mst2_metadata_namespace_check, + ADD COLUMN metadata_storage_uuid text, + ADD COLUMN implementation_fingerprint bytea, + ADD COLUMN catalog_fingerprint bytea, + ADD CONSTRAINT mst2_namespace_family_pair CHECK (( + (singleton=1 AND family_identity='v3-generic-session-1' AND graph_domain='generic-v1' + AND admission_state='G_ADMITTED_Q_CLOSED' AND collector_state='CLOSED' AND core_schema=metadata_schema + AND core_schema_oid=metadata_schema_oid AND metadata_storage_uuid IS NULL + AND implementation_fingerprint IS NULL AND catalog_fingerprint IS NULL) + OR (singleton IS NULL AND family_identity='v3-rooted-qualified-1' AND graph_domain='qualified-v1' + AND admission_state='ROOTED_Q_ADMITTED' AND collector_state='ENABLED' AND core_schema<>metadata_schema + AND core_schema_oid<>metadata_schema_oid AND metadata_storage_uuid IS NOT NULL + AND octet_length(implementation_fingerprint)=32 AND octet_length(catalog_fingerprint)=32) + ) IS TRUE); +CREATE UNIQUE INDEX idx_mst2_namespace_generic_slot ON mst2_metadata_namespace(singleton) + WHERE singleton IS NOT NULL; +CREATE UNIQUE INDEX idx_mst2_namespace_domain ON mst2_metadata_namespace(graph_domain); +CREATE UNIQUE INDEX idx_mst2_namespace_metadata_schema ON mst2_metadata_namespace(metadata_schema_oid); + +CREATE OR REPLACE FUNCTION mst2_route_enter(caller_schema text) RETURNS uuid LANGUAGE plpgsql VOLATILE +SET search_path=$CORE_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE n mst2_metadata_namespace%ROWTYPE; g mst2_metadata_namespace%ROWTYPE; mono_held boolean; route_held boolean; +BEGIN + IF NOT mst2_route_scope_valid(caller_schema) OR (SELECT count(*) FROM mst2_metadata_namespace)>2 THEN + RAISE EXCEPTION 'storage route captured primary scope is unavailable'; + END IF; + SELECT * INTO STRICT g FROM mst2_metadata_namespace WHERE singleton=1; + mono_held:=mst2_route_lock_held(1297043024,g.mono_lock_key2); + route_held:=mst2_route_lock_held(1296718001,pg_catalog.hashtext(g.core_schema)); + IF mst2_route_lock_held(1296718001,pg_catalog.hashtext(g.core_schema),false) AND NOT mono_held THEN + RAISE EXCEPTION 'storage route lock was acquired before core mono'; + END IF; + IF EXISTS(SELECT 1 FROM mst2_metadata_namespace x + WHERE mst2_route_lock_held(1296717362,pg_catalog.hashtext(x.metadata_schema),false)) + AND NOT (mono_held AND route_held) THEN + RAISE EXCEPTION 'storage route cannot acquire core locks after retention'; + END IF; + PERFORM pg_catalog.set_config('lock_timeout','5000ms',true); + PERFORM pg_catalog.pg_advisory_xact_lock(1297043024,g.mono_lock_key2); + PERFORM pg_catalog.pg_advisory_xact_lock(1296718001,pg_catalog.hashtext(g.core_schema)); + FOR n IN SELECT * FROM mst2_metadata_namespace ORDER BY namespace_uuid LOOP + PERFORM pg_catalog.pg_advisory_xact_lock(1296717362,pg_catalog.hashtext(n.metadata_schema)); + END LOOP; + RETURN g.namespace_uuid; +END $$; + +-- The new namespace is included before acquiring any retention lock. +CREATE FUNCTION mst2_route_family_candidate_enter(caller_schema text,candidate uuid,candidate_schema text) +RETURNS void LANGUAGE plpgsql VOLATILE SET search_path=$CORE_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE g mst2_metadata_namespace%ROWTYPE; r record; +BEGIN + IF NOT mst2_route_scope_valid(caller_schema) OR candidate IS NULL + OR substr(candidate::text,15,1)<>'4' OR substr(candidate::text,20,1) NOT IN ('8','9','a','b') + OR candidate_schema IS DISTINCT FROM 'mst2q_'||replace(candidate::text,'-','') THEN + RAISE EXCEPTION 'qualified provisioning candidate is not a fresh server namespace'; + END IF; + IF EXISTS(SELECT 1 FROM pg_catalog.pg_locks WHERE locktype='advisory' AND pid=pg_catalog.pg_backend_pid() + AND database=(SELECT oid FROM pg_catalog.pg_database WHERE datname=pg_catalog.current_database()) + AND classid=1296717362::oid AND objsubid=2 AND granted) THEN + RAISE EXCEPTION 'qualified provisioning cannot extend a previously locked retention set'; + END IF; + SELECT * INTO STRICT g FROM mst2_metadata_namespace WHERE singleton=1; + IF mst2_route_lock_held(1296718001,pg_catalog.hashtext(g.core_schema),false) + AND NOT mst2_route_lock_held(1297043024,g.mono_lock_key2) THEN + RAISE EXCEPTION 'storage route lock was acquired before core mono'; + END IF; + PERFORM pg_catalog.set_config('lock_timeout','5000ms',true); + PERFORM pg_catalog.pg_advisory_xact_lock(1297043024,g.mono_lock_key2); + PERFORM pg_catalog.pg_advisory_xact_lock(1296718001,pg_catalog.hashtext(g.core_schema)); + -- A racing successful bootstrap can register while this caller waits. + IF EXISTS(SELECT 1 FROM mst2_metadata_namespace WHERE graph_domain='qualified-v1') THEN + RAISE EXCEPTION 'qualified provisioning candidate lost bootstrap serialization'; + END IF; + FOR r IN SELECT namespace_uuid,metadata_schema FROM mst2_metadata_namespace + UNION ALL SELECT candidate,candidate_schema ORDER BY namespace_uuid LOOP + PERFORM pg_catalog.pg_advisory_xact_lock(1296717362,pg_catalog.hashtext(r.metadata_schema)); + END LOOP; +END $$; + +CREATE FUNCTION mst2_route_family_catalog(c_oid oid,q_oid oid,exempt_q_oid oid DEFAULT 0::oid) RETURNS bytea LANGUAGE sql VOLATILE +SET search_path=pg_catalog,pg_temp AS $catalog$ $CATALOG_SQL$ $catalog$; +CREATE FUNCTION mst2_route_family_shape(q_oid oid,n_uuid uuid,s_uuid text) RETURNS bytea LANGUAGE sql VOLATILE +SET search_path=pg_catalog,pg_temp AS $shape$ $SHAPE_SQL$ $shape$; +CREATE TABLE mst2_qualified_family_policy ( + singleton smallint PRIMARY KEY CHECK(singleton=1),implementation_fingerprint bytea NOT NULL CHECK(octet_length(implementation_fingerprint)=32), + expected_shape bytea NOT NULL CHECK(octet_length(expected_shape)=32), + authority_catalog bytea NOT NULL CHECK(octet_length(authority_catalog)=32) +); +CREATE TRIGGER mst2_route_family_policy_immutable BEFORE UPDATE OR DELETE ON mst2_qualified_family_policy + FOR EACH ROW EXECUTE FUNCTION mst2_route_immutable(); +CREATE TRIGGER mst2_route_family_policy_truncate_guard BEFORE TRUNCATE ON mst2_qualified_family_policy + FOR EACH STATEMENT EXECUTE FUNCTION mst2_route_immutable(); + +CREATE FUNCTION mst2_route_family_registration_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$CORE_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE g mst2_metadata_namespace%ROWTYPE; stamp record; +BEGIN + SELECT * INTO STRICT g FROM mst2_metadata_namespace WHERE singleton=1; + IF (SELECT count(*) FROM mst2_metadata_namespace)<>1 OR NEW.singleton IS NOT NULL + OR substr(NEW.namespace_uuid::text,15,1)<>'4' OR substr(NEW.namespace_uuid::text,20,1) NOT IN ('8','9','a','b') + OR NEW.graph_domain<>'qualified-v1' OR NEW.family_identity<>'v3-rooted-qualified-1' + OR NEW.admission_state<>'ROOTED_Q_ADMITTED' OR NEW.collector_state<>'ENABLED' + OR ROW(NEW.core_schema,NEW.core_schema_oid,NEW.database_name,NEW.database_oid,NEW.storage_uuid, + NEW.server_address,NEW.server_port,NEW.mono_lock_key2) IS DISTINCT FROM + ROW(g.core_schema,g.core_schema_oid,g.database_name,g.database_oid,g.storage_uuid, + g.server_address,g.server_port,g.mono_lock_key2) + OR NEW.metadata_schema IS DISTINCT FROM 'mst2q_'||replace(NEW.namespace_uuid::text,'-','') + OR NEW.metadata_storage_uuid IS NOT DISTINCT FROM g.storage_uuid + OR NEW.metadata_storage_uuid IS DISTINCT FROM (NEW.metadata_storage_uuid::uuid)::text + OR substr(NEW.metadata_storage_uuid,15,1)<>'4' OR substr(NEW.metadata_storage_uuid,20,1) NOT IN ('8','9','a','b') + OR NOT EXISTS(SELECT 1 FROM pg_catalog.pg_namespace n + WHERE n.oid=NEW.metadata_schema_oid AND n.nspname=NEW.metadata_schema) + OR NEW.implementation_fingerprint IS DISTINCT FROM decode('$IMPLEMENTATION_SHA$','hex') + OR NOT mst2_route_lock_held(1297043024,g.mono_lock_key2) + OR NOT mst2_route_lock_held(1296718001,pg_catalog.hashtext(g.core_schema)) + OR NOT mst2_route_lock_held(1296717362,pg_catalog.hashtext(NEW.metadata_schema)) + OR NOT mst2_route_lock_held(1296717362,pg_catalog.hashtext(g.metadata_schema)) THEN + RAISE EXCEPTION 'qualified namespace registration has no exact admitted rooted family scope and lock set'; + END IF; + EXECUTE pg_catalog.format('SELECT * FROM %I.mst2_metadata_family_identity WHERE singleton=1',NEW.metadata_schema) + INTO STRICT stamp; + IF ROW(stamp.namespace_uuid,stamp.storage_uuid,stamp.core_schema_oid,stamp.metadata_schema_oid, + stamp.family_identity,stamp.implementation_fingerprint) IS DISTINCT FROM + ROW(NEW.namespace_uuid,NEW.metadata_storage_uuid,NEW.core_schema_oid,NEW.metadata_schema_oid, + NEW.family_identity,NEW.implementation_fingerprint) + OR NEW.catalog_fingerprint IS DISTINCT FROM mst2_route_family_catalog(NEW.core_schema_oid,NEW.metadata_schema_oid) THEN + RAISE EXCEPTION 'qualified namespace registration fingerprint disagrees with its physical family'; + END IF; + IF NOT EXISTS(SELECT 1 FROM mst2_qualified_family_policy p WHERE p.singleton=1 + AND p.implementation_fingerprint=NEW.implementation_fingerprint + AND p.authority_catalog=mst2_route_family_catalog(NEW.core_schema_oid,0::oid,NEW.metadata_schema_oid) + AND p.expected_shape=mst2_route_family_shape(NEW.metadata_schema_oid,NEW.namespace_uuid,NEW.metadata_storage_uuid)) THEN + RAISE EXCEPTION 'qualified namespace does not have the trusted complete physical family shape'; + END IF; + RETURN NEW; +END $$; +DROP TRIGGER mst2_route_namespace_registration_closed ON mst2_metadata_namespace; +CREATE TRIGGER mst2_route_namespace_registration_guard BEFORE INSERT ON mst2_metadata_namespace + FOR EACH ROW EXECUTE FUNCTION mst2_route_family_registration_guard(); + +CREATE FUNCTION mst2_metadata_has_generic_overlap(p bytea) RETURNS boolean LANGUAGE sql VOLATILE +SET search_path=$CORE_SCHEMA$,pg_catalog,pg_temp AS $$ + SELECT EXISTS(SELECT 1 FROM mst2_retention_node WHERE node_id='page:sha256:'||encode(p,'hex')) + OR EXISTS(SELECT 1 FROM mst2_retention_gc_op WHERE node_id='page:sha256:'||encode(p,'hex')) +$$; + +-- Existing generic route derivation must never select the newly registered Q row. +$GENERIC_INSERT_GUARD$ +$SOURCE_REVISION_SQL$ +$ROOTED_ROUTES_SQL$ diff --git a/src/jupiter/migration/m20261008_000200_rooted_routes.sql b/src/jupiter/migration/m20261008_000200_rooted_routes.sql new file mode 100644 index 00000000..e4cef146 --- /dev/null +++ b/src/jupiter/migration/m20261008_000200_rooted_routes.sql @@ -0,0 +1,171 @@ +CREATE FUNCTION mst2_route_qualified_namespace_valid(id uuid) RETURNS boolean LANGUAGE sql VOLATILE +SET search_path=$CORE_SCHEMA$,pg_catalog,pg_temp AS $$ + SELECT mst2_route_scope_valid($CORE_LITERAL$) AND EXISTS(SELECT 1 FROM mst2_metadata_namespace q + JOIN mst2_metadata_namespace g ON g.singleton=1 + JOIN mst2_qualified_family_policy policy ON policy.singleton=1 + JOIN pg_catalog.pg_namespace physical ON physical.oid=q.metadata_schema_oid AND physical.nspname=q.metadata_schema + WHERE q.namespace_uuid=id AND q.singleton IS NULL AND q.graph_domain='qualified-v1' + AND q.family_identity='v3-rooted-qualified-1' AND q.admission_state='ROOTED_Q_ADMITTED' AND q.collector_state='ENABLED' + AND q.metadata_schema='mst2q_'||replace(q.namespace_uuid::text,'-','') + AND q.metadata_schema_oid<>q.core_schema_oid + AND ROW(q.core_schema,q.core_schema_oid,q.database_name,q.database_oid,q.storage_uuid, + q.server_address,q.server_port,q.mono_lock_key2) IS NOT DISTINCT FROM + ROW(g.core_schema,g.core_schema_oid,g.database_name,g.database_oid,g.storage_uuid, + g.server_address,g.server_port,g.mono_lock_key2) + AND q.implementation_fingerprint=policy.implementation_fingerprint + AND policy.authority_catalog=mst2_route_family_catalog(q.core_schema_oid,0::oid,q.metadata_schema_oid) + AND q.catalog_fingerprint=mst2_route_family_catalog(q.core_schema_oid,q.metadata_schema_oid) + AND policy.expected_shape=mst2_route_family_shape(q.metadata_schema_oid,q.namespace_uuid,q.metadata_storage_uuid)) +$$; + +CREATE OR REPLACE FUNCTION mst2_route_statement_barrier() RETURNS trigger LANGUAGE plpgsql VOLATILE AS $$ +DECLARE q_id uuid; caller text:=pg_catalog.current_schema(); +BEGIN + IF TG_TABLE_SCHEMA<>$CORE_LITERAL$ OR NOT EXISTS(SELECT 1 FROM pg_catalog.pg_class + WHERE oid=TG_RELID AND relnamespace=$CORE_OID$) THEN + RAISE EXCEPTION 'storage route mutation is outside its captured core schema'; + END IF; + IF caller IS DISTINCT FROM $CORE_LITERAL$ THEN + SELECT n.namespace_uuid INTO q_id FROM $CORE_SCHEMA$.mst2_metadata_namespace n + WHERE n.metadata_schema=caller AND n.graph_domain='qualified-v1'; + IF NOT FOUND OR NOT $CORE_SCHEMA$.mst2_route_qualified_namespace_valid(q_id) THEN + RAISE EXCEPTION 'storage route mutation caller has no exact captured physical family'; END IF; + END IF; + PERFORM $CORE_SCHEMA$.mst2_route_enter($CORE_LITERAL$); + RETURN NULL; +END $$; + +CREATE FUNCTION mst2_route_qualified_snapshot_proof(sid text) RETURNS boolean LANGUAGE plpgsql VOLATILE +SET search_path=$CORE_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE r mst2_snapshot_storage_route%ROWTYPE; n mst2_metadata_namespace%ROWTYPE; valid boolean; +BEGIN + SELECT * INTO r FROM mst2_snapshot_storage_route WHERE snapshot_id=sid; + IF NOT FOUND THEN RETURN false; END IF; + SELECT * INTO n FROM mst2_metadata_namespace WHERE namespace_uuid=r.namespace_uuid; + IF NOT FOUND OR NOT mst2_route_qualified_namespace_valid(n.namespace_uuid) THEN RETURN false; END IF; + EXECUTE pg_catalog.format('SELECT %I.mst2_metadata_snapshot_route_proof($1,$2,$3,$4,$5,$6,$7)',n.metadata_schema) + INTO valid USING r.snapshot_id,r.canonical_descriptor,r.instance_id,r.commit_oid,r.root_tree_oid,r.metadata_root,r.source_profile; + RETURN coalesce(valid,false); +END $$; + +CREATE FUNCTION mst2_route_qualified_lease_proof(lid text) RETURNS boolean LANGUAGE plpgsql VOLATILE +SET search_path=$CORE_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE r mst2_lease_storage_route%ROWTYPE; n mst2_metadata_namespace%ROWTYPE; valid boolean; +BEGIN + SELECT * INTO r FROM mst2_lease_storage_route WHERE lease_id=lid; + IF NOT FOUND THEN RETURN false; END IF; + SELECT * INTO n FROM mst2_metadata_namespace WHERE namespace_uuid=r.namespace_uuid; + IF NOT FOUND OR NOT mst2_route_qualified_namespace_valid(n.namespace_uuid) THEN RETURN false; END IF; + EXECUTE pg_catalog.format('SELECT %I.mst2_metadata_lease_route_proof($1,$2,$3,$4,$5,$6,$7,$8,$9,$10)',n.metadata_schema) + INTO valid USING r.lease_id,r.snapshot_id,r.namespace_uuid,r.session_incarnation,r.prepare_id,r.metadata_root, + r.authorization_epoch,r.publication_sequence,r.writer_epoch,r.certificate_receipt_id; + RETURN coalesce(valid,false); +END $$; + +-- Existing G proof functions remain authoritative for their original family. +CREATE FUNCTION mst2_route_family_for_snapshot(sid text,caller_schema text) +RETURNS TABLE(namespace_uuid uuid,graph_domain text) LANGUAGE plpgsql VOLATILE +SET search_path=$CORE_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE route mst2_snapshot_storage_route%ROWTYPE; namespace mst2_metadata_namespace%ROWTYPE; +BEGIN + IF NOT mst2_route_scope_valid(caller_schema) THEN RAISE EXCEPTION 'storage route captured primary scope is unavailable'; END IF; + SELECT * INTO route FROM mst2_snapshot_storage_route WHERE snapshot_id=sid; + IF NOT FOUND THEN RETURN; END IF; + SELECT * INTO namespace FROM mst2_metadata_namespace n WHERE n.namespace_uuid=route.namespace_uuid; + IF NOT FOUND OR namespace.graph_domain='generic-v1' AND NOT mst2_route_snapshot_proof(sid) + OR namespace.graph_domain='qualified-v1' AND NOT mst2_route_qualified_snapshot_proof(sid) + OR namespace.graph_domain NOT IN ('generic-v1','qualified-v1') THEN + RAISE EXCEPTION 'permanent snapshot route has a corrupt exact family or fixed source binding'; END IF; + RETURN QUERY SELECT namespace.namespace_uuid,namespace.graph_domain; +END $$; + +CREATE FUNCTION mst2_route_family_for_lease(lid text,caller_schema text) +RETURNS TABLE(namespace_uuid uuid,graph_domain text) LANGUAGE plpgsql VOLATILE +SET search_path=$CORE_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE route mst2_lease_storage_route%ROWTYPE; namespace mst2_metadata_namespace%ROWTYPE; +BEGIN + IF NOT mst2_route_scope_valid(caller_schema) THEN RAISE EXCEPTION 'storage route captured primary scope is unavailable'; END IF; + SELECT * INTO route FROM mst2_lease_storage_route WHERE lease_id=lid; + IF NOT FOUND THEN RETURN; END IF; + SELECT * INTO namespace FROM mst2_metadata_namespace n WHERE n.namespace_uuid=route.namespace_uuid; + IF NOT FOUND OR namespace.graph_domain='generic-v1' AND NOT mst2_route_lease_proof(lid) + OR namespace.graph_domain='qualified-v1' AND NOT mst2_route_qualified_lease_proof(lid) + OR namespace.graph_domain NOT IN ('generic-v1','qualified-v1') THEN + RAISE EXCEPTION 'permanent lease route has a corrupt exact family or fixed source binding'; END IF; + RETURN QUERY SELECT namespace.namespace_uuid,namespace.graph_domain; +END $$; + +CREATE OR REPLACE FUNCTION mst2_route_insert_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$CORE_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE namespace mst2_metadata_namespace%ROWTYPE; valid boolean; +BEGIN + SELECT * INTO namespace FROM mst2_metadata_namespace WHERE namespace_uuid=NEW.namespace_uuid; + IF NOT FOUND THEN RAISE EXCEPTION 'storage route target namespace is not registered'; END IF; + IF namespace.graph_domain='qualified-v1' THEN + IF NOT mst2_route_qualified_namespace_valid(namespace.namespace_uuid) THEN + RAISE EXCEPTION 'qualified route has no exact complete physical family catalog'; END IF; + IF TG_TABLE_NAME='mst2_snapshot_storage_route' THEN + EXECUTE pg_catalog.format('SELECT %I.mst2_metadata_snapshot_candidate($1,$2,$3,$4,$5,$6,$7)',namespace.metadata_schema) + INTO valid USING NEW.snapshot_id,NEW.canonical_descriptor,NEW.instance_id,NEW.commit_oid, + NEW.root_tree_oid,NEW.metadata_root,NEW.source_profile; + ELSIF TG_TABLE_NAME='mst2_lease_storage_route' THEN + EXECUTE pg_catalog.format('SELECT %I.mst2_metadata_lease_route_proof($1,$2,$3,$4,$5,$6,$7,$8,$9,$10)',namespace.metadata_schema) + INTO valid USING NEW.lease_id,NEW.snapshot_id,NEW.namespace_uuid,NEW.session_incarnation,NEW.prepare_id, + NEW.metadata_root,NEW.authorization_epoch,NEW.publication_sequence,NEW.writer_epoch,NEW.certificate_receipt_id; + ELSE RAISE EXCEPTION 'qualified route cannot use a generic binding relation'; END IF; + IF NOT coalesce(valid,false) THEN RAISE EXCEPTION 'qualified route is not independently derived from its exact source'; END IF; + RETURN NEW; + END IF; + IF namespace.singleton IS DISTINCT FROM 1 OR namespace.graph_domain<>'generic-v1' THEN + RAISE EXCEPTION 'storage route target family is unsupported'; END IF; + IF TG_TABLE_NAME='mst2_snapshot_storage_route' THEN + IF NOT EXISTS(SELECT 1 FROM mst2_snapshot_context s JOIN mst2_metadata_prepare p ON p.prepare_id=s.prepare_id + JOIN mst2_metadata_namespace n ON n.namespace_uuid=NEW.namespace_uuid + WHERE n.singleton=1 AND n.graph_domain='generic-v1' AND s.snapshot_id=NEW.snapshot_id AND NEW.canonical_descriptor=s.canonical_descriptor + AND NEW.instance_id=s.instance_id AND NEW.commit_oid=s.commit_oid AND NEW.root_tree_oid=s.root_tree_oid + AND NEW.metadata_root=s.metadata_root AND NEW.source_profile=mst2_route_profile(p.source_domain,p.tagged_root_tree_oid, + p.scope,p.schema_version,p.metadata_codec,p.materialization_policy,p.fs_semantics,p.access_projection, + p.verification_revision,p.projection_revision) + AND p.state='COMMITTED' AND p.metadata_root=s.metadata_root AND p.source_domain='native-git' + AND (p.graph_domain IS NULL OR p.graph_domain='generic-v1') + AND p.tagged_root_tree_oid IN ('sha1:'||s.root_tree_oid,'sha256:'||s.root_tree_oid)) THEN + RAISE EXCEPTION 'storage route is not derived from its actual generic context'; + END IF; + ELSIF TG_TABLE_NAME='mst2_generic_session_storage_binding' THEN + IF NOT EXISTS(SELECT 1 FROM mst2_snapshot_context s JOIN mst2_snapshot_storage_route r USING(snapshot_id) + WHERE s.snapshot_id=NEW.snapshot_id AND r.namespace_uuid=NEW.namespace_uuid + AND s.prepare_id=NEW.prepare_id AND s.metadata_root=NEW.metadata_root) THEN + RAISE EXCEPTION 'storage route generic incarnation is not its actual context'; + END IF; + ELSIF TG_TABLE_NAME='mst2_lease_storage_route' THEN + IF NOT EXISTS(SELECT 1 FROM mst2_snapshot_lease l JOIN mst2_generic_session_storage_binding b USING(snapshot_id) + WHERE l.lease_id=NEW.lease_id AND l.snapshot_id=NEW.snapshot_id AND b.namespace_uuid=NEW.namespace_uuid + AND b.session_incarnation=NEW.session_incarnation AND b.prepare_id=NEW.prepare_id AND b.metadata_root=NEW.metadata_root + AND l.authorization_epoch=NEW.authorization_epoch AND l.publication_sequence=NEW.publication_sequence + AND l.writer_epoch=NEW.writer_epoch AND l.certificate_receipt_id=NEW.certificate_receipt_id + AND mst2_route_snapshot_proof(l.snapshot_id)) THEN + RAISE EXCEPTION 'storage route lease is not its exact generic incarnation and source'; + END IF; + ELSE RAISE EXCEPTION 'storage route insertion target is not registered'; END IF; + RETURN NEW; +END $$; + +CREATE FUNCTION mst2_route_permanent_complete() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$CORE_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE domain text; valid boolean; +BEGIN + SELECT graph_domain INTO domain FROM mst2_metadata_namespace WHERE namespace_uuid=NEW.namespace_uuid; + IF TG_TABLE_NAME='mst2_snapshot_storage_route' THEN + valid:=CASE domain WHEN 'generic-v1' THEN mst2_route_snapshot_proof(NEW.snapshot_id) + WHEN 'qualified-v1' THEN mst2_route_qualified_snapshot_proof(NEW.snapshot_id) ELSE false END; + ELSE + valid:=CASE domain WHEN 'generic-v1' THEN mst2_route_lease_proof(NEW.lease_id) + WHEN 'qualified-v1' THEN mst2_route_qualified_lease_proof(NEW.lease_id) ELSE false END; + END IF; + IF NOT coalesce(valid,false) THEN RAISE EXCEPTION 'permanent storage route cannot commit without its exact actual incarnation'; END IF; + RETURN NULL; +END $$; +CREATE CONSTRAINT TRIGGER mst2_route_permanent_complete AFTER INSERT ON mst2_snapshot_storage_route + DEFERRABLE INITIALLY DEFERRED FOR EACH ROW EXECUTE FUNCTION mst2_route_permanent_complete(); +CREATE CONSTRAINT TRIGGER mst2_route_permanent_complete AFTER INSERT ON mst2_lease_storage_route + DEFERRABLE INITIALLY DEFERRED FOR EACH ROW EXECUTE FUNCTION mst2_route_permanent_complete(); diff --git a/src/jupiter/migration/m20261008_000300_add_mst2_chunk_map_retention.rs b/src/jupiter/migration/m20261008_000300_add_mst2_chunk_map_retention.rs new file mode 100644 index 00000000..81e58229 --- /dev/null +++ b/src/jupiter/migration/m20261008_000300_add_mst2_chunk_map_retention.rs @@ -0,0 +1,26 @@ +use sea_orm::{ConnectionTrait, DbBackend}; +use sea_orm_migration::prelude::*; + +#[derive(DeriveMigrationName)] +pub struct Migration; + +#[async_trait::async_trait] +impl MigrationTrait for Migration { + async fn up(&self, manager: &SchemaManager) -> Result<(), DbErr> { + if manager.get_database_backend() != DbBackend::Postgres { + return Err(DbErr::Custom( + "chunk-map retention requires primary PostgreSQL".into(), + )); + } + manager + .get_connection() + .execute_unprepared(include_str!("m20261008_000300_chunk_map_retention.sql")) + .await?; + Ok(()) + } + + async fn down(&self, _manager: &SchemaManager) -> Result<(), DbErr> { + // Never discard generation fences or late-create reconciliation history. + Ok(()) + } +} diff --git a/src/jupiter/migration/m20261008_000300_chunk_map_retention.sql b/src/jupiter/migration/m20261008_000300_chunk_map_retention.sql new file mode 100644 index 00000000..f967d440 --- /dev/null +++ b/src/jupiter/migration/m20261008_000300_chunk_map_retention.sql @@ -0,0 +1,262 @@ +-- Physical receipt generations are never reused. APPLIED history remains +-- bounded and is rechecked: a timed-out backend create can complete late. +CREATE SEQUENCE mst2_chunk_generation_sequence START 2; +ALTER TABLE mst2_chunk_map_source ADD COLUMN receipt_generation bigint NOT NULL DEFAULT 1 CHECK(receipt_generation>0); +ALTER TABLE mst2_chunk_map_source ADD COLUMN receipt_key text GENERATED ALWAYS AS + (CASE WHEN receipt_generation=1 THEN pg_catalog.encode(source_id,'hex') + ELSE pg_catalog.encode(source_id,'hex')||'-'||pg_catalog.lpad(pg_catalog.to_hex(receipt_generation),16,'0') END) STORED; +CREATE UNIQUE INDEX mst2_chunk_map_source_receipt_key ON mst2_chunk_map_source(receipt_key); + +CREATE TABLE mst2_chunk_receipt_generation ( + receipt_key text PRIMARY KEY CHECK(receipt_key ~ '^[0-9a-f]{64}(-[0-9a-f]{16})?$'), + source_id bytea NOT NULL CHECK(pg_catalog.octet_length(source_id)=32), + generation bigint NOT NULL CHECK(generation>0), + source_bytes bytea NOT NULL CHECK(pg_catalog.octet_length(source_bytes) BETWEEN 1 AND 2048), + primary_scope bytea NOT NULL CHECK(pg_catalog.octet_length(primary_scope) BETWEEN 1 AND 1024), + map_id bytea CHECK(map_id IS NULL OR pg_catalog.octet_length(map_id)=32), + map_generation bigint CHECK(map_generation IS NULL OR map_generation>0), + receipt_bytes bytea CHECK(receipt_bytes IS NULL OR pg_catalog.octet_length(receipt_bytes) BETWEEN 129 AND 4096), + create_completed boolean NOT NULL DEFAULT false, + create_completed_at timestamptz, + owner uuid, + state text NOT NULL CHECK(state IN ('RESERVED','CREATING','LIVE','DELETING','APPLIED')), + reserved_pages integer NOT NULL CHECK(reserved_pages BETWEEN 0 AND 32768), + reserved_nodes integer NOT NULL CHECK(reserved_nodes BETWEEN 0 AND 65535), + reserved_bytes bigint NOT NULL CHECK(reserved_bytes BETWEEN 0 AND 536870912), + deadline timestamptz, + created_at timestamptz NOT NULL DEFAULT pg_catalog.clock_timestamp(), + last_progress timestamptz NOT NULL DEFAULT pg_catalog.clock_timestamp(), + checked_at timestamptz NOT NULL DEFAULT '-infinity', + UNIQUE(source_id,generation), + CHECK(receipt_key=CASE WHEN generation=1 THEN pg_catalog.encode(source_id,'hex') ELSE pg_catalog.encode(source_id,'hex')||'-'||pg_catalog.lpad(pg_catalog.to_hex(generation),16,'0') END), + CHECK(state NOT IN ('CREATING','LIVE') OR (map_id IS NOT NULL AND receipt_bytes IS NOT NULL)), + CHECK(state NOT IN ('RESERVED','CREATING') OR (owner IS NOT NULL AND deadline IS NOT NULL)) +); +CREATE INDEX mst2_chunk_receipt_generation_maintenance ON mst2_chunk_receipt_generation(state,checked_at,receipt_key); +CREATE INDEX mst2_chunk_receipt_generation_created ON mst2_chunk_receipt_generation(created_at) WHERE receipt_bytes IS NOT NULL; +CREATE INDEX mst2_chunk_receipt_generation_completed ON mst2_chunk_receipt_generation(create_completed_at) WHERE receipt_bytes IS NOT NULL; +CREATE INDEX mst2_chunk_receipt_generation_install ON mst2_chunk_receipt_generation(map_id,map_generation,deadline) WHERE state='CREATING'; +CREATE TABLE mst2_chunk_map_lifetime ( + map_id bytea PRIMARY KEY REFERENCES mst2_chunk_map(map_id) DEFERRABLE INITIALLY DEFERRED, + generation bigint NOT NULL CHECK(generation>0), + state text NOT NULL CHECK(state IN ('LIVE','DELETING')), + last_used timestamptz NOT NULL DEFAULT pg_catalog.clock_timestamp(), + UNIQUE(map_id,generation) +); +CREATE TABLE mst2_chunk_reader ( + owner uuid PRIMARY KEY, + receipt_key text NOT NULL REFERENCES mst2_chunk_receipt_generation(receipt_key), + source_generation bigint NOT NULL CHECK(source_generation>0), + map_id bytea NOT NULL CHECK(pg_catalog.octet_length(map_id)=32), + map_generation bigint NOT NULL CHECK(map_generation>0), + deadline timestamptz NOT NULL, + last_progress timestamptz NOT NULL DEFAULT pg_catalog.clock_timestamp() +); +CREATE INDEX mst2_chunk_reader_receipt ON mst2_chunk_reader(receipt_key,source_generation,deadline); +CREATE INDEX mst2_chunk_reader_map ON mst2_chunk_reader(map_id,map_generation,deadline); +CREATE TABLE mst2_chunk_map_gc ( + map_id bytea NOT NULL CHECK(pg_catalog.octet_length(map_id)=32), + generation bigint NOT NULL CHECK(generation>0), + state text NOT NULL CHECK(state IN ('PENDING','APPLIED')), + created_at timestamptz NOT NULL DEFAULT pg_catalog.clock_timestamp(), + applied_at timestamptz, + PRIMARY KEY(map_id,generation) +); + +-- Bootstrap existing #77 data from actual bounded rows. SQL does not infer +-- backing receipt presence; repository admission separately inventories it. +DO $$ DECLARE maps bigint; sources bigint; leaves bigint; nodes bigint; bytes bigint; BEGIN + SELECT pg_catalog.count(*) INTO maps FROM mst2_chunk_map; + SELECT pg_catalog.count(*) INTO sources FROM mst2_chunk_map_source; + SELECT pg_catalog.count(*) INTO leaves FROM mst2_chunk_map_leaf; + SELECT pg_catalog.count(*) INTO nodes FROM mst2_chunk_map_node; + SELECT COALESCE(pg_catalog.sum(pg_catalog.pg_total_relation_size(c.oid)),0) INTO bytes + FROM pg_catalog.pg_class c JOIN pg_catalog.pg_namespace n ON n.oid=c.relnamespace + WHERE n.nspname=pg_catalog.current_schema() AND c.relkind='r' + AND c.relname IN ('mst2_chunk_map','mst2_chunk_map_source','mst2_chunk_map_leaf','mst2_chunk_map_node'); + IF maps>4096 OR sources>16384 OR leaves>65536 OR nodes>131072 OR bytes>536870912 THEN + RAISE EXCEPTION 'existing chunk-map indexes exceed bounded retention bootstrap'; + END IF; +END $$; + +INSERT INTO mst2_chunk_map_lifetime(map_id,generation,state) SELECT map_id,1,'LIVE' FROM mst2_chunk_map; +INSERT INTO mst2_chunk_receipt_generation(receipt_key,source_id,generation,source_bytes,primary_scope,map_id,map_generation,receipt_bytes,create_completed,create_completed_at,state,reserved_pages,reserved_nodes,reserved_bytes) + SELECT s.receipt_key,s.source_id,1,s.source_bytes,s.primary_scope,s.map_id,1, + pg_catalog.convert_to('MST2-CHUNK-MAP-RECEIPT','UTF8')||pg_catalog.decode('00','hex')|| + pg_catalog.int4send(pg_catalog.octet_length(s.primary_scope))||s.primary_scope|| + pg_catalog.int4send(pg_catalog.octet_length(s.source_bytes))||s.source_bytes||m.descriptor, + true,pg_catalog.clock_timestamp(),'LIVE',0,0,0 FROM mst2_chunk_map_source s JOIN mst2_chunk_map m ON m.map_id=s.map_id; + +CREATE FUNCTION mst2_chunk_retention_barrier() RETURNS void LANGUAGE plpgsql AS $$ +BEGIN + IF pg_catalog.pg_is_in_recovery() OR pg_catalog.current_setting('transaction_isolation')<>'read committed' THEN + RAISE EXCEPTION 'chunk retention requires primary READ COMMITTED'; + END IF; + PERFORM pg_catalog.pg_advisory_xact_lock(1296717363,pg_catalog.hashtext(pg_catalog.current_schema())); +END $$; + +CREATE FUNCTION mst2_chunk_retention_bytes() RETURNS bigint LANGUAGE sql AS $$ + SELECT COALESCE(pg_catalog.sum(pg_catalog.pg_total_relation_size(c.oid)),0)::bigint + FROM pg_catalog.pg_class c JOIN pg_catalog.pg_namespace n ON n.oid=c.relnamespace + WHERE n.nspname=pg_catalog.current_schema() AND c.relkind='r' + AND c.relname IN ('mst2_chunk_map','mst2_chunk_map_source','mst2_chunk_map_leaf','mst2_chunk_map_node', + 'mst2_chunk_receipt_generation','mst2_chunk_map_lifetime','mst2_chunk_reader','mst2_chunk_map_gc') +$$; +DO $$ BEGIN + IF mst2_chunk_retention_bytes()>536870912 THEN RAISE EXCEPTION 'bounded chunk retention bootstrap exceeds actual durable byte quota'; END IF; +END $$; + +CREATE OR REPLACE FUNCTION mst2_chunk_map_primary() RETURNS trigger LANGUAGE plpgsql AS $$ +BEGIN + IF pg_catalog.current_schema() IS DISTINCT FROM TG_TABLE_SCHEMA THEN + RAISE EXCEPTION 'chunk resident insertion requires its actual primary schema'; + END IF; + PERFORM mst2_chunk_retention_barrier(); + IF mst2_chunk_retention_bytes()+1048576>536870912 THEN + RAISE EXCEPTION 'chunk resident insertion lacks actual durable byte headroom'; + END IF; + RETURN NULL; +END $$; + +-- Preserve the original immutable/source-completeness oracles. Only exact +-- terminal generations with no live owners may lose their resident indexes. +CREATE OR REPLACE FUNCTION mst2_chunk_map_immutable() RETURNS trigger LANGUAGE plpgsql AS $$ +DECLARE permitted boolean:=false; +BEGIN + IF TG_OP<>'DELETE' OR pg_catalog.current_schema() IS DISTINCT FROM TG_TABLE_SCHEMA THEN + RAISE EXCEPTION 'chunk map indexes and source receipts are immutable'; + END IF; + PERFORM mst2_chunk_retention_barrier(); + IF TG_TABLE_NAME='mst2_chunk_map_source' THEN + EXECUTE pg_catalog.format('SELECT EXISTS(SELECT 1 FROM %I.mst2_chunk_receipt_generation g WHERE g.receipt_key=$1 AND g.source_id=$2 AND g.generation=$3 AND g.map_id=$4 AND g.state=''APPLIED'') AND NOT EXISTS(SELECT 1 FROM %I.mst2_chunk_reader r WHERE r.receipt_key=$1 AND r.source_generation=$3 AND r.deadline>pg_catalog.clock_timestamp())',TG_TABLE_SCHEMA,TG_TABLE_SCHEMA) + INTO permitted USING OLD.receipt_key,OLD.source_id,OLD.receipt_generation,OLD.map_id; + ELSE + EXECUTE pg_catalog.format('SELECT EXISTS(SELECT 1 FROM %I.mst2_chunk_map_lifetime l JOIN %I.mst2_chunk_map_gc g ON g.map_id=l.map_id AND g.generation=l.generation WHERE l.map_id=$1 AND l.state=''DELETING'' AND g.state=''PENDING'') AND NOT EXISTS(SELECT 1 FROM %I.mst2_chunk_map_source s WHERE s.map_id=$1) AND NOT EXISTS(SELECT 1 FROM %I.mst2_chunk_reader r JOIN %I.mst2_chunk_map_lifetime l ON l.map_id=r.map_id AND l.generation=r.map_generation WHERE r.map_id=$1 AND r.deadline>pg_catalog.clock_timestamp())',TG_TABLE_SCHEMA,TG_TABLE_SCHEMA,TG_TABLE_SCHEMA,TG_TABLE_SCHEMA,TG_TABLE_SCHEMA) + INTO permitted USING OLD.map_id; + END IF; + IF NOT permitted THEN RAISE EXCEPTION 'chunk map deletion lacks an exact retired generation'; END IF; + RETURN OLD; +END $$; + +CREATE FUNCTION mst2_chunk_generation_guard() RETURNS trigger LANGUAGE plpgsql AS $$ +DECLARE owned boolean; +BEGIN + IF pg_catalog.current_schema() IS DISTINCT FROM TG_TABLE_SCHEMA THEN + RAISE EXCEPTION 'chunk retention mutation requires its actual primary schema'; + END IF; + PERFORM mst2_chunk_retention_barrier(); + IF TG_OP='INSERT' THEN + EXECUTE pg_catalog.format('SELECT pg_catalog.count(*)>=131072 FROM %I.mst2_chunk_receipt_generation',TG_TABLE_SCHEMA) INTO owned; + IF owned THEN RAISE EXCEPTION 'chunk receipt history quota exhausted'; END IF; + END IF; + IF TG_OP='DELETE' OR TG_OP='TRUNCATE' THEN + RAISE EXCEPTION 'chunk receipt generations and collection history are retained'; + END IF; + IF mst2_chunk_retention_bytes()+(CASE WHEN NEW.state IN ('DELETING','APPLIED') THEN 262144 ELSE 1048576 END)>536870912 THEN + RAISE EXCEPTION 'chunk receipt mutation lacks actual durable byte headroom'; + END IF; + IF TG_OP='UPDATE' AND (NEW.receipt_key IS DISTINCT FROM OLD.receipt_key OR NEW.source_id IS DISTINCT FROM OLD.source_id + OR NEW.generation IS DISTINCT FROM OLD.generation OR NEW.source_bytes IS DISTINCT FROM OLD.source_bytes + OR NEW.primary_scope IS DISTINCT FROM OLD.primary_scope OR NEW.created_at IS DISTINCT FROM OLD.created_at + OR (OLD.receipt_bytes IS NOT NULL AND NEW.receipt_bytes IS DISTINCT FROM OLD.receipt_bytes) + OR (OLD.map_id IS NOT NULL AND NEW.map_id IS DISTINCT FROM OLD.map_id) + OR (OLD.map_generation IS NOT NULL AND NEW.map_generation IS DISTINCT FROM OLD.map_generation) + OR (OLD.create_completed AND NOT NEW.create_completed) + OR (OLD.create_completed_at IS NOT NULL AND NEW.create_completed_at IS DISTINCT FROM OLD.create_completed_at) + OR (OLD.state='APPLIED' AND NEW.state<>'APPLIED') + OR (OLD.state='DELETING' AND NEW.state NOT IN ('DELETING','APPLIED')) + OR (OLD.state='LIVE' AND NEW.state NOT IN ('LIVE','DELETING')) + OR (OLD.state='CREATING' AND NEW.state NOT IN ('CREATING','LIVE','DELETING')) + OR (OLD.state='RESERVED' AND NEW.state NOT IN ('RESERVED','CREATING','DELETING','APPLIED'))) THEN + RAISE EXCEPTION 'chunk receipt generation identity or transition changed'; + END IF; + IF NEW.state='DELETING' AND (TG_OP='INSERT' OR OLD.state<>'DELETING') THEN + EXECUTE pg_catalog.format('SELECT EXISTS(SELECT 1 FROM %I.mst2_chunk_reader WHERE receipt_key=$1 AND source_generation=$2 AND deadline>pg_catalog.clock_timestamp())',TG_TABLE_SCHEMA) + INTO owned USING NEW.receipt_key,NEW.generation; + IF owned THEN RAISE EXCEPTION 'receipt retirement cannot cross a live reader'; END IF; + END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_chunk_generation_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_chunk_receipt_generation + FOR EACH ROW EXECUTE FUNCTION mst2_chunk_generation_guard(); +CREATE TRIGGER mst2_chunk_generation_no_truncate BEFORE TRUNCATE ON mst2_chunk_receipt_generation + FOR EACH STATEMENT EXECUTE FUNCTION mst2_chunk_generation_guard(); + +CREATE FUNCTION mst2_chunk_gc_history_guard() RETURNS trigger LANGUAGE plpgsql AS $$ +DECLARE full_history boolean; +BEGIN + IF pg_catalog.current_schema() IS DISTINCT FROM TG_TABLE_SCHEMA THEN RAISE EXCEPTION 'wrong chunk history schema'; END IF; + PERFORM mst2_chunk_retention_barrier(); + IF TG_OP='INSERT' THEN + EXECUTE pg_catalog.format('SELECT pg_catalog.count(*)>=131072 FROM %I.mst2_chunk_map_gc',TG_TABLE_SCHEMA) INTO full_history; + IF full_history THEN RAISE EXCEPTION 'chunk collection history quota exhausted'; END IF; + END IF; + IF TG_OP='DELETE' OR TG_OP='TRUNCATE' THEN RAISE EXCEPTION 'chunk collection history is retained'; END IF; + IF mst2_chunk_retention_bytes()+262144>536870912 THEN RAISE EXCEPTION 'chunk collection history lacks actual durable byte headroom'; END IF; + IF TG_OP='UPDATE' AND (NEW.map_id IS DISTINCT FROM OLD.map_id OR NEW.generation IS DISTINCT FROM OLD.generation + OR NEW.created_at IS DISTINCT FROM OLD.created_at OR OLD.state='APPLIED' OR NEW.state<>'APPLIED') THEN + RAISE EXCEPTION 'chunk collection history is immutable except terminal completion'; + END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_chunk_gc_history_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_chunk_map_gc + FOR EACH ROW EXECUTE FUNCTION mst2_chunk_gc_history_guard(); +CREATE TRIGGER mst2_chunk_gc_no_truncate BEFORE TRUNCATE ON mst2_chunk_map_gc + FOR EACH STATEMENT EXECUTE FUNCTION mst2_chunk_gc_history_guard(); + +CREATE FUNCTION mst2_chunk_reader_guard() RETURNS trigger LANGUAGE plpgsql AS $$ +DECLARE live boolean; +BEGIN + IF pg_catalog.current_schema() IS DISTINCT FROM TG_TABLE_SCHEMA THEN RAISE EXCEPTION 'wrong chunk reader schema'; END IF; + PERFORM mst2_chunk_retention_barrier(); + IF TG_OP='TRUNCATE' THEN RAISE EXCEPTION 'chunk readers cannot be truncated'; END IF; + IF TG_OP='DELETE' THEN RETURN OLD; END IF; + IF mst2_chunk_retention_bytes()+1048576>536870912 THEN RAISE EXCEPTION 'chunk reader mutation lacks actual durable byte headroom'; END IF; + IF TG_OP='INSERT' THEN + EXECUTE pg_catalog.format('SELECT pg_catalog.count(*)>=4096 FROM %I.mst2_chunk_reader',TG_TABLE_SCHEMA) INTO live; + IF live THEN RAISE EXCEPTION 'chunk reader quota exhausted'; END IF; + END IF; + IF NEW.deadline>pg_catalog.clock_timestamp()+interval '60 seconds' OR NEW.deadline<=pg_catalog.clock_timestamp() + OR (TG_OP='UPDATE' AND (OLD.deadline<=pg_catalog.clock_timestamp() OR NEW.owner IS DISTINCT FROM OLD.owner + OR NEW.receipt_key IS DISTINCT FROM OLD.receipt_key OR NEW.source_generation IS DISTINCT FROM OLD.source_generation + OR NEW.map_id IS DISTINCT FROM OLD.map_id OR NEW.map_generation IS DISTINCT FROM OLD.map_generation)) THEN + RAISE EXCEPTION 'chunk reader cannot renew an expired or different owner'; + END IF; + EXECUTE pg_catalog.format('SELECT EXISTS(SELECT 1 FROM %I.mst2_chunk_receipt_generation g JOIN %I.mst2_chunk_map_lifetime l ON l.map_id=g.map_id AND l.generation=g.map_generation JOIN %I.mst2_chunk_map_source s ON s.receipt_key=g.receipt_key AND s.receipt_generation=g.generation WHERE g.receipt_key=$1 AND g.generation=$2 AND g.map_id=$3 AND g.map_generation=$4 AND g.state=''LIVE'' AND l.state=''LIVE'')',TG_TABLE_SCHEMA,TG_TABLE_SCHEMA,TG_TABLE_SCHEMA) + INTO live USING NEW.receipt_key,NEW.source_generation,NEW.map_id,NEW.map_generation; + IF NOT live THEN RAISE EXCEPTION 'chunk reader lacks exact live source and map generations'; END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_chunk_reader_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_chunk_reader + FOR EACH ROW EXECUTE FUNCTION mst2_chunk_reader_guard(); +CREATE TRIGGER mst2_chunk_reader_no_truncate BEFORE TRUNCATE ON mst2_chunk_reader + FOR EACH STATEMENT EXECUTE FUNCTION mst2_chunk_reader_guard(); + +CREATE FUNCTION mst2_chunk_map_lifetime_guard() RETURNS trigger LANGUAGE plpgsql AS $$ +DECLARE permitted boolean; +BEGIN + IF pg_catalog.current_schema() IS DISTINCT FROM TG_TABLE_SCHEMA THEN RAISE EXCEPTION 'wrong chunk map lifetime schema'; END IF; + PERFORM mst2_chunk_retention_barrier(); + IF TG_OP='TRUNCATE' THEN RAISE EXCEPTION 'chunk map lifetimes cannot be truncated'; END IF; + IF TG_OP='DELETE' THEN + EXECUTE pg_catalog.format('SELECT $2=''DELETING'' AND EXISTS(SELECT 1 FROM %I.mst2_chunk_map_gc g WHERE g.map_id=$1 AND g.generation=$3 AND g.state=''PENDING'') AND NOT EXISTS(SELECT 1 FROM %I.mst2_chunk_map_source s WHERE s.map_id=$1) AND NOT EXISTS(SELECT 1 FROM %I.mst2_chunk_reader r WHERE r.map_id=$1 AND r.map_generation=$3 AND r.deadline>pg_catalog.clock_timestamp())',TG_TABLE_SCHEMA,TG_TABLE_SCHEMA,TG_TABLE_SCHEMA) + INTO permitted USING OLD.map_id,OLD.state,OLD.generation; + IF NOT permitted THEN RAISE EXCEPTION 'map lifetime deletion lacks exact reader-free collection claim'; END IF; + RETURN OLD; + END IF; + IF mst2_chunk_retention_bytes()+(CASE WHEN NEW.state='DELETING' THEN 262144 ELSE 1048576 END)>536870912 THEN + RAISE EXCEPTION 'chunk map lifetime mutation lacks actual durable byte headroom'; + END IF; + IF TG_OP='UPDATE' AND (NEW.map_id IS DISTINCT FROM OLD.map_id OR NEW.generation IS DISTINCT FROM OLD.generation + OR (OLD.state='DELETING' AND NEW.state<>'DELETING')) THEN RAISE EXCEPTION 'map lifetime cannot reincarnate in place'; END IF; + IF TG_OP='UPDATE' AND NEW.state='DELETING' AND OLD.state='LIVE' THEN + EXECUTE pg_catalog.format('SELECT EXISTS(SELECT 1 FROM %I.mst2_chunk_map_gc g WHERE g.map_id=$1 AND g.generation=$2 AND g.state=''PENDING'') AND NOT EXISTS(SELECT 1 FROM %I.mst2_chunk_map_source WHERE map_id=$1) AND NOT EXISTS(SELECT 1 FROM %I.mst2_chunk_reader WHERE map_id=$1 AND map_generation=$2 AND deadline>pg_catalog.clock_timestamp()) AND NOT EXISTS(SELECT 1 FROM %I.mst2_chunk_receipt_generation WHERE map_id=$1 AND map_generation=$2 AND state=''CREATING'' AND deadline>pg_catalog.clock_timestamp())',TG_TABLE_SCHEMA,TG_TABLE_SCHEMA,TG_TABLE_SCHEMA,TG_TABLE_SCHEMA) + INTO permitted USING OLD.map_id,OLD.generation; + IF NOT permitted THEN RAISE EXCEPTION 'map collection cannot cross a reader or active partial installation'; END IF; + END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_chunk_map_lifetime_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_chunk_map_lifetime + FOR EACH ROW EXECUTE FUNCTION mst2_chunk_map_lifetime_guard(); +CREATE TRIGGER mst2_chunk_map_lifetime_no_truncate BEFORE TRUNCATE ON mst2_chunk_map_lifetime + FOR EACH STATEMENT EXECUTE FUNCTION mst2_chunk_map_lifetime_guard(); diff --git a/src/jupiter/migration/m20261008_000400_add_mst2_reader_retention.rs b/src/jupiter/migration/m20261008_000400_add_mst2_reader_retention.rs new file mode 100644 index 00000000..674c3a90 --- /dev/null +++ b/src/jupiter/migration/m20261008_000400_add_mst2_reader_retention.rs @@ -0,0 +1,21 @@ +//! Upgrade the trusted v3 reader lifecycle and its native plan decoder. + +use sea_orm_migration::prelude::*; + +#[derive(DeriveMigrationName)] +pub struct Migration; + +#[async_trait::async_trait] +impl MigrationTrait for Migration { + async fn up(&self, manager: &SchemaManager) -> Result<(), DbErr> { + super::qualified_native_runtime_upgrade::upgrade(manager, true).await + } + + async fn down(&self, _manager: &SchemaManager) -> Result<(), DbErr> { + Ok(()) + } + + fn use_transaction(&self) -> Option { + Some(true) + } +} diff --git a/src/jupiter/migration/m20261008_000400_reader_retention_upgrade.sql b/src/jupiter/migration/m20261008_000400_reader_retention_upgrade.sql new file mode 100644 index 00000000..cb7b6876 --- /dev/null +++ b/src/jupiter/migration/m20261008_000400_reader_retention_upgrade.sql @@ -0,0 +1,46 @@ +SET LOCAL search_path=$Q_SCHEMA$,pg_catalog,pg_temp; +ALTER TABLE mst2_metadata_reader_operation DISABLE TRIGGER mst2_00_family_barrier; +ALTER TABLE mst2_metadata_reader_operation DISABLE TRIGGER mst2_metadata_reader_guard; +ALTER TABLE mst2_metadata_reader_operation DISABLE TRIGGER mst2_metadata_reader_complete; +ALTER TABLE mst2_metadata_root_anchor DISABLE TRIGGER mst2_00_family_barrier; +ALTER TABLE mst2_metadata_root_anchor DISABLE TRIGGER mst2_metadata_root_anchor_guard; + +ALTER TABLE mst2_metadata_reader_operation + ADD COLUMN reader_issuance bigint NOT NULL DEFAULT 0 CHECK(reader_issuance>=0), + ADD COLUMN terminal_xid bigint; +UPDATE mst2_metadata_reader_operation SET terminal_xid=0 WHERE state IN ('FINISHED','EXPIRED'); +ALTER TABLE mst2_metadata_reader_operation ALTER COLUMN reader_issuance DROP DEFAULT; +ALTER TABLE mst2_metadata_reader_operation + ADD CONSTRAINT mst2_metadata_reader_identity UNIQUE(operation_id,reader_issuance), + ADD CONSTRAINT mst2_metadata_reader_terminal_state CHECK((state='ACTIVE')=(terminal_xid IS NULL)); +CREATE TABLE mst2_metadata_reader_issuance ( + singleton smallint PRIMARY KEY CHECK(singleton=1),high_water bigint NOT NULL CHECK(high_water>=0) +); +INSERT INTO mst2_metadata_reader_issuance VALUES(1,0); +CREATE INDEX mst2_metadata_reader_terminal ON mst2_metadata_reader_operation(terminal_xid,operation_id) WHERE state IN ('FINISHED','EXPIRED'); +ALTER TABLE mst2_metadata_root_anchor ADD COLUMN reader_issuance bigint; +UPDATE mst2_metadata_root_anchor SET reader_issuance=0 WHERE reader_operation_id IS NOT NULL; +ALTER TABLE mst2_metadata_root_anchor + DROP CONSTRAINT mst2_metadata_root_anchor_reader_operation_id_fkey, + ADD CONSTRAINT mst2_metadata_root_anchor_reader_identity FOREIGN KEY(reader_operation_id,reader_issuance) + REFERENCES mst2_metadata_reader_operation(operation_id,reader_issuance) MATCH FULL, + ADD CONSTRAINT mst2_metadata_root_anchor_reader_kind CHECK((anchor_kind IN ('REQUEST','READER'))=(reader_operation_id IS NOT NULL)), + ADD CONSTRAINT mst2_metadata_root_anchor_reader_fields CHECK((reader_operation_id IS NULL)=(reader_issuance IS NULL)); +CREATE INDEX mst2_metadata_root_anchor_reader ON mst2_metadata_root_anchor(reader_operation_id,reader_issuance) WHERE reader_operation_id IS NOT NULL; +DROP FUNCTION mst2_metadata_begin_reader(text,text,text); +DROP FUNCTION mst2_metadata_finish_reader(uuid); +DROP FUNCTION mst2_metadata_read_source_entries(uuid,uuid,bytea,bigint,bytea,jsonb); + +$READER_FUNCTIONS_SQL$ + +CREATE TRIGGER mst2_00_family_barrier BEFORE INSERT OR UPDATE OR DELETE ON mst2_metadata_reader_issuance + FOR EACH STATEMENT EXECUTE FUNCTION mst2_metadata_dml_barrier(); +CREATE TRIGGER mst2_metadata_truncate_guard BEFORE TRUNCATE ON mst2_metadata_reader_issuance + FOR EACH STATEMENT EXECUTE FUNCTION mst2_metadata_immutable(); +CREATE TRIGGER mst2_metadata_reader_issuance_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_metadata_reader_issuance + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_reader_issuance_guard(); +ALTER TABLE mst2_metadata_reader_operation ENABLE TRIGGER mst2_00_family_barrier; +ALTER TABLE mst2_metadata_reader_operation ENABLE TRIGGER mst2_metadata_reader_guard; +ALTER TABLE mst2_metadata_reader_operation ENABLE TRIGGER mst2_metadata_reader_complete; +ALTER TABLE mst2_metadata_root_anchor ENABLE TRIGGER mst2_00_family_barrier; +ALTER TABLE mst2_metadata_root_anchor ENABLE TRIGGER mst2_metadata_root_anchor_guard; diff --git a/src/jupiter/migration/m20261008_000500_fix_mst2_native_runtime.rs b/src/jupiter/migration/m20261008_000500_fix_mst2_native_runtime.rs new file mode 100644 index 00000000..aed57614 --- /dev/null +++ b/src/jupiter/migration/m20261008_000500_fix_mst2_native_runtime.rs @@ -0,0 +1,21 @@ +//! Repair the exact previously admitted v3 decoder without rewriting history. + +use sea_orm_migration::prelude::*; + +#[derive(DeriveMigrationName)] +pub struct Migration; + +#[async_trait::async_trait] +impl MigrationTrait for Migration { + async fn up(&self, manager: &SchemaManager) -> Result<(), DbErr> { + super::qualified_native_runtime_upgrade::upgrade(manager, false).await + } + + async fn down(&self, _manager: &SchemaManager) -> Result<(), DbErr> { + Ok(()) + } + + fn use_transaction(&self) -> Option { + Some(true) + } +} diff --git a/src/jupiter/migration/m20261008_000500_previous_reader_family.sql b/src/jupiter/migration/m20261008_000500_previous_reader_family.sql new file mode 100644 index 00000000..c76bfded --- /dev/null +++ b/src/jupiter/migration/m20261008_000500_previous_reader_family.sql @@ -0,0 +1,425 @@ +-- Captured exactly from c1ff280281440891e732ecfaed4732ee59c41a90; migration verification only. +-- READER DDL BEGIN +CREATE TABLE mst2_metadata_reader_operation ( + operation_id uuid PRIMARY KEY,lease_id text NOT NULL REFERENCES mst2_qualified_lease_binding(lease_id), + snapshot_id text NOT NULL,session_incarnation uuid NOT NULL,root_page bytea NOT NULL,root_generation bigint NOT NULL, + lease_epoch bigint NOT NULL CHECK(lease_epoch>0),hard_deadline_unix bigint NOT NULL, + state text NOT NULL CHECK(state IN ('ACTIVE','FINISHED','EXPIRED')), + FOREIGN KEY(snapshot_id,session_incarnation) REFERENCES mst2_qualified_session_incarnation(snapshot_id,session_incarnation) +); +CREATE INDEX mst2_metadata_reader_active_lease ON mst2_metadata_reader_operation(lease_id,state,operation_id); +CREATE INDEX mst2_metadata_reader_active_deadline ON mst2_metadata_reader_operation(hard_deadline_unix,operation_id) WHERE state='ACTIVE'; +CREATE TABLE mst2_metadata_root_anchor ( + anchor_id uuid PRIMARY KEY,anchor_kind text NOT NULL CHECK(anchor_kind IN ('PREPARE','REUSE','SESSION','LEASE','REQUEST','READER')), + owner_key text NOT NULL CHECK(octet_length(owner_key) BETWEEN 1 AND 512), + root_page bytea NOT NULL,root_generation bigint NOT NULL,root_certificate_digest bytea NOT NULL, + prepare_id text REFERENCES mst2_metadata_prepare(prepare_id), + snapshot_id text,session_incarnation uuid,lease_id text REFERENCES mst2_qualified_lease_binding(lease_id), + reader_operation_id uuid REFERENCES mst2_metadata_reader_operation(operation_id), + UNIQUE(anchor_kind,owner_key,root_page,root_generation), + FOREIGN KEY(root_page,root_generation) REFERENCES mst2_metadata_graph_node(page_id,generation), + FOREIGN KEY(root_page,root_generation,root_certificate_digest) + REFERENCES mst2_metadata_page_certificate(page_id,generation,certificate_digest), + FOREIGN KEY(snapshot_id,session_incarnation) REFERENCES mst2_qualified_session_incarnation(snapshot_id,session_incarnation) +); +CREATE INDEX mst2_metadata_root_anchor_page ON mst2_metadata_root_anchor(root_page,root_generation,anchor_kind,owner_key); +CREATE INDEX mst2_metadata_root_anchor_prepare ON mst2_metadata_root_anchor(prepare_id,anchor_kind,anchor_id); +CREATE INDEX mst2_metadata_root_anchor_lease ON mst2_metadata_root_anchor(lease_id,anchor_kind,anchor_id); + +-- READER DDL END + +CREATE FUNCTION mst2_metadata_dml_barrier() RETURNS trigger LANGUAGE plpgsql VOLATILE AS $$ +BEGIN + IF pg_catalog.current_schema() IS DISTINCT FROM $Q_LITERAL$ OR TG_TABLE_SCHEMA IS DISTINCT FROM $Q_LITERAL$ + OR NOT EXISTS(SELECT 1 FROM pg_catalog.pg_class WHERE oid=TG_RELID AND relnamespace='$Q_OID$'::oid) + OR NOT EXISTS(SELECT 1 FROM pg_catalog.pg_namespace WHERE oid='$Q_OID$'::oid AND nspname=$Q_LITERAL$) + OR pg_catalog.pg_is_in_recovery() OR pg_catalog.current_setting('transaction_isolation')<>'read committed' THEN + RAISE EXCEPTION 'qualified mutation requires its captured primary family and READ COMMITTED'; + END IF; + PERFORM $CORE_SCHEMA$.mst2_route_enter($CORE_LITERAL$); + IF NOT EXISTS(SELECT 1 FROM $CORE_SCHEMA$.mst2_metadata_namespace n + WHERE n.namespace_uuid='$NAMESPACE_UUID$'::uuid AND n.metadata_schema=$Q_LITERAL$ + AND n.metadata_schema_oid='$Q_OID$'::oid AND n.core_schema_oid=$CORE_OID$ + AND n.metadata_storage_uuid='$STORAGE_UUID$' AND n.admission_state='ROOTED_Q_ADMITTED' + AND n.collector_state='ENABLED' AND n.implementation_fingerprint=pg_catalog.decode('$IMPLEMENTATION_SHA$','hex') + AND n.catalog_fingerprint=$CORE_SCHEMA$.mst2_route_family_catalog($CORE_OID$,'$Q_OID$'::oid)) THEN + RAISE EXCEPTION 'qualified physical family catalog fingerprint is unavailable'; + END IF; + RETURN NULL; +END $$; + +CREATE FUNCTION mst2_metadata_gc_enabled() RETURNS boolean LANGUAGE sql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ + SELECT NOT pg_is_in_recovery() AND current_setting('transaction_isolation')='read committed' + AND EXISTS(SELECT 1 FROM $CORE_SCHEMA$.mst2_metadata_namespace n WHERE n.namespace_uuid='$NAMESPACE_UUID$'::uuid + AND n.graph_domain='qualified-v1' AND n.metadata_schema=$Q_LITERAL$ AND n.metadata_schema_oid='$Q_OID$'::oid + AND n.core_schema_oid=$CORE_OID$ AND n.metadata_storage_uuid='$STORAGE_UUID$' + AND n.admission_state='ROOTED_Q_ADMITTED' AND n.collector_state='ENABLED' + AND n.implementation_fingerprint=decode('$IMPLEMENTATION_SHA$','hex') + AND n.catalog_fingerprint=$CORE_SCHEMA$.mst2_route_family_catalog($CORE_OID$,'$Q_OID$'::oid)) +$$; + +CREATE FUNCTION mst2_metadata_root_anchor_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + IF TG_OP='UPDATE' THEN RAISE EXCEPTION 'rooted anchors cannot retarget immutable owner or root identities'; END IF; + IF TG_OP='DELETE' THEN + IF OLD.anchor_kind IN ('PREPARE','REUSE') THEN + IF NOT EXISTS(SELECT 1 FROM mst2_metadata_prepare WHERE prepare_id=OLD.prepare_id AND state IN ('PREPARING','COMMITTED','ABORTED')) THEN + RAISE EXCEPTION 'temporary anchor owner history is missing'; + END IF; + ELSIF OLD.anchor_kind='SESSION' THEN + IF NOT EXISTS(SELECT 1 FROM mst2_qualified_session_incarnation s + WHERE s.snapshot_id=OLD.snapshot_id AND s.session_incarnation=OLD.session_incarnation AND s.state='RETIRED') + OR EXISTS(SELECT 1 FROM mst2_qualified_lease_binding l WHERE l.snapshot_id=OLD.snapshot_id + AND l.session_incarnation=OLD.session_incarnation AND l.state='ACTIVE') THEN + RAISE EXCEPTION 'session root still has its active incarnation or leases'; + END IF; + ELSIF OLD.anchor_kind='LEASE' THEN + IF NOT EXISTS(SELECT 1 FROM mst2_qualified_lease_binding WHERE lease_id=OLD.lease_id AND state IN ('RELEASED','EXPIRED')) + OR EXISTS(SELECT 1 FROM mst2_metadata_reader_operation WHERE lease_id=OLD.lease_id AND state='ACTIVE') THEN + RAISE EXCEPTION 'lease root still has its active lease or readers'; + END IF; + ELSE + IF NOT EXISTS(SELECT 1 FROM mst2_metadata_reader_operation r WHERE r.operation_id=OLD.reader_operation_id + AND (r.state='FINISHED' OR r.state='EXPIRED' AND r.hard_deadline_unix<=floor(extract(epoch FROM clock_timestamp()))::bigint)) THEN + RAISE EXCEPTION 'reader root still has an active operation'; + END IF; + END IF; + RETURN OLD; + END IF; + IF NOT EXISTS(SELECT 1 FROM mst2_metadata_graph_node node + JOIN mst2_metadata_current cur USING(page_id,generation) + JOIN mst2_metadata_lifetime life USING(page_id,generation) + JOIN mst2_metadata_page_certificate proof USING(page_id,generation) + WHERE node.page_id=NEW.root_page AND node.generation=NEW.root_generation AND node.state='LIVE' + AND node.certificate_digest=NEW.root_certificate_digest AND proof.certificate_digest=NEW.root_certificate_digest + AND life.state IN ('RESERVED','LIVE') AND life.graph_domain='qualified-v1' + AND NOT EXISTS(SELECT 1 FROM mst2_metadata_gc_op gc WHERE gc.page_id=node.page_id AND gc.generation=node.generation)) THEN + RAISE EXCEPTION 'rooted anchor does not protect its exact canonical current graph'; + END IF; + IF NEW.anchor_kind IN ('PREPARE','REUSE') THEN + IF NEW.owner_key IS DISTINCT FROM NEW.prepare_id OR NEW.snapshot_id IS NOT NULL OR NEW.session_incarnation IS NOT NULL + OR NEW.lease_id IS NOT NULL OR NEW.reader_operation_id IS NOT NULL + OR NOT EXISTS(SELECT 1 FROM mst2_metadata_prepare WHERE prepare_id=NEW.prepare_id + AND state IN ('PREPARING','COMMITTED') AND coverage_retired_at IS NULL) + OR NEW.anchor_kind='PREPARE' AND NOT (EXISTS(SELECT 1 FROM mst2_metadata_prepare_page + WHERE prepare_id=NEW.prepare_id AND page_id=NEW.root_page AND generation=NEW.root_generation) + OR EXISTS(SELECT 1 FROM mst2_metadata_prepare q JOIN mst2_metadata_prepare_reuse_root r USING(prepare_id) + WHERE q.prepare_id=NEW.prepare_id AND q.plan_kind='ROOTED' AND q.metadata_root=NEW.root_page + AND r.root_page=NEW.root_page AND r.root_generation=NEW.root_generation)) + OR NEW.anchor_kind='REUSE' AND NOT EXISTS(SELECT 1 FROM mst2_metadata_prepare_reuse_root + WHERE prepare_id=NEW.prepare_id AND root_page=NEW.root_page AND root_generation=NEW.root_generation) THEN + RAISE EXCEPTION 'temporary anchor differs from its immutable delta or reused-root owner'; + END IF; + ELSIF NEW.anchor_kind='SESSION' THEN + IF NEW.owner_key IS DISTINCT FROM NEW.snapshot_id||':'||NEW.session_incarnation::text OR NEW.prepare_id IS NOT NULL + OR NEW.lease_id IS NOT NULL OR NEW.reader_operation_id IS NOT NULL + OR NOT EXISTS(SELECT 1 FROM mst2_qualified_session_incarnation s WHERE s.snapshot_id=NEW.snapshot_id + AND s.session_incarnation=NEW.session_incarnation AND s.metadata_root=NEW.root_page AND s.root_generation=NEW.root_generation AND s.state='READY') THEN + RAISE EXCEPTION 'session anchor differs from its exact ready incarnation'; + END IF; + ELSIF NEW.anchor_kind='LEASE' THEN + IF NEW.owner_key IS DISTINCT FROM NEW.lease_id OR NEW.prepare_id IS NOT NULL OR NEW.reader_operation_id IS NOT NULL + OR NOT EXISTS(SELECT 1 FROM mst2_qualified_lease_binding l WHERE l.lease_id=NEW.lease_id + AND l.snapshot_id=NEW.snapshot_id AND l.session_incarnation=NEW.session_incarnation + AND l.metadata_root=NEW.root_page AND l.root_generation=NEW.root_generation AND l.state='ACTIVE' + AND l.expires_at_unix>floor(extract(epoch FROM clock_timestamp()))::bigint) THEN + RAISE EXCEPTION 'lease anchor differs from its exact active lease'; + END IF; + ELSE + IF NEW.owner_key IS DISTINCT FROM NEW.reader_operation_id::text OR NEW.prepare_id IS NOT NULL + OR NOT EXISTS(SELECT 1 FROM mst2_metadata_reader_operation r JOIN mst2_qualified_lease_binding l USING(lease_id) + WHERE r.operation_id=NEW.reader_operation_id AND r.lease_id=NEW.lease_id AND r.snapshot_id=NEW.snapshot_id + AND r.session_incarnation=NEW.session_incarnation AND r.root_page=NEW.root_page AND r.root_generation=NEW.root_generation + AND r.state='ACTIVE' AND l.state='ACTIVE' AND l.lease_epoch=r.lease_epoch + AND r.hard_deadline_unix<=l.expires_at_unix AND r.hard_deadline_unix>floor(extract(epoch FROM clock_timestamp()))::bigint) THEN + RAISE EXCEPTION 'reader anchor differs from its exact active lease operation'; + END IF; + END IF; + RETURN NEW; +END $$; + +CREATE FUNCTION mst2_metadata_serving_complete() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE s mst2_qualified_session_incarnation%ROWTYPE; l mst2_qualified_lease_binding%ROWTYPE; + r mst2_metadata_reader_operation%ROWTYPE; +BEGIN + IF TG_TABLE_NAME='mst2_qualified_session_incarnation' THEN + SELECT * INTO STRICT s FROM mst2_qualified_session_incarnation + WHERE snapshot_id=NEW.snapshot_id AND session_incarnation=NEW.session_incarnation; + IF NOT mst2_metadata_incarnation_proof(s.snapshot_id,s.session_incarnation,s.state='READY') + OR s.state='READY' AND NOT EXISTS(SELECT 1 FROM mst2_qualified_lease_binding lease + WHERE lease.snapshot_id=s.snapshot_id AND lease.session_incarnation=s.session_incarnation AND lease.state='ACTIVE') + OR s.state='RETIRED' AND EXISTS(SELECT 1 FROM mst2_metadata_root_anchor a + WHERE a.anchor_kind='SESSION' AND a.snapshot_id=s.snapshot_id AND a.session_incarnation=s.session_incarnation) THEN + RAISE EXCEPTION 'qualified session cannot commit without its exact final serving roots'; END IF; + ELSIF TG_TABLE_NAME='mst2_qualified_lease_binding' THEN + SELECT * INTO STRICT l FROM mst2_qualified_lease_binding WHERE lease_id=NEW.lease_id; + IF NOT EXISTS(SELECT 1 FROM $CORE_SCHEMA$.mst2_lease_storage_route route + WHERE route.lease_id=l.lease_id AND route.namespace_uuid=l.namespace_uuid AND route.snapshot_id=l.snapshot_id + AND route.session_incarnation=l.session_incarnation AND route.prepare_id=l.prepare_id AND route.metadata_root=l.metadata_root + AND route.authorization_epoch=l.authorization_epoch AND route.publication_sequence=l.publication_sequence + AND route.writer_epoch=l.writer_epoch AND route.certificate_receipt_id=l.certificate_receipt_id) + OR l.state='ACTIVE' AND (l.expires_at_unix<=floor(extract(epoch FROM clock_timestamp()))::bigint + OR NOT mst2_metadata_incarnation_proof(l.snapshot_id,l.session_incarnation,true) + OR NOT EXISTS(SELECT 1 FROM mst2_metadata_root_anchor a WHERE a.anchor_kind='LEASE' AND a.owner_key=l.lease_id + AND a.lease_id=l.lease_id AND a.root_page=l.metadata_root AND a.root_generation=l.root_generation)) + OR l.state<>'ACTIVE' AND NOT EXISTS(SELECT 1 FROM mst2_metadata_reader_operation operation + WHERE operation.lease_id=l.lease_id AND operation.state='ACTIVE') + AND EXISTS(SELECT 1 FROM mst2_metadata_root_anchor a WHERE a.anchor_kind='LEASE' AND a.lease_id=l.lease_id) THEN + RAISE EXCEPTION 'qualified lease cannot commit without its exact final route and owned protection'; END IF; + ELSE + SELECT * INTO STRICT r FROM mst2_metadata_reader_operation WHERE operation_id=NEW.operation_id; + IF r.state='ACTIVE' AND (r.hard_deadline_unix<=floor(extract(epoch FROM clock_timestamp()))::bigint + OR (SELECT count(*) FROM mst2_metadata_root_anchor a WHERE a.reader_operation_id=r.operation_id + AND a.anchor_kind IN ('REQUEST','READER') AND a.owner_key=r.operation_id::text + AND a.lease_id=r.lease_id AND a.snapshot_id=r.snapshot_id AND a.session_incarnation=r.session_incarnation + AND a.root_page=r.root_page AND a.root_generation=r.root_generation)<>2) + OR r.state<>'ACTIVE' AND EXISTS(SELECT 1 FROM mst2_metadata_root_anchor a WHERE a.reader_operation_id=r.operation_id) THEN + RAISE EXCEPTION 'qualified reader cannot commit without both exact owned roots or definitive cleanup'; END IF; + END IF; + RETURN NULL; +END $$; + +CREATE FUNCTION mst2_metadata_reader_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE l mst2_qualified_lease_binding%ROWTYPE; now_unix bigint:=floor(extract(epoch FROM clock_timestamp()))::bigint; +BEGIN + IF TG_OP='DELETE' THEN RAISE EXCEPTION 'qualified reader operation history is immutable'; END IF; + IF TG_OP='UPDATE' THEN + IF (to_jsonb(NEW)-'state') IS DISTINCT FROM (to_jsonb(OLD)-'state') + OR OLD.state<>'ACTIVE' AND NEW.state<>OLD.state OR NEW.state NOT IN ('ACTIVE','FINISHED','EXPIRED') + OR NEW.state='EXPIRED' AND OLD.hard_deadline_unix>now_unix THEN + RAISE EXCEPTION 'qualified reader identity cannot change or be prematurely expired'; END IF; + RETURN NEW; + END IF; + SELECT * INTO l FROM mst2_qualified_lease_binding WHERE lease_id=NEW.lease_id AND state='ACTIVE' AND expires_at_unix>now_unix; + IF NOT FOUND OR NEW.state<>'ACTIVE' OR substr(NEW.operation_id::text,15,1)<>'4' + OR substr(NEW.operation_id::text,20,1) NOT IN ('8','9','a','b') + OR ROW(NEW.snapshot_id,NEW.session_incarnation,NEW.root_page,NEW.root_generation,NEW.lease_epoch) IS DISTINCT FROM + ROW(l.snapshot_id,l.session_incarnation,l.metadata_root,l.root_generation,l.lease_epoch) + OR NEW.hard_deadline_unix<=now_unix OR NEW.hard_deadline_unix>least(l.expires_at_unix,now_unix+60) + OR NOT mst2_metadata_incarnation_proof(l.snapshot_id,l.session_incarnation,true) THEN + RAISE EXCEPTION 'qualified reader lacks its exact active lease and bounded deadline'; END IF; + RETURN NEW; +END $$; + +CREATE FUNCTION mst2_metadata_begin_reader(sid text,lid text,instance text) +RETURNS TABLE(operation_id uuid,root_generation bigint,certificate_digest bytea) LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE l mst2_qualified_lease_binding%ROWTYPE; s record; op uuid:=gen_random_uuid(); deadline bigint; kind text; +BEGIN + PERFORM $CORE_SCHEMA$.mst2_route_enter($CORE_LITERAL$); + PERFORM mst2_metadata_cleanup_expired(64); + SELECT * INTO s FROM mst2_metadata_session_row(sid,lid,instance); + IF NOT FOUND THEN RETURN; END IF; + SELECT * INTO STRICT l FROM mst2_qualified_lease_binding WHERE lease_id=lid; + deadline:=least(l.expires_at_unix,floor(extract(epoch FROM clock_timestamp()))::bigint+60); + INSERT INTO mst2_metadata_reader_operation(operation_id,lease_id,snapshot_id,session_incarnation,root_page,root_generation, + lease_epoch,hard_deadline_unix,state) VALUES(op,lid,sid,l.session_incarnation,l.metadata_root,l.root_generation,l.lease_epoch,deadline,'ACTIVE'); + FOREACH kind IN ARRAY ARRAY['REQUEST','READER'] LOOP + INSERT INTO mst2_metadata_root_anchor(anchor_id,anchor_kind,owner_key,root_page,root_generation,root_certificate_digest, + snapshot_id,session_incarnation,lease_id,reader_operation_id) VALUES(gen_random_uuid(),kind,op::text, + l.metadata_root,l.root_generation,s.certificate_digest,sid,l.session_incarnation,lid,op); + END LOOP; + RETURN QUERY SELECT op,l.root_generation,s.certificate_digest::bytea; +END $$; + +CREATE FUNCTION mst2_metadata_finish_reader(op uuid) RETURNS void LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE r mst2_metadata_reader_operation%ROWTYPE; +BEGIN + PERFORM $CORE_SCHEMA$.mst2_route_enter($CORE_LITERAL$); + SELECT * INTO r FROM mst2_metadata_reader_operation WHERE operation_id=op; + IF NOT FOUND THEN RETURN; END IF; + IF r.state='ACTIVE' THEN UPDATE mst2_metadata_reader_operation SET state='FINISHED' WHERE operation_id=op; END IF; + DELETE FROM mst2_metadata_root_anchor WHERE reader_operation_id=op AND anchor_kind IN ('REQUEST','READER'); + PERFORM mst2_metadata_cleanup_lease(r.lease_id); +END $$; + +CREATE FUNCTION mst2_metadata_read_source_entries(op uuid,source_id uuid,p bytea,g bigint,c bytea,names jsonb) +RETURNS TABLE(name bytea,git_oid text,kind smallint,byte_size bigint,content_digest bytea,child_root bytea, + child_generation bigint,child_certificate_digest bytea,child_attestation_id uuid,fact_state text) +LANGUAGE plpgsql VOLATILE SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE a record; profile jsonb; +BEGIN + IF jsonb_typeof(names)<>'array' OR jsonb_array_length(names)>256 + OR EXISTS(SELECT 1 FROM jsonb_array_elements_text(names) wanted + WHERE wanted !~ '^([0-9a-f]{2}){1,255}$') THEN + RAISE EXCEPTION 'qualified source-name read exceeds its bounded exact request'; END IF; + SELECT source.namespace_uuid,source.source_profile,source.tagged_tree_oid,source.source_body_digest,source.source_revision, + source.root_page,source.root_generation,source.root_certificate_digest + INTO a FROM mst2_metadata_source_root_attestation source + JOIN mst2_metadata_prepare origin ON origin.prepare_id=source.origin_prepare_id + WHERE source.attestation_id=source_id AND origin.state='COMMITTED' + AND source.root_page=p AND source.root_generation=g AND source.root_certificate_digest=c + AND source.namespace_uuid='$NAMESPACE_UUID$'::uuid; + IF NOT FOUND THEN RAISE EXCEPTION 'qualified selected directory has no definitive source attestation'; END IF; + SELECT mst2_metadata_native_profile(session.prepare_id) INTO profile + FROM mst2_metadata_reader_operation reader JOIN mst2_qualified_session_incarnation session + ON session.snapshot_id=reader.snapshot_id AND session.session_incarnation=reader.session_incarnation + WHERE reader.operation_id=op AND reader.state='ACTIVE' + AND reader.hard_deadline_unix>floor(extract(epoch FROM clock_timestamp()))::bigint + AND EXISTS(SELECT 1 FROM mst2_metadata_root_anchor anchor WHERE anchor.reader_operation_id=reader.operation_id + AND anchor.anchor_kind='READER' AND anchor.root_page=reader.root_page AND anchor.root_generation=reader.root_generation); + IF NOT FOUND OR profile IS DISTINCT FROM a.source_profile + OR NOT mst2_metadata_root_live(a.root_page,a.root_generation,a.root_certificate_digest) THEN + RAISE EXCEPTION 'qualified source read lost its exact reader profile and current directory'; END IF; + IF NOT $CORE_SCHEMA$.mst2_route_source_tree_matches(split_part(a.tagged_tree_oid,':',2),a.source_revision,a.source_body_digest) THEN + RAISE EXCEPTION 'qualified selected directory source body changed'; END IF; + RETURN QUERY SELECT reference.name,reference.git_oid,reference.kind,reference.byte_size,reference.content_digest, + reference.child_root,reference.child_generation,reference.child_certificate_digest,child.attestation_id, + CASE WHEN reference.kind=4 THEN CASE WHEN child.attestation_id IS NULL THEN 'SOURCE_UNAVAILABLE' ELSE 'READY' END + WHEN fact.git_oid IS NULL THEN 'MISSING' + WHEN fact.state<>'VERIFIED' OR fact.verification_version NOT IN (1,2) + OR fact.size NOT BETWEEN 0 AND 8796093022208 OR octet_length(fact.raw_sha256)<>32 THEN 'INVALID' + WHEN fact.verification_version=1 THEN 'MISSING' + WHEN fact.size IS DISTINCT FROM reference.byte_size OR reference.kind=3 AND fact.size NOT BETWEEN 1 AND 4095 + OR CASE WHEN octet_length(fact.raw_sha256)=32 THEN fact.raw_sha256 ELSE NULL END + IS DISTINCT FROM reference.content_digest THEN 'INVALID' + ELSE 'READY' END + FROM (SELECT DISTINCT decode(value,'hex') AS name FROM jsonb_array_elements_text(names)) wanted + JOIN mst2_metadata_source_entry_reference reference ON reference.attestation_id=source_id AND reference.name=wanted.name + LEFT JOIN LATERAL ( + SELECT verified.git_oid,verified.state,verified.verification_version,verified.size,verified.raw_sha256 + FROM $CORE_SCHEMA$.mst2_verified_object verified WHERE reference.kind<>4 AND verified.storage_domain='git' + AND verified.object_kind='blob' AND verified.git_oid=split_part(reference.git_oid,':',2) + FOR SHARE OF verified NOWAIT + ) fact ON true + LEFT JOIN LATERAL ( + SELECT candidate.attestation_id FROM mst2_metadata_source_root_attestation candidate + JOIN mst2_metadata_prepare origin ON origin.prepare_id=candidate.origin_prepare_id + JOIN $CORE_SCHEMA$.mega_tree source_tree ON source_tree.tree_id=split_part(candidate.tagged_tree_oid,':',2) + WHERE reference.kind=4 AND candidate.namespace_uuid=a.namespace_uuid AND origin.state='COMMITTED' + AND candidate.tagged_tree_oid=reference.git_oid AND candidate.source_profile=a.source_profile + AND candidate.root_page=reference.child_root AND candidate.root_generation=reference.child_generation + AND candidate.root_certificate_digest=reference.child_certificate_digest + AND $CORE_SCHEMA$.mst2_route_source_tree_matches(split_part(candidate.tagged_tree_oid,':',2),candidate.source_revision,candidate.source_body_digest) + AND mst2_metadata_root_live(candidate.root_page,candidate.root_generation,candidate.root_certificate_digest) + ORDER BY candidate.attestation_id LIMIT 1 + ) child ON true ORDER BY reference.name; +END $$; + +CREATE FUNCTION mst2_metadata_cleanup_expired(maximum integer DEFAULT 64) RETURNS void LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE item record; now_unix bigint; bound integer:=least(64,greatest(0,maximum)); +BEGIN + PERFORM $CORE_SCHEMA$.mst2_route_enter($CORE_LITERAL$); + now_unix:=floor(extract(epoch FROM clock_timestamp()))::bigint; + FOR item IN SELECT * FROM ( + (SELECT 'READER'::text AS kind,operation_id::text AS owner,lease_id,hard_deadline_unix AS deadline + FROM mst2_metadata_reader_operation WHERE state='ACTIVE' AND hard_deadline_unix<=now_unix + ORDER BY hard_deadline_unix,operation_id LIMIT bound) + UNION ALL + (SELECT 'LEASE'::text,lease_id,lease_id,expires_at_unix + FROM mst2_qualified_lease_binding WHERE state='ACTIVE' AND expires_at_unix<=now_unix + ORDER BY expires_at_unix,lease_id LIMIT bound) + ) expired ORDER BY deadline,kind,owner LIMIT bound LOOP + IF item.kind='READER' THEN + UPDATE mst2_metadata_reader_operation SET state='EXPIRED' WHERE operation_id=item.owner::uuid AND state='ACTIVE'; + DELETE FROM mst2_metadata_root_anchor WHERE reader_operation_id=item.owner::uuid AND anchor_kind IN ('REQUEST','READER'); + ELSE + UPDATE mst2_qualified_lease_binding SET state='EXPIRED',lease_epoch=lease_epoch+1 WHERE lease_id=item.owner AND state='ACTIVE'; + END IF; + PERFORM mst2_metadata_cleanup_lease(item.lease_id); + END LOOP; +END $$; + +CREATE FUNCTION mst2_metadata_gc_owner_cleanup(maximum integer) +RETURNS TABLE(examined bigint,readers_expired bigint,leases_expired bigint,prepares_aborted bigint, + handovers_retired bigint,orphans_retired bigint) +LANGUAGE plpgsql VOLATILE SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE bound integer:=maximum; item record; q mst2_metadata_prepare%ROWTYPE; now_unix bigint; +BEGIN + PERFORM $CORE_SCHEMA$.mst2_route_enter($CORE_LITERAL$); + IF maximum IS NULL OR maximum NOT BETWEEN 0 AND 64 THEN RAISE EXCEPTION 'qualified owner cleanup budget must be 0..=64'; END IF; + examined:=0; readers_expired:=0; leases_expired:=0; prepares_aborted:=0; handovers_retired:=0; orphans_retired:=0; + now_unix:=floor(extract(epoch FROM clock_timestamp()))::bigint; + FOR item IN SELECT * FROM ( + (SELECT 'READER'::text AS kind,operation_id::text AS owner,lease_id,hard_deadline_unix AS deadline + FROM mst2_metadata_reader_operation WHERE state='ACTIVE' AND hard_deadline_unix<=now_unix + ORDER BY hard_deadline_unix,operation_id LIMIT bound) + UNION ALL + (SELECT 'LEASE'::text,lease_id,lease_id,expires_at_unix FROM mst2_qualified_lease_binding + WHERE state='ACTIVE' AND expires_at_unix<=now_unix ORDER BY expires_at_unix,lease_id LIMIT bound) + UNION ALL + (SELECT 'PREPARE'::text,prepare_id,NULL::text,floor(extract(epoch FROM orphan_expires_at))::bigint + FROM mst2_metadata_prepare WHERE (state='PREPARING' OR state='COMMITTED' AND coverage_retired_at IS NULL) + AND orphan_expires_at<=clock_timestamp() ORDER BY orphan_expires_at,prepare_id LIMIT bound) + ) expired ORDER BY deadline,kind,owner LIMIT bound LOOP + examined:=examined+1; + IF item.kind='READER' THEN + UPDATE mst2_metadata_reader_operation SET state='EXPIRED' WHERE operation_id=item.owner::uuid AND state='ACTIVE'; + IF NOT FOUND THEN RAISE EXCEPTION 'qualified expired reader changed behind its mutation barrier'; END IF; + readers_expired:=readers_expired+1; + DELETE FROM mst2_metadata_root_anchor WHERE reader_operation_id=item.owner::uuid AND anchor_kind IN ('REQUEST','READER'); + PERFORM mst2_metadata_cleanup_lease(item.lease_id); + ELSIF item.kind='LEASE' THEN + UPDATE mst2_qualified_lease_binding SET state='EXPIRED',lease_epoch=lease_epoch+1 WHERE lease_id=item.owner AND state='ACTIVE'; + IF NOT FOUND THEN RAISE EXCEPTION 'qualified expired lease changed behind its mutation barrier'; END IF; + leases_expired:=leases_expired+1; PERFORM mst2_metadata_cleanup_lease(item.lease_id); + ELSE + SELECT * INTO STRICT q FROM mst2_metadata_prepare WHERE prepare_id=item.owner; + IF q.state='PREPARING' THEN + DELETE FROM mst2_metadata_graph_root WHERE prepare_id=q.prepare_id; + UPDATE mst2_metadata_prepare SET state='ABORTED',aborted_at=clock_timestamp() WHERE prepare_id=q.prepare_id; + DELETE FROM mst2_metadata_root_anchor WHERE prepare_id=q.prepare_id AND anchor_kind IN ('PREPARE','REUSE'); + prepares_aborted:=prepares_aborted+1; + ELSIF mst2_metadata_session_covers_prepare(q.prepare_id) THEN + UPDATE mst2_metadata_prepare SET coverage_retired_at=clock_timestamp() WHERE prepare_id=q.prepare_id; + DELETE FROM mst2_metadata_root_anchor WHERE prepare_id=q.prepare_id AND anchor_kind IN ('PREPARE','REUSE'); + handovers_retired:=handovers_retired+1; + ELSIF mst2_metadata_orphan_prepare_eligible(q.prepare_id) THEN + UPDATE mst2_metadata_prepare SET coverage_retired_at=clock_timestamp() WHERE prepare_id=q.prepare_id; + DELETE FROM mst2_metadata_root_anchor WHERE prepare_id=q.prepare_id AND anchor_kind IN ('PREPARE','REUSE'); + orphans_retired:=orphans_retired+1; + END IF; + END IF; + END LOOP; + RETURN NEXT; +END $$; + +CREATE FUNCTION mst2_route_family_registration_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$CORE_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE g mst2_metadata_namespace%ROWTYPE; stamp record; +BEGIN + SELECT * INTO STRICT g FROM mst2_metadata_namespace WHERE singleton=1; + IF (SELECT count(*) FROM mst2_metadata_namespace)<>1 OR NEW.singleton IS NOT NULL + OR substr(NEW.namespace_uuid::text,15,1)<>'4' OR substr(NEW.namespace_uuid::text,20,1) NOT IN ('8','9','a','b') + OR NEW.graph_domain<>'qualified-v1' OR NEW.family_identity<>'v3-rooted-qualified-1' + OR NEW.admission_state<>'ROOTED_Q_ADMITTED' OR NEW.collector_state<>'ENABLED' + OR ROW(NEW.core_schema,NEW.core_schema_oid,NEW.database_name,NEW.database_oid,NEW.storage_uuid, + NEW.server_address,NEW.server_port,NEW.mono_lock_key2) IS DISTINCT FROM + ROW(g.core_schema,g.core_schema_oid,g.database_name,g.database_oid,g.storage_uuid, + g.server_address,g.server_port,g.mono_lock_key2) + OR NEW.metadata_schema IS DISTINCT FROM 'mst2q_'||replace(NEW.namespace_uuid::text,'-','') + OR NEW.metadata_storage_uuid IS NOT DISTINCT FROM g.storage_uuid + OR NEW.metadata_storage_uuid IS DISTINCT FROM (NEW.metadata_storage_uuid::uuid)::text + OR substr(NEW.metadata_storage_uuid,15,1)<>'4' OR substr(NEW.metadata_storage_uuid,20,1) NOT IN ('8','9','a','b') + OR NOT EXISTS(SELECT 1 FROM pg_catalog.pg_namespace n + WHERE n.oid=NEW.metadata_schema_oid AND n.nspname=NEW.metadata_schema) + OR NEW.implementation_fingerprint IS DISTINCT FROM decode('$IMPLEMENTATION_SHA$','hex') + OR NOT mst2_route_lock_held(1297043024,g.mono_lock_key2) + OR NOT mst2_route_lock_held(1296718001,pg_catalog.hashtext(g.core_schema)) + OR NOT mst2_route_lock_held(1296717362,pg_catalog.hashtext(NEW.metadata_schema)) + OR NOT mst2_route_lock_held(1296717362,pg_catalog.hashtext(g.metadata_schema)) THEN + RAISE EXCEPTION 'qualified namespace registration has no exact admitted rooted family scope and lock set'; + END IF; + EXECUTE pg_catalog.format('SELECT * FROM %I.mst2_metadata_family_identity WHERE singleton=1',NEW.metadata_schema) + INTO STRICT stamp; + IF ROW(stamp.namespace_uuid,stamp.storage_uuid,stamp.core_schema_oid,stamp.metadata_schema_oid, + stamp.family_identity,stamp.implementation_fingerprint) IS DISTINCT FROM + ROW(NEW.namespace_uuid,NEW.metadata_storage_uuid,NEW.core_schema_oid,NEW.metadata_schema_oid, + NEW.family_identity,NEW.implementation_fingerprint) + OR NEW.catalog_fingerprint IS DISTINCT FROM mst2_route_family_catalog(NEW.core_schema_oid,NEW.metadata_schema_oid) THEN + RAISE EXCEPTION 'qualified namespace registration fingerprint disagrees with its physical family'; + END IF; + IF NOT EXISTS(SELECT 1 FROM mst2_qualified_family_policy p WHERE p.singleton=1 + AND p.implementation_fingerprint=NEW.implementation_fingerprint + AND p.authority_catalog=mst2_route_family_catalog(NEW.core_schema_oid,0::oid,NEW.metadata_schema_oid) + AND p.expected_shape=mst2_route_family_shape(NEW.metadata_schema_oid,NEW.namespace_uuid,NEW.metadata_storage_uuid)) THEN + RAISE EXCEPTION 'qualified namespace does not have the trusted complete physical family shape'; + END IF; + RETURN NEW; +END $$; diff --git a/src/jupiter/migration/m20261008_000500_previous_rooted_decoder.sql b/src/jupiter/migration/m20261008_000500_previous_rooted_decoder.sql new file mode 100644 index 00000000..94ca2612 --- /dev/null +++ b/src/jupiter/migration/m20261008_000500_previous_rooted_decoder.sql @@ -0,0 +1,110 @@ +CREATE FUNCTION mst2_metadata_decode_rooted_plan(b bytea) RETURNS jsonb +LANGUAGE plpgsql IMMUTABLE STRICT SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE domain bytea:=convert_to('mega.mst2.rooted-install.v1','UTF8')||decode('00','hex'); + cursor_pos integer; part jsonb; source_domain text; tree_oid text; scope text; hash_kind text; identity jsonb; + root bytea; count numeric; i integer; page bytea; parent bytea; child bytea; previous bytea; previous_tree text; + size numeric; generation numeric; attestation bytea; attestation_digest bytea; certificate_digest bytea; + delta_rows jsonb[]:=ARRAY[]::jsonb[]; edge_rows jsonb[]:=ARRAY[]::jsonb[]; + reuse_rows jsonb[]:=ARRAY[]::jsonb[]; source_rows jsonb[]:=ARRAY[]::jsonb[]; total_bytes bigint:=0; + delta_index jsonb; node_index jsonb; adjacency jsonb; +BEGIN + IF octet_length(b)>2097152 OR substring(b FROM 1 FOR octet_length(domain))<>domain THEN + RAISE EXCEPTION 'rooted preparation domain or byte budget is invalid'; + END IF; + cursor_pos:=octet_length(domain); + IF mst2_metadata_read_be(b,cursor_pos,2)<>1 THEN RAISE EXCEPTION 'rooted preparation version is unsupported'; END IF; + cursor_pos:=cursor_pos+2; + part:=mst2_metadata_rooted_string(b,cursor_pos,64); cursor_pos:=(part->>'end')::integer; source_domain:=part->>'text'; + part:=mst2_metadata_rooted_string(b,cursor_pos,128); cursor_pos:=(part->>'end')::integer; tree_oid:=part->>'text'; + part:=mst2_metadata_rooted_string(b,cursor_pos,4096); cursor_pos:=(part->>'end')::integer; scope:=part->>'text'; + IF source_domain<>'native-git' OR tree_oid !~ '^(sha1:[0-9a-f]{40}|sha256:[0-9a-f]{64}|blake3:[0-9a-f]{64})$' + OR scope NOT LIKE '/%' OR scope<>'/' AND (scope LIKE '%/' OR scope LIKE '%//%' OR EXISTS( + SELECT 1 FROM unnest(string_to_array(substring(scope FROM 2),'/')) name WHERE name IN ('','.','..') + OR octet_length(name)>255) OR cardinality(string_to_array(substring(scope FROM 2),'/'))>256) THEN + RAISE EXCEPTION 'rooted source identity or scope is invalid'; END IF; + hash_kind:=split_part(tree_oid,':',1); + identity:=jsonb_build_object('source_domain',source_domain,'tagged_root_tree_oid',tree_oid,'scope',scope, + 'schema_version',mst2_metadata_read_be(b,cursor_pos,2), + 'metadata_codec',mst2_metadata_read_be(b,cursor_pos+2,2), + 'materialization_policy',mst2_metadata_read_be(b,cursor_pos+4,2), + 'fs_semantics',mst2_metadata_read_be(b,cursor_pos+6,2), + 'access_projection',mst2_metadata_read_be(b,cursor_pos+8,2), + 'verification_revision',mst2_metadata_read_be(b,cursor_pos+10,4), + 'projection_revision',mst2_metadata_read_be(b,cursor_pos+14,2)); + cursor_pos:=cursor_pos+16; + IF (identity->>'schema_version')::integer<>2 OR (identity->>'metadata_codec')::integer<>1 + OR (identity->>'materialization_policy')::integer<>1 OR (identity->>'fs_semantics')::integer<>1 + OR (identity->>'access_projection')::integer<>0 OR (identity->>'verification_revision')::integer<>2 + OR (identity->>'projection_revision')::integer<>1 THEN RAISE EXCEPTION 'rooted source profile is not current native'; END IF; + IF cursor_pos>octet_length(b)-32 THEN RAISE EXCEPTION 'rooted metadata root is truncated'; END IF; + root:=substring(b FROM cursor_pos+1 FOR 32); cursor_pos:=cursor_pos+32; + count:=mst2_metadata_read_be(b,cursor_pos,4); cursor_pos:=cursor_pos+4; + IF count>4096 THEN RAISE EXCEPTION 'rooted delta exceeds its node budget'; END IF; + IF count>0 THEN FOR i IN 1..count::integer LOOP + IF cursor_pos>octet_length(b)-40 THEN RAISE EXCEPTION 'rooted delta member is truncated'; END IF; + page:=substring(b FROM cursor_pos+1 FOR 32); size:=mst2_metadata_read_be(b,cursor_pos+32,8); cursor_pos:=cursor_pos+40; + IF previous IS NOT NULL AND previous>=page OR size NOT BETWEEN 20 AND 16384 THEN + RAISE EXCEPTION 'rooted delta is not exactly ordered or has invalid size'; END IF; + previous:=page; total_bytes:=total_bytes+size::bigint; + IF total_bytes>67108864 THEN RAISE EXCEPTION 'rooted delta exceeds its metadata byte budget'; END IF; + delta_rows:=array_append(delta_rows,jsonb_build_object('page',encode(page,'hex'),'size',size)); + END LOOP; END IF; + previous:=NULL; count:=mst2_metadata_read_be(b,cursor_pos,4); cursor_pos:=cursor_pos+4; + IF count>16384 THEN RAISE EXCEPTION 'rooted delta exceeds its edge budget'; END IF; + IF count>0 THEN FOR i IN 1..count::integer LOOP + IF cursor_pos>octet_length(b)-64 THEN RAISE EXCEPTION 'rooted edge is truncated'; END IF; + parent:=substring(b FROM cursor_pos+1 FOR 32); child:=substring(b FROM cursor_pos+33 FOR 32); cursor_pos:=cursor_pos+64; + IF previous IS NOT NULL AND previous>=parent||child OR parent=child THEN RAISE EXCEPTION 'rooted edges are not unique and ordered'; END IF; + previous:=parent||child; + edge_rows:=array_append(edge_rows,jsonb_build_object('parent',encode(parent,'hex'),'child',encode(child,'hex'))); + END LOOP; END IF; + previous:=NULL; count:=mst2_metadata_read_be(b,cursor_pos,4); cursor_pos:=cursor_pos+4; + IF count>4096 OR coalesce(array_length(delta_rows,1),0)+count>4096 THEN RAISE EXCEPTION 'rooted delta and boundaries exceed node budget'; END IF; + IF count>0 THEN FOR i IN 1..count::integer LOOP + IF cursor_pos>octet_length(b)-120 THEN RAISE EXCEPTION 'rooted reuse boundary is truncated'; END IF; + page:=substring(b FROM cursor_pos+1 FOR 32); generation:=mst2_metadata_read_be(b,cursor_pos+32,8); + attestation:=substring(b FROM cursor_pos+41 FOR 16); attestation_digest:=substring(b FROM cursor_pos+57 FOR 32); + certificate_digest:=substring(b FROM cursor_pos+89 FOR 32); cursor_pos:=cursor_pos+120; + IF previous IS NOT NULL AND previous>=page OR generation NOT BETWEEN 1 AND 9223372036854775807 THEN + RAISE EXCEPTION 'rooted reuse boundaries are not exact positive ordered lifetimes'; END IF; + previous:=page; + reuse_rows:=array_append(reuse_rows,jsonb_build_object('page',encode(page,'hex'),'generation',generation, + 'attestation_id',encode(attestation,'hex')::uuid,'attestation_digest',encode(attestation_digest,'hex'), + 'certificate_digest',encode(certificate_digest,'hex'))); + END LOOP; END IF; + count:=mst2_metadata_read_be(b,cursor_pos,4); cursor_pos:=cursor_pos+4; + IF count NOT BETWEEN 1 AND 4096 THEN RAISE EXCEPTION 'rooted source-root budget is invalid'; END IF; + FOR i IN 1..count::integer LOOP + part:=mst2_metadata_rooted_string(b,cursor_pos,128); cursor_pos:=(part->>'end')::integer; tree_oid:=part->>'text'; + IF cursor_pos>octet_length(b)-32 THEN RAISE EXCEPTION 'rooted source root is truncated'; END IF; + page:=substring(b FROM cursor_pos+1 FOR 32); cursor_pos:=cursor_pos+32; + IF tree_oid !~ '^(sha1:[0-9a-f]{40}|sha256:[0-9a-f]{64}|blake3:[0-9a-f]{64})$' + OR split_part(tree_oid,':',1)<>hash_kind OR previous_tree IS NOT NULL AND convert_to(previous_tree,'UTF8')>=convert_to(tree_oid,'UTF8') THEN + RAISE EXCEPTION 'rooted source roots are not unique ordered same-profile identities'; END IF; + previous_tree:=tree_oid; source_rows:=array_append(source_rows,jsonb_build_object('tree_oid',tree_oid,'page',encode(page,'hex'))); + END LOOP; + IF cursor_pos<>octet_length(b) THEN RAISE EXCEPTION 'rooted preparation has trailing bytes'; END IF; + SELECT coalesce(jsonb_object_agg(value->>'page',true),'{}'::jsonb) INTO delta_index FROM unnest(delta_rows) d(value); + SELECT coalesce(jsonb_object_agg(value->>'page',true),'{}'::jsonb) INTO node_index FROM ( + SELECT d.value FROM unnest(delta_rows) d(value) UNION ALL SELECT r.value FROM unnest(reuse_rows) r(value)) nodes; + IF coalesce(array_length(delta_rows,1),0)+coalesce(array_length(reuse_rows,1),0)=0 + OR EXISTS(SELECT 1 FROM unnest(reuse_rows) r(value) WHERE delta_index ? (r.value->>'page')) + OR NOT node_index ? encode(root,'hex') + OR EXISTS(SELECT 1 FROM unnest(edge_rows) e(value) WHERE NOT delta_index ? (e.value->>'parent') + OR NOT node_index ? (e.value->>'child')) + OR EXISTS(SELECT 1 FROM unnest(source_rows) s(value) WHERE NOT node_index ? (s.value->>'page')) + OR NOT EXISTS(SELECT 1 FROM unnest(source_rows) s(value) WHERE value->>'page'=encode(root,'hex')) THEN + RAISE EXCEPTION 'rooted plan has overlap or an unbound graph/source endpoint'; + END IF; + SELECT coalesce(jsonb_object_agg(parent,children),'{}'::jsonb) INTO adjacency FROM ( + SELECT value->>'parent' AS parent,jsonb_agg(value->>'child') AS children FROM unnest(edge_rows) e(value) + GROUP BY value->>'parent') grouped; + IF EXISTS(WITH RECURSIVE reached(page) AS (SELECT encode(root,'hex') UNION + SELECT child FROM reached r CROSS JOIN LATERAL jsonb_array_elements_text(adjacency->r.page) children(child)) + SELECT 1 FROM (SELECT d.value FROM unnest(delta_rows) d(value) UNION ALL SELECT r.value FROM unnest(reuse_rows) r(value)) nodes + WHERE NOT EXISTS(SELECT 1 FROM reached r WHERE r.page=nodes.value->>'page')) THEN + RAISE EXCEPTION 'rooted plan includes members outside its bounded delta and boundary closure'; + END IF; + RETURN identity||jsonb_build_object('root',encode(root,'hex'),'delta',to_jsonb(delta_rows),'edges',to_jsonb(edge_rows), + 'reused',to_jsonb(reuse_rows),'source_roots',to_jsonb(source_rows),'total_delta_bytes',total_bytes); +END $$; \ No newline at end of file diff --git a/src/jupiter/migration/m20261008_000600_fix_mst2_descriptor_wire.rs b/src/jupiter/migration/m20261008_000600_fix_mst2_descriptor_wire.rs new file mode 100644 index 00000000..7efcf97f --- /dev/null +++ b/src/jupiter/migration/m20261008_000600_fix_mst2_descriptor_wire.rs @@ -0,0 +1,21 @@ +//! Repair canonical descriptor integer encoding without rewriting ownership. + +use sea_orm_migration::prelude::*; + +#[derive(DeriveMigrationName)] +pub struct Migration; + +#[async_trait::async_trait] +impl MigrationTrait for Migration { + async fn up(&self, manager: &SchemaManager) -> Result<(), DbErr> { + super::qualified_native_runtime_upgrade::upgrade_descriptor(manager).await + } + + async fn down(&self, _manager: &SchemaManager) -> Result<(), DbErr> { + Ok(()) + } + + fn use_transaction(&self) -> Option { + Some(true) + } +} diff --git a/src/jupiter/migration/m20261008_000600_previous_descriptor.sql b/src/jupiter/migration/m20261008_000600_previous_descriptor.sql new file mode 100644 index 00000000..dab92001 --- /dev/null +++ b/src/jupiter/migration/m20261008_000600_previous_descriptor.sql @@ -0,0 +1,27 @@ +CREATE FUNCTION mst2_metadata_descriptor(pid text,instance text,commit_id text,tree_id text) +RETURNS bytea LANGUAGE plpgsql VOLATILE STRICT SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE p record; scope_bytes bytea; view_digest bytea; instance_bytes bytea; +BEGIN + SELECT tagged_root_tree_oid,scope,metadata_root INTO p FROM mst2_metadata_prepare WHERE prepare_id=pid AND plan_kind='ROOTED' + AND state='COMMITTED' AND graph_domain='qualified-v1' AND mst2_metadata_scope_matches(primary_scope); + IF NOT FOUND OR instance IS DISTINCT FROM (instance::uuid)::text + OR split_part(p.tagged_root_tree_oid,':',2) IS DISTINCT FROM tree_id + OR NOT EXISTS(SELECT 1 FROM $CORE_SCHEMA$.mega_commit c WHERE c.commit_id=$3 AND c.tree=$4) + OR octet_length(commit_id)<>octet_length(tree_id) + OR commit_id !~ '^([0-9a-f]{40}|[0-9a-f]{64})$' THEN + RAISE EXCEPTION 'qualified descriptor has no exact canonical instance and fixed source'; + END IF; + PERFORM mst2_metadata_native_profile(pid); + scope_bytes:=convert_to(p.scope,'UTF8'); + IF p.scope !~ '^/' OR octet_length(scope_bytes)>4096 OR p.scope<>'/' AND + (p.scope ~ '/$|//|/(\.|\.\.)(/|$)' OR cardinality(string_to_array(substring(p.scope FROM 2),'/'))>256) + OR EXISTS(SELECT 1 FROM unnest(string_to_array(substring(p.scope FROM 2),'/')) component + WHERE octet_length(component)>255) THEN + RAISE EXCEPTION 'qualified descriptor scope is not canonical'; + END IF; + instance_bytes:=decode(replace(instance,'-',''),'hex'); + view_digest:=sha256(convert_to('mega.mst2.namespaceview','UTF8')||decode('00','hex')||convert_to(commit_id,'UTF8')); + RETURN convert_to('MSD2','UTF8')||decode('00020001','hex')||instance_bytes||view_digest + ||decode(lpad(to_hex(octet_length(scope_bytes)),4,'0'),'hex')||scope_bytes + ||decode('0001000100000000','hex')||p.metadata_root; +END $$; diff --git a/src/jupiter/migration/mod.rs b/src/jupiter/migration/mod.rs index 61d94c37..e3a737a5 100644 --- a/src/jupiter/migration/mod.rs +++ b/src/jupiter/migration/mod.rs @@ -139,11 +139,81 @@ mod m20260921_000100_fix_git_tag_unique; mod m20260923_000100_import_repo_cleanups; mod m20260923_000200_canonicalize_import_repo_paths; mod m20260925_000100_media_paging; +mod m20261005_000100_add_mst2_publication_request_digest; +mod m20261005_000100_add_mst2_retention_durability; +mod m20261005_000200_add_mst2_native_head; +mod m20261005_000200_harden_mst2_retention_graph; +mod m20261005_000300_add_mst2_metadata_install; mod m20261006_000100_add_view_tables; +mod m20261007_000100_add_mst2_snapshot_sessions; +mod m20261007_000200_add_mst2_metadata_generations; +mod m20261007_000300_add_mst2_metadata_lifetime_history; +mod m20261007_000400_add_mst2_qualified_metadata_gc; +mod m20261007_000500_add_mst2_install_capability; +mod m20261007_000600_add_mst2_storage_routes; +mod m20261008_000100_add_mst2_chunk_maps; +mod m20261008_000200_add_mst2_rooted_qualified_family; +mod m20261008_000300_add_mst2_chunk_map_retention; +mod m20261008_000400_add_mst2_reader_retention; +mod m20261008_000500_fix_mst2_native_runtime; +mod m20261008_000600_fix_mst2_descriptor_wire; +pub(crate) mod qualified_native_runtime_upgrade; mod runner; pub use m20260905_000100_add_push_queue::ensure_queue_control_seed; pub use runner::apply_migrations; +#[cfg(test)] +pub(crate) async fn test_upgrade_reader_retention( + connection: &sea_orm::DatabaseConnection, +) -> Result<(), sea_orm::DbErr> { + use sea_orm::TransactionTrait; + use sea_orm_migration::{MigrationTrait, SchemaManager}; + + let txn = connection.begin().await?; + let result = async { + m20261008_000400_add_mst2_reader_retention::Migration + .up(&SchemaManager::new(&txn)) + .await?; + m20261008_000600_fix_mst2_descriptor_wire::Migration + .up(&SchemaManager::new(&txn)) + .await + } + .await; + match result { + Ok(()) => txn.commit().await, + Err(error) => { + txn.rollback().await?; + Err(error) + } + } +} + +#[cfg(test)] +pub(crate) async fn test_upgrade_native_runtime( + connection: &sea_orm::DatabaseConnection, +) -> Result<(), sea_orm::DbErr> { + use sea_orm::TransactionTrait; + use sea_orm_migration::{MigrationTrait, SchemaManager}; + + let txn = connection.begin().await?; + let result = async { + m20261008_000500_fix_mst2_native_runtime::Migration + .up(&SchemaManager::new(&txn)) + .await?; + m20261008_000600_fix_mst2_descriptor_wire::Migration + .up(&SchemaManager::new(&txn)) + .await + } + .await; + match result { + Ok(()) => txn.commit().await, + Err(error) => { + txn.rollback().await?; + Err(error) + } + } +} + /// Primary key `BIGINT` (not DB auto-increment); the application assigns `id` (e.g. `idgenerator::IdInstance::next_id`). fn pk_bigint(name: T) -> ColumnDef { big_integer(name).primary_key().take() @@ -267,7 +337,24 @@ impl MigratorTrait for Migrator { Box::new(m20260923_000100_import_repo_cleanups::Migration), Box::new(m20260923_000200_canonicalize_import_repo_paths::Migration), Box::new(m20260925_000100_media_paging::Migration), + Box::new(m20261005_000100_add_mst2_publication_request_digest::Migration), + Box::new(m20261005_000200_add_mst2_native_head::Migration), + Box::new(m20261005_000100_add_mst2_retention_durability::Migration), + Box::new(m20261005_000200_harden_mst2_retention_graph::Migration), + Box::new(m20261005_000300_add_mst2_metadata_install::Migration), Box::new(m20261006_000100_add_view_tables::Migration), + Box::new(m20261007_000100_add_mst2_snapshot_sessions::Migration), + Box::new(m20261007_000200_add_mst2_metadata_generations::Migration), + Box::new(m20261007_000300_add_mst2_metadata_lifetime_history::Migration), + Box::new(m20261007_000400_add_mst2_qualified_metadata_gc::Migration), + Box::new(m20261007_000500_add_mst2_install_capability::Migration), + Box::new(m20261007_000600_add_mst2_storage_routes::Migration), + Box::new(m20261008_000100_add_mst2_chunk_maps::Migration), + Box::new(m20261008_000200_add_mst2_rooted_qualified_family::Migration), + Box::new(m20261008_000300_add_mst2_chunk_map_retention::Migration), + Box::new(m20261008_000400_add_mst2_reader_retention::Migration), + Box::new(m20261008_000500_fix_mst2_native_runtime::Migration), + Box::new(m20261008_000600_fix_mst2_descriptor_wire::Migration), ] } } @@ -1177,16 +1264,34 @@ mod tests { #[tokio::test] async fn import_repo_alias_rows_canonicalized() { let names = migration_names(); + let expected = [ + "m20260923_000200_canonicalize_import_repo_paths", + "m20260925_000100_media_paging", + "m20261005_000100_add_mst2_publication_request_digest", + "m20261005_000200_add_mst2_native_head", + "m20261005_000100_add_mst2_retention_durability", + "m20261005_000200_harden_mst2_retention_graph", + "m20261005_000300_add_mst2_metadata_install", + VIEW_MIGRATION_NAME, + "m20261007_000100_add_mst2_snapshot_sessions", + "m20261007_000200_add_mst2_metadata_generations", + "m20261007_000300_add_mst2_metadata_lifetime_history", + "m20261007_000400_add_mst2_qualified_metadata_gc", + "m20261007_000500_add_mst2_install_capability", + "m20261007_000600_add_mst2_storage_routes", + "m20261008_000100_add_mst2_chunk_maps", + "m20261008_000200_add_mst2_rooted_qualified_family", + "m20261008_000300_add_mst2_chunk_map_retention", + "m20261008_000400_add_mst2_reader_retention", + "m20261008_000500_fix_mst2_native_runtime", + "m20261008_000600_fix_mst2_descriptor_wire", + ] + .map(str::to_owned); assert_eq!( - &names[names.len() - 3..], - &[ - "m20260923_000200_canonicalize_import_repo_paths".to_string(), - "m20260925_000100_media_paging".to_string(), - VIEW_MIGRATION_NAME.to_string(), - ], - "view tables are registered last" + &names[names.len() - expected.len()..], + expected.as_slice(), + "the complete native migration suffix follows media paging in registered order" ); - let db = alias_db().await; insert_repo(&db, 1, "/third-party//a").await; insert_repo(&db, 2, "/third-party/b/").await; diff --git a/src/jupiter/migration/qualified_native_runtime_upgrade.rs b/src/jupiter/migration/qualified_native_runtime_upgrade.rs new file mode 100644 index 00000000..85cd256b --- /dev/null +++ b/src/jupiter/migration/qualified_native_runtime_upgrade.rs @@ -0,0 +1,654 @@ +//! Trusted forward repair of the physical v3 Q family, preserving history. + +use sea_orm::{ConnectionTrait, DbBackend, Statement, Value}; +use sea_orm_migration::prelude::*; +use sha2::{Digest, Sha256}; + +use crate::jupiter::storage::qualified_metadata_family::{ + implementation_fingerprint, render_family, +}; + +pub(crate) const PREVIOUS_IMPLEMENTATION: &str = + "6d7095dc052e60bf6de21a987b72e23fdfbfe819355f0e847d2a7f3c1cbd3f27"; + +pub(crate) const RETENTION_IMPLEMENTATION: &str = + "900e7a340e7365c9f31995e070edbfe2076b870ec09fe89e4dc43e56f26b91ab"; +pub(crate) const NATIVE_RUNTIME_IMPLEMENTATION: &str = + "87b104e6b5593b1d6e5904a4e4be94e6ba0814dbb2c25dd8b38a0ea950d42307"; +const PREVIOUS_DECODER: &str = include_str!("m20261008_000500_previous_rooted_decoder.sql"); +const PREVIOUS_READERS: &str = include_str!("m20261008_000500_previous_reader_family.sql"); +const PREVIOUS_DESCRIPTOR: &str = include_str!("m20261008_000600_previous_descriptor.sql"); + +fn rejected(message: &str) -> DbErr { + DbErr::Custom(message.into()) +} + +fn identifier(value: &str) -> String { + format!("\"{}\"", value.replace('"', "\"\"")) +} + +fn literal(value: &str) -> String { + format!("'{}'", value.replace('\'', "''")) +} + +fn trusted_function(source: &str, name: &str) -> Result { + let prefix = format!("CREATE FUNCTION {name}("); + let start = source + .find(&prefix) + .ok_or_else(|| rejected("reader migration trusted function is missing"))?; + let function = &source[start..]; + let body = function + .find("AS $") + .ok_or_else(|| rejected("reader migration function body is missing"))? + + 3; + let tag_end = function[body + 1..] + .find('$') + .ok_or_else(|| rejected("reader migration function delimiter is missing"))? + + body + + 2; + let tag = &function[body..tag_end]; + let end = function[tag_end..] + .find(tag) + .ok_or_else(|| rejected("reader migration function terminator is missing"))? + + tag_end + + tag.len(); + if function.as_bytes().get(end) != Some(&b';') { + return Err(rejected("reader migration function terminator is invalid")); + } + Ok(function[..=end].replacen("CREATE FUNCTION", "CREATE OR REPLACE FUNCTION", 1)) +} + +async fn digest( + connection: &C, + sql: String, + values: Vec, +) -> Result, DbErr> { + connection + .query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + sql, + values, + )) + .await? + .ok_or_else(|| rejected("reader migration physical fingerprint is missing"))? + .try_get("", "fingerprint") +} + +async fn catalog( + connection: &C, + core_oid: i64, + q_oid: i64, + exempt_oid: i64, +) -> Result, DbErr> { + let source = include_str!("../storage/qualified_family_catalog.sql") + .replace("$CORE_OID$", "$1::bigint::oid") + .replace("$Q_OID$", "$2::bigint::oid") + .replace("$EXEMPT_Q_OID$", "$3::bigint::oid"); + digest( + connection, + source, + vec![core_oid.into(), q_oid.into(), exempt_oid.into()], + ) + .await +} + +async fn shape( + connection: &C, + q_oid: i64, + namespace: &str, + storage: &str, +) -> Result, DbErr> { + let search_path: String = connection + .query_one_raw(Statement::from_string( + DbBackend::Postgres, + "SELECT pg_catalog.current_setting('search_path') AS search_path", + )) + .await? + .ok_or_else(|| rejected("reader migration caller search path is missing"))? + .try_get("", "search_path")?; + // Catalog deparsing must use the same path as the trusted shape helper. + connection + .execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT pg_catalog.set_config('search_path',$1,true)", + ["pg_catalog,pg_temp".into()], + )) + .await?; + let source = include_str!("../storage/qualified_family_shape.sql") + .replace("q_oid", "$1::bigint::oid") + .replace("n_uuid", "$2::uuid") + .replace("s_uuid", "$3::text"); + let fingerprint = digest( + connection, + source, + vec![q_oid.into(), namespace.into(), storage.into()], + ) + .await; + let restored = connection + .execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT pg_catalog.set_config('search_path',$1,true)", + [search_path.into()], + )) + .await; + let fingerprint = fingerprint?; + restored?; + Ok(fingerprint) +} + +fn replace_function(source: &mut String, name: &str, captured: &str) -> Result<(), DbErr> { + let plain = |sql: &str| -> Result { + Ok(trusted_function(sql, name)?.replacen( + "CREATE OR REPLACE FUNCTION", + "CREATE FUNCTION", + 1, + )) + }; + let current = plain(source)?; + let previous = plain(captured)?; + *source = source.replacen(¤t, &previous, 1); + Ok(()) +} + +/// Historical source is used only to authenticate forward migration templates. +pub(crate) fn render_prior_family( + core: &str, + core_oid: i64, + q: &str, + q_oid: i64, + namespace: &str, + storage: &str, + implementation: &str, +) -> Result { + if ![ + PREVIOUS_IMPLEMENTATION, + RETENTION_IMPLEMENTATION, + NATIVE_RUNTIME_IMPLEMENTATION, + ] + .contains(&implementation) + { + return Err(rejected("native runtime template version is unsupported")); + } + let decoder = PREVIOUS_DECODER.replace("\r\n", "\n"); + let readers = PREVIOUS_READERS.replace("\r\n", "\n"); + let descriptor = PREVIOUS_DESCRIPTOR.replace("\r\n", "\n"); + if hex::encode(Sha256::digest(decoder.as_bytes())) + != "f14d024cfb821f9d6b7348ab9b69f48c09862ff7dd3fbb2b360b253f949e35a9" + || hex::encode(Sha256::digest(readers.as_bytes())) + != "ec9b422440da111e5e047e1a4c85a28a33053cf0e320a56ea944ebefeaa8ea37" + || hex::encode(Sha256::digest(descriptor.as_bytes())) + != "c833666fcb1b3c9fd0bf186e5ad732f5e8b9b9dbf4360ce8d93a0d91d17d5c1a" + { + return Err(rejected("native runtime historical source capture changed")); + } + let mut source = + render_family(core, core_oid, q, q_oid, namespace, storage).replace("\r\n", "\n"); + replace_function(&mut source, "mst2_metadata_descriptor", &descriptor)?; + if implementation != NATIVE_RUNTIME_IMPLEMENTATION { + replace_function(&mut source, "mst2_metadata_decode_rooted_plan", &decoder)?; + } + if implementation == PREVIOUS_IMPLEMENTATION { + let start = source + .find("CREATE TABLE mst2_metadata_reader_operation (") + .ok_or_else(|| rejected("native runtime reader template is missing"))?; + let end = source[start..] + .find("CREATE FUNCTION mst2_metadata_session_covers_prepare(") + .ok_or_else(|| rejected("native runtime reader template boundary is missing"))? + + start; + let captured_start = readers + .find("-- READER DDL BEGIN\n") + .ok_or_else(|| rejected("native runtime captured reader DDL is missing"))? + + "-- READER DDL BEGIN\n".len(); + let captured_end = readers + .find("-- READER DDL END") + .ok_or_else(|| rejected("native runtime captured reader DDL boundary is missing"))?; + source.replace_range(start..end, &readers[captured_start..captured_end]); + for name in [ + "mst2_metadata_dml_barrier", + "mst2_metadata_gc_enabled", + "mst2_metadata_root_anchor_guard", + "mst2_metadata_serving_complete", + "mst2_metadata_reader_guard", + "mst2_metadata_begin_reader", + "mst2_metadata_finish_reader", + "mst2_metadata_read_source_entries", + "mst2_metadata_cleanup_expired", + "mst2_metadata_gc_owner_cleanup", + ] { + replace_function(&mut source, name, &readers)?; + } + for name in [ + "mst2_metadata_next_reader_issuance", + "mst2_metadata_reader_issuance_guard", + "mst2_metadata_prune_readers", + ] { + let function = trusted_function(&source, name)?.replacen( + "CREATE OR REPLACE FUNCTION", + "CREATE FUNCTION", + 1, + ); + source = source.replacen(&function, "", 1); + } + source = source.replace( + "CREATE TRIGGER mst2_metadata_reader_issuance_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_metadata_reader_issuance\n FOR EACH ROW EXECUTE FUNCTION mst2_metadata_reader_issuance_guard();", "", + ).replace("'mst2_metadata_reader_issuance',", ""); + } + Ok(source + .replace(&hex::encode(implementation_fingerprint()), implementation) + .replace("$CORE_SCHEMA$", &identifier(core)) + .replace("$CORE_LITERAL$", &literal(core)) + .replace("$Q_SCHEMA$", &identifier(q)) + .replace("$Q_LITERAL$", &literal(q)) + .replace("$CORE_OID$", &core_oid.to_string()) + .replace("$Q_OID$", &q_oid.to_string()) + .replace("$NAMESPACE_UUID$", namespace) + .replace("$STORAGE_UUID$", storage) + .replace("$IMPLEMENTATION_SHA$", implementation)) +} + +struct TrustedShape { + canonical: Vec, + legacy: Option>, +} + +async fn template_shapes( + connection: &C, + core: &str, + core_oid: i64, + implementation: &[u8], +) -> Result { + let namespace = uuid::Uuid::new_v4().to_string(); + let schema = format!("mst2q_{}", namespace.replace('-', "")); + let storage = uuid::Uuid::new_v4().to_string(); + let q = identifier(&schema); + let c = identifier(core); + connection + .execute_unprepared(&format!("CREATE SCHEMA {q}")) + .await?; + let oid: i64 = connection + .query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT oid::bigint FROM pg_catalog.pg_namespace WHERE nspname=$1", + [schema.clone().into()], + )) + .await? + .ok_or_else(|| rejected("native runtime trusted template schema is missing"))? + .try_get("", "oid")?; + let version = hex::encode(implementation); + let source = if implementation == implementation_fingerprint().as_slice() { + render_family(core, core_oid, &schema, oid, &namespace, &storage) + } else { + render_prior_family(core, core_oid, &schema, oid, &namespace, &storage, &version)? + }; + connection.execute_unprepared(&source).await?; + let canonical = shape(connection, oid, &namespace, &storage).await?; + // Reproduce only the exact historical Q-visible deparsing defect. The + // exception is admitted below solely for the known 900e policy without Q. + let legacy = if version == RETENTION_IMPLEMENTATION { + let source = include_str!("../storage/qualified_family_shape.sql") + .replace("q_oid", "$1::bigint::oid") + .replace("n_uuid", "$2::uuid") + .replace("s_uuid", "$3::text"); + Some( + digest( + connection, + source, + vec![oid.into(), namespace.into(), storage.into()], + ) + .await?, + ) + } else { + None + }; + connection + .execute_unprepared(&format!( + "SET LOCAL search_path={c},pg_catalog,pg_temp; DROP SCHEMA {q} CASCADE" + )) + .await?; + Ok(TrustedShape { canonical, legacy }) +} + +struct Family { + schema: String, + oid: i64, + namespace: String, + storage: String, + catalog: Vec, +} + +pub(super) async fn upgrade( + manager: &SchemaManager<'_>, + allow_previous_readers: bool, +) -> Result<(), DbErr> { + upgrade_to(manager, allow_previous_readers, false).await +} + +pub(super) async fn upgrade_descriptor(manager: &SchemaManager<'_>) -> Result<(), DbErr> { + // Finish an interrupted reader/runtime upgrade before the independent + // descriptor repair, without replaying reader DDL on an admitted family. + upgrade(manager, true).await?; + upgrade_to(manager, false, true).await +} + +async fn upgrade_to( + manager: &SchemaManager<'_>, + allow_previous_readers: bool, + repair_descriptor: bool, +) -> Result<(), DbErr> { + if manager.get_database_backend() != DbBackend::Postgres { + return Err(rejected( + "qualified reader retention requires primary PostgreSQL", + )); + } + let connection = manager.get_connection(); + let captured = connection.query_one_raw(Statement::from_string(DbBackend::Postgres, + "SELECT n.nspname,n.oid::bigint,registry.mono_lock_key2 FROM pg_catalog.pg_namespace n + JOIN mst2_metadata_namespace registry ON registry.singleton=1 AND registry.core_schema=n.nspname + AND registry.core_schema_oid=n.oid WHERE n.nspname=current_schema()")) + .await?.ok_or_else(|| rejected("reader migration requires its captured core schema"))?; + let core: String = captured.try_get("", "nspname")?; + let core_oid: i64 = captured.try_get("", "oid")?; + let mono: i32 = captured.try_get("", "mono_lock_key2")?; + let c = identifier(&core); + connection + .execute_unprepared("SELECT pg_catalog.set_config('lock_timeout','5000ms',true)") + .await?; + connection + .execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT pg_catalog.pg_advisory_xact_lock(1297043024,$1)", + [mono.into()], + )) + .await?; + connection + .execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT pg_catalog.pg_advisory_xact_lock(1296718001,pg_catalog.hashtext($1))", + [core.clone().into()], + )) + .await?; + let namespaces = connection + .query_all_raw(Statement::from_string( + DbBackend::Postgres, + format!( + "SELECT metadata_schema FROM {c}.mst2_metadata_namespace ORDER BY namespace_uuid" + ), + )) + .await?; + if !(1..=2).contains(&namespaces.len()) { + return Err(rejected( + "reader migration namespace registry is not bounded", + )); + } + let namespace_count = namespaces.len(); + for namespace in namespaces { + let schema: String = namespace.try_get("", "metadata_schema")?; + connection + .execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT pg_catalog.pg_advisory_xact_lock(1296717362,pg_catalog.hashtext($1))", + [schema.into()], + )) + .await?; + } + connection.execute_unprepared(&format!( + "LOCK TABLE {c}.mst2_metadata_namespace,{c}.mst2_qualified_family_policy IN ACCESS EXCLUSIVE MODE" + )).await?; + let policy=connection.query_one_raw(Statement::from_string(DbBackend::Postgres, + format!("SELECT implementation_fingerprint,expected_shape,authority_catalog FROM {c}.mst2_qualified_family_policy WHERE singleton=1"))) + .await?.ok_or_else(|| rejected("reader migration trusted policy is missing"))?; + let implementation: Vec = policy.try_get("", "implementation_fingerprint")?; + let compiled = implementation_fingerprint(); + let runtime = + hex::decode(NATIVE_RUNTIME_IMPLEMENTATION).map_err(|error| rejected(&error.to_string()))?; + let current = if repair_descriptor { + compiled.clone() + } else { + runtime.clone() + }; + let previous = + hex::decode(PREVIOUS_IMPLEMENTATION).map_err(|error| rejected(&error.to_string()))?; + let retention = + hex::decode(RETENTION_IMPLEMENTATION).map_err(|error| rejected(&error.to_string()))?; + let supported = implementation == compiled + || implementation == runtime + || !repair_descriptor + && (implementation == retention + || allow_previous_readers && implementation == previous); + if !supported { + return Err(rejected( + "reader migration refuses an unsupported prior implementation", + )); + } + let rows=connection.query_all_raw(Statement::from_string(DbBackend::Postgres,format!( + "SELECT namespace_uuid::text,metadata_schema,metadata_schema_oid::bigint,metadata_storage_uuid, + implementation_fingerprint,catalog_fingerprint FROM {c}.mst2_metadata_namespace WHERE graph_domain='qualified-v1'"))).await?; + let family = match rows.as_slice() { + [] => None, + [row] => Some(Family { + schema: row.try_get("", "metadata_schema")?, + oid: row.try_get("", "metadata_schema_oid")?, + namespace: row.try_get("", "namespace_uuid")?, + storage: row.try_get("", "metadata_storage_uuid")?, + catalog: row.try_get("", "catalog_fingerprint")?, + }), + _ => return Err(rejected("reader migration requires at most one Q family")), + }; + let q_oid = family.as_ref().map_or(0, |family| family.oid); + if namespace_count != if family.is_some() { 2 } else { 1 } { + return Err(rejected( + "native runtime namespace registry identity is not exact", + )); + } + if policy.try_get::>("", "authority_catalog")? + != catalog(connection, core_oid, 0, q_oid).await? + { + return Err(rejected( + "reader migration prior core authority catalog changed", + )); + } + let scope = connection + .query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + format!("SELECT {c}.mst2_route_scope_valid($1) AS valid"), + [core.clone().into()], + )) + .await? + .ok_or_else(|| rejected("reader migration core scope is missing"))?; + if !scope.try_get::("", "valid")? { + return Err(rejected("reader migration prior core scope changed")); + } + if let Some(family) = &family { + for identity in [&family.namespace, &family.storage] { + let parsed = uuid::Uuid::parse_str(identity) + .map_err(|_| rejected("native runtime prior Q UUID is invalid"))?; + if parsed.get_version_num() != 4 + || parsed.get_variant() != uuid::Variant::RFC4122 + || parsed.to_string() != *identity + { + return Err(rejected("native runtime prior Q UUID is not canonical v4")); + } + } + let q = identifier(&family.schema); + let identity=connection.query_one_raw(Statement::from_sql_and_values(DbBackend::Postgres,format!( + "SELECT EXISTS(SELECT 1 FROM {c}.mst2_metadata_namespace n + JOIN {c}.mst2_metadata_namespace g ON g.singleton=1 + JOIN {q}.mst2_metadata_family_identity i ON i.singleton=1 + JOIN {q}.mst2_metadata_storage_scope s ON s.singleton=1 + JOIN pg_catalog.pg_namespace physical ON physical.oid=n.metadata_schema_oid AND physical.nspname=n.metadata_schema + WHERE n.namespace_uuid=$1::uuid AND n.metadata_schema='mst2q_'||replace(n.namespace_uuid::text,'-','') + AND n.singleton IS NULL AND n.family_identity='v3-rooted-qualified-1' AND n.graph_domain='qualified-v1' + AND n.admission_state='ROOTED_Q_ADMITTED' AND n.collector_state='ENABLED' + AND ROW(n.core_schema,n.core_schema_oid,n.database_name,n.database_oid,n.storage_uuid,n.server_address,n.server_port,n.mono_lock_key2) + IS NOT DISTINCT FROM ROW(g.core_schema,g.core_schema_oid,g.database_name,g.database_oid,g.storage_uuid,g.server_address,g.server_port,g.mono_lock_key2) + AND n.metadata_storage_uuid<>g.storage_uuid AND n.implementation_fingerprint=$2 + AND ROW(i.namespace_uuid,i.storage_uuid,i.core_schema_oid,i.metadata_schema_oid,i.family_identity,i.implementation_fingerprint) + IS NOT DISTINCT FROM ROW(n.namespace_uuid,n.metadata_storage_uuid,n.core_schema_oid,n.metadata_schema_oid,n.family_identity,n.implementation_fingerprint) + AND s.storage_uuid=i.storage_uuid) AS valid"),[family.namespace.clone().into(),implementation.clone().into()])).await? + .ok_or_else(|| rejected("reader migration prior Q identity is missing"))?; + if !identity.try_get::("", "valid")? + || catalog(connection, core_oid, family.oid, 0).await? != family.catalog + || shape(connection, family.oid, &family.namespace, &family.storage).await? + != policy.try_get::>("", "expected_shape")? + { + return Err(rejected( + "reader migration prior Q physical stamp or shape changed", + )); + } + } + let prior = template_shapes(connection, &core, core_oid, &implementation).await?; + let recorded_shape: Vec = policy.try_get("", "expected_shape")?; + if recorded_shape != prior.canonical + && !(implementation == retention + && family.is_none() + && prior.legacy.as_ref() == Some(&recorded_shape)) + { + return Err(rejected( + "native runtime migration prior policy shape is not trusted", + )); + } + if implementation == compiled || implementation == current { + return Ok(()); + } + + let registration = include_str!("m20261008_000200_rooted_qualified_family.sql") + .replace("$CORE_SCHEMA$", &c) + .replace("$CORE_LITERAL$", &literal(&core)) + .replace("$IMPLEMENTATION_SHA$", &hex::encode(¤t)); + connection + .execute_unprepared(&trusted_function( + ®istration, + "mst2_route_family_registration_guard", + )?) + .await?; + + let expected_shape = template_shapes(connection, &core, core_oid, ¤t) + .await? + .canonical; + + if let Some(family) = &family { + let q = identifier(&family.schema); + connection.execute_unprepared(&format!("LOCK TABLE {q}.mst2_metadata_reader_operation,{q}.mst2_metadata_root_anchor IN ACCESS EXCLUSIVE MODE")).await?; + let rendered = if repair_descriptor { + render_family( + &core, + core_oid, + &family.schema, + family.oid, + &family.namespace, + &family.storage, + ) + } else { + render_prior_family( + &core, + core_oid, + &family.schema, + family.oid, + &family.namespace, + &family.storage, + NATIVE_RUNTIME_IMPLEMENTATION, + )? + }; + let mut functions = String::new(); + let replacements: &[&str] = if repair_descriptor { + &[ + "mst2_metadata_descriptor", + "mst2_metadata_dml_barrier", + "mst2_metadata_gc_enabled", + ] + } else if implementation == previous { + &[ + "mst2_metadata_decode_rooted_plan", + "mst2_metadata_dml_barrier", + "mst2_metadata_gc_enabled", + "mst2_metadata_root_anchor_guard", + "mst2_metadata_serving_complete", + "mst2_metadata_next_reader_issuance", + "mst2_metadata_reader_issuance_guard", + "mst2_metadata_reader_guard", + "mst2_metadata_prune_readers", + "mst2_metadata_begin_reader", + "mst2_metadata_finish_reader", + "mst2_metadata_read_source_entries", + "mst2_metadata_cleanup_expired", + "mst2_metadata_gc_owner_cleanup", + ] + } else { + &[ + "mst2_metadata_decode_rooted_plan", + "mst2_metadata_dml_barrier", + "mst2_metadata_gc_enabled", + ] + }; + for name in replacements { + functions.push_str(&trusted_function(&rendered, name)?); + functions.push('\n'); + } + if implementation == previous { + let upgrade = include_str!("m20261008_000400_reader_retention_upgrade.sql") + .replace("$Q_SCHEMA$", &q) + .replace("$READER_FUNCTIONS_SQL$", &functions); + connection.execute_unprepared(&upgrade).await?; + } else { + connection + .execute_unprepared(&format!("SET LOCAL search_path={q},pg_catalog,pg_temp")) + .await?; + connection.execute_unprepared(&functions).await?; + } + if shape(connection, family.oid, &family.namespace, &family.storage).await? + != expected_shape + { + return Err(rejected( + "reader migration upgraded Q differs from the trusted complete shape", + )); + } + } + + // Fingerprints include trigger identities/enabled state, but not table + // values. Preserve trigger OIDs while writing only these controlled stamps. + let authority = catalog(connection, core_oid, 0, q_oid).await?; + let full = if q_oid == 0 { + None + } else { + Some(catalog(connection, core_oid, q_oid, 0).await?) + }; + connection.execute_unprepared(&format!("ALTER TABLE {c}.mst2_qualified_family_policy DISABLE TRIGGER mst2_route_family_policy_immutable")).await?; + connection.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres,format!( + "UPDATE {c}.mst2_qualified_family_policy SET implementation_fingerprint=$1,expected_shape=$2,authority_catalog=$3 WHERE singleton=1"), + [current.clone().into(),expected_shape.clone().into(),authority.clone().into()])).await?; + connection.execute_unprepared(&format!("ALTER TABLE {c}.mst2_qualified_family_policy ENABLE TRIGGER mst2_route_family_policy_immutable")).await?; + if let (Some(family), Some(full)) = (&family, &full) { + let q = identifier(&family.schema); + connection.execute_unprepared(&format!( + "ALTER TABLE {q}.mst2_metadata_family_identity DISABLE TRIGGER mst2_00_family_barrier; + ALTER TABLE {q}.mst2_metadata_family_identity DISABLE TRIGGER mst2_metadata_identity_immutable; + ALTER TABLE {c}.mst2_metadata_namespace DISABLE TRIGGER mst2_00_route_statement_barrier; + ALTER TABLE {c}.mst2_metadata_namespace DISABLE TRIGGER mst2_route_immutable")).await?; + connection.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres, + format!("UPDATE {q}.mst2_metadata_family_identity SET implementation_fingerprint=$1 WHERE singleton=1"),[current.clone().into()])).await?; + connection.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres,format!( + "UPDATE {c}.mst2_metadata_namespace SET implementation_fingerprint=$1,catalog_fingerprint=$2 WHERE namespace_uuid=$3::uuid"), + [current.clone().into(),full.clone().into(),family.namespace.clone().into()])).await?; + connection.execute_unprepared(&format!( + "ALTER TABLE {q}.mst2_metadata_family_identity ENABLE TRIGGER mst2_00_family_barrier; + ALTER TABLE {q}.mst2_metadata_family_identity ENABLE TRIGGER mst2_metadata_identity_immutable; + ALTER TABLE {c}.mst2_metadata_namespace ENABLE TRIGGER mst2_00_route_statement_barrier; + ALTER TABLE {c}.mst2_metadata_namespace ENABLE TRIGGER mst2_route_immutable")).await?; + } + connection + .execute_unprepared(&format!("SET LOCAL search_path={c},pg_catalog,pg_temp")) + .await?; + if catalog(connection, core_oid, 0, q_oid).await? != authority { + return Err(rejected( + "reader migration final core authority catalog changed", + )); + } + if let (Some(family), Some(full)) = (&family, &full) + && (catalog(connection, core_oid, q_oid, 0).await? != *full + || shape(connection, family.oid, &family.namespace, &family.storage).await? + != expected_shape) + { + return Err(rejected("reader migration final Q fingerprint changed")); + } + Ok(()) +} diff --git a/src/jupiter/service/git_service.rs b/src/jupiter/service/git_service.rs index 17cf70f4..8652cc97 100644 --- a/src/jupiter/service/git_service.rs +++ b/src/jupiter/service/git_service.rs @@ -107,6 +107,44 @@ impl GitService { Ok(data) } + pub async fn get_object_stream(&self, hash: &str) -> Result { + Ok(self.get_object_stream_with_meta(hash).await?.0) + } + + pub async fn get_object_stream_with_meta( + &self, + hash: &str, + ) -> Result<(ObjectByteStream, ObjectMeta), MegaError> { + if !is_full_hex_object_id(hash) { + return Err(MegaError::Other("Invalid object ID format".to_string())); + } + let key = ObjectKey { + namespace: ObjectNamespace::Git, + key: hash.to_string(), + }; + Ok(self.obj_storage.inner.get_stream(&key).await?) + } + + pub async fn get_object_range_stream_exact( + &self, + hash: &str, + start: u64, + end: u64, + ) -> Result, MegaError> { + if !is_full_hex_object_id(hash) { + return Err(MegaError::Other("Invalid object ID format".to_string())); + } + let key = ObjectKey { + namespace: ObjectNamespace::Git, + key: hash.to_string(), + }; + Ok(self + .obj_storage + .inner + .get_range_stream_exact(&key, start, end) + .await?) + } + pub fn get_objects_stream(&self, hashes: Vec) -> MultiObjectByteStream<'_> { // Filter out obviously invalid object ids early to avoid spurious backend requests. // Callers that need strict validation should validate up-front and return 4xx. diff --git a/src/jupiter/service/mono_service.rs b/src/jupiter/service/mono_service.rs index 8574a734..a94668a9 100644 --- a/src/jupiter/service/mono_service.rs +++ b/src/jupiter/service/mono_service.rs @@ -3,7 +3,7 @@ use std::sync::Arc; use futures::{StreamExt, stream}; use git_internal::internal::{ metadata::{EntryMeta, MetaAttached}, - object::blob::Blob, + object::{ObjectTrait, blob::Blob}, pack::entry::Entry, }; use sea_orm::{ActiveModelTrait, ConnectionTrait, IntoActiveModel, TransactionTrait}; @@ -19,7 +19,9 @@ use crate::{ base_storage::{BaseStorage, StorageConnector}, mono_storage::MonoStorage, }, - utils::converter::{IntoMegaModel, MegaModelConverter, MegaObjectModel, process_entry}, + utils::converter::{ + BootstrapCommitTime, IntoMegaModel, MegaModelConverter, MegaObjectModel, process_entry, + }, }, }; @@ -84,7 +86,7 @@ impl MonoService { pub async fn init_monorepo(&self, mono_config: &MonoConfig) -> Result<(), MegaError> { mono_config.ensure_normal_service_object_format()?; - self.initialize_monorepo(mono_config).await + self.initialize_monorepo(mono_config, None).await } /// Initializes an empty Monorepo for a controlled, one-shot bootstrap. @@ -97,12 +99,69 @@ impl MonoService { mono_config: &MonoConfig, ) -> Result<(), MegaError> { mono_config.object_hash_kind()?; - self.initialize_monorepo(mono_config).await + self.initialize_monorepo(mono_config, None).await + } + + pub(crate) async fn bootstrap_monorepo_with_commit_time( + &self, + mono_config: &MonoConfig, + commit_time: Option, + ) -> Result<(), MegaError> { + mono_config.object_hash_kind()?; + self.initialize_monorepo(mono_config, commit_time).await } - async fn initialize_monorepo(&self, mono_config: &MonoConfig) -> Result<(), MegaError> { + async fn ensure_existing_root_matches_expected( + &self, + txn: &sea_orm::DatabaseTransaction, + root_ref: &crate::callisto::mega_refs::Model, + expected: &MegaModelConverter, + ) -> Result<(), MegaError> { + let mismatch = || { + MegaError::Other( + "existing Monorepo root does not match the configured initial graph and commit time; refusing to modify it" + .to_string(), + ) + }; + if root_ref.ref_commit_hash != expected.commit.id.to_string() + || root_ref.ref_tree_hash != expected.root_tree.id.to_string() + { + return Err(mismatch()); + } + let commits = self + .mono_storage + .get_commits_by_hashes_fallible(txn, std::slice::from_ref(&root_ref.ref_commit_hash)) + .await?; + let commit = commits.first().ok_or_else(mismatch)?; + let tree = self + .mono_storage + .get_tree_by_hash_in_txn(&root_ref.ref_tree_hash, txn) + .await? + .ok_or_else(mismatch)?; + if commit.tree != expected.commit.tree_id.to_string() + || commit.parents_id != serde_json::json!([]) + || commit.author.as_deref().map(str::as_bytes) + != Some(expected.commit.author.to_data()?.as_slice()) + || commit.committer.as_deref().map(str::as_bytes) + != Some(expected.commit.committer.to_data()?.as_slice()) + || commit.content.as_deref() != Some(expected.commit.message.as_str()) + || tree.sub_trees != expected.root_tree.to_data()? + { + return Err(mismatch()); + } + Ok(()) + } + + async fn initialize_monorepo( + &self, + mono_config: &MonoConfig, + commit_time: Option, + ) -> Result<(), MegaError> { let txn = self.mono_storage.get_connection().begin().await?; acquire_monorepo_initialization_lock(&txn).await?; + let expected = commit_time + .map(|time| MegaModelConverter::init_with_commit_time(mono_config, Some(time))) + .transpose()?; if let Some(root_ref) = self.mono_storage.get_main_ref_in_txn("/", &txn).await? { ensure_existing_root_ref_matches_config( @@ -110,11 +169,18 @@ impl MonoService { &root_ref.ref_commit_hash, &root_ref.ref_tree_hash, )?; + if let Some(expected) = &expected { + self.ensure_existing_root_matches_expected(&txn, &root_ref, expected) + .await?; + } txn.commit().await?; tracing::info!("Monorepo Directory Already Inited, skip init process!"); return Ok(()); } - let converter = MegaModelConverter::init(mono_config)?; + let converter = match expected { + Some(expected) => expected, + None => MegaModelConverter::init(mono_config)?, + }; let commit = converter .commit .into_mega_model(EntryMeta::default()) diff --git a/src/jupiter/service/native_publication_push_tests.rs b/src/jupiter/service/native_publication_push_tests.rs new file mode 100644 index 00000000..eb746d1c --- /dev/null +++ b/src/jupiter/service/native_publication_push_tests.rs @@ -0,0 +1,429 @@ +use bytes::Bytes; + +use crate::callisto::{ + mst2_native_head, mst2_native_publication, mst2_publication, mst2_publication_outbox, +}; +use sea_orm::{ColumnTrait, ConnectionTrait, EntityTrait, PaginatorTrait, QueryFilter, Statement}; +use crate::jupiter::utils::converter::FromMegaModel; + +const NATIVE_INSTANCE: &str = "6ab219b0-4275-45ba-9d7b-7b0b633018cd"; + +async fn native_fixture() -> ( + tempfile::TempDir, + crate::jupiter::storage::Storage, + git_internal::internal::object::commit::Commit, + String, +) { + let temp = tempfile::tempdir().unwrap(); + let mut config = crate::config::testing::isolated_config(temp.path().join("config")); + config.monorepo.push_policy = PushPolicy::Trunk; + config.mst2.enabled = true; + config.mst2.publication_enabled = true; + config.mst2.instance_uuid = Some(NATIVE_INSTANCE.to_owned()); + let (database, schema) = crate::jupiter::tests::test_db_config(temp.path()).await; + config.database = database; + let mut connection = crate::jupiter::storage::init::database_connection(&config.database) + .await + .unwrap(); + connection.set_metric_callback(move |_| { + let _held_by_the_connection = &schema; + }); + let mut storage = crate::jupiter::storage::Storage::new_with_connection( + Arc::new(config), + Arc::new(connection), + crate::jupiter::storage::object_storage::mock_object_storage(), + ) + .await + .unwrap(); + storage.push_queue_service = storage + .push_queue_service + .with_timeouts(Duration::from_secs(30), Duration::from_millis(20)); + let storage = crate::jupiter::tests::with_test_vault(storage, temp.path()).await; + let name = format!("native-{}", uuid::Uuid::new_v4().simple()); + use git_internal::internal::object::{ + commit::Commit, + tree::{Tree, TreeItem, TreeItemMode}, + }; + let keep_oid = storage.git_service.save_object_from_raw(Bytes::from_static(b"keep")).await.unwrap(); + let old_oid = storage.git_service.save_object_from_raw(Bytes::from_static(b"old")).await.unwrap(); + let child = Tree::from_tree_items(vec![wh03_blob_item("x.txt", &old_oid)]).unwrap(); + let root_tree = Tree::from_tree_items(vec![ + wh03_blob_item(".gitkeep", &keep_oid), + TreeItem::new(TreeItemMode::Tree, child.id, name.clone()), + ]).unwrap(); + let root_commit = Commit::from_tree_id(root_tree.id, vec![], "root"); + let tip = Commit::from_tree_id(child.id, vec![], "path tip"); + let mono = storage.mono_storage(); + mono.save_mega_trees(vec![child.clone(), root_tree.clone()], root_commit.id, None).await.unwrap(); + mono.save_mega_commits(vec![root_commit.clone(), tip.clone()], None).await.unwrap(); + mono.save_refs(mega_refs::Model::new( + "/", MEGA_BRANCH_NAME.to_owned(), root_commit.id.to_string(), root_tree.id.to_string(), false, + ), None).await.unwrap(); + let path = format!("/{name}"); + mono.save_refs(mega_refs::Model::new( + path.clone(), MEGA_BRANCH_NAME.to_owned(), tip.id.to_string(), child.id.to_string(), false, + ), None).await.unwrap(); + storage.push_queue_service.push_queue_storage.set_control_flags(Some(true), None, None).await.unwrap(); + mono.initialize_native_publication_for_maintenance( + NATIVE_INSTANCE, + &crate::jupiter::storage::native_publication_storage::NativeRoot { + commit: root_commit.id.to_string(), tree: root_tree.id.to_string(), + }, + ).await.unwrap(); + assert!(mono.read_native_publication_head(NATIVE_INSTANCE).await.is_err()); + storage.push_queue_service.push_queue_storage.set_control_flags(Some(false), None, None).await.unwrap(); + (temp, storage, tip, path) +} + +async fn save_same_tree_commit(storage: &crate::jupiter::storage::Storage, parent: &git_internal::internal::object::commit::Commit) -> (String, PushPayload) { + let commit = git_internal::internal::object::commit::Commit::from_tree_id(parent.tree_id, vec![parent.id], "same native tree n1"); + storage.mono_storage().save_mega_commits(vec![commit.clone()], None).await.unwrap(); + let oid = commit.id.to_string(); + (oid.clone(), PushPayload { commits: vec![oid], fork_base: Some(parent.id.to_string()), n: 1 }) +} + +#[tokio::test] +async fn real_same_tree_push_advances_one_global_certificate_and_n0_replay_stays_fixed() { + let (_temp, storage, tip, path) = native_fixture().await; + let mono = storage.mono_storage(); + let before = mono.get_main_ref("/").await.unwrap().unwrap(); + let (new, payload) = save_same_tree_commit(&storage, &tip).await; + let id = wh03_enqueue_push(&storage, &path, &tip.id.to_string(), &new, &payload).await; + assert!(matches!(wh03_exec(&storage,id).await, ExecuteOutcome::Done { root_cas_writes:1,.. })); + let after = mono.get_main_ref("/").await.unwrap().unwrap(); + assert_eq!((after.ref_commit_hash,after.ref_tree_hash),(before.ref_commit_hash.clone(),before.ref_tree_hash.clone())); + let head = mono.read_native_publication_head(NATIVE_INSTANCE).await.unwrap(); + assert_eq!(head.token.sequence,1); + let receipt = mst2_publication::Entity::find().filter(mst2_publication::Column::OperationId.eq(format!("mst2:trunk-queue:{id}"))) + .one(mono.get_connection()).await.unwrap().unwrap(); + assert_eq!(receipt.namespace,path); + assert_eq!(receipt.new_oid,new); + assert_eq!(receipt.native_certificate_version,Some(1)); + assert_ne!(receipt.new_oid,head.root.commit); + let noop = PushPayload { commits:vec![],fork_base:None,n:0 }; + let noop_id = wh03_enqueue_push(&storage,&path,&new,&new,&noop).await; + assert!(matches!(wh03_exec(&storage,noop_id).await, ExecuteOutcome::Done { root_cas_writes:1,.. })); + assert_eq!(mono.read_native_publication_head(NATIVE_INSTANCE).await.unwrap().token,head.token); + mono.get_connection().execute_raw(Statement::from_sql_and_values(mono.get_connection().get_database_backend(), + "UPDATE push_queue SET status='Running',landed_commit_id=NULL WHERE id=$1",[id.into()], + )).await.unwrap(); + assert!(matches!(wh03_exec(&storage,id).await, ExecuteOutcome::Done { root_cas_writes:0,.. })); + assert_eq!(mono.read_native_publication_head(NATIVE_INSTANCE).await.unwrap().token,head.token); + assert_eq!(mst2_native_publication::Entity::find().count(mono.get_connection()).await.unwrap(),1); +} + +#[tokio::test] +async fn maintenance_cannot_reinitialize_published_history_after_head_loss() { + let (_temp, storage, tip, path) = native_fixture().await; + let mono = storage.mono_storage(); + let (new, payload) = save_same_tree_commit(&storage, &tip).await; + let id = wh03_enqueue_push(&storage, &path, &tip.id.to_string(), &new, &payload).await; + assert!(matches!(wh03_exec(&storage, id).await, ExecuteOutcome::Done { .. })); + let head = mono.read_native_publication_head(NATIVE_INSTANCE).await.unwrap(); + let certificates = mst2_native_publication::Entity::find().all(mono.get_connection()).await.unwrap(); + let receipts = mst2_publication::Entity::find().all(mono.get_connection()).await.unwrap(); + let outbox = mst2_publication_outbox::Entity::find().all(mono.get_connection()).await.unwrap(); + assert_eq!(certificates.len(), 1); + storage.push_queue_service.push_queue_storage.set_control_flags(Some(true), None, None).await.unwrap(); + mono.get_connection().execute_unprepared("DELETE FROM mst2_native_head").await.unwrap(); + for instance in [NATIVE_INSTANCE, "11111111-2222-4333-8444-555555555555"] { + let error = mono.initialize_native_publication_for_maintenance(instance, &head.root).await.unwrap_err(); + assert!(error.to_string().contains("native publication history exists"), "rejected initialization for {instance}: {error}"); + let released = mono.get_connection().begin().await.unwrap(); + assert!(PushQueueStorage::try_mono_write_lock(&released).await.unwrap(), "rejected initialization must release the mono write lock before returning"); + released.rollback().await.unwrap(); + } + assert_eq!(mst2_native_head::Entity::find().count(mono.get_connection()).await.unwrap(), 0); + assert_eq!(mst2_native_publication::Entity::find().all(mono.get_connection()).await.unwrap(), certificates); + assert_eq!(mst2_publication::Entity::find().all(mono.get_connection()).await.unwrap(), receipts); + assert_eq!(mst2_publication_outbox::Entity::find().all(mono.get_connection()).await.unwrap(), outbox); + assert!(storage.push_queue_service.push_queue_storage.get_control().await.unwrap().paused); + mono.get_connection().execute_unprepared("DELETE FROM mst2_native_publication").await.unwrap(); + for instance in [NATIVE_INSTANCE, "11111111-2222-4333-8444-555555555555"] { + let error = mono.initialize_native_publication_for_maintenance(instance, &head.root).await.unwrap_err(); + assert!(error.to_string().contains("native publication history exists"), "rejected initialization for {instance}: {error}"); + let released = mono.get_connection().begin().await.unwrap(); + assert!(PushQueueStorage::try_mono_write_lock(&released).await.unwrap(), "rejected initialization must release the mono write lock before returning"); + released.rollback().await.unwrap(); + } + assert_eq!(mst2_native_head::Entity::find().count(mono.get_connection()).await.unwrap(), 0); + assert_eq!(mst2_native_publication::Entity::find().count(mono.get_connection()).await.unwrap(), 0); + assert_eq!(mst2_publication::Entity::find().all(mono.get_connection()).await.unwrap(), receipts); + assert_eq!(mst2_publication_outbox::Entity::find().all(mono.get_connection()).await.unwrap(), outbox); + assert!(storage.push_queue_service.push_queue_storage.get_control().await.unwrap().paused); + let root = mono.get_main_ref("/").await.unwrap().unwrap(); + assert_eq!((root.ref_commit_hash, root.ref_tree_hash), (head.root.commit, head.root.tree)); +} + +#[tokio::test] +async fn real_merge_fails_closed_when_native_publication_is_enabled() { + let (_temp, storage, tip, path) = native_fixture().await; + let mono = storage.mono_storage(); + let root_before = mono.get_main_ref("/").await.unwrap().unwrap(); + let head_before = mst2_native_head::Entity::find_by_id("/") + .one(mono.get_connection()) + .await + .unwrap() + .unwrap(); + + let cl_link = format!("CL-NATIVE-GUARD-{}", uuid::Uuid::new_v4().simple()); + let EnqueueOutcome::Inserted { id } = storage + .push_queue_service + .enqueue(EnqueueRequest { + kind: PushQueueKindEnum::Merge, + operation_id: merge_operation_id(&cl_link), + path: path.clone(), + old_id: tip.id.to_string(), + new_id: tip.id.to_string(), + requester: Some("native-guard-test".to_owned()), + payload: serde_json::json!({ + "cl_link": cl_link, + "authz_principal": "native-guard-test", + "execution_actor": "native-guard-test" + }), + ref_name: None, + is_delete: false, + }) + .await + .unwrap() + else { + panic!("merge enqueue did not insert"); + }; + assert_eq!( + storage + .push_queue_service + .storage() + .claim_for_execution(id) + .await + .unwrap(), + ClaimOutcome::Claimed + ); + + let merge_ctx = MergeExecContext { + storage: storage.clone(), + git_object_cache: Arc::new(crate::ceres::api_service::cache::GitObjectCache { + connection: crate::jupiter::tests::test_redis_manager().await, + prefix: String::new(), + }), + abort_before_cl_status: false, + pause_after_apply: Duration::ZERO, + pause_after_apply_barrier: None, + }; + let outcome = storage + .push_queue_service + .execute_b3( + ExecuteRequest { + id, + ..Default::default() + }, + None, + Some(&merge_ctx), + None, + ) + .await + .unwrap(); + assert!(matches!( + outcome, + ExecuteOutcome::Failed { + ref failure, + ref message, + .. + } if failure == "Conflict" + && message == "MST2 native publication does not cover merge writer" + )); + assert_eq!(wh03_row_status(&storage, id).await, PushQueueStatusEnum::Failed); + + let root_after = mono.get_main_ref("/").await.unwrap().unwrap(); + assert_eq!(root_after.ref_commit_hash, root_before.ref_commit_hash); + assert_eq!(root_after.ref_tree_hash, root_before.ref_tree_hash); + let head_after = mst2_native_head::Entity::find_by_id("/") + .one(mono.get_connection()) + .await + .unwrap() + .unwrap(); + assert_eq!(head_after, head_before); + assert_eq!(mst2_publication::Entity::find().count(mono.get_connection()).await.unwrap(), 0); + assert_eq!( + mst2_native_publication::Entity::find() + .count(mono.get_connection()) + .await + .unwrap(), + 0 + ); + assert_eq!( + mst2_publication_outbox::Entity::find() + .count(mono.get_connection()) + .await + .unwrap(), + 0 + ); +} + +#[tokio::test] +async fn real_push_rejects_same_root_changed_publication_token_before_any_ref_write() { + let (_temp, storage, tip, path) = native_fixture().await; + let (new,payload)=save_same_tree_commit(&storage,&tip).await; + let id=wh03_enqueue_push(&storage,&path,&tip.id.to_string(),&new,&payload).await; + let mono=storage.mono_storage(); + // A separate writer transaction changes only the publication generation. + mono.get_connection().execute_unprepared("UPDATE mst2_native_head SET sequence=1 WHERE namespace='/'").await.unwrap(); + let outcome=wh03_exec(&storage,id).await; + assert!(matches!(outcome,ExecuteOutcome::Failed { ref failure,ref message,.. } + if failure=="Conflict" && message.contains("publication token is stale"))); + assert_eq!(mono.get_main_ref(&path).await.unwrap().unwrap().ref_commit_hash,tip.id.to_string()); + assert_eq!(mst2_native_publication::Entity::find().count(mono.get_connection()).await.unwrap(),0); + assert_eq!(mst2_publication::Entity::find().count(mono.get_connection()).await.unwrap(),0); +} + +async fn api_state(storage: crate::jupiter::storage::Storage) -> crate::api::MonoApiServiceState { + crate::api::MonoApiServiceState { + entity_store: storage.entity_store.clone(),storage, + session_store:crate::api::oauth::api_store::BrowserSessionStore::Anonymous, + git_object_cache:Arc::new(crate::ceres::api_service::cache::GitObjectCache { + connection:crate::jupiter::tests::test_redis_manager().await,prefix:String::new(), + }),listen_addr:"127.0.0.1:0".to_owned(), + } +} + +async fn http_resolve(state: crate::api::MonoApiServiceState,scope:&str) -> (u16,serde_json::Value) { + use tower::ServiceExt; + let app=crate::api::router::snapshot_router::routers(state.clone()).with_state(state); + let response=app.oneshot(axum::http::Request::builder().method("POST").uri("/snapshots/resolve") + .header("content-type","application/json").body(axum::body::Body::from(serde_json::json!({"target":{"kind":"latest"},"scope":scope}).to_string())).unwrap()).await.unwrap(); + let status=response.status().as_u16(); + let body=axum::body::to_bytes(response.into_body(),1_048_576).await.unwrap(); + (status,serde_json::from_slice(&body).unwrap()) +} + +#[tokio::test] +async fn http_resolve_captured_before_commit_never_mixes_new_sequence_with_old_descriptor() { + let (_temp,storage,tip,path)=native_fixture().await; + #[cfg(unix)] + let writer=crate::ceres::snapshot::projection_writer::ProjectionObservationSink::start(_temp.path()).unwrap(); + #[cfg(unix)] + let storage={ let mut storage=storage; storage.projection_observation_sink=Some(writer.clone()); storage }; + let (first,payload)=save_same_tree_commit(&storage,&tip).await; + let id=wh03_enqueue_push(&storage,&path,&tip.id.to_string(),&first,&payload).await; + assert!(matches!(wh03_exec(&storage,id).await,ExecuteOutcome::Done{..})); + let state=api_state(storage.clone()).await; + let ((status,v1),first_observations)=crate::ceres::snapshot::projection_observation::with_observations(http_resolve(state.clone(),"/")).await; + assert_eq!(status,200,"{v1}"); + assert_eq!(first_observations.len(),1); + let first_identity=first_observations[0].test_identity(); + assert_eq!(first_identity["snapshot_id"],v1["descriptor"]["snapshot_id"]); + assert_eq!(first_identity["metadata_root"],v1["descriptor"]["metadata_root"]); + assert_eq!(first_identity["namespace_view_id"],v1["descriptor"]["namespace_view_id"]); + assert_eq!(first_identity["instance_id"],v1["descriptor"]["instance_id"]); + assert_eq!(first_identity["native_publication_sequence"],v1["publication_sequence"]); + assert_eq!(first_identity["native_writer_epoch"],v1["writer_epoch"]); + assert!(first_identity["native_certificate_receipt_id"].as_u64().unwrap()>0); + assert!(!first_identity["request_id"].as_str().unwrap().is_empty()); + let (status,subscope)=http_resolve(state.clone(),&path).await; + assert_eq!(status,200,"{subscope}"); + assert_eq!(v1["publication_sequence"],subscope["publication_sequence"]); + assert_eq!(v1["writer_epoch"],subscope["writer_epoch"]); + let root_v1=storage.mono_storage().get_main_ref("/").await.unwrap().unwrap(); + assert_eq!(first_identity["root_commit_oid"],git_internal::hash::ObjectHash::from_hex_for_kind(git_internal::hash::get_hash_kind(),&root_v1.ref_commit_hash).unwrap().to_tagged_string()); + assert_eq!(first_identity["root_tree_oid"],git_internal::hash::ObjectHash::from_hex_for_kind(git_internal::hash::get_hash_kind(),&root_v1.ref_tree_hash).unwrap().to_tagged_string()); + let native_v1=storage.mono_storage().read_native_publication_head(storage.config().mst2.instance_uuid.as_deref().unwrap()).await.unwrap(); + assert_eq!(first_identity["native_certificate_receipt_id"].as_u64(),Some(native_v1.token.certificate.unwrap() as u64)); + let captured=Arc::new(tokio::sync::Barrier::new(2)); + let release=Arc::new(tokio::sync::Barrier::new(2)); + let task_state=state.clone();let task_captured=captured.clone();let task_release=release.clone(); + let mut reader=tokio::spawn(async move { + crate::ceres::snapshot::projection_observation::with_observations(crate::api::router::snapshot_router::with_native_resolve_barriers(task_captured,task_release,http_resolve(task_state,"/"))).await + }); + tokio::time::timeout(Duration::from_secs(5),captured.wait()).await.unwrap(); + let first_commit=storage.mono_storage().get_commit_by_hash(&first).await.unwrap().unwrap(); + let first_commit=git_internal::internal::object::commit::Commit::from_mega_model(first_commit); + let next_blob = storage.git_service.save_object_from_raw(Bytes::from_static(b"next projection")).await.unwrap(); + let (next,next_payload)=wh03_save_n1_commit(&storage,first_commit.id,&next_blob,"next projection").await; + let id=wh03_enqueue_push(&storage,&path,&first,&next,&next_payload).await; + assert!(matches!(wh03_exec(&storage,id).await,ExecuteOutcome::Done{..})); + tokio::time::timeout(Duration::from_secs(5),release.wait()).await.unwrap(); + let ((status,delayed),delayed_observations)=tokio::time::timeout(Duration::from_secs(10),&mut reader).await.unwrap().unwrap(); + assert_eq!(status,503,"{delayed}"); + assert_eq!(delayed["error"]["code"],"SNAPSHOT_NOT_READY"); + assert_eq!(delayed["error"]["retryable"],true); + assert!(delayed.get("descriptor").is_none()); + assert!(delayed.get("publication_sequence").is_none()); + assert!(delayed_observations.is_empty(),"rejected stale handoff emitted a success observation"); + use tower::ServiceExt; + let old_app=crate::api::router::snapshot_router::routers(state.clone()).with_state(state.clone()); + let old=old_app.oneshot(axum::http::Request::builder().method("GET") + .uri(format!("/snapshots/{}/descriptor",v1["descriptor"]["snapshot_id"].as_str().unwrap())) + .header("x-mega-snapshot-lease",v1["lease_id"].as_str().unwrap()) + .body(axum::body::Body::empty()).unwrap()).await.unwrap(); + assert_eq!(old.status(),200); + let old=axum::body::to_bytes(old.into_body(),1_048_576).await.unwrap(); + let old:serde_json::Value=serde_json::from_slice(&old).unwrap(); + assert_eq!(old["descriptor"],v1["descriptor"],"already protected old SID changed with the writer"); + let ((status,v2),next_observations)=crate::ceres::snapshot::projection_observation::with_observations(http_resolve(state.clone(),"/")).await; + assert_eq!(status,200,"{v2}");assert_eq!(v2["publication_sequence"],"2"); + assert_ne!(v2["descriptor"]["snapshot_id"],v1["descriptor"]["snapshot_id"]); + assert_eq!(next_observations.len(),1); + let next_identity=next_observations[0].test_identity(); + assert_eq!(next_identity["native_publication_sequence"],v2["publication_sequence"]); + assert_eq!(next_identity["snapshot_id"],v2["descriptor"]["snapshot_id"]); + assert_eq!(next_identity["metadata_root"],v2["descriptor"]["metadata_root"]); + assert_eq!(next_identity["namespace_view_id"],v2["descriptor"]["namespace_view_id"]); + assert_eq!(next_identity["instance_id"],v2["descriptor"]["instance_id"]); + assert_eq!(next_identity["native_writer_epoch"],v2["writer_epoch"]); + let native_v2=storage.mono_storage().read_native_publication_head(storage.config().mst2.instance_uuid.as_deref().unwrap()).await.unwrap(); + assert_eq!(next_identity["root_commit_oid"],git_internal::hash::ObjectHash::from_hex_for_kind(git_internal::hash::get_hash_kind(),&native_v2.root.commit).unwrap().to_tagged_string()); + assert_eq!(next_identity["root_tree_oid"],git_internal::hash::ObjectHash::from_hex_for_kind(git_internal::hash::get_hash_kind(),&native_v2.root.tree).unwrap().to_tagged_string()); + assert_eq!(next_identity["native_certificate_receipt_id"].as_u64(),Some(native_v2.token.certificate.unwrap() as u64)); + assert_ne!(next_identity["native_certificate_receipt_id"],first_identity["native_certificate_receipt_id"]); + let ((status,_),failed_observations)=crate::ceres::snapshot::projection_observation::with_observations(http_resolve(state,"/absent-observation-scope")).await; + assert_eq!(status,404); + assert!(failed_observations.is_empty(),"failed projection emitted success observation"); + #[cfg(unix)] + { + writer.shutdown(std::time::Instant::now()+Duration::from_secs(5)).await.unwrap(); + let directory=writer.test_directory(_temp.path()); + let records=std::fs::read_to_string(directory.join("records.jsonl")).unwrap(); + let records:Vec=records.lines().map(|line|serde_json::from_str(line).unwrap()).collect(); + assert_eq!(records.len(),3,"failed resolve must not be a successful observation"); + for (record,observation) in [(&records[0],&first_observations[0]),(&records[2],&next_observations[0])] { + let typed=serde_json::to_value(observation.wire_record()).unwrap(); + assert_eq!(record["payload"],typed,"writer must preserve the full validated operation tuple"); + assert_eq!(record["payload"].as_object().unwrap().len(),38); + } + assert_ne!(records[0]["payload"]["request_id"],records[2]["payload"]["request_id"]); + let status:serde_json::Value=serde_json::from_slice(&std::fs::read(directory.join("status.json")).unwrap()).unwrap(); + assert_eq!(status["closed"],true); + assert_eq!(status["accepted_records"],3); + assert_eq!(status["written_records"],3); + assert_eq!(status["first_error_code"],0); + } +} + +#[cfg(unix)] +#[tokio::test] +async fn rejected_native_observation_source_does_not_change_ready_resolve_and_is_a_sticky_writer_failure() { + let (temp,mut storage,tip,path)=native_fixture().await; + let (first,payload)=save_same_tree_commit(&storage,&tip).await; + let id=wh03_enqueue_push(&storage,&path,&tip.id.to_string(),&first,&payload).await; + assert!(matches!(wh03_exec(&storage,id).await,ExecuteOutcome::Done{..})); + let head=storage.mono_storage().read_native_publication_head(NATIVE_INSTANCE).await.unwrap(); + assert_eq!(head.token.sequence,1); + assert!(head.token.certificate.unwrap()>0); + let writer=crate::ceres::snapshot::projection_writer::ProjectionObservationSink::start(temp.path()).unwrap(); + storage.projection_observation_sink=Some(writer.clone()); + let state=api_state(storage).await; + // Corrupt only the observation's captured certificate. The actual native + // head is READY; INITIALIZING heads remain correctly rejected with 503. + let ((status,response),observations)=crate::ceres::snapshot::projection_observation::with_observations( + crate::api::router::snapshot_router::with_rejected_native_observation_source(http_resolve(state,"/")) + ).await; + assert_eq!(status,200,"{response}"); + assert_eq!(response["publication_sequence"],"1"); + assert!(observations.is_empty()); + assert!(writer.shutdown(std::time::Instant::now()+Duration::from_secs(5)).await.is_err()); + let directory=writer.test_directory(temp.path()); + assert!(std::fs::read(directory.join("records.jsonl")).unwrap().is_empty()); + let status:serde_json::Value=serde_json::from_slice(&std::fs::read(directory.join("status.json")).unwrap()).unwrap(); + assert_eq!(status["first_error_code"],crate::ceres::snapshot::projection_writer::WriterFailure::ObservationBindingRejected as u8); + assert_eq!(status["accepted_records"],0); + assert_eq!(status["closed"],true); +} diff --git a/src/jupiter/service/push_queue_service.rs b/src/jupiter/service/push_queue_service.rs index 61dedbce..643cc859 100644 --- a/src/jupiter/service/push_queue_service.rs +++ b/src/jupiter/service/push_queue_service.rs @@ -31,6 +31,11 @@ use crate::{ blob_path_index::BlobPathIndexMode, git_db_storage::GitDbStorage, mono_storage::MonoStorage, + mst2_publication_storage::{ + PreparedPublication, PublicationPreparation, PublicationReceiptError, + PublicationRequest, + }, + native_publication_storage::PreparedNativePublication, push_queue_storage::{ ClaimOutcome, EnqueueOutcome, EnqueueParams, EnqueueRejectReason, PushQueueStorage, }, @@ -419,7 +424,7 @@ pub struct PushExecContext { pub storage: crate::jupiter::storage::Storage, pub git_object_cache: std::sync::Arc, /// Test-only: two-phase sync inside the B3 txn right before - /// `apply_push_in_txn` (after the baseline read): the executor signals + /// `apply_push_in_txn` or the n=0 root CAS (after the baseline read): the executor signals /// "race window open" on the enter barrier, then blocks on the release /// barrier until the test's queue-bypassing writer has committed — a /// deterministic apply-time root CAS miss (WH-03). @@ -428,6 +433,12 @@ pub struct PushExecContext { pub pre_apply_release_barrier: Option>, } +struct PushRefBaseline<'a> { + cur_commit: Option<&'a str>, + cur_tree: Option<&'a str>, + root: Option<&'a crate::callisto::mega_refs::Model>, +} + /// Context required to execute a claimed `kind=merge` round under B3. pub struct MergeExecContext { pub storage: crate::jupiter::storage::Storage, @@ -697,6 +708,11 @@ impl PushQueueService { self } + pub(crate) fn with_native_publication(mut self, enabled: bool) -> Self { + self.push_queue_storage = self.push_queue_storage.with_native_publication(enabled); + self + } + pub fn with_max_push_commits(mut self, max_push_commits: usize) -> Self { self.max_push_commits = max_push_commits; self @@ -1277,8 +1293,56 @@ impl PushQueueService { return Ok(ExecuteOutcome::HardStopped { id: req.id }); } + // Receipt replay precedes baseline/ref writes: the root may already + // have advanced since a committed operation lost its response. + let publication = if let Some(ctx) = push_ctx + && ctx.storage.config().mst2.publication_enabled + && row.kind != PushQueueKindEnum::Attach + && !(row.kind == PushQueueKindEnum::Merge && merge_ctx.is_some()) + { + let request = match PublicationRequest::from_trunk_queue(&row) { + Ok(request) => request, + Err(error) => { + return self + .b3_fail_merge(txn, req.id, "Conflict", error.to_string()) + .await; + } + }; + let preparation = match self + .mono_storage + .begin_publication_in_txn(&txn, request) + .await + { + Ok(preparation) => preparation, + Err(PublicationReceiptError::Database(error)) => return Err(error.into()), + Err(error) => { + return self + .b3_fail_merge(txn, req.id, "Conflict", error.to_string()) + .await; + } + }; + match preparation { + PublicationPreparation::Prepared(prepared) => Some(prepared), + PublicationPreparation::AlreadyCommitted(committed) => { + return self + .b3_replay_committed_operation(txn, req.id, committed.receipt.new_oid) + .await; + } + PublicationPreparation::AlreadyCommittedNoop(receipt) => { + return self + .b3_replay_committed_operation(txn, req.id, receipt.landed_commit_id) + .await; + } + } + } else { + None + }; + // Expected-root baseline compare (NULL-safe). - let root = self.mono_storage.get_main_ref_in_txn("/", &txn).await?; + let root = self + .mono_storage + .get_native_main_ref_in_txn("/", &txn) + .await?; let (cur_commit, cur_tree) = match &root { Some(r) => ( Some(r.ref_commit_hash.as_str()), @@ -1309,6 +1373,55 @@ impl PushQueueService { return Ok(ExecuteOutcome::BypassDetected { id: req.id }); } + // Native publication currently has no merge projection. Refuse the + // real merge writer before it can mutate refs, commits, or CL state; + // keeping this gate ahead of dispatch makes publication fail closed + // while the writer matrix is still incomplete. + if row.kind == PushQueueKindEnum::Merge + && merge_ctx.is_some_and(|ctx| ctx.storage.config().mst2.publication_enabled) + { + return self + .b3_fail_merge( + txn, + req.id, + "Conflict", + "MST2 native publication does not cover merge writer".into(), + ) + .await; + } + + let native_publication = if publication.is_some() && row.kind == PushQueueKindEnum::Push { + let ctx = push_ctx.ok_or_else(|| { + MegaError::Other("native publication requires the real push writer".into()) + })?; + let config = ctx.storage.config(); + let Some(instance) = config.mst2.instance_uuid.as_deref() else { + return self + .b3_fail_merge( + txn, + req.id, + "Conflict", + "native publication instance missing".into(), + ) + .await; + }; + match self + .mono_storage + .reserve_native_publication_in_txn(&txn, &row, instance) + .await + { + Ok(prepared) => Some(prepared), + Err(PublicationReceiptError::Database(error)) => return Err(error.into()), + Err(error) => { + return self + .b3_fail_merge(txn, req.id, "Conflict", error.to_string()) + .await; + } + } + } else { + None + }; + if req.force_conflict() { return self .b4_conflict_requeue_holding_lock(txn, req.id, &row) @@ -1359,7 +1472,18 @@ impl PushQueueService { } if let Some(ctx) = push_ctx { return self - .b3_execute_push(txn, &row, ctx, cur_commit, cur_tree, root.as_ref()) + .b3_execute_push( + txn, + &row, + ctx, + PushRefBaseline { + cur_commit, + cur_tree, + root: root.as_ref(), + }, + publication, + native_publication, + ) .await; } tracing::trace!(policy = ?self.push_policy, id = req.id, "B3 push stub"); @@ -1462,26 +1586,16 @@ impl PushQueueService { return Ok(self.outcome_claim_lost(req.id)); } - // T05 (spec 09 §1): record the publication receipt, outbox event and - // per-namespace sequence in the *same* transaction as the root CAS - // above, so the visible change, its sequence and its outbox commit - // atomically. Gated off by default (`mst2.publication_enabled`), so - // an unconfigured deployment keeps the exact prior behavior. - if let Some(ctx) = push_ctx - && ctx.storage.config().mst2.publication_enabled - { - let old_oid = cur_commit.unwrap_or(ZERO_ID); - let namespace = self.mono_storage.normalize_namespace(&row.path); + if let Some(prepared) = publication { self.mono_storage .record_publication_in_txn( &txn, - &row.operation_id, - &namespace, - old_oid, + prepared, + cur_commit.unwrap_or(ZERO_ID), &new_commit, - "trunk_push", ) - .await?; + .await + .map_err(|error| MegaError::Other(error.to_string()))?; } PushQueueStorage::notify_mono_write_queue(&txn).await?; @@ -1494,6 +1608,25 @@ impl PushQueueService { }) } + async fn b3_replay_committed_operation( + &self, + txn: sea_orm::DatabaseTransaction, + id: i64, + landed_commit_id: String, + ) -> Result { + if !PushQueueStorage::mark_done_if_running_in_txn(&txn, id, &landed_commit_id).await? { + txn.rollback().await?; + return Ok(self.outcome_claim_lost(id)); + } + PushQueueStorage::notify_mono_write_queue(&txn).await?; + txn.commit().await?; + Ok(ExecuteOutcome::Done { + id, + landed_commit_id, + root_cas_writes: 0, + }) + } + /// TP-08: attach kind under held `MONO_WRITE_LOCK` — sole root write is /// `attach_to_monorepo_parent_in_txn` (no generic CAS stub, no retry loop). /// @@ -2321,13 +2454,13 @@ impl PushQueueService { txn: sea_orm::DatabaseTransaction, row: &push_queue::Model, ctx: &PushExecContext, - cur_commit: Option<&str>, - cur_tree: Option<&str>, - root: Option<&crate::callisto::mega_refs::Model>, + baseline: PushRefBaseline<'_>, + publication: Option, + native_publication: Option, ) -> Result { let id = row.id; match self - .b3_execute_push_inner(txn, row, ctx, cur_commit, cur_tree, root) + .b3_execute_push_inner(txn, row, ctx, baseline, publication, native_publication) .await { Ok(outcome) => Ok(outcome), @@ -2370,10 +2503,15 @@ impl PushQueueService { txn: sea_orm::DatabaseTransaction, row: &push_queue::Model, ctx: &PushExecContext, - cur_commit: Option<&str>, - cur_tree: Option<&str>, - root: Option<&crate::callisto::mega_refs::Model>, + baseline: PushRefBaseline<'_>, + publication: Option, + native_publication: Option, ) -> Result { + let PushRefBaseline { + cur_commit, + cur_tree, + root, + } = baseline; use std::path::PathBuf; use git_internal::{ @@ -2423,7 +2561,7 @@ impl PushQueueService { let path_row = self .mono_storage - .get_main_ref_in_txn(&row.path, &txn) + .get_native_main_ref_in_txn(&row.path, &txn) .await?; if path_row.is_none() { @@ -2494,6 +2632,12 @@ impl PushQueueService { ) .await; } + if let Some(barrier) = &ctx.pre_apply_enter_barrier { + barrier.wait().await; + } + if let Some(barrier) = &ctx.pre_apply_release_barrier { + barrier.wait().await; + } PushQueueStorage::savepoint(&txn, "b3_kind").await?; let cas_ok = self .mono_storage @@ -2530,6 +2674,24 @@ impl PushQueueService { txn.rollback().await?; return Ok(self.outcome_claim_lost(row.id)); } + if let Some(prepared) = native_publication { + self.mono_storage + .finish_native_noop_in_txn(&txn, prepared) + .await + .map_err(|error| MegaError::Other(error.to_string()))?; + } + if let Some(prepared) = publication { + self.mono_storage + .record_noop_operation_in_txn( + &txn, + prepared, + &root.ref_commit_hash, + &root.ref_tree_hash, + &pref.ref_commit_hash, + ) + .await + .map_err(|error| MegaError::Other(error.to_string()))?; + } PushQueueStorage::notify_mono_write_queue(&txn).await?; txn.commit().await?; self.run_c_segment_index(row.id, &row.path).await; @@ -2799,26 +2961,23 @@ impl PushQueueService { return Ok(self.outcome_claim_lost(row.id)); } - // T05 (spec 09 §1/§7): record the publication receipt, outbox event - // and per-namespace sequence in the *same* transaction as the ref CAS - // above, so the visible change, its sequence and its outbox commit - // atomically (trunk push = the primary default-namespace writer of - // spec 09 §5). Gated off by default (`mst2.publication_enabled`): - // a deployment that has not passed the §9 shadow comparison keeps - // the exact prior behavior. Replaying the same operation id lands - // the receipt once (PUB-11). - if ctx.storage.config().mst2.publication_enabled { - let namespace = self.mono_storage.normalize_namespace(&normalized); - self.mono_storage + if let Some(prepared) = publication { + let committed = self + .mono_storage .record_publication_in_txn( &txn, - &row.operation_id, - &namespace, + prepared, cur_commit.unwrap_or(ZERO_ID), &landed_commit_id, - "trunk_push", ) - .await?; + .await + .map_err(|error| MegaError::Other(error.to_string()))?; + let native = native_publication + .ok_or_else(|| MegaError::Other("native push reservation missing".into()))?; + self.mono_storage + .record_native_publication_in_txn(&txn, native, &committed) + .await + .map_err(|error| MegaError::Other(error.to_string()))?; } PushQueueStorage::notify_mono_write_queue(&txn).await?; @@ -3398,10 +3557,10 @@ fn reject_to_error(reason: EnqueueRejectReason) -> MegaError { #[cfg(test)] mod tests { - use sea_orm::ConnectionTrait; use serde_json::json; use super::*; + include!("native_publication_push_tests.rs"); use crate::{ callisto::{mega_refs, sea_orm_active_enums::PushQueueFailureEnum}, jupiter::{ @@ -4830,6 +4989,503 @@ mod tests { .unwrap() } + #[tokio::test] + async fn mst2_receipt_real_push_replays_before_root_baseline_and_rejects_changed_request() { + use sea_orm::{ + ColumnTrait, ConnectionTrait, EntityTrait, PaginatorTrait, QueryFilter, Statement, + }; + + use crate::callisto::{mst2_publication, mst2_publication_outbox, mst2_queue_noop_receipt}; + + let temp = tempfile::tempdir().unwrap(); + let mut config = crate::config::testing::isolated_config(temp.path().join("config")); + config.monorepo.push_policy = PushPolicy::Trunk; + config.mst2.publication_enabled = true; + config.mst2.instance_uuid = Some("6ab219b0-4275-45ba-9d7b-7b0b633018cd".to_owned()); + let storage = crate::jupiter::tests::test_storage_with_config(temp.path(), config).await; + let storage = crate::jupiter::tests::with_test_vault(storage, temp.path()).await; + let (old_commit, path) = wh03_path_fixture(&storage, "receipt").await; + storage + .mono_storage() + .initialize_native_publication("6ab219b0-4275-45ba-9d7b-7b0b633018cd") + .await + .unwrap(); + let old_id = old_commit.id.to_string(); + let (new_id, payload) = wh03_save_n1_commit( + &storage, + old_commit.id, + "eeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeee", + "receipt new", + ) + .await; + let id = wh03_enqueue_push(&storage, &path, &old_id, &new_id, &payload).await; + assert!(matches!( + wh03_exec(&storage, id).await, + ExecuteOutcome::Done { + root_cas_writes: 1, + .. + } + )); + let mono = storage.mono_storage(); + let root_after = mono.get_main_ref("/").await.unwrap().unwrap(); + let operation_id = format!("mst2:trunk-queue:{id}"); + let receipt = mst2_publication::Entity::find() + .filter(mst2_publication::Column::OperationId.eq(&operation_id)) + .one(mono.get_connection()) + .await + .unwrap() + .unwrap(); + assert_eq!(receipt.request_digest_version, Some(1)); + assert!(receipt.request_digest.is_some()); + assert_eq!(receipt.new_oid, new_id); + assert_eq!( + mst2_publication_outbox::Entity::find() + .count(mono.get_connection()) + .await + .unwrap(), + 1 + ); + + let retry = |requester: Option, payload: JsonValue| EnqueueRequest { + kind: PushQueueKindEnum::Push, + operation_id: push_operation_id(&old_id, &new_id), + path: path.clone(), + old_id: old_id.clone(), + new_id: new_id.clone(), + requester, + payload, + ref_name: Some(MEGA_BRANCH_NAME.to_owned()), + is_delete: false, + }; + assert!( + matches!(storage.push_queue_service.enqueue(retry(None, payload.to_json())).await.unwrap(), + EnqueueOutcome::Replay { id: replay_id, .. } if replay_id == id) + ); + let error = storage + .push_queue_service + .enqueue(retry(Some("different-actor".to_owned()), payload.to_json())) + .await + .unwrap_err(); + assert!(error.to_string().contains("MST2_PUBLICATION_CONFLICT")); + let mut different = payload.to_json(); + different["fork_base"] = serde_json::json!("different-base"); + let error = storage + .push_queue_service + .enqueue(retry(None, different)) + .await + .unwrap_err(); + assert!(error.to_string().contains("MST2_PUBLICATION_CONFLICT")); + + // Model recovery of a Running queue row after COMMIT lost its response. + // Original claim baselines remain old, so a late replay would fail them. + mono.get_connection() + .execute_raw(Statement::from_sql_and_values( + mono.get_connection().get_database_backend(), + "UPDATE push_queue SET status = 'Running', landed_commit_id = NULL WHERE id = $1", + [id.into()], + )) + .await + .unwrap(); + assert_eq!( + wh03_exec(&storage, id).await, + ExecuteOutcome::Done { + id, + landed_commit_id: new_id.clone(), + root_cas_writes: 0, + } + ); + let root = mono.get_main_ref("/").await.unwrap().unwrap(); + assert_eq!(root.ref_commit_hash, root_after.ref_commit_hash); + assert_eq!(root.ref_tree_hash, root_after.ref_tree_hash); + let check_txn = mono.get_connection().begin().await.unwrap(); + assert!( + !PushQueueStorage::is_hard_stopped_in_txn(&check_txn) + .await + .unwrap() + ); + check_txn.rollback().await.unwrap(); + assert_eq!(mono.publication_sequence(&path).await.unwrap(), 1); + assert_eq!( + mst2_publication::Entity::find() + .count(mono.get_connection()) + .await + .unwrap(), + 1 + ); + + mono.get_connection().execute_raw(Statement::from_sql_and_values( + mono.get_connection().get_database_backend(), + "UPDATE push_queue SET status = 'Running', requester = 'changed-actor' WHERE id = $1", + [id.into()], + )).await.unwrap(); + assert!( + matches!(wh03_exec(&storage, id).await, ExecuteOutcome::Failed { ref failure, ref message, .. } + if failure == "Conflict" && message.contains("MST2_PUBLICATION_CONFLICT")) + ); + assert_eq!( + mono.get_main_ref("/") + .await + .unwrap() + .unwrap() + .ref_commit_hash, + root_after.ref_commit_hash + ); + assert_eq!(mono.publication_sequence(&path).await.unwrap(), 1); + assert_eq!( + mst2_publication_outbox::Entity::find() + .count(mono.get_connection()) + .await + .unwrap(), + 1 + ); + + // Distinct internal no-op operations do not create publications. + let n0_payload = PushPayload { + commits: Vec::new(), + fork_base: None, + n: 0, + }; + let mut noop_ids = Vec::new(); + for round in 0..3 { + let operation_id = format!("internal-noop-{round}"); + let outcome = storage + .push_queue_service + .enqueue(EnqueueRequest { + kind: PushQueueKindEnum::Push, + operation_id: operation_id.clone(), + path: path.clone(), + old_id: new_id.clone(), + new_id: new_id.clone(), + requester: None, + payload: n0_payload.to_json(), + ref_name: Some(MEGA_BRANCH_NAME.to_owned()), + is_delete: false, + }) + .await + .unwrap(); + let EnqueueOutcome::Inserted { id: n0_id } = outcome else { + panic!("{outcome:?}"); + }; + assert_eq!( + storage + .push_queue_service + .storage() + .claim_for_execution(n0_id) + .await + .unwrap(), + ClaimOutcome::Claimed + ); + assert!(matches!( + wh03_exec(&storage, n0_id).await, + ExecuteOutcome::Done { + root_cas_writes: 1, + .. + } + )); + let noop = mst2_queue_noop_receipt::Entity::find() + .filter( + mst2_queue_noop_receipt::Column::OperationId + .eq(format!("mst2:trunk-queue:{n0_id}")), + ) + .one(mono.get_connection()) + .await + .unwrap() + .unwrap(); + assert_eq!(noop.observed_sequence, 1); + assert_eq!(noop.observed_root_commit, root_after.ref_commit_hash); + assert_eq!(noop.observed_root_tree, root_after.ref_tree_hash); + assert_eq!(noop.landed_commit_id, new_id); + assert_eq!(mono.publication_sequence(&path).await.unwrap(), 1); + assert_eq!( + mst2_publication::Entity::find() + .count(mono.get_connection()) + .await + .unwrap(), + 1 + ); + assert_eq!( + mst2_publication_outbox::Entity::find() + .count(mono.get_connection()) + .await + .unwrap(), + 1 + ); + assert!(matches!(storage.push_queue_service.enqueue(EnqueueRequest { + kind: PushQueueKindEnum::Push, operation_id: operation_id.clone(), + path: path.clone(), old_id: new_id.clone(), new_id: new_id.clone(), + requester: None, payload: n0_payload.to_json(), + ref_name: Some(MEGA_BRANCH_NAME.to_owned()), is_delete: false, + }).await.unwrap(), EnqueueOutcome::Replay { id: replay_id, .. } if replay_id == n0_id)); + let error = storage + .push_queue_service + .enqueue(EnqueueRequest { + kind: PushQueueKindEnum::Push, + operation_id: operation_id.clone(), + path: path.clone(), + old_id: new_id.clone(), + new_id: new_id.clone(), + requester: Some("changed-actor".to_owned()), + payload: n0_payload.to_json(), + ref_name: Some(MEGA_BRANCH_NAME.to_owned()), + is_delete: false, + }) + .await + .unwrap_err(); + assert!(error.to_string().contains("MST2_PUBLICATION_CONFLICT")); + let mut changed_payload = n0_payload.to_json(); + changed_payload["fork_base"] = serde_json::json!("changed-intent"); + let error = storage + .push_queue_service + .enqueue(EnqueueRequest { + kind: PushQueueKindEnum::Push, + operation_id, + path: path.clone(), + old_id: new_id.clone(), + new_id: new_id.clone(), + requester: None, + payload: changed_payload, + ref_name: Some(MEGA_BRANCH_NAME.to_owned()), + is_delete: false, + }) + .await + .unwrap_err(); + assert!(error.to_string().contains("MST2_PUBLICATION_CONFLICT")); + noop_ids.push(n0_id); + } + let (later_id, later_payload) = wh03_save_n1_commit( + &storage, + new_id.parse().unwrap(), + "ffffffffffffffffffffffffffffffffffffffff", + "after no-op", + ) + .await; + let later_queue_id = + wh03_enqueue_push(&storage, &path, &new_id, &later_id, &later_payload).await; + assert!(matches!( + wh03_exec(&storage, later_queue_id).await, + ExecuteOutcome::Done { + root_cas_writes: 1, + .. + } + )); + let later_root = mono.get_main_ref("/").await.unwrap().unwrap(); + assert_ne!(later_root.ref_commit_hash, root_after.ref_commit_hash); + for &n0_id in &noop_ids { + mono.get_connection().execute_raw(Statement::from_sql_and_values( + mono.get_connection().get_database_backend(), + "UPDATE push_queue SET status = 'Running', landed_commit_id = NULL WHERE id = $1", + [n0_id.into()], + )).await.unwrap(); + assert_eq!( + wh03_exec(&storage, n0_id).await, + ExecuteOutcome::Done { + id: n0_id, + landed_commit_id: new_id.clone(), + root_cas_writes: 0, + } + ); + } + let root = mono.get_main_ref("/").await.unwrap().unwrap(); + assert_eq!(root.ref_commit_hash, later_root.ref_commit_hash); + assert_eq!(root.ref_tree_hash, later_root.ref_tree_hash); + assert_eq!(mono.publication_sequence(&path).await.unwrap(), 2); + assert_eq!( + mst2_publication::Entity::find() + .count(mono.get_connection()) + .await + .unwrap(), + 2 + ); + assert_eq!( + mst2_publication_outbox::Entity::find() + .count(mono.get_connection()) + .await + .unwrap(), + 2 + ); + assert_eq!( + mst2_queue_noop_receipt::Entity::find() + .count(mono.get_connection()) + .await + .unwrap(), + 3 + ); + let n0_id = noop_ids[0]; + mono.get_connection().execute_raw(Statement::from_sql_and_values( + mono.get_connection().get_database_backend(), + "UPDATE push_queue SET status = 'Running', requester = 'changed-actor' WHERE id = $1", + [n0_id.into()], + )).await.unwrap(); + assert!( + matches!(wh03_exec(&storage, n0_id).await, ExecuteOutcome::Failed { ref failure, ref message, .. } + if failure == "Conflict" && message.contains("MST2_PUBLICATION_CONFLICT")) + ); + assert_eq!( + mono.get_main_ref("/") + .await + .unwrap() + .unwrap() + .ref_commit_hash, + later_root.ref_commit_hash + ); + assert_eq!(mono.publication_sequence(&path).await.unwrap(), 2); + assert_eq!( + mst2_queue_noop_receipt::Entity::find() + .count(mono.get_connection()) + .await + .unwrap(), + 3 + ); + } + + #[tokio::test] + async fn mst2_noop_real_push_checks_epoch_baseline_and_net_zero_cas_fences() { + use sea_orm::{ConnectionTrait, EntityTrait, PaginatorTrait, Statement}; + + use crate::callisto::{mst2_publication, mst2_publication_outbox, mst2_queue_noop_receipt}; + + for fence in ["epoch", "baseline", "cas"] { + let temp = tempfile::tempdir().unwrap(); + let mut config = crate::config::testing::isolated_config(temp.path().join("config")); + config.monorepo.push_policy = PushPolicy::Trunk; + config.mst2.publication_enabled = true; + config.mst2.instance_uuid = Some("6ab219b0-4275-45ba-9d7b-7b0b633018cd".to_owned()); + let storage = + crate::jupiter::tests::test_storage_with_config(temp.path(), config).await; + let storage = crate::jupiter::tests::with_test_vault(storage, temp.path()).await; + let (tip, path) = wh03_path_fixture(&storage, &format!("noop-{fence}")).await; + storage + .mono_storage() + .initialize_native_publication("6ab219b0-4275-45ba-9d7b-7b0b633018cd") + .await + .unwrap(); + let tip_id = tip.id.to_string(); + let payload = PushPayload { + commits: Vec::new(), + fork_base: None, + n: 0, + }; + let id = wh03_enqueue_push(&storage, &path, &tip_id, &tip_id, &payload).await; + let mono = storage.mono_storage(); + let root_before = mono.get_main_ref("/").await.unwrap().unwrap(); + let outcome = if fence == "epoch" { + mono.get_connection().execute_raw(Statement::from_sql_and_values( + mono.get_connection().get_database_backend(), + "INSERT INTO mst2_namespace_seq (namespace, sequence, epoch) VALUES ($1, 0, 2)", + [path.clone().into()], + )).await.unwrap(); + wh03_exec(&storage, id).await + } else if fence == "baseline" { + mono.get_connection() + .execute_raw(Statement::from_sql_and_values( + mono.get_connection().get_database_backend(), + "UPDATE push_queue SET expected_commit_hash = $1 WHERE id = $2", + ["f".repeat(40).into(), id.into()], + )) + .await + .unwrap(); + wh03_exec(&storage, id).await + } else { + let enter = Arc::new(tokio::sync::Barrier::new(2)); + let release = Arc::new(tokio::sync::Barrier::new(2)); + let ctx = PushExecContext { + storage: storage.clone(), + git_object_cache: Arc::new(crate::ceres::api_service::cache::GitObjectCache { + connection: crate::jupiter::tests::test_redis_manager().await, + prefix: String::new(), + }), + pre_apply_enter_barrier: Some(Arc::clone(&enter)), + pre_apply_release_barrier: Some(Arc::clone(&release)), + }; + let exec_storage = storage.clone(); + let mut worker = tokio::spawn(async move { + exec_storage + .push_queue_service + .execute_b3( + ExecuteRequest { + id, + ..Default::default() + }, + None, + None, + Some(&ctx), + ) + .await + }); + let handshake = async { + tokio::time::timeout(Duration::from_secs(5), enter.wait()).await.map_err(|_| "no-op did not reach CAS".to_owned())?; + tokio::time::timeout(Duration::from_secs(5), mono.get_connection().execute_raw(Statement::from_sql_and_values( + mono.get_connection().get_database_backend(), + "UPDATE mega_refs SET ref_commit_hash = $1 WHERE path = '/' AND ref_name = $2", + ["f".repeat(40).into(), MEGA_BRANCH_NAME.into()], + ))).await.map_err(|_| "bypass writer timed out".to_owned())?.map_err(|error| error.to_string())?; + tokio::time::timeout(Duration::from_secs(5), release.wait()).await.map_err(|_| "no-op CAS release timed out".to_owned()) + }.await; + if let Err(error) = handshake { + worker.abort(); + let _ = tokio::time::timeout(Duration::from_secs(5), &mut worker).await; + panic!("{error}"); + } + match tokio::time::timeout(Duration::from_secs(10), &mut worker).await { + Ok(result) => result.unwrap().unwrap(), + Err(_) => { + worker.abort(); + let _ = tokio::time::timeout(Duration::from_secs(5), &mut worker).await; + panic!("no-op CAS worker did not finish"); + } + } + }; + if fence == "epoch" { + assert!( + matches!(outcome, ExecuteOutcome::Failed { ref failure, ref message, .. } + if failure == "Conflict" && message.contains("writer epoch is fenced")) + ); + } else { + assert_eq!(outcome, ExecuteOutcome::BypassDetected { id }); + } + let root = mono.get_main_ref("/").await.unwrap().unwrap(); + assert_eq!( + root.ref_commit_hash, + if fence == "cas" { + "f".repeat(40) + } else { + root_before.ref_commit_hash + } + ); + assert_eq!(root.ref_tree_hash, root_before.ref_tree_hash); + assert_eq!(mono.publication_sequence(&path).await.unwrap(), 0); + assert_eq!( + mst2_publication::Entity::find() + .count(mono.get_connection()) + .await + .unwrap(), + 0 + ); + assert_eq!( + mst2_publication_outbox::Entity::find() + .count(mono.get_connection()) + .await + .unwrap(), + 0 + ); + assert_eq!( + mst2_queue_noop_receipt::Entity::find() + .count(mono.get_connection()) + .await + .unwrap(), + 0 + ); + let txn = mono.get_connection().begin().await.unwrap(); + assert_eq!( + PushQueueStorage::is_hard_stopped_in_txn(&txn) + .await + .unwrap(), + fence != "epoch" + ); + txn.rollback().await.unwrap(); + } + } + async fn assert_view_signal(signal: &ViewSignal, context: &str) { tokio::time::timeout(Duration::from_secs(1), signal.notified()) .await diff --git a/src/jupiter/storage/init.rs b/src/jupiter/storage/init.rs index d8480141..47044027 100644 --- a/src/jupiter/storage/init.rs +++ b/src/jupiter/storage/init.rs @@ -10,6 +10,21 @@ use crate::{ jupiter::migration::{apply_migrations, ensure_queue_control_seed}, }; +#[cfg(test)] +tokio::task_local! { + static GENERIC_HISTORY_BOOTSTRAP: (); +} + +#[cfg(test)] +pub(crate) async fn with_generic_history_bootstrap(future: F) -> F::Output { + GENERIC_HISTORY_BOOTSTRAP.scope((), future).await +} + +#[cfg(test)] +pub(super) fn generic_history_bootstrap_active() -> bool { + GENERIC_HISTORY_BOOTSTRAP.try_with(|_| ()).is_ok() +} + /// Create a PostgreSQL database connection. /// /// After a successful connection, applies any pending database migrations and @@ -25,6 +40,11 @@ pub async fn database_connection(db_config: &DbConfig) -> Result, + /// Derived native projection memoization, scoped to this storage assembly. + /// Clones share it; independent databases/backends never share entries. + pub(crate) native_projection_cache: Arc, + pub(crate) native_chunk_maps: + Arc>, + pub(crate) native_snapshot_sessions: + Arc>, + #[cfg(test)] + pub(crate) shadow_qualified_metadata: + Arc>, + pub(crate) rooted_qualified_metadata: Arc< + tokio::sync::OnceCell>, + >, + pub(crate) projection_observation_sink: + Option>, pub cl_service: CLService, pub push_queue_service: PushQueueService, pub artifact_service: ArtifactService, @@ -179,6 +201,46 @@ pub struct Storage { } impl Storage { + pub(crate) async fn rooted_qualified_metadata_writer( + &self, + ) -> Result<&qualified_metadata_family::RootedQualifiedMetadataRepository, MegaError> { + self.rooted_qualified_metadata + .get_or_try_init(|| async { + let repository = Arc::new( + qualified_metadata_family::RootedQualifiedMetadataRepository::open( + self.mono_storage().get_connection(), + &self.config().database, + ) + .await?, + ); + repository + .maintenance_tick(64) + .await + .map_err(|error| MegaError::Other(error.to_string()))?; + qualified_metadata_family::RootedQualifiedMetadataRepository::start_maintenance( + &repository, + ); + Ok::<_, MegaError>(repository) + }) + .await + .map(Arc::as_ref) + } + + #[cfg(test)] + pub(crate) async fn shadow_qualified_metadata_writer( + &self, + ) -> Result<&qualified_metadata_family::ShadowQualifiedMetadataWriter, MegaError> { + self.shadow_qualified_metadata + .get_or_try_init(|| async { + qualified_metadata_family::ShadowQualifiedMetadataWriter::open( + self.mono_storage().get_connection(), + &self.config().database, + ) + .await + }) + .await + } + pub async fn new( config: Arc, object_store: MegaObjectStorageWrapper, @@ -224,7 +286,8 @@ impl Storage { }; let commit_binding_storage = CommitBindingStorage { base: base.clone() }; - let push_queue_storage = PushQueueStorage::new(base.clone()); + let push_queue_storage = PushQueueStorage::new(base.clone()) + .with_native_publication(config.mst2.publication_enabled); let buck_storage = BuckStorage { base: base.clone() }; let bots_storage = BotsStorage { base: base.clone() }; @@ -291,7 +354,8 @@ impl Storage { let push_queue_service = PushQueueService::new(base.clone(), config.monorepo.push_policy.clone()) .with_view_signal(view_runtime.signal()) - .with_max_push_commits(config.monorepo.max_push_commits); + .with_max_push_commits(config.monorepo.max_push_commits) + .with_native_publication(config.mst2.publication_enabled); let artifact_service = ArtifactService::new(base.clone(), object_store.clone()); let buck_service = BuckService::new( base.clone(), @@ -308,8 +372,15 @@ impl Storage { let storage_event_emitter = crate::jupiter::service::storage_event_emitter::StorageEventEmitter::from_config_disabled(&config); - Ok(Storage { + let storage = Storage { app_service: app_service.into(), + native_projection_cache: Arc::default(), + native_snapshot_sessions: Arc::default(), + native_chunk_maps: Arc::default(), + #[cfg(test)] + shadow_qualified_metadata: Arc::default(), + rooted_qualified_metadata: Arc::default(), + projection_observation_sink: None, config_handle, config, cl_service: CLService::new(base.clone()), @@ -328,7 +399,13 @@ impl Storage { view_runtime, entity_store: Arc::new(SharedEntityStore::default()), vault: None, - }) + }; + #[cfg(test)] + if init::generic_history_bootstrap_active() { + return Ok(storage); + } + storage.rooted_qualified_metadata_writer().await?; + Ok(storage) } pub fn config_handle(&self) -> ConfigHandle { @@ -685,6 +762,13 @@ impl Storage { Storage { app_service, + native_projection_cache: Arc::default(), + native_snapshot_sessions: Arc::default(), + native_chunk_maps: Arc::default(), + #[cfg(test)] + shadow_qualified_metadata: Arc::default(), + rooted_qualified_metadata: Arc::default(), + projection_observation_sink: None, // app_service: AppService::mock(), cl_service: CLService::mock(), push_queue_service: PushQueueService::new( diff --git a/src/jupiter/storage/mono_storage.rs b/src/jupiter/storage/mono_storage.rs index ff2fb954..5228c6fc 100644 --- a/src/jupiter/storage/mono_storage.rs +++ b/src/jupiter/storage/mono_storage.rs @@ -26,7 +26,7 @@ use sea_orm::{ use crate::{ callisto::{ mega_blob, mega_cl, mega_commit, mega_ref_tombstones, mega_refs, mega_tag, mega_tree, - mst2_publication, mst2_publication_outbox, mst2_verified_object, + mst2_verified_object, }, common::{ errors::MegaError, @@ -274,6 +274,24 @@ impl MonoStorage { Ok(result) } + pub(crate) async fn get_native_main_ref_in_txn( + &self, + path: &str, + txn: &DatabaseTransaction, + ) -> Result, MegaError> { + let rows = mega_refs::Entity::find() + .filter(mega_refs::Column::Path.eq(path)) + .filter(mega_refs::Column::RefName.eq(MEGA_BRANCH_NAME)) + .filter(mega_refs::Column::IsCl.eq(false)) + .all(txn) + .await?; + match rows.len() { + 0 => Ok(None), + 1 => Ok(rows.into_iter().next()), + _ => Err(MegaError::Other("selected native ref is ambiguous".into())), + } + } + pub async fn get_main_ref_in_txn( &self, path: &str, @@ -1362,6 +1380,7 @@ impl MonoStorage { updated_at = $3 WHERE path = '/' AND ref_name = $4 + AND is_cl = false AND ref_commit_hash IS NOT DISTINCT FROM $5 AND ref_tree_hash IS NOT DISTINCT FROM $6 "#, @@ -1770,12 +1789,43 @@ impl MonoStorage { if oids.is_empty() { return Ok(HashMap::new()); } - let rows = mst2_verified_object::Entity::find() - .filter(mst2_verified_object::Column::StorageDomain.eq("git")) - .filter(mst2_verified_object::Column::ObjectKind.eq("blob")) - .filter(mst2_verified_object::Column::GitOid.is_in(oids)) - .all(self.get_connection()) - .await?; + let connection = self.get_connection(); + let rows = if connection.get_database_backend() == sea_orm::DbBackend::Postgres { + use sea_orm::{DbBackend, FromQueryResult, Statement}; + let scope = connection.query_one_raw(Statement::from_string(DbBackend::Postgres, + "SELECT n.nspname AS schema FROM pg_catalog.pg_namespace n JOIN pg_catalog.pg_class c ON c.relnamespace=n.oid AND c.relname='mst2_verified_object' AND c.relkind='r' WHERE n.nspname=pg_catalog.current_schema() AND n.nspname NOT LIKE 'pg_temp_%'")) + .await?.ok_or_else(|| MegaError::Other("actual verified blob relation is missing".into()))?; + let schema: String = scope.try_get("", "schema")?; + let relation = format!("\"{}\".mst2_verified_object", schema.replace('"', "\"\"")); + let oids = serde_json::to_string(&oids) + .map_err(|error| MegaError::Other(error.to_string()))?; + let sql = format!( + "SELECT v.id,v.storage_domain,CASE WHEN pg_catalog.octet_length(v.git_oid) IN (40,64) THEN v.git_oid ELSE NULL END AS git_oid,v.object_kind,CASE WHEN pg_catalog.octet_length(v.raw_sha256)=32 THEN v.raw_sha256 ELSE NULL END AS raw_sha256,CASE WHEN v.size BETWEEN 0 AND {MST2_MAX_FILE_SIZE} THEN v.size ELSE NULL END AS size,CASE WHEN v.verification_version IN (1,{MST2_VERIFICATION_VERSION}) THEN v.verification_version ELSE NULL END AS verification_version,CASE WHEN v.state='VERIFIED' THEN v.state ELSE NULL END AS state,v.created_at FROM {relation} v WHERE v.storage_domain='git' AND v.object_kind='blob' AND v.git_oid IN (SELECT value FROM pg_catalog.jsonb_array_elements_text($1::jsonb))" + ); + connection + .query_all_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + sql, + [oids.into()], + )) + .await? + .iter() + .map(|row| { + mst2_verified_object::Model::from_query_result(row, "").map_err(|_| { + MegaError::ObjStorageInconsistent( + "invalid bounded MST/2 verified blob record".into(), + ) + }) + }) + .collect::, _>>()? + } else { + mst2_verified_object::Entity::find() + .filter(mst2_verified_object::Column::StorageDomain.eq("git")) + .filter(mst2_verified_object::Column::ObjectKind.eq("blob")) + .filter(mst2_verified_object::Column::GitOid.is_in(oids)) + .all(connection) + .await? + }; for row in &rows { if row.state != "VERIFIED" || ![1, MST2_VERIFICATION_VERSION].contains(&row.verification_version) @@ -1872,112 +1922,6 @@ impl MonoStorage { } } - /// Record one namespace publication inside the caller's transaction - /// (spec 09 §1/§7): bump the per-namespace sequence, insert the - /// receipt and append the outbox event — atomically with the ref CAS - /// the same transaction performs. - /// - /// Idempotent per (namespace, operation_id): a retried writer gets - /// back the original sequence and neither the counter nor the outbox - /// advances twice (PUB-11). The uniqueness is scoped to the namespace - /// because push operation ids (`old→new`) are only unique within one - /// repo: two namespaces landing the same (old,new) pair are distinct - /// publications, not replays. - pub async fn record_publication_in_txn( - &self, - txn: &DatabaseTransaction, - operation_id: &str, - namespace: &str, - old_oid: &str, - new_oid: &str, - writer_kind: &str, - ) -> Result { - // One SELECT by operation id answers both questions: the same id - // under this namespace is a replay (PUB-11) and returns the original - // sequence; the same id under a different namespace means the id is - // not actually unique to this publication — refuse rather than - // silently double-bind it. - if let Some(existing) = mst2_publication::Entity::find() - .filter(mst2_publication::Column::OperationId.eq(operation_id)) - .one(txn) - .await? - { - if existing.namespace == namespace { - return Ok(existing.sequence); - } - return Err(MegaError::Other(format!( - "operation id {operation_id} already published under a different namespace" - ))); - } - // Upsert the sequence row and bump it atomically; the unique - // namespace key makes this a single-winner counter. - let backend = txn.get_database_backend(); - let bump = sea_orm::Statement::from_sql_and_values( - backend, - r#"INSERT INTO mst2_namespace_seq ("namespace", "sequence", "epoch") - VALUES ($1, 1, 1) - ON CONFLICT ("namespace") DO UPDATE SET "sequence" = "mst2_namespace_seq"."sequence" + 1 - RETURNING "sequence""#, - [namespace.into()], - ); - let row = txn - .query_one_raw(bump) - .await? - .ok_or_else(|| MegaError::Other("mst2_namespace_seq upsert returned no row".into()))?; - let sequence: i64 = row.try_get_by_index(0)?; - let now = chrono::Utc::now().fixed_offset(); - - // Concurrency-safe insert: the receipt is unique per - // (namespace, operation_id). The replay check above covers a - // committed prior publication; a truly concurrent same-operation - // writer loses at the outbox insert's unique operation_id below, - // rolling this transaction back rather than double-publishing. - let insert = mst2_publication::ActiveModel { - id: sea_orm::ActiveValue::NotSet, - operation_id: Set(operation_id.to_string()), - namespace: Set(namespace.to_string()), - sequence: Set(sequence), - old_oid: Set(old_oid.to_string()), - new_oid: Set(new_oid.to_string()), - writer_epoch: Set(1), - writer_kind: Set(writer_kind.to_string()), - created_at: Set(now), - }; - match mst2_publication::Entity::insert(insert) - .on_conflict( - OnConflict::columns([ - mst2_publication::Column::Namespace, - mst2_publication::Column::OperationId, - ]) - .do_nothing() - .to_owned(), - ) - .exec(txn) - .await - { - // RecordNotInserted means a receipt with this identity already - // exists; see the outbox insert below for how a racing writer - // of the same operation is resolved. - Ok(_) | Err(DbErr::RecordNotInserted) => {} - Err(e) => return Err(e.into()), - } - - // The replay check above already returned for a pre-existing receipt, - // so reaching here means this call owns `sequence`; the outbox row - // inserts normally. - mst2_publication_outbox::ActiveModel { - id: sea_orm::ActiveValue::NotSet, - operation_id: Set(operation_id.to_string()), - namespace: Set(namespace.to_string()), - sequence: Set(sequence), - state: Set("PENDING".to_string()), - created_at: Set(now), - } - .insert(txn) - .await?; - Ok(sequence) - } - /// Canonical publication namespace for a repository path: a leading /// slash, no trailing slash, root is `/`. Publication identities are /// keyed on this so `/p/` and `/p` never count as two namespaces. @@ -3491,7 +3435,7 @@ mod tests { // Rollback ⇒ nothing visible: the receipt commits only with the // ref CAS in the same transaction (spec 09 §1). let txn = mono.get_connection().begin().await.unwrap(); - mono.record_publication_in_txn(&txn, "op-roll", "/", "", "aaa", "trunk_push") + mono.record_test_publication_in_txn(&txn, "op-roll", "/", "", "aaa", "trunk_push") .await .unwrap(); txn.rollback().await.unwrap(); @@ -3511,7 +3455,7 @@ mod tests { for (op, old, new) in [("op-1", "", "a1"), ("op-2", "a1", "a2")] { let txn = mono.get_connection().begin().await.unwrap(); let seq = mono - .record_publication_in_txn(&txn, op, "/", old, new, "trunk_push") + .record_test_publication_in_txn(&txn, op, "/", old, new, "trunk_push") .await .unwrap(); txn.commit().await.unwrap(); @@ -3530,7 +3474,7 @@ mod tests { // adds nothing (PUB-11) — the push retry path depends on this. let txn = mono.get_connection().begin().await.unwrap(); let again = mono - .record_publication_in_txn(&txn, "op-1", "/", "", "a1", "trunk_push") + .record_test_publication_in_txn(&txn, "op-1", "/", "", "a1", "trunk_push") .await .unwrap(); txn.commit().await.unwrap(); @@ -3547,7 +3491,7 @@ mod tests { // Distinct namespaces keep independent sequences. let txn = mono.get_connection().begin().await.unwrap(); let other = mono - .record_publication_in_txn(&txn, "op-child", "/child", "", "c1", "trunk_push") + .record_test_publication_in_txn(&txn, "op-child", "/child", "", "c1", "trunk_push") .await .unwrap(); txn.commit().await.unwrap(); @@ -3564,13 +3508,13 @@ mod tests { // replay: refuse rather than silently double-bind the id (the // reviewer's P3.1 — a silent no-op left /b's counter at 0 forever). let txn = mono.get_connection().begin().await.unwrap(); - mono.record_publication_in_txn(&txn, "shared-op", "/", "", "a1", "trunk_push") + mono.record_test_publication_in_txn(&txn, "shared-op", "/", "", "a1", "trunk_push") .await .unwrap(); txn.commit().await.unwrap(); let txn = mono.get_connection().begin().await.unwrap(); let err = mono - .record_publication_in_txn(&txn, "shared-op", "/b", "", "a1", "trunk_push") + .record_test_publication_in_txn(&txn, "shared-op", "/b", "", "a1", "trunk_push") .await; assert!(err.is_err(), "cross-namespace reuse must be refused"); drop(txn); // the failed transaction rolls back on drop @@ -3593,7 +3537,7 @@ mod tests { let m = base.mono_storage(); let txn = m.get_connection().begin().await.unwrap(); let seq = m - .record_publication_in_txn( + .record_test_publication_in_txn( &txn, &format!("op-c{i}"), "/", diff --git a/src/jupiter/storage/mst2_publication_storage.rs b/src/jupiter/storage/mst2_publication_storage.rs new file mode 100644 index 00000000..a7111c44 --- /dev/null +++ b/src/jupiter/storage/mst2_publication_storage.rs @@ -0,0 +1,514 @@ +//! SQL receipt identity for the existing trunk queue adapter. +//! +//! This is not a complete namespace/projection publication implementation. + +use sea_orm::{ + ActiveModelTrait, ActiveValue::Set, ColumnTrait, ConnectionTrait, DatabaseTransaction, DbErr, + EntityTrait, QueryFilter, Statement, +}; +use sha2::{Digest, Sha256}; + +use crate::{ + callisto::{ + mst2_publication, mst2_publication_outbox, mst2_queue_noop_receipt, push_queue, + sea_orm_active_enums::PushQueueKindEnum, + }, + common::utils::MEGA_BRANCH_NAME, + jupiter::storage::mono_storage::MonoStorage, +}; + +const REQUEST_DIGEST_VERSION: i32 = 1; +const WRITER_EPOCH: i64 = 1; + +#[derive(Debug, thiserror::Error)] +pub(crate) enum PublicationReceiptError { + #[error("MST2_PUBLICATION_CONFLICT: {0}")] + Conflict(String), + #[error("MST2_PUBLICATION_LEGACY_RECEIPT: operation {0} has no supported request digest")] + LegacyReceipt(String), + #[error("MST2_PUBLICATION_INTEGRITY_ERROR: {0}")] + Integrity(String), + #[error(transparent)] + Database(#[from] DbErr), +} + +/// Constructed from a server-persisted queue row, never a client digest. +#[derive(Debug, Clone)] +pub(crate) struct PublicationRequest { + operation_id: String, + namespace: String, + writer_kind: String, + request_digest: String, + legacy_operation_id: Option, + is_noop: bool, +} + +impl PublicationRequest { + pub(crate) fn from_trunk_queue( + row: &push_queue::Model, + ) -> Result { + if row.id <= 0 || row.operation_id.is_empty() { + return Err(PublicationReceiptError::Integrity( + "invalid queue operation identity".into(), + )); + } + mst2_codec::descriptor::validate_scope(&row.path) + .map_err(|error| PublicationReceiptError::Integrity(error.to_string()))?; + let writer_kind = match &row.kind { + PushQueueKindEnum::Push => "trunk_push", + PushQueueKindEnum::Merge => "trunk_merge_stub", + PushQueueKindEnum::Attach => { + return Err(PublicationReceiptError::Integrity( + "attach is not covered by this adapter".into(), + )); + } + }; + let operation_id = format!("mst2:trunk-queue:{}", row.id); + let mut hash = request_hasher(&operation_id, &row.path, writer_kind); + hash_field(&mut hash, row.operation_id.as_bytes()); + hash_field(&mut hash, MEGA_BRANCH_NAME.as_bytes()); + hash_field(&mut hash, row.old_id.as_bytes()); + hash_field(&mut hash, row.new_id.as_bytes()); + hash.update([u8::from(row.requester.is_some())]); + if let Some(actor) = &row.requester { + hash_field(&mut hash, actor.as_bytes()); + } + hash_json(&mut hash, &row.payload); + Ok(Self { + operation_id, + namespace: row.path.clone(), + writer_kind: writer_kind.to_string(), + request_digest: format!("sha256:{}", hex::encode(hash.finalize())), + legacy_operation_id: Some(row.operation_id.clone()), + is_noop: row.kind == PushQueueKindEnum::Push + && row.old_id == row.new_id + && row.payload.get("n").and_then(serde_json::Value::as_u64) == Some(0), + }) + } + + #[cfg(test)] + fn for_test( + operation_id: &str, + namespace: &str, + old: &str, + new: &str, + writer_kind: &str, + ) -> Self { + let mut hash = request_hasher(operation_id, namespace, writer_kind); + hash_field(&mut hash, old.as_bytes()); + hash_field(&mut hash, new.as_bytes()); + Self { + operation_id: operation_id.to_owned(), + namespace: namespace.to_owned(), + writer_kind: writer_kind.to_owned(), + request_digest: format!("sha256:{}", hex::encode(hash.finalize())), + legacy_operation_id: None, + is_noop: false, + } + } +} + +fn request_hasher(operation_id: &str, namespace: &str, writer_kind: &str) -> Sha256 { + let mut hash = Sha256::new(); + hash.update(b"mega.mst2.trunk-queue-request\0"); + hash.update(REQUEST_DIGEST_VERSION.to_le_bytes()); + hash.update(WRITER_EPOCH.to_le_bytes()); + hash_field(&mut hash, operation_id.as_bytes()); + hash_field(&mut hash, namespace.as_bytes()); + hash_field(&mut hash, writer_kind.as_bytes()); + hash +} + +fn hash_field(hash: &mut Sha256, bytes: &[u8]) { + hash.update((bytes.len() as u64).to_le_bytes()); + hash.update(bytes); +} + +fn hash_json(hash: &mut Sha256, value: &serde_json::Value) { + use serde_json::Value; + match value { + Value::Null => hash.update([0]), + Value::Bool(value) => hash.update([1, u8::from(*value)]), + Value::Number(value) => { + hash.update([2]); + hash_field(hash, value.to_string().as_bytes()); + } + Value::String(value) => { + hash.update([3]); + hash_field(hash, value.as_bytes()); + } + Value::Array(values) => { + hash.update([4]); + hash.update((values.len() as u64).to_le_bytes()); + for value in values { + hash_json(hash, value); + } + } + Value::Object(values) => { + hash.update([5]); + hash.update((values.len() as u64).to_le_bytes()); + let mut keys: Vec<_> = values.keys().collect(); + keys.sort_unstable(); + for key in keys { + hash_field(hash, key.as_bytes()); + hash_json(hash, &values[key]); + } + } + } +} + +#[derive(Debug)] +pub(crate) struct CommittedPublication { + pub(crate) receipt: mst2_publication::Model, + pub(crate) outbox: mst2_publication_outbox::Model, +} + +#[derive(Debug)] +pub(crate) enum PublicationPreparation { + Prepared(PreparedPublication), + AlreadyCommitted(CommittedPublication), + AlreadyCommittedNoop(mst2_queue_noop_receipt::Model), +} + +/// The SQL transaction id prevents finalizing a reservation in another txn. +#[derive(Debug)] +pub(crate) struct PreparedPublication { + request: PublicationRequest, + sequence: i64, + transaction_id: i64, +} + +async fn validate_committed_publication( + txn: &DatabaseTransaction, + request: &PublicationRequest, + receipt: mst2_publication::Model, +) -> Result { + if receipt.namespace != request.namespace { + return Err(PublicationReceiptError::Conflict( + "operation belongs to another namespace".into(), + )); + } + if receipt.request_digest_version != Some(REQUEST_DIGEST_VERSION) + || receipt.request_digest.is_none() + { + return Err(PublicationReceiptError::LegacyReceipt( + request.operation_id.clone(), + )); + } + if receipt.writer_kind != request.writer_kind + || receipt.writer_epoch != WRITER_EPOCH + || receipt.request_digest.as_deref() != Some(request.request_digest.as_str()) + { + return Err(PublicationReceiptError::Conflict( + "operation request changed".into(), + )); + } + let outbox = mst2_publication_outbox::Entity::find() + .filter(mst2_publication_outbox::Column::OperationId.eq(&request.operation_id)) + .one(txn) + .await? + .ok_or_else(|| PublicationReceiptError::Integrity("receipt outbox missing".into()))?; + if outbox.namespace != receipt.namespace || outbox.sequence != receipt.sequence { + return Err(PublicationReceiptError::Integrity( + "receipt outbox identity mismatch".into(), + )); + } + Ok(CommittedPublication { receipt, outbox }) +} + +async fn reject_legacy_queue_receipt( + txn: &DatabaseTransaction, + legacy_operation_id: Option<&str>, + namespace: &str, +) -> Result<(), PublicationReceiptError> { + if let Some(legacy_id) = legacy_operation_id + && let Some(legacy) = mst2_publication::Entity::find() + .filter(mst2_publication::Column::OperationId.eq(legacy_id)) + .filter(mst2_publication::Column::Namespace.eq(namespace)) + .one(txn) + .await? + && (legacy.request_digest_version != Some(REQUEST_DIGEST_VERSION) + || legacy.request_digest.is_none()) + { + return Err(PublicationReceiptError::LegacyReceipt(legacy_id.to_owned())); + } + Ok(()) +} + +enum RecordedOperation { + Publication(mst2_publication::Model), + Noop(mst2_queue_noop_receipt::Model), +} + +async fn find_committed_operation( + txn: &DatabaseTransaction, + operation_id: &str, +) -> Result, PublicationReceiptError> { + let publication = mst2_publication::Entity::find() + .filter(mst2_publication::Column::OperationId.eq(operation_id)) + .one(txn) + .await?; + let noop = mst2_queue_noop_receipt::Entity::find() + .filter(mst2_queue_noop_receipt::Column::OperationId.eq(operation_id)) + .one(txn) + .await?; + match (publication, noop) { + (Some(_), Some(_)) => Err(PublicationReceiptError::Integrity( + "operation has both publication and no-op receipts".into(), + )), + (Some(receipt), None) => Ok(Some(RecordedOperation::Publication(receipt))), + (None, Some(receipt)) => Ok(Some(RecordedOperation::Noop(receipt))), + (None, None) => Ok(None), + } +} + +async fn validate_committed_operation( + txn: &DatabaseTransaction, + request: &PublicationRequest, + existing: RecordedOperation, +) -> Result { + match existing { + RecordedOperation::Publication(receipt) => Ok(PublicationPreparation::AlreadyCommitted( + validate_committed_publication(txn, request, receipt).await?, + )), + RecordedOperation::Noop(receipt) => { + if receipt.namespace != request.namespace + || receipt.writer_kind != request.writer_kind + || receipt.writer_epoch != WRITER_EPOCH + || !request.is_noop + || receipt.request_digest != request.request_digest + { + return Err(PublicationReceiptError::Conflict( + "operation request changed".into(), + )); + } + if receipt.request_digest_version != REQUEST_DIGEST_VERSION { + return Err(PublicationReceiptError::LegacyReceipt( + request.operation_id.clone(), + )); + } + if mst2_publication_outbox::Entity::find() + .filter(mst2_publication_outbox::Column::OperationId.eq(&request.operation_id)) + .one(txn) + .await? + .is_some() + { + return Err(PublicationReceiptError::Integrity( + "no-op operation has a publication outbox".into(), + )); + } + Ok(PublicationPreparation::AlreadyCommittedNoop(receipt)) + } + } +} + +impl MonoStorage { + /// Queue admission can replay Done without executing B3; audit that path too. + pub(crate) async fn validate_queue_publication_replay_in_txn( + &self, + txn: &DatabaseTransaction, + candidate: &push_queue::Model, + ) -> Result<(), PublicationReceiptError> { + let operation_id = format!("mst2:trunk-queue:{}", candidate.id); + let Some(existing) = find_committed_operation(txn, &operation_id).await? else { + reject_legacy_queue_receipt(txn, Some(&candidate.operation_id), &candidate.path) + .await?; + return Ok(()); + }; + let request = PublicationRequest::from_trunk_queue(candidate)?; + if let RecordedOperation::Publication(receipt) = &existing { + self.validate_native_historical_receipt_in_txn(txn, receipt) + .await?; + } + validate_committed_operation(txn, &request, existing).await?; + Ok(()) + } + + /// Lock before any business ref write, then recheck immutable receipt identity. + pub(crate) async fn begin_publication_in_txn( + &self, + txn: &DatabaseTransaction, + request: PublicationRequest, + ) -> Result { + txn.execute_raw(Statement::from_sql_and_values( + txn.get_database_backend(), + "INSERT INTO mst2_namespace_seq (namespace, sequence, epoch) VALUES ($1, 0, $2) \ + ON CONFLICT (namespace) DO NOTHING", + [request.namespace.clone().into(), WRITER_EPOCH.into()], + )) + .await?; + let row = txn + .query_one_raw(Statement::from_sql_and_values( + txn.get_database_backend(), + "SELECT sequence, epoch, txid_current() AS transaction_id \ + FROM mst2_namespace_seq WHERE namespace = $1 FOR UPDATE", + [request.namespace.clone().into()], + )) + .await? + .ok_or_else(|| PublicationReceiptError::Integrity("namespace row missing".into()))?; + let sequence: i64 = row.try_get("", "sequence")?; + let epoch: i64 = row.try_get("", "epoch")?; + if let Some(existing) = find_committed_operation(txn, &request.operation_id).await? { + if let RecordedOperation::Publication(receipt) = &existing { + self.validate_native_historical_receipt_in_txn(txn, receipt) + .await?; + } + return validate_committed_operation(txn, &request, existing).await; + } + if epoch != WRITER_EPOCH || sequence < 0 { + return Err(PublicationReceiptError::Conflict( + "writer epoch is fenced".into(), + )); + } + reject_legacy_queue_receipt( + txn, + request.legacy_operation_id.as_deref(), + &request.namespace, + ) + .await?; + Ok(PublicationPreparation::Prepared(PreparedPublication { + request, + sequence, + transaction_id: row.try_get("", "transaction_id")?, + })) + } + + /// Finalize only the request reserved before this transaction's business writes. + pub(crate) async fn record_publication_in_txn( + &self, + txn: &DatabaseTransaction, + prepared: PreparedPublication, + old_oid: &str, + new_oid: &str, + ) -> Result { + let request = prepared.request; + if request.is_noop { + return Err(PublicationReceiptError::Integrity( + "no-op requests do not publish".into(), + )); + } + let row = txn + .query_one_raw(Statement::from_sql_and_values( + txn.get_database_backend(), + "UPDATE mst2_namespace_seq SET sequence = sequence + 1 \ + WHERE namespace = $1 AND sequence = $2 AND epoch = $3 AND txid_current() = $4 \ + RETURNING sequence", + [ + request.namespace.clone().into(), + prepared.sequence.into(), + WRITER_EPOCH.into(), + prepared.transaction_id.into(), + ], + )) + .await? + .ok_or_else(|| { + PublicationReceiptError::Conflict( + "publication reservation is stale or belongs to another transaction".into(), + ) + })?; + let sequence: i64 = row.try_get("", "sequence")?; + let now = chrono::Utc::now().fixed_offset(); + let receipt = mst2_publication::ActiveModel { + id: sea_orm::ActiveValue::NotSet, + operation_id: Set(request.operation_id.clone()), + namespace: Set(request.namespace.clone()), + sequence: Set(sequence), + old_oid: Set(old_oid.to_owned()), + new_oid: Set(new_oid.to_owned()), + writer_epoch: Set(WRITER_EPOCH), + writer_kind: Set(request.writer_kind), + request_digest: Set(Some(request.request_digest)), + request_digest_version: Set(Some(REQUEST_DIGEST_VERSION)), + native_certificate_version: Set(None), + created_at: Set(now), + } + .insert(txn) + .await?; + #[cfg(all(test, unix))] + tests::crash_checkpoint("receipt-written"); + let outbox = mst2_publication_outbox::ActiveModel { + id: sea_orm::ActiveValue::NotSet, + operation_id: Set(request.operation_id), + namespace: Set(request.namespace), + sequence: Set(sequence), + state: Set("PENDING".to_owned()), + created_at: Set(now), + } + .insert(txn) + .await?; + #[cfg(all(test, unix))] + tests::crash_checkpoint("outbox-written"); + Ok(CommittedPublication { receipt, outbox }) + } + + pub(crate) async fn record_noop_operation_in_txn( + &self, + txn: &DatabaseTransaction, + prepared: PreparedPublication, + root_commit: &str, + root_tree: &str, + landed_commit_id: &str, + ) -> Result { + if !prepared.request.is_noop { + return Err(PublicationReceiptError::Integrity( + "request is not a trunk push no-op".into(), + )); + } + txn.query_one_raw(Statement::from_sql_and_values( + txn.get_database_backend(), + "SELECT sequence FROM mst2_namespace_seq \ + WHERE namespace = $1 AND sequence = $2 AND epoch = $3 AND txid_current() = $4 FOR UPDATE", + [prepared.request.namespace.clone().into(), prepared.sequence.into(), WRITER_EPOCH.into(), prepared.transaction_id.into()], + )).await?.ok_or_else(|| PublicationReceiptError::Conflict("no-op reservation is stale or belongs to another transaction".into()))?; + let request = prepared.request; + let receipt = mst2_queue_noop_receipt::ActiveModel { + id: sea_orm::ActiveValue::NotSet, + operation_id: Set(request.operation_id), + namespace: Set(request.namespace), + request_digest: Set(request.request_digest), + request_digest_version: Set(REQUEST_DIGEST_VERSION), + writer_epoch: Set(WRITER_EPOCH), + writer_kind: Set(request.writer_kind), + observed_sequence: Set(prepared.sequence), + observed_root_commit: Set(root_commit.to_owned()), + observed_root_tree: Set(root_tree.to_owned()), + landed_commit_id: Set(landed_commit_id.to_owned()), + created_at: Set(chrono::Utc::now().fixed_offset()), + } + .insert(txn) + .await?; + #[cfg(all(test, unix))] + tests::crash_checkpoint("noop-receipt-written"); + Ok(receipt) + } + + #[cfg(test)] + pub(crate) async fn record_test_publication_in_txn( + &self, + txn: &DatabaseTransaction, + operation_id: &str, + namespace: &str, + old: &str, + new: &str, + writer_kind: &str, + ) -> Result { + let request = PublicationRequest::for_test(operation_id, namespace, old, new, writer_kind); + let committed = match self.begin_publication_in_txn(txn, request).await? { + PublicationPreparation::AlreadyCommitted(committed) => committed, + PublicationPreparation::AlreadyCommittedNoop(_) => { + return Err(PublicationReceiptError::Integrity( + "test publication replayed a no-op".into(), + )); + } + PublicationPreparation::Prepared(prepared) => { + self.record_publication_in_txn(txn, prepared, old, new) + .await? + } + }; + Ok(committed.receipt.sequence) + } +} + +#[cfg(test)] +#[path = "mst2_publication_tests.rs"] +mod tests; diff --git a/src/jupiter/storage/mst2_publication_tests.rs b/src/jupiter/storage/mst2_publication_tests.rs new file mode 100644 index 00000000..0937a625 --- /dev/null +++ b/src/jupiter/storage/mst2_publication_tests.rs @@ -0,0 +1,1091 @@ +use std::sync::Arc; + +use sea_orm::{PaginatorTrait, TransactionTrait}; +use sea_orm_migration::MigratorTrait; + +use super::*; +use crate::jupiter::{ + migration::{Migrator, apply_migrations}, + storage::{ + base_storage::{BaseStorage, StorageConnector}, + push_queue_storage::{ClaimOutcome, EnqueueOutcome, EnqueueParams, PushQueueStorage}, + }, + tests::test_db_connection, +}; + +fn queue_row() -> push_queue::Model { + let now = chrono::Utc::now().fixed_offset(); + push_queue::Model { + id: 7, + kind: PushQueueKindEnum::Push, + operation_id: "old→new".to_owned(), + path: "/project".to_owned(), + old_id: "old".to_owned(), + new_id: "new".to_owned(), + payload: serde_json::json!({"commits": ["new"], "n": 1, "fork_base": "old"}), + landed_commit_id: None, + status: crate::callisto::sea_orm_active_enums::PushQueueStatusEnum::Running, + requester: Some("alice".to_owned()), + failure_type: None, + error_message: None, + heartbeat_at: now, + superseded_by: None, + expected_commit_hash: Some("root-before".to_owned()), + expected_tree_hash: Some("tree-before".to_owned()), + expected_native_sequence: None, + expected_native_epoch: None, + expected_native_certificate: None, + pending_action: None, + enqueued_at: now, + started_at: Some(now), + finished_at: None, + updated_at: now, + } +} + +fn noop_queue_row() -> push_queue::Model { + let mut row = queue_row(); + row.old_id = "tip".to_owned(); + row.new_id = "tip".to_owned(); + row.payload = serde_json::json!({"commits": [], "fork_base": null, "n": 0}); + row +} + +#[test] +fn queue_digest_is_versioned_and_binds_trusted_intent_not_retry_baseline() { + let row = queue_row(); + let original = PublicationRequest::from_trunk_queue(&row).unwrap(); + assert_eq!(original.operation_id, "mst2:trunk-queue:7"); + let mut retry = row.clone(); + retry.expected_commit_hash = Some("later-root".to_owned()); + retry.expected_tree_hash = Some("later-tree".to_owned()); + retry.started_at = None; + retry.payload = serde_json::from_str(r#"{"n":1,"fork_base":"old","commits":["new"]}"#).unwrap(); + assert_eq!( + PublicationRequest::from_trunk_queue(&retry) + .unwrap() + .request_digest, + original.request_digest + ); + for change in ["actor", "path", "old", "new", "payload", "identity"] { + let mut changed = row.clone(); + match change { + "actor" => changed.requester = Some("bob".to_owned()), + "path" => changed.path = "/other".to_owned(), + "old" => changed.old_id = "other-old".to_owned(), + "new" => changed.new_id = "other-new".to_owned(), + "payload" => changed.payload["commits"] = serde_json::json!(["new", "extra"]), + "identity" => changed.id += 1, + _ => unreachable!(), + } + assert_ne!( + PublicationRequest::from_trunk_queue(&changed) + .unwrap() + .request_digest, + original.request_digest, + "{change}" + ); + } +} + +async fn storage() -> (tempfile::TempDir, MonoStorage) { + let temp = tempfile::tempdir().unwrap(); + let db = test_db_connection(temp.path()).await; + apply_migrations(&db, false).await.unwrap(); + ( + temp, + MonoStorage { + base: BaseStorage::new(Arc::new(db)), + }, + ) +} + +async fn seed_root(mono: &MonoStorage) { + mono.get_connection().execute_raw(Statement::from_sql_and_values( + mono.get_connection().get_database_backend(), + "INSERT INTO mega_refs (id, path, ref_name, ref_commit_hash, ref_tree_hash, created_at, updated_at, is_cl) \ + VALUES (1, '/', $1, $2, $3, now(), now(), false)", + [MEGA_BRANCH_NAME.into(), "a".repeat(40).into(), "b".repeat(40).into()], + )).await.unwrap(); +} + +async fn prepared( + mono: &MonoStorage, + txn: &DatabaseTransaction, + request: PublicationRequest, +) -> PreparedPublication { + match mono.begin_publication_in_txn(txn, request).await.unwrap() { + PublicationPreparation::Prepared(prepared) => prepared, + PublicationPreparation::AlreadyCommitted(_) => panic!("fresh operation"), + PublicationPreparation::AlreadyCommittedNoop(_) => panic!("fresh no-op operation"), + } +} + +async fn advance(mono: &MonoStorage, txn: &DatabaseTransaction) -> bool { + mono.cas_update_root_main_ref_in_txn( + txn, + Some(&"a".repeat(40)), + Some(&"b".repeat(40)), + &"c".repeat(40), + &"d".repeat(40), + ) + .await + .unwrap() +} + +async fn counts(mono: &MonoStorage) -> (u64, u64) { + ( + mst2_publication::Entity::find() + .count(mono.get_connection()) + .await + .unwrap(), + mst2_publication_outbox::Entity::find() + .count(mono.get_connection()) + .await + .unwrap(), + ) +} + +#[tokio::test] +async fn merge_stub_admission_replays_same_receipt_and_rejects_changed_intent() { + let (_temp, mono) = storage().await; + seed_root(&mono).await; + let queue = PushQueueStorage::new(mono.base.clone()); + let payload = serde_json::json!({"steps": ["first"], "mode": "stub"}); + let admitted = queue + .enqueue_atomic(EnqueueParams { + kind: PushQueueKindEnum::Merge, + operation_id: "merge-stub-admission", + path: "/project", + old_id: "old", + new_id: "new", + requester: Some("alice"), + payload: payload.clone(), + }) + .await + .unwrap(); + let EnqueueOutcome::Inserted { id } = admitted else { + panic!("{admitted:?}"); + }; + // With no committed receipt, admission retains the existing adoption + // behavior and the original persisted intent remains the execution input. + assert_eq!( + queue + .enqueue_atomic(EnqueueParams { + kind: PushQueueKindEnum::Merge, + operation_id: "merge-stub-admission", + path: "/project", + old_id: "old", + new_id: "new", + requester: Some("legacy-retry"), + payload: serde_json::json!({"different": "no receipt yet"}), + }) + .await + .unwrap(), + EnqueueOutcome::Adopted { id } + ); + assert_eq!( + queue.claim_for_execution(id).await.unwrap(), + ClaimOutcome::Claimed + ); + let row = push_queue::Entity::find_by_id(id) + .one(mono.get_connection()) + .await + .unwrap() + .unwrap(); + assert_eq!(row.requester.as_deref(), Some("alice")); + assert_eq!(row.payload, payload); + let txn = mono.get_connection().begin().await.unwrap(); + let reservation = prepared( + &mono, + &txn, + PublicationRequest::from_trunk_queue(&row).unwrap(), + ) + .await; + assert!(advance(&mono, &txn).await); + let landed = "c".repeat(40); + assert!( + PushQueueStorage::mark_done_if_running_in_txn(&txn, id, &landed) + .await + .unwrap() + ); + let original = mono + .record_publication_in_txn(&txn, reservation, &"a".repeat(40), &landed) + .await + .unwrap(); + assert_eq!(original.receipt.writer_kind, "trunk_merge_stub"); + txn.commit().await.unwrap(); + assert_eq!( + queue + .enqueue_atomic(EnqueueParams { + kind: PushQueueKindEnum::Merge, + operation_id: "merge-stub-admission", + path: "/project", + old_id: "old", + new_id: "new", + requester: Some("alice"), + payload: payload.clone(), + }) + .await + .unwrap(), + EnqueueOutcome::Replay { + id, + landed_commit_id: Some(landed.clone()) + } + ); + for change in ["actor", "payload"] { + let (requester, retry_payload) = if change == "actor" { + ("mallory", payload.clone()) + } else { + ( + "alice", + serde_json::json!({"steps": ["first", "changed"], "mode": "stub"}), + ) + }; + let error = queue + .enqueue_atomic(EnqueueParams { + kind: PushQueueKindEnum::Merge, + operation_id: "merge-stub-admission", + path: "/project", + old_id: "old", + new_id: "new", + requester: Some(requester), + payload: retry_payload, + }) + .await + .unwrap_err(); + assert!( + error.to_string().contains("MST2_PUBLICATION_CONFLICT"), + "{change}" + ); + } + assert_eq!( + mst2_publication::Entity::find() + .all(mono.get_connection()) + .await + .unwrap(), + vec![original.receipt] + ); + assert_eq!( + mst2_publication_outbox::Entity::find() + .all(mono.get_connection()) + .await + .unwrap(), + vec![original.outbox] + ); + assert_eq!(mono.publication_sequence("/project").await.unwrap(), 1); + assert_eq!( + mono.get_main_ref("/") + .await + .unwrap() + .unwrap() + .ref_commit_hash, + landed + ); +} + +#[tokio::test] +async fn same_request_concurrently_returns_one_original_receipt_and_outbox() { + let (_temp, mono) = storage().await; + seed_root(&mono).await; + let request = PublicationRequest::for_test("same-op", "/", "a", "c", "test"); + let barrier = Arc::new(tokio::sync::Barrier::new(2)); + let mut workers = Vec::new(); + for _ in 0..2 { + let mono = mono.clone(); + let request = request.clone(); + let barrier = Arc::clone(&barrier); + workers.push(tokio::spawn(async move { + let txn = mono.get_connection().begin().await.unwrap(); + barrier.wait().await; + let (committed, wrote) = + match mono.begin_publication_in_txn(&txn, request).await.unwrap() { + PublicationPreparation::AlreadyCommitted(committed) => (committed, false), + PublicationPreparation::AlreadyCommittedNoop(_) => { + panic!("publication request") + } + PublicationPreparation::Prepared(prepared) => { + assert!(advance(&mono, &txn).await); + ( + mono.record_publication_in_txn(&txn, prepared, "a", "c") + .await + .unwrap(), + true, + ) + } + }; + txn.commit().await.unwrap(); + (committed, wrote) + })); + } + let mut results = Vec::new(); + for worker in workers { + results.push( + tokio::time::timeout(std::time::Duration::from_secs(30), worker) + .await + .unwrap() + .unwrap(), + ); + } + assert_eq!(results.iter().filter(|(_, wrote)| *wrote).count(), 1); + assert_eq!(results[0].0.receipt, results[1].0.receipt); + assert_eq!(results[0].0.outbox, results[1].0.outbox); + assert_eq!(mono.publication_sequence("/").await.unwrap(), 1); + assert_eq!(counts(&mono).await, (1, 1)); +} + +#[tokio::test] +async fn changed_request_conflicts_before_ref_writes() { + let (_temp, mono) = storage().await; + seed_root(&mono).await; + let txn = mono.get_connection().begin().await.unwrap(); + let request = PublicationRequest::from_trunk_queue(&queue_row()).unwrap(); + let reserved = prepared(&mono, &txn, request).await; + assert!(advance(&mono, &txn).await); + mono.record_publication_in_txn(&txn, reserved, "a", "c") + .await + .unwrap(); + txn.commit().await.unwrap(); + for change in ["actor", "new", "payload"] { + let mut changed = queue_row(); + match change { + "actor" => changed.requester = Some("mallory".to_owned()), + "new" => changed.new_id = "overwrite".to_owned(), + "payload" => changed.payload["commits"] = serde_json::json!(["injected"]), + _ => unreachable!(), + } + let txn = mono.get_connection().begin().await.unwrap(); + assert!(matches!( + mono.begin_publication_in_txn( + &txn, + PublicationRequest::from_trunk_queue(&changed).unwrap() + ) + .await, + Err(PublicationReceiptError::Conflict(_)) + )); + txn.rollback().await.unwrap(); + assert_eq!( + mono.get_main_ref("/") + .await + .unwrap() + .unwrap() + .ref_commit_hash, + "c".repeat(40) + ); + assert_eq!(mono.publication_sequence("/project").await.unwrap(), 1); + assert_eq!(counts(&mono).await, (1, 1)); + } +} + +#[tokio::test] +async fn different_operations_compete_on_actual_root_cas_without_extra_sequence() { + let (_temp, mono) = storage().await; + seed_root(&mono).await; + let barrier = Arc::new(tokio::sync::Barrier::new(2)); + let mut workers = Vec::new(); + for i in 0..2 { + let mono = mono.clone(); + let barrier = Arc::clone(&barrier); + workers.push(tokio::spawn(async move { + let txn = mono.get_connection().begin().await.unwrap(); + barrier.wait().await; + let request = + PublicationRequest::for_test(&format!("competitor-{i}"), "/", "a", "c", "test"); + let reserved = prepared(&mono, &txn, request).await; + if !advance(&mono, &txn).await { + txn.rollback().await.unwrap(); + return false; + } + mono.record_publication_in_txn(&txn, reserved, "a", "c") + .await + .unwrap(); + txn.commit().await.unwrap(); + true + })); + } + let mut winners = 0; + for worker in workers { + winners += usize::from( + tokio::time::timeout(std::time::Duration::from_secs(30), worker) + .await + .unwrap() + .unwrap(), + ); + } + assert_eq!(winners, 1); + assert_eq!(counts(&mono).await, (1, 1)); + assert_eq!(mono.publication_sequence("/").await.unwrap(), 1); +} + +#[tokio::test] +async fn reservation_cannot_be_finalized_from_another_transaction() { + let (_temp, mono) = storage().await; + // Keep the namespace/sequence unchanged after rollback: only the txn id differs. + mono.get_connection() + .execute_unprepared( + "INSERT INTO mst2_namespace_seq (namespace, sequence, epoch) VALUES ('/', 0, 1)", + ) + .await + .unwrap(); + let txn = mono.get_connection().begin().await.unwrap(); + let reserved = prepared( + &mono, + &txn, + PublicationRequest::for_test("txn-op", "/", "a", "c", "test"), + ) + .await; + txn.rollback().await.unwrap(); + let other = mono.get_connection().begin().await.unwrap(); + assert!(matches!( + mono.record_publication_in_txn(&other, reserved, "a", "c") + .await, + Err(PublicationReceiptError::Conflict(_)) + )); + other.rollback().await.unwrap(); + assert_eq!(counts(&mono).await, (0, 0)); + assert_eq!(mono.publication_sequence("/").await.unwrap(), 0); +} + +#[tokio::test] +async fn legacy_receipt_and_queue_alias_do_not_guess_request_digest() { + let (_temp, mono) = storage().await; + let row = queue_row(); + mono.get_connection().execute_raw(Statement::from_sql_and_values( + mono.get_connection().get_database_backend(), + "INSERT INTO mst2_publication (operation_id, namespace, sequence, old_oid, new_oid, writer_epoch, writer_kind, created_at) \ + VALUES ($1, $2, 1, 'old', 'new', 1, 'trunk_push', now())", + [row.operation_id.clone().into(), row.path.clone().into()], + )).await.unwrap(); + for request in [ + PublicationRequest::for_test(&row.operation_id, &row.path, "old", "new", "trunk_push"), + PublicationRequest::from_trunk_queue(&row).unwrap(), + ] { + let txn = mono.get_connection().begin().await.unwrap(); + assert!(matches!( + mono.begin_publication_in_txn(&txn, request).await, + Err(PublicationReceiptError::LegacyReceipt(_)) + )); + txn.rollback().await.unwrap(); + } + let txn = mono.get_connection().begin().await.unwrap(); + assert!(matches!( + mono.validate_queue_publication_replay_in_txn(&txn, &row) + .await, + Err(PublicationReceiptError::LegacyReceipt(_)) + )); + txn.rollback().await.unwrap(); + let legacy = mst2_publication::Entity::find() + .one(mono.get_connection()) + .await + .unwrap() + .unwrap(); + assert!(legacy.request_digest.is_none()); + assert!(legacy.request_digest_version.is_none()); + assert_eq!(counts(&mono).await, (1, 0)); +} + +#[tokio::test] +async fn incomplete_outbox_refuses_both_receipt_and_queue_replay() { + for corruption in ["missing", "sequence", "namespace"] { + let (_temp, mono) = storage().await; + seed_root(&mono).await; + let row = queue_row(); + let request = PublicationRequest::from_trunk_queue(&row).unwrap(); + let txn = mono.get_connection().begin().await.unwrap(); + let reserved = prepared(&mono, &txn, request.clone()).await; + assert!(advance(&mono, &txn).await); + mono.record_publication_in_txn(&txn, reserved, "a", "c") + .await + .unwrap(); + txn.commit().await.unwrap(); + let sql = match corruption { + "missing" => "DELETE FROM mst2_publication_outbox", + "sequence" => "UPDATE mst2_publication_outbox SET sequence = sequence + 1", + "namespace" => "UPDATE mst2_publication_outbox SET namespace = '/other'", + _ => unreachable!(), + }; + mono.get_connection().execute_unprepared(sql).await.unwrap(); + let before = counts(&mono).await; + let txn = mono.get_connection().begin().await.unwrap(); + assert!( + matches!( + mono.begin_publication_in_txn(&txn, request).await, + Err(PublicationReceiptError::Integrity(_)) + ), + "{corruption}" + ); + txn.rollback().await.unwrap(); + let txn = mono.get_connection().begin().await.unwrap(); + assert!( + matches!( + mono.validate_queue_publication_replay_in_txn(&txn, &row) + .await, + Err(PublicationReceiptError::Integrity(_)) + ), + "{corruption}" + ); + txn.rollback().await.unwrap(); + assert_eq!( + mono.get_main_ref("/") + .await + .unwrap() + .unwrap() + .ref_commit_hash, + "c".repeat(40) + ); + assert_eq!(mono.publication_sequence(&row.path).await.unwrap(), 1); + assert_eq!(counts(&mono).await, before); + } +} + +#[tokio::test] +async fn concurrent_noop_operations_keep_the_publication_sequence_and_outbox() { + let (_temp, mono) = storage().await; + seed_root(&mono).await; + let barrier = Arc::new(tokio::sync::Barrier::new(2)); + let mut workers = Vec::new(); + for _ in 0..2 { + let mono = mono.clone(); + let barrier = Arc::clone(&barrier); + workers.push(tokio::spawn(async move { + let txn = mono.get_connection().begin().await.unwrap(); + barrier.wait().await; + let (receipt, wrote) = match mono + .begin_publication_in_txn( + &txn, + PublicationRequest::from_trunk_queue(&noop_queue_row()).unwrap(), + ) + .await + .unwrap() + { + PublicationPreparation::Prepared(reserved) => { + assert!( + mono.cas_update_root_main_ref_in_txn( + &txn, + Some(&"a".repeat(40)), + Some(&"b".repeat(40)), + &"a".repeat(40), + &"b".repeat(40) + ) + .await + .unwrap() + ); + ( + mono.record_noop_operation_in_txn( + &txn, + reserved, + &"a".repeat(40), + &"b".repeat(40), + "tip", + ) + .await + .unwrap(), + true, + ) + } + PublicationPreparation::AlreadyCommittedNoop(receipt) => (receipt, false), + PublicationPreparation::AlreadyCommitted(_) => panic!("no-op request"), + }; + txn.commit().await.unwrap(); + (receipt, wrote) + })); + } + let mut results = Vec::new(); + for worker in workers { + results.push( + tokio::time::timeout(std::time::Duration::from_secs(30), worker) + .await + .unwrap() + .unwrap(), + ); + } + assert_eq!(results.iter().filter(|(_, wrote)| *wrote).count(), 1); + assert_eq!(results[0].0, results[1].0); + assert_eq!(results[0].0.observed_sequence, 0); + assert_eq!(mono.publication_sequence("/project").await.unwrap(), 0); + assert_eq!(counts(&mono).await, (0, 0)); + assert_eq!( + mst2_queue_noop_receipt::Entity::find() + .count(mono.get_connection()) + .await + .unwrap(), + 1 + ); + for change in ["actor", "payload"] { + let mut row = noop_queue_row(); + if change == "actor" { + row.requester = Some("other".to_owned()); + } else { + row.payload["fork_base"] = serde_json::json!("other"); + } + let txn = mono.get_connection().begin().await.unwrap(); + assert!(matches!( + mono.begin_publication_in_txn( + &txn, + PublicationRequest::from_trunk_queue(&row).unwrap() + ) + .await, + Err(PublicationReceiptError::Conflict(_)) + )); + txn.rollback().await.unwrap(); + let txn = mono.get_connection().begin().await.unwrap(); + assert!(matches!( + mono.validate_queue_publication_replay_in_txn(&txn, &row) + .await, + Err(PublicationReceiptError::Conflict(_)) + )); + txn.rollback().await.unwrap(); + } +} + +#[tokio::test] +async fn noop_receipt_replay_rejects_unsupported_version_and_publication_corruption() { + for corruption in ["version", "publication", "outbox"] { + let (_temp, mono) = storage().await; + let row = noop_queue_row(); + let request = PublicationRequest::from_trunk_queue(&row).unwrap(); + let txn = mono.get_connection().begin().await.unwrap(); + let reserved = prepared(&mono, &txn, request.clone()).await; + mono.record_noop_operation_in_txn(&txn, reserved, "root", "tree", "tip") + .await + .unwrap(); + txn.commit().await.unwrap(); + let sql = match corruption { + "version" => "UPDATE mst2_queue_noop_receipt SET request_digest_version = 2", + "publication" => { + "INSERT INTO mst2_publication (operation_id, namespace, sequence, old_oid, new_oid, writer_epoch, writer_kind, created_at) \ + VALUES ('mst2:trunk-queue:7', '/project', 1, 'a', 'b', 1, 'trunk_push', now())" + } + "outbox" => { + "INSERT INTO mst2_publication_outbox (operation_id, namespace, sequence, state, created_at) \ + VALUES ('mst2:trunk-queue:7', '/project', 1, 'PENDING', now())" + } + _ => unreachable!(), + }; + mono.get_connection().execute_unprepared(sql).await.unwrap(); + let publications_before = mst2_publication::Entity::find() + .all(mono.get_connection()) + .await + .unwrap(); + let outbox_before = mst2_publication_outbox::Entity::find() + .all(mono.get_connection()) + .await + .unwrap(); + for admission in [false, true] { + let txn = mono.get_connection().begin().await.unwrap(); + let result = if admission { + mono.validate_queue_publication_replay_in_txn(&txn, &row) + .await + } else { + mono.begin_publication_in_txn(&txn, request.clone()) + .await + .map(|_| ()) + }; + if corruption == "version" { + assert!(matches!( + result, + Err(PublicationReceiptError::LegacyReceipt(_)) + )); + } else { + assert!(matches!(result, Err(PublicationReceiptError::Integrity(_)))); + } + txn.rollback().await.unwrap(); + let namespace = mono + .get_connection() + .query_one_raw(Statement::from_sql_and_values( + mono.get_connection().get_database_backend(), + "SELECT sequence FROM mst2_namespace_seq WHERE namespace = $1", + [row.path.clone().into()], + )) + .await + .unwrap() + .unwrap(); + assert_eq!(namespace.try_get::("", "sequence").unwrap(), 0); + assert_eq!( + mst2_publication::Entity::find() + .all(mono.get_connection()) + .await + .unwrap(), + publications_before + ); + assert_eq!( + mst2_publication_outbox::Entity::find() + .all(mono.get_connection()) + .await + .unwrap(), + outbox_before + ); + } + assert_eq!( + mst2_queue_noop_receipt::Entity::find() + .count(mono.get_connection()) + .await + .unwrap(), + 1 + ); + } +} + +#[tokio::test] +async fn noop_reservation_cannot_move_to_another_transaction() { + let (_temp, mono) = storage().await; + mono.get_connection() + .execute_unprepared( + "INSERT INTO mst2_namespace_seq (namespace, sequence, epoch) VALUES ('/project', 0, 1)", + ) + .await + .unwrap(); + let txn = mono.get_connection().begin().await.unwrap(); + let reserved = prepared( + &mono, + &txn, + PublicationRequest::from_trunk_queue(&noop_queue_row()).unwrap(), + ) + .await; + txn.rollback().await.unwrap(); + let other = mono.get_connection().begin().await.unwrap(); + assert!(matches!( + mono.record_noop_operation_in_txn(&other, reserved, "root", "tree", "tip") + .await, + Err(PublicationReceiptError::Conflict(_)) + )); + other.rollback().await.unwrap(); + assert_eq!(mono.publication_sequence("/project").await.unwrap(), 0); + assert_eq!(counts(&mono).await, (0, 0)); + assert_eq!( + mst2_queue_noop_receipt::Entity::find() + .count(mono.get_connection()) + .await + .unwrap(), + 0 + ); +} + +#[tokio::test] +async fn migration_preserves_legacy_receipts_and_validates_digest_columns() { + let temp = tempfile::tempdir().unwrap(); + let db = test_db_connection(temp.path()).await; + let migrations = Migrator::migrations(); + let receipt_at = migrations + .iter() + .position(|migration| { + migration.name() == "m20261005_000100_add_mst2_publication_request_digest" + }) + .expect("publication receipt migration registered"); + let old_count = u32::try_from(receipt_at).unwrap(); + Migrator::up(&db, Some(old_count)).await.unwrap(); + db.execute_unprepared( + "INSERT INTO mst2_publication (operation_id, namespace, sequence, old_oid, new_oid, writer_epoch, writer_kind, created_at) \ + VALUES ('legacy', '/', 1, 'a', 'b', 1, 'test', now())" + ).await.unwrap(); + apply_migrations(&db, false).await.unwrap(); + let txn = db.begin().await.unwrap(); + migrations[receipt_at] + .up(&sea_orm_migration::SchemaManager::new(&txn)) + .await + .unwrap(); + txn.commit().await.unwrap(); + let legacy = mst2_publication::Entity::find() + .one(&db) + .await + .unwrap() + .unwrap(); + assert!(legacy.request_digest.is_none()); + assert!(legacy.request_digest_version.is_none()); + for (digest, version) in [ + (Some(format!("sha256:{}", "a".repeat(64))), Some(1)), + (Some(format!("sha256:{}", "A".repeat(64))), Some(1)), + (Some("broken".to_owned()), Some(1)), + (None, Some(1)), + (Some(format!("sha256:{}", "a".repeat(64))), None), + (Some(format!("sha256:{}", "a".repeat(64))), Some(0)), + ] { + let valid = digest.as_deref() == Some(format!("sha256:{}", "a".repeat(64)).as_str()) + && version == Some(1); + let txn = db.begin().await.unwrap(); + let result = txn.execute_raw(Statement::from_sql_and_values( + txn.get_database_backend(), + "UPDATE mst2_publication SET request_digest = $1, request_digest_version = $2 WHERE operation_id = 'legacy'", + [digest.into(), version.into()], + )).await; + assert_eq!(result.is_ok(), valid); + txn.rollback().await.unwrap(); + } + let legacy = mst2_publication::Entity::find() + .one(&db) + .await + .unwrap() + .unwrap(); + assert!(legacy.request_digest.is_none()); +} + +#[cfg(unix)] +pub(super) fn crash_checkpoint(phase: &str) { + if std::env::var("MEGA_MST2_RECEIPT_CRASH_PHASE") + .ok() + .as_deref() + == Some(phase) + { + unsafe { + libc::raise(libc::SIGKILL); + } + panic!("SIGKILL did not terminate the worker"); + } +} + +#[cfg(unix)] +#[test] +fn receipt_crash_worker() { + let Ok(url) = std::env::var("MEGA_MST2_RECEIPT_WORKER_DB") else { + return; + }; + tokio::runtime::Builder::new_current_thread() + .enable_all() + .build() + .unwrap() + .block_on(async { + let db = sea_orm::Database::connect(url).await.unwrap(); + let mono = MonoStorage { + base: BaseStorage::new(Arc::new(db)), + }; + let txn = mono.get_connection().begin().await.unwrap(); + let noop = std::env::var("MEGA_MST2_RECEIPT_WORKER_NOOP").is_ok(); + let request = if noop { + PublicationRequest::from_trunk_queue(&noop_queue_row()).unwrap() + } else { + PublicationRequest::for_test("process-op", "/", "a", "c", "test") + }; + let reserved = prepared(&mono, &txn, request).await; + if noop { + assert!( + mono.cas_update_root_main_ref_in_txn( + &txn, + Some(&"a".repeat(40)), + Some(&"b".repeat(40)), + &"a".repeat(40), + &"b".repeat(40) + ) + .await + .unwrap() + ); + } else { + assert!(advance(&mono, &txn).await); + } + crash_checkpoint("ref-written"); + if noop { + mono.record_noop_operation_in_txn( + &txn, + reserved, + &"a".repeat(40), + &"b".repeat(40), + "tip", + ) + .await + .unwrap(); + } else { + mono.record_publication_in_txn(&txn, reserved, "a", "c") + .await + .unwrap(); + } + txn.commit().await.unwrap(); + crash_checkpoint("commit-complete"); + }); +} + +#[cfg(unix)] +#[tokio::test] +async fn sigkill_recovery_and_commit_without_response_use_original_receipt() { + use std::os::unix::process::ExitStatusExt; + + for phase in [ + "ref-written", + "receipt-written", + "outbox-written", + "commit-complete", + ] { + let temp = tempfile::tempdir().unwrap(); + let (config, _schema) = crate::jupiter::tests::test_db_config(temp.path()).await; + let db = crate::jupiter::storage::init::database_connection(&config) + .await + .unwrap(); + let mono = MonoStorage { + base: BaseStorage::new(Arc::new(db)), + }; + seed_root(&mono).await; + let mut worker = std::process::Command::new(std::env::current_exe().unwrap()) + .args([ + "--exact", + "jupiter::storage::mst2_publication_storage::tests::receipt_crash_worker", + "--nocapture", + ]) + .env("MEGA_MST2_RECEIPT_WORKER_DB", &config.db_url) + .env("MEGA_MST2_RECEIPT_CRASH_PHASE", phase) + .env_remove("MEGA_MST2_RECEIPT_WORKER_NOOP") + .stdout(std::process::Stdio::null()) + .stderr(std::process::Stdio::inherit()) + .spawn() + .unwrap(); + let deadline = std::time::Instant::now() + std::time::Duration::from_secs(30); + let status = loop { + if let Some(status) = worker.try_wait().unwrap() { + break status; + } + if std::time::Instant::now() >= deadline { + worker.kill().unwrap(); + worker.wait().unwrap(); + panic!("receipt worker timed out at {phase}"); + } + tokio::time::sleep(std::time::Duration::from_millis(10)).await; + }; + assert_eq!(status.signal(), Some(libc::SIGKILL), "{phase}: {status}"); + let committed = phase == "commit-complete"; + let root = mono.get_main_ref("/").await.unwrap().unwrap(); + assert_eq!( + root.ref_commit_hash, + if committed { "c" } else { "a" }.repeat(40) + ); + assert_eq!(counts(&mono).await, if committed { (1, 1) } else { (0, 0) }); + assert_eq!( + mono.publication_sequence("/").await.unwrap(), + i64::from(committed) + ); + let retry_db = sea_orm::Database::connect(config.db_url.as_str()) + .await + .unwrap(); + let retry_mono = MonoStorage { + base: BaseStorage::new(Arc::new(retry_db)), + }; + let txn = retry_mono.get_connection().begin().await.unwrap(); + let request = PublicationRequest::for_test("process-op", "/", "a", "c", "test"); + match retry_mono + .begin_publication_in_txn(&txn, request) + .await + .unwrap() + { + PublicationPreparation::AlreadyCommitted(original) => { + assert!(committed); + assert_eq!(original.receipt.sequence, 1); + assert_eq!(original.receipt.new_oid, "c"); + assert_eq!(original.outbox.sequence, 1); + txn.commit().await.unwrap(); + } + PublicationPreparation::Prepared(_) => { + assert!(!committed); + txn.rollback().await.unwrap(); + } + PublicationPreparation::AlreadyCommittedNoop(_) => panic!("publication request"), + } + assert_eq!(counts(&mono).await, if committed { (1, 1) } else { (0, 0) }); + } +} + +#[cfg(unix)] +#[tokio::test] +async fn noop_sigkill_and_lost_commit_response_never_create_a_publication() { + use std::os::unix::process::ExitStatusExt; + + for phase in ["ref-written", "noop-receipt-written", "commit-complete"] { + let temp = tempfile::tempdir().unwrap(); + let (config, _schema) = crate::jupiter::tests::test_db_config(temp.path()).await; + let db = crate::jupiter::storage::init::database_connection(&config) + .await + .unwrap(); + let mono = MonoStorage { + base: BaseStorage::new(Arc::new(db)), + }; + seed_root(&mono).await; + let mut worker = std::process::Command::new(std::env::current_exe().unwrap()) + .args([ + "--exact", + "jupiter::storage::mst2_publication_storage::tests::receipt_crash_worker", + "--nocapture", + ]) + .env("MEGA_MST2_RECEIPT_WORKER_DB", &config.db_url) + .env("MEGA_MST2_RECEIPT_CRASH_PHASE", phase) + .env("MEGA_MST2_RECEIPT_WORKER_NOOP", "1") + .stdout(std::process::Stdio::null()) + .stderr(std::process::Stdio::inherit()) + .spawn() + .unwrap(); + let deadline = std::time::Instant::now() + std::time::Duration::from_secs(30); + let status = loop { + if let Some(status) = worker.try_wait().unwrap() { + break status; + } + if std::time::Instant::now() >= deadline { + worker.kill().unwrap(); + worker.wait().unwrap(); + panic!("no-op receipt worker timed out at {phase}"); + } + tokio::time::sleep(std::time::Duration::from_millis(10)).await; + }; + assert_eq!(status.signal(), Some(libc::SIGKILL), "{phase}: {status}"); + let committed = phase == "commit-complete"; + assert_eq!( + mono.get_main_ref("/") + .await + .unwrap() + .unwrap() + .ref_commit_hash, + "a".repeat(40) + ); + assert_eq!(counts(&mono).await, (0, 0)); + assert_eq!(mono.publication_sequence("/project").await.unwrap(), 0); + assert_eq!( + mst2_queue_noop_receipt::Entity::find() + .count(mono.get_connection()) + .await + .unwrap(), + u64::from(committed) + ); + let original = mst2_queue_noop_receipt::Entity::find() + .one(mono.get_connection()) + .await + .unwrap(); + if committed { + let txn = mono.get_connection().begin().await.unwrap(); + assert!(advance(&mono, &txn).await); + txn.commit().await.unwrap(); + } + // Reconnect after the worker died; committed no-op identity survives a later root. + let retry_db = sea_orm::Database::connect(config.db_url.as_str()) + .await + .unwrap(); + let retry_mono = MonoStorage { + base: BaseStorage::new(Arc::new(retry_db)), + }; + let txn = retry_mono.get_connection().begin().await.unwrap(); + match retry_mono + .begin_publication_in_txn( + &txn, + PublicationRequest::from_trunk_queue(&noop_queue_row()).unwrap(), + ) + .await + .unwrap() + { + PublicationPreparation::AlreadyCommittedNoop(receipt) => { + assert!(committed); + assert_eq!(Some(receipt.clone()), original); + assert_eq!(receipt.observed_sequence, 0); + assert_eq!(receipt.observed_root_commit, "a".repeat(40)); + assert_eq!(receipt.landed_commit_id, "tip"); + txn.commit().await.unwrap(); + } + PublicationPreparation::Prepared(_) => { + assert!(!committed); + txn.rollback().await.unwrap(); + } + PublicationPreparation::AlreadyCommitted(_) => panic!("no-op must not publish"), + } + assert_eq!(counts(&mono).await, (0, 0)); + assert_eq!(mono.publication_sequence("/project").await.unwrap(), 0); + assert_eq!( + mono.get_main_ref("/") + .await + .unwrap() + .unwrap() + .ref_commit_hash, + if committed { "c" } else { "a" }.repeat(40) + ); + } +} diff --git a/src/jupiter/storage/mst2_retention.rs b/src/jupiter/storage/mst2_retention.rs new file mode 100644 index 00000000..3931fe86 --- /dev/null +++ b/src/jupiter/storage/mst2_retention.rs @@ -0,0 +1,680 @@ +//! Asynchronous PostgreSQL retention graph repository (T06-B, spec 10 §6). +//! +//! This repository is not yet wired into the in-process snapshot runtime. +//! Graph mutations serialize on a schema-scoped transaction advisory lock; +//! callers can include them in a publication transaction. Each mutation uses +//! a savepoint, so a rejected group cannot leave partial data in that caller's +//! READ COMMITTED transaction. Bytes must be complete and verified before +//! retaining a node. +//! +//! A GC claim commits DELETING and a PENDING REMOVE intent together. A worker +//! may repeat the idempotent physical deletion after a crash, then call +//! `complete_gc`. The outgoing edges, child counters and APPLIED receipt commit +//! together, so replay cannot subtract references twice. No method here deletes +//! bytes, and shared Git raw deletion remains disabled until GitRetentionPort +//! is integrated. Physical workers and durable lease/publication wiring remain +//! separate integration work. Removed identities are tombstoned by their GC +//! receipts: rebuilding one requires future generation/fencing integration, +//! otherwise a stale physical worker could delete its newly reconstructed bytes. + +use std::collections::{BTreeMap, BTreeSet, VecDeque}; + +use sea_orm::{ + ColumnTrait, ConnectionTrait, DatabaseConnection, DatabaseTransaction, DbBackend, EntityTrait, + QueryFilter, QueryOrder, QuerySelect, Statement, TransactionTrait, +}; +use serde_json::json; + +use crate::{ + callisto::{mst2_retention_gc_op, mst2_retention_node}, + ceres::snapshot::{ + error::{SnapshotError, SnapshotErrorCode}, + retention::{NodeState, RetentionEdge, RetentionNode, RetentionRoot}, + }, +}; + +const MAX_NODES: usize = 4096; +const MAX_EDGES: usize = 16_384; +const MAX_ROOTS: usize = 16; +const MAX_PENDING_BATCH: u64 = 1000; +pub(crate) const RETENTION_LOCK_KEY: i32 = 1_296_717_362; + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum GcClaim { + /// This transaction marked the node and recorded its pending operation. + Marked, + /// The same operation is committed and awaits physical deletion/ack. + Pending, + /// The same operation already removed the node and released its edges. + Applied, + /// A root, an incoming edge, another claim or absence prevents collection. + Unavailable, +} + +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct GcCompletion { + pub replayed: bool, + /// Newly zero-reference children; roots must still be checked when claimed. + pub zero_reference_children: Vec, +} + +#[derive(Clone)] +pub struct PostgresRetentionRepository { + connection: DatabaseConnection, +} + +impl PostgresRetentionRepository { + pub fn new(connection: DatabaseConnection) -> Self { + Self { connection } + } + + pub async fn node( + &self, + id: &str, + ) -> Result, SnapshotError> { + mst2_retention_node::Entity::find_by_id(id.to_owned()) + .one(&self.connection) + .await + .map_err(internal) + } + + /// Bounded replay scan. PENDING intents survive process/worker replacement. + pub async fn pending_gc( + &self, + limit: u64, + ) -> Result, SnapshotError> { + if limit == 0 || limit > MAX_PENDING_BATCH { + return Err(limit_error("pending GC batch must be 1..=1000")); + } + mst2_retention_gc_op::Entity::find() + .filter(mst2_retention_gc_op::Column::State.eq("PENDING")) + .order_by_asc(mst2_retention_gc_op::Column::CreatedAt) + .order_by_asc(mst2_retention_gc_op::Column::OperationId) + .limit(limit) + .all(&self.connection) + .await + .map_err(internal) + } + + pub async fn retain_group( + &self, + nodes: &[RetentionNode], + edges: &[RetentionEdge], + roots: &[RetentionRoot], + ) -> Result<(), SnapshotError> { + let txn = self.connection.begin().await.map_err(internal)?; + let result = Self::retain_group_in_txn(&txn, nodes, edges, roots).await; + finish(txn, result).await + } + + /// Retain an immutable group atomically. Roots cover each supplied node; + /// callers select that group rather than scanning the reachable graph. + /// An existing parent's edges may only be replayed, not extended; a new + /// identity describes a new graph. + pub async fn retain_group_in_txn( + txn: &DatabaseTransaction, + nodes: &[RetentionNode], + edges: &[RetentionEdge], + roots: &[RetentionRoot], + ) -> Result<(), SnapshotError> { + let group = PreparedGroup::new(nodes, edges, roots)?; + let savepoint = begin_graph(txn).await?; + let result = retain_locked(&savepoint, &group).await; + finish(savepoint, result).await + } + + pub async fn release_root(&self, root: &RetentionRoot) -> Result<(), SnapshotError> { + let txn = self.connection.begin().await.map_err(internal)?; + let result = Self::release_root_in_txn(&txn, root).await; + finish(txn, result).await + } + + /// Attach a root to an already verified immutable DAG root. Incoming edges + /// retain descendants; this does not copy every page for each lease. + pub(crate) async fn acquire_existing_roots_in_txn( + txn: &DatabaseTransaction, + node_id: &str, + roots: &[RetentionRoot], + ) -> Result<(), SnapshotError> { + validate_id(node_id)?; + if roots.is_empty() || roots.len() > MAX_ROOTS { + return Err(limit_error( + "existing-root acquisition requires 1..=16 roots", + )); + } + let identities = roots + .iter() + .map(root_identity) + .collect::, _>>()?; + let roots: Vec<_> = identities + .iter() + .map(|(key, kind)| json!({"key":key,"kind":kind})) + .collect(); + let encoded = serde_json::to_string(&roots).map_err(internal)?; + let savepoint = begin_graph(txn).await?; + let result=async { + let row=savepoint.query_one_raw(statement( + "SELECT state FROM mst2_retention_node WHERE node_id=$1 FOR UPDATE",[node_id.into()] + )).await.map_err(internal)?.ok_or_else(|| unavailable("metadata root is missing"))?; + if row.try_get::("","state").map_err(internal)?!="LIVE" { + return Err(unavailable("metadata root is not LIVE")); + } + if savepoint.query_one_raw(statement( + "SELECT r.node_id FROM mst2_retention_root r JOIN jsonb_to_recordset($2::jsonb) + AS p(key text,kind text) ON r.root_key=p.key WHERE r.node_id<>$1 OR r.root_kind<>p.kind LIMIT 1", + [node_id.into(),encoded.clone().into()], + )).await.map_err(internal)?.is_some() { + return Err(integrity("retention root is bound to another DAG or kind")); + } + savepoint.execute_raw(statement( + "INSERT INTO mst2_retention_root(node_id,root_key,root_kind,created_at) + SELECT $1,p.key,p.kind,now() FROM jsonb_to_recordset($2::jsonb) AS p(key text,kind text) + ON CONFLICT(node_id,root_key) DO NOTHING",[node_id.into(),encoded.into()], + )).await.map_err(internal)?; + Ok(()) + }.await; + finish(savepoint, result).await + } + + pub async fn release_root_in_txn( + txn: &DatabaseTransaction, + root: &RetentionRoot, + ) -> Result<(), SnapshotError> { + let (key, _) = root_identity(root)?; + let savepoint = begin_graph(txn).await?; + let result = savepoint + .execute_raw(statement( + "DELETE FROM mst2_retention_root WHERE root_key = $1", + [key.into()], + )) + .await + .map(|_| ()) + .map_err(internal); + finish(savepoint, result).await + } + + pub async fn mark_deleting( + &self, + operation_id: &str, + node_id: &str, + ) -> Result { + let txn = self.connection.begin().await.map_err(internal)?; + let result = Self::mark_deleting_in_txn(&txn, operation_id, node_id).await; + finish(txn, result).await + } + + /// The LIVE check, zero-reference check, CAS and replay intent share the + /// retention mutation lock. Acquisition either precedes this CAS or fails. + pub async fn mark_deleting_in_txn( + txn: &DatabaseTransaction, + operation_id: &str, + node_id: &str, + ) -> Result { + validate_id(operation_id)?; + validate_id(node_id)?; + let savepoint = begin_graph(txn).await?; + let result = mark_locked(&savepoint, operation_id, node_id).await; + finish(savepoint, result).await + } + + /// Acknowledge an idempotent, durable physical delete. Until this succeeds + /// DELETING parents retain every child. This changes graph rows only. + pub async fn complete_gc(&self, operation_id: &str) -> Result { + let txn = self.connection.begin().await.map_err(internal)?; + let result = Self::complete_gc_in_txn(&txn, operation_id).await; + finish(txn, result).await + } + + pub async fn complete_gc_in_txn( + txn: &DatabaseTransaction, + operation_id: &str, + ) -> Result { + validate_id(operation_id)?; + let savepoint = begin_graph(txn).await?; + let result = complete_locked(&savepoint, operation_id).await; + finish(savepoint, result).await + } +} + +struct PreparedGroup { + nodes_json: String, + edges_json: String, + roots_json: String, +} + +impl PreparedGroup { + fn new( + nodes: &[RetentionNode], + edges: &[RetentionEdge], + roots: &[RetentionRoot], + ) -> Result { + if nodes.len() > MAX_NODES || edges.len() > MAX_EDGES || roots.len() > MAX_ROOTS { + return Err(limit_error("retention group exceeds bounded batch limits")); + } + let mut unique_nodes = BTreeMap::new(); + for node in nodes { + validate_id(&node.id)?; + if node.state != NodeState::Live { + return Err(unavailable("retention acquisition requires LIVE nodes")); + } + let bytes = i64::try_from(node.bytes) + .map_err(|_| limit_error("retention bytes exceed signed database range"))?; + let definition = (node.kind.as_str(), bytes); + if unique_nodes + .insert(node.id.as_str(), definition) + .is_some_and(|old| old != definition) + { + return Err(integrity("conflicting retention node definitions")); + } + } + let mut unique_edges = BTreeSet::new(); + for edge in edges { + validate_id(&edge.parent)?; + validate_id(&edge.child)?; + unique_edges.insert((edge.parent.as_str(), edge.child.as_str())); + } + reject_cycle(&unique_edges)?; + let roots: BTreeSet<_> = roots.iter().map(root_identity).collect::>()?; + Ok(Self { + nodes_json: json!( + unique_nodes + .iter() + .map(|(id, (kind, bytes))| { + json!({"node_id": id, "kind": kind, "bytes": bytes}) + }) + .collect::>() + ) + .to_string(), + edges_json: json!( + unique_edges + .iter() + .map(|(parent, child)| { json!({"parent_id": parent, "child_id": child}) }) + .collect::>() + ) + .to_string(), + roots_json: json!( + roots + .iter() + .map(|(key, kind)| { json!({"root_key": key, "root_kind": kind}) }) + .collect::>() + ) + .to_string(), + }) + } +} + +fn reject_cycle(edges: &BTreeSet<(&str, &str)>) -> Result<(), SnapshotError> { + let mut incoming: BTreeMap<&str, usize> = BTreeMap::new(); + let mut children: BTreeMap<&str, Vec<&str>> = BTreeMap::new(); + for &(parent, child) in edges { + incoming.entry(parent).or_default(); + *incoming.entry(child).or_default() += 1; + children.entry(parent).or_default().push(child); + } + let mut ready: VecDeque<_> = incoming + .iter() + .filter_map(|(&id, &count)| (count == 0).then_some(id)) + .collect(); + let mut processed = 0; + while let Some(id) = ready.pop_front() { + processed += 1; + if let Some(children) = children.get(id) { + for child in children { + if let Some(count) = incoming.get_mut(child) { + *count -= 1; + if *count == 0 { + ready.push_back(child); + } + } + } + } + } + if processed != incoming.len() { + return Err(integrity("retention graph must be acyclic")); + } + Ok(()) +} + +async fn retain_locked( + txn: &DatabaseTransaction, + group: &PreparedGroup, +) -> Result<(), SnapshotError> { + if txn + .query_one_raw(statement( + "SELECT i.node_id FROM jsonb_to_recordset($1::jsonb) AS i(node_id text) \ + WHERE NOT EXISTS (SELECT 1 FROM mst2_retention_node n WHERE n.node_id = i.node_id) \ + AND EXISTS (SELECT 1 FROM mst2_retention_gc_op op WHERE op.node_id = i.node_id) LIMIT 1", + [group.nodes_json.clone().into()], + )) + .await + .map_err(internal)? + .is_some() + { + return Err(unavailable( + "removed node identity requires generation-fenced reconstruction", + )); + } + // Batch comparisons happen before any externally visible commit. An + // identity never changes kind, size or DELETING state on replay. + if txn + .query_one_raw(statement( + "SELECT n.node_id, n.state FROM mst2_retention_node n \ + JOIN jsonb_to_recordset($1::jsonb) AS i(node_id text, kind text, bytes bigint) \ + ON n.node_id = i.node_id \ + WHERE n.state <> 'LIVE' OR n.kind <> i.kind OR n.bytes <> i.bytes LIMIT 1", + [group.nodes_json.clone().into()], + )) + .await + .map_err(internal)? + .is_some() + { + return Err(unavailable( + "existing node is DELETING or conflicts with immutable identity", + )); + } + if txn + .query_one_raw(statement( + "SELECT e.parent_id FROM \ + jsonb_to_recordset($1::jsonb) AS e(parent_id text, child_id text) \ + JOIN mst2_retention_node n ON n.node_id = e.parent_id \ + WHERE NOT EXISTS (SELECT 1 FROM mst2_retention_edge old \ + WHERE old.parent_id = e.parent_id AND old.child_id = e.child_id) LIMIT 1", + [group.edges_json.clone().into()], + )) + .await + .map_err(internal)? + .is_some() + { + return Err(integrity( + "an existing retention parent's edges are immutable", + )); + } + txn.execute_raw(statement( + "INSERT INTO mst2_retention_node (node_id, kind, state, bytes, incoming_refs, created_at) \ + SELECT node_id, kind, 'LIVE', bytes, 0, now() FROM \ + jsonb_to_recordset($1::jsonb) AS i(node_id text, kind text, bytes bigint) \ + ON CONFLICT (node_id) DO NOTHING", + [group.nodes_json.clone().into()], + )) + .await + .map_err(internal)?; + if txn + .query_one_raw(statement( + "SELECT e.parent_id FROM \ + jsonb_to_recordset($1::jsonb) AS e(parent_id text, child_id text) \ + LEFT JOIN mst2_retention_node p ON p.node_id = e.parent_id \ + LEFT JOIN mst2_retention_node c ON c.node_id = e.child_id \ + WHERE p.node_id IS NULL OR c.node_id IS NULL \ + OR p.state <> 'LIVE' OR c.state <> 'LIVE' LIMIT 1", + [group.edges_json.clone().into()], + )) + .await + .map_err(internal)? + .is_some() + { + return Err(unavailable( + "retention edge requires two known LIVE endpoints", + )); + } + txn.execute_raw(statement( + "WITH added AS ( \ + INSERT INTO mst2_retention_edge (parent_id, child_id, created_at) \ + SELECT parent_id, child_id, now() FROM \ + jsonb_to_recordset($1::jsonb) AS e(parent_id text, child_id text) \ + ON CONFLICT (parent_id, child_id) DO NOTHING RETURNING child_id \ + ), delta AS (SELECT child_id, count(*) AS refs FROM added GROUP BY child_id) \ + UPDATE mst2_retention_node n SET incoming_refs = n.incoming_refs + delta.refs \ + FROM delta WHERE n.node_id = delta.child_id", + [group.edges_json.clone().into()], + )) + .await + .map_err(internal)?; + txn.execute_raw(statement( + "INSERT INTO mst2_retention_root (node_id, root_key, root_kind, created_at) \ + SELECT n.node_id, r.root_key, r.root_kind, now() FROM \ + jsonb_to_recordset($1::jsonb) AS n(node_id text) CROSS JOIN \ + jsonb_to_recordset($2::jsonb) AS r(root_key text, root_kind text) \ + ON CONFLICT (node_id, root_key) DO NOTHING", + [ + group.nodes_json.clone().into(), + group.roots_json.clone().into(), + ], + )) + .await + .map_err(internal)?; + Ok(()) +} + +async fn mark_locked( + txn: &DatabaseTransaction, + operation_id: &str, + node_id: &str, +) -> Result { + if let Some(op) = mst2_retention_gc_op::Entity::find_by_id(operation_id.to_owned()) + .one(txn) + .await + .map_err(internal)? + { + if op.node_id != node_id || op.operation != "REMOVE" { + return Err(integrity("GC operation id is bound to different work")); + } + return match op.state.as_str() { + "PENDING" => Ok(GcClaim::Pending), + "APPLIED" => Ok(GcClaim::Applied), + _ => Err(integrity("GC operation is not replayable")), + }; + } + let row = txn + .query_one_raw(statement( + "SELECT n.state, n.incoming_refs, \ + (SELECT count(*) FROM mst2_retention_edge e WHERE e.child_id = n.node_id) AS actual_refs \ + FROM mst2_retention_node n WHERE n.node_id = $1", + [node_id.into()], + )) + .await + .map_err(internal)?; + let Some(row) = row else { + return Ok(GcClaim::Unavailable); + }; + let refs: i64 = row.try_get("", "incoming_refs").map_err(internal)?; + let actual: i64 = row.try_get("", "actual_refs").map_err(internal)?; + if refs != actual || refs < 0 { + return Err(integrity("retention reference audit failed; GC stopped")); + } + if row.try_get::("", "state").map_err(internal)? != "LIVE" || refs != 0 { + return Ok(GcClaim::Unavailable); + } + let changed = txn + .execute_raw(statement( + "UPDATE mst2_retention_node n SET state = 'DELETING' \ + WHERE node_id = $1 AND state = 'LIVE' AND incoming_refs = 0 \ + AND NOT EXISTS (SELECT 1 FROM mst2_retention_root r WHERE r.node_id = n.node_id) \ + AND NOT EXISTS (SELECT 1 FROM mst2_retention_edge e WHERE e.child_id = n.node_id)", + [node_id.into()], + )) + .await + .map_err(internal)?; + if changed.rows_affected() == 0 { + return Ok(GcClaim::Unavailable); + } + txn.execute_raw(statement( + "INSERT INTO mst2_retention_gc_op (operation_id, node_id, operation, state, attempts, created_at) \ + VALUES ($1, $2, 'REMOVE', 'PENDING', 0, now())", + [operation_id.into(), node_id.into()], + )) + .await + .map_err(internal)?; + Ok(GcClaim::Marked) +} + +async fn complete_locked( + txn: &DatabaseTransaction, + operation_id: &str, +) -> Result { + let op = mst2_retention_gc_op::Entity::find_by_id(operation_id.to_owned()) + .one(txn) + .await + .map_err(internal)? + .ok_or_else(|| integrity("unknown GC operation"))?; + if op.operation != "REMOVE" { + return Err(integrity("GC operation is not a removal")); + } + if op.state == "APPLIED" { + return Ok(GcCompletion { + replayed: true, + zero_reference_children: Vec::new(), + }); + } + if op.state != "PENDING" { + return Err(integrity("GC operation is not replayable")); + } + let node = mst2_retention_node::Entity::find_by_id(op.node_id.clone()) + .one(txn) + .await + .map_err(internal)? + .ok_or_else(|| integrity("pending GC node is missing"))?; + if node.state != "DELETING" || node.incoming_refs != 0 { + return Err(integrity("pending GC node is not unreferenced DELETING")); + } + if txn + .query_one_raw(statement( + "SELECT node_id FROM mst2_retention_root WHERE node_id = $1 \ + UNION ALL SELECT child_id FROM mst2_retention_edge WHERE child_id = $1 LIMIT 1", + [op.node_id.clone().into()], + )) + .await + .map_err(internal)? + .is_some() + { + return Err(integrity("pending GC node gained a reference; GC stopped")); + } + // Audit before subtracting. Even a damaged stored counter must not cause + // a child's data to be treated as unreferenced. + if txn.query_one_raw(statement( + "SELECT c.node_id FROM mst2_retention_edge e \ + JOIN mst2_retention_node c ON c.node_id = e.child_id \ + WHERE e.parent_id = $1 AND (c.state <> 'LIVE' OR c.incoming_refs <= 0 OR \ + c.incoming_refs <> (SELECT count(*) FROM mst2_retention_edge i WHERE i.child_id = c.node_id)) \ + LIMIT 1", + [op.node_id.clone().into()], + )).await.map_err(internal)?.is_some() { + return Err(integrity("child reference audit failed; GC stopped")); + } + let rows = txn.query_all_raw(statement( + "WITH removed AS (DELETE FROM mst2_retention_edge WHERE parent_id = $1 RETURNING child_id), \ + delta AS (SELECT child_id, count(*) AS refs FROM removed GROUP BY child_id) \ + UPDATE mst2_retention_node n SET incoming_refs = n.incoming_refs - delta.refs \ + FROM delta WHERE n.node_id = delta.child_id RETURNING n.node_id, n.incoming_refs", + [op.node_id.clone().into()], + )).await.map_err(internal)?; + let mut zero_reference_children = Vec::new(); + for row in rows { + if row.try_get::("", "incoming_refs").map_err(internal)? == 0 { + zero_reference_children.push(row.try_get("", "node_id").map_err(internal)?); + } + } + zero_reference_children.sort(); + txn.execute_raw(statement( + "DELETE FROM mst2_retention_node WHERE node_id = $1 AND state = 'DELETING'", + [op.node_id.into()], + )) + .await + .map_err(internal)?; + txn.execute_raw(statement( + "UPDATE mst2_retention_gc_op SET state = 'APPLIED', completed_at = now(), attempts = attempts + 1 \ + WHERE operation_id = $1 AND state = 'PENDING'", + [operation_id.into()], + )).await.map_err(internal)?; + Ok(GcCompletion { + replayed: false, + zero_reference_children, + }) +} + +async fn begin_graph(txn: &DatabaseTransaction) -> Result { + if txn.get_database_backend() != DbBackend::Postgres { + return Err(internal("retention repository requires PostgreSQL")); + } + let isolation = txn + .query_one_raw(statement("SHOW transaction_isolation", [])) + .await + .map_err(internal)? + .ok_or_else(|| internal("transaction isolation query returned no row"))? + .try_get_by_index::(0) + .map_err(internal)?; + if isolation != "read committed" { + return Err(internal( + "retention mutations require READ COMMITTED transaction isolation", + )); + } + let savepoint = txn.begin().await.map_err(internal)?; + let result = savepoint + .execute_raw(statement( + "SELECT pg_advisory_xact_lock($1, hashtext(current_schema()))", + [RETENTION_LOCK_KEY.into()], + )) + .await + .map_err(internal); + if let Err(err) = result { + return finish(savepoint, Err(err)).await; + } + Ok(savepoint) +} + +async fn finish( + txn: DatabaseTransaction, + result: Result, +) -> Result { + match result { + Ok(value) => { + txn.commit().await.map_err(internal)?; + Ok(value) + } + Err(err) => { + txn.rollback().await.map_err(internal)?; + Err(err) + } + } +} + +fn statement(sql: &str, values: [sea_orm::Value; N]) -> Statement { + Statement::from_sql_and_values(DbBackend::Postgres, sql, values) +} + +fn validate_id(id: &str) -> Result<(), SnapshotError> { + if id.is_empty() || id.len() > 255 || id.contains('\0') { + return Err(limit_error( + "retention identifiers require 1..=255 bytes without NUL", + )); + } + Ok(()) +} + +fn root_identity(root: &RetentionRoot) -> Result<(String, &'static str), SnapshotError> { + let (id, kind) = match root { + RetentionRoot::Lease(id) => (id, "lease"), + RetentionRoot::Pin(id) => (id, "pin"), + RetentionRoot::Prepare(id) => (id, "prepare"), + }; + validate_id(id)?; + let key = format!("{kind}:{id}"); + validate_id(&key)?; + Ok((key, kind)) +} + +fn internal(error: impl std::fmt::Display) -> SnapshotError { + SnapshotError::new(SnapshotErrorCode::Internal, error.to_string()) +} +fn integrity(message: &str) -> SnapshotError { + SnapshotError::new(SnapshotErrorCode::IntegrityError, message) +} +fn unavailable(message: &str) -> SnapshotError { + SnapshotError::new(SnapshotErrorCode::ObjectUnavailable, message) +} +fn limit_error(message: &str) -> SnapshotError { + SnapshotError::new(SnapshotErrorCode::LimitExceeded, message) +} + +#[cfg(test)] +#[path = "mst2_retention_tests.rs"] +mod tests; diff --git a/src/jupiter/storage/mst2_retention_tests.rs b/src/jupiter/storage/mst2_retention_tests.rs new file mode 100644 index 00000000..a8f7c866 --- /dev/null +++ b/src/jupiter/storage/mst2_retention_tests.rs @@ -0,0 +1,629 @@ +use std::time::Duration; + +use sea_orm::{ + ConnectionTrait, DatabaseConnection, EntityTrait, IsolationLevel, PaginatorTrait, + TransactionTrait, +}; +use sea_orm_migration::MigratorTrait; + +use super::*; +use crate::{ + callisto::{mst2_retention_edge, mst2_retention_root}, + ceres::snapshot::retention::RetainedKind, + jupiter::{migration::Migrator, tests::test_db_connection}, +}; + +fn node(id: &str) -> RetentionNode { + RetentionNode { + id: id.into(), + kind: RetainedKind::Page, + state: NodeState::Live, + bytes: 7, + } +} + +fn edge(parent: &str, child: &str) -> RetentionEdge { + RetentionEdge { + parent: parent.into(), + child: child.into(), + } +} + +async fn fixture() -> (DatabaseConnection, PostgresRetentionRepository) { + let temp = tempfile::TempDir::new().expect("temp directory"); + let db = test_db_connection(temp.path()).await; + Migrator::up(&db, None) + .await + .expect("isolated schema migrations"); + let repository = PostgresRetentionRepository::new(db.clone()); + (db, repository) +} + +#[test] +fn t06b_rejects_cycles_conflicting_identities_and_unbounded_groups() { + let a = node("a"); + let b = node("b"); + assert_eq!( + PreparedGroup::new(&[a.clone(), b], &[edge("a", "b"), edge("b", "a")], &[]) + .err() + .expect("cycle rejected") + .code, + SnapshotErrorCode::IntegrityError + ); + let mut conflicting = a.clone(); + conflicting.bytes += 1; + assert!(PreparedGroup::new(&[a.clone(), conflicting], &[], &[]).is_err()); + assert_eq!( + PreparedGroup::new(&vec![a; MAX_NODES + 1], &[], &[]) + .err() + .expect("batch rejected") + .code, + SnapshotErrorCode::LimitExceeded + ); +} + +#[tokio::test] +async fn t06b_retains_unique_edges_and_independent_roots_once() { + let (db, repository) = fixture().await; + let roots = [ + RetentionRoot::Lease("one".into()), + RetentionRoot::Pin("two".into()), + ]; + let nodes = [node("parent"), node("child")]; + let edges = [edge("parent", "child"), edge("parent", "child")]; + for _ in 0..2 { + repository + .retain_group(&nodes, &edges, &roots) + .await + .expect("idempotent retain"); + } + assert_eq!( + repository + .node("child") + .await + .unwrap() + .unwrap() + .incoming_refs, + 1 + ); + assert_eq!( + mst2_retention_edge::Entity::find() + .count(&db) + .await + .unwrap(), + 1 + ); + assert_eq!( + mst2_retention_root::Entity::find() + .count(&db) + .await + .unwrap(), + 4 + ); + repository.release_root(&roots[0]).await.unwrap(); + repository.release_root(&roots[0]).await.unwrap(); + assert_eq!( + mst2_retention_root::Entity::find() + .count(&db) + .await + .unwrap(), + 2 + ); + assert_eq!( + repository + .mark_deleting("gc-parent", "parent") + .await + .unwrap(), + GcClaim::Unavailable + ); + repository.release_root(&roots[1]).await.unwrap(); + assert_eq!( + repository + .mark_deleting("gc-parent", "parent") + .await + .unwrap(), + GcClaim::Marked + ); + assert_eq!( + repository.mark_deleting("gc-child", "child").await.unwrap(), + GcClaim::Unavailable + ); +} + +#[tokio::test] +async fn t06b_group_failure_rolls_back_savepoint_even_if_caller_commits() { + let (db, repository) = fixture().await; + repository + .retain_group(&[node("old")], &[], &[]) + .await + .unwrap(); + assert_eq!( + repository.mark_deleting("old-gc", "old").await.unwrap(), + GcClaim::Marked + ); + let txn = db.begin().await.unwrap(); + let error = PostgresRetentionRepository::retain_group_in_txn( + &txn, + &[node("new")], + &[edge("new", "old")], + &[RetentionRoot::Lease("partial".into())], + ) + .await + .expect_err("DELETING child rejects whole group"); + assert_eq!(error.code, SnapshotErrorCode::ObjectUnavailable); + txn.commit().await.unwrap(); + assert!(repository.node("new").await.unwrap().is_none()); + assert_eq!( + repository.node("old").await.unwrap().unwrap().state, + "DELETING" + ); + assert_eq!( + mst2_retention_root::Entity::find() + .count(&db) + .await + .unwrap(), + 0 + ); + assert_eq!( + mst2_retention_edge::Entity::find() + .count(&db) + .await + .unwrap(), + 0 + ); +} + +#[tokio::test] +async fn t06b_publication_rollback_removes_retention_acquisition() { + let (db, repository) = fixture().await; + let txn = db.begin().await.unwrap(); + PostgresRetentionRepository::retain_group_in_txn( + &txn, + &[node("publication")], + &[], + &[RetentionRoot::Prepare("writer".into())], + ) + .await + .unwrap(); + txn.rollback().await.unwrap(); + assert!(repository.node("publication").await.unwrap().is_none()); + assert_eq!( + mst2_retention_root::Entity::find() + .count(&db) + .await + .unwrap(), + 0 + ); +} + +#[tokio::test] +async fn t06b_root_release_rollback_preserves_coverage() { + let (db, repository) = fixture().await; + let root = RetentionRoot::Lease("reader".into()); + repository + .retain_group(&[node("page")], &[], std::slice::from_ref(&root)) + .await + .unwrap(); + let txn = db.begin().await.unwrap(); + PostgresRetentionRepository::release_root_in_txn(&txn, &root) + .await + .unwrap(); + txn.rollback().await.unwrap(); + assert_eq!( + repository.mark_deleting("page-op", "page").await.unwrap(), + GcClaim::Unavailable + ); + assert_eq!( + mst2_retention_root::Entity::find() + .count(&db) + .await + .unwrap(), + 1 + ); +} + +#[tokio::test] +async fn t06b_rejects_snapshot_isolation_without_mutating_outer_transaction() { + let (db, repository) = fixture().await; + let txn = db + .begin_with_config(Some(IsolationLevel::RepeatableRead), None) + .await + .unwrap(); + let error = PostgresRetentionRepository::retain_group_in_txn( + &txn, + &[node("page")], + &[], + &[RetentionRoot::Lease("reader".into())], + ) + .await + .expect_err("old MVCC snapshots cannot safely follow an advisory lock"); + assert_eq!(error.code, SnapshotErrorCode::Internal); + assert!(error.message.contains("READ COMMITTED")); + txn.commit().await.unwrap(); + assert!(repository.node("page").await.unwrap().is_none()); + assert_eq!( + mst2_retention_root::Entity::find() + .count(&db) + .await + .unwrap(), + 0 + ); +} + +#[tokio::test] +async fn t06b_existing_parent_and_node_definition_are_immutable() { + let (db, repository) = fixture().await; + repository + .retain_group(&[node("parent"), node("child")], &[], &[]) + .await + .unwrap(); + let mut changed = node("parent"); + changed.bytes += 1; + assert!(repository.retain_group(&[changed], &[], &[]).await.is_err()); + assert!( + repository + .retain_group(&[node("parent")], &[edge("parent", "child")], &[]) + .await + .is_err() + ); + assert_eq!(repository.node("parent").await.unwrap().unwrap().bytes, 7); + assert_eq!( + mst2_retention_edge::Entity::find() + .count(&db) + .await + .unwrap(), + 0 + ); +} + +#[tokio::test] +async fn t06b_gc_replay_subtracts_edges_once_and_tombstones_removed_identity() { + let (db, repository) = fixture().await; + repository + .retain_group( + &[node("parent"), node("child")], + &[edge("parent", "child")], + &[], + ) + .await + .unwrap(); + assert_eq!( + repository + .mark_deleting("parent-op", "parent") + .await + .unwrap(), + GcClaim::Marked + ); + let restarted = PostgresRetentionRepository::new(db.clone()); + let pending = restarted.pending_gc(10).await.unwrap(); + assert_eq!(pending.len(), 1); + assert_eq!(pending[0].operation_id, "parent-op"); + assert_eq!( + restarted + .mark_deleting("parent-op", "parent") + .await + .unwrap(), + GcClaim::Pending + ); + assert_eq!( + restarted + .mark_deleting("another-op", "parent") + .await + .unwrap(), + GcClaim::Unavailable + ); + assert!( + restarted + .mark_deleting("parent-op", "different") + .await + .is_err() + ); + + // Simulated crash/rollback during bookkeeping: the PENDING intent and + // child's reference survive and can be replayed after reaper replacement. + let txn = db.begin().await.unwrap(); + let provisional = PostgresRetentionRepository::complete_gc_in_txn(&txn, "parent-op") + .await + .unwrap(); + assert_eq!(provisional.zero_reference_children, vec!["child"]); + txn.rollback().await.unwrap(); + assert_eq!( + restarted + .node("child") + .await + .unwrap() + .unwrap() + .incoming_refs, + 1 + ); + assert_eq!(restarted.pending_gc(10).await.unwrap().len(), 1); + assert_eq!( + restarted.node("parent").await.unwrap().unwrap().state, + "DELETING" + ); + + let completed = restarted.complete_gc("parent-op").await.unwrap(); + assert!(!completed.replayed); + assert_eq!(completed.zero_reference_children, vec!["child"]); + assert!(restarted.node("parent").await.unwrap().is_none()); + assert_eq!( + restarted + .node("child") + .await + .unwrap() + .unwrap() + .incoming_refs, + 0 + ); + assert!(restarted.complete_gc("parent-op").await.unwrap().replayed); + assert_eq!( + restarted + .mark_deleting("parent-op", "parent") + .await + .unwrap(), + GcClaim::Applied + ); + assert!(restarted.pending_gc(10).await.unwrap().is_empty()); + let receipt = mst2_retention_gc_op::Entity::find_by_id("parent-op".to_owned()) + .one(&db) + .await + .unwrap() + .unwrap(); + assert_eq!(receipt.attempts, 1); + assert!(receipt.completed_at.is_some()); + + // Old pending workers must not be able to delete same-id reconstructed + // bytes. Reconstruction requires a future generation/fencing protocol. + assert_eq!( + restarted + .retain_group(&[node("parent")], &[], &[]) + .await + .unwrap_err() + .code, + SnapshotErrorCode::ObjectUnavailable + ); + assert!(restarted.complete_gc("parent-op").await.unwrap().replayed); + assert!(restarted.node("parent").await.unwrap().is_none()); +} + +#[tokio::test] +async fn t06b_counter_audit_stops_gc_and_preserves_pending_edges() { + let (db, repository) = fixture().await; + repository + .retain_group( + &[node("parent"), node("child")], + &[edge("parent", "child")], + &[], + ) + .await + .unwrap(); + db.execute_unprepared( + "UPDATE mst2_retention_node SET incoming_refs = 0 WHERE node_id = 'child'", + ) + .await + .unwrap(); + assert_eq!( + repository + .mark_deleting("child-op", "child") + .await + .unwrap_err() + .code, + SnapshotErrorCode::IntegrityError + ); + assert_eq!( + repository + .mark_deleting("parent-op", "parent") + .await + .unwrap(), + GcClaim::Marked + ); + assert_eq!( + repository.complete_gc("parent-op").await.unwrap_err().code, + SnapshotErrorCode::IntegrityError + ); + assert_eq!( + mst2_retention_edge::Entity::find() + .count(&db) + .await + .unwrap(), + 1 + ); + assert_eq!(repository.pending_gc(10).await.unwrap().len(), 1); + assert_eq!( + repository.node("child").await.unwrap().unwrap().state, + "LIVE" + ); + db.execute_unprepared( + "UPDATE mst2_retention_node SET incoming_refs = 1 WHERE node_id = 'child'", + ) + .await + .unwrap(); + repository.complete_gc("parent-op").await.unwrap(); + assert_eq!( + repository + .node("child") + .await + .unwrap() + .unwrap() + .incoming_refs, + 0 + ); +} + +#[tokio::test] +async fn t06b_root_acquisition_wins_before_gc_cas() { + let (db, repository) = fixture().await; + repository + .retain_group(&[node("page")], &[], &[]) + .await + .unwrap(); + let root_txn = db.begin().await.unwrap(); + PostgresRetentionRepository::retain_group_in_txn( + &root_txn, + &[node("page")], + &[], + &[RetentionRoot::Lease("winner".into())], + ) + .await + .unwrap(); + let (started_tx, started_rx) = tokio::sync::oneshot::channel(); + let gc_repository = repository.clone(); + let mut gc = tokio::spawn(async move { + started_tx.send(()).unwrap(); + gc_repository.mark_deleting("gc-op", "page").await + }); + started_rx.await.unwrap(); + assert!( + tokio::time::timeout(Duration::from_millis(50), &mut gc) + .await + .is_err() + ); + root_txn.commit().await.unwrap(); + assert_eq!( + tokio::time::timeout(Duration::from_secs(5), gc) + .await + .unwrap() + .unwrap() + .unwrap(), + GcClaim::Unavailable + ); + assert_eq!( + repository.node("page").await.unwrap().unwrap().state, + "LIVE" + ); + assert!(repository.pending_gc(10).await.unwrap().is_empty()); +} + +#[tokio::test] +async fn t06b_gc_cas_wins_and_new_root_cannot_reference_deleting_node() { + let (db, repository) = fixture().await; + repository + .retain_group(&[node("page")], &[], &[]) + .await + .unwrap(); + let gc_txn = db.begin().await.unwrap(); + assert_eq!( + PostgresRetentionRepository::mark_deleting_in_txn(&gc_txn, "gc-op", "page") + .await + .unwrap(), + GcClaim::Marked + ); + let (started_tx, started_rx) = tokio::sync::oneshot::channel(); + let acquire_repository = repository.clone(); + let mut acquire = tokio::spawn(async move { + started_tx.send(()).unwrap(); + acquire_repository + .retain_group(&[node("page")], &[], &[RetentionRoot::Lease("late".into())]) + .await + }); + started_rx.await.unwrap(); + assert!( + tokio::time::timeout(Duration::from_millis(50), &mut acquire) + .await + .is_err() + ); + gc_txn.commit().await.unwrap(); + assert_eq!( + tokio::time::timeout(Duration::from_secs(5), acquire) + .await + .unwrap() + .unwrap() + .unwrap_err() + .code, + SnapshotErrorCode::ObjectUnavailable + ); + assert_eq!( + mst2_retention_root::Entity::find() + .count(&db) + .await + .unwrap(), + 0 + ); + assert_eq!(repository.pending_gc(10).await.unwrap().len(), 1); +} + +#[tokio::test] +async fn t06b_forward_migration_backfills_old_graph_and_enforces_constraints() { + let temp = tempfile::TempDir::new().unwrap(); + let db = test_db_connection(temp.path()).await; + let at = Migrator::migrations() + .iter() + .position(|migration| migration.name() == "m20261005_000200_harden_mst2_retention_graph") + .expect("hardening migration registered"); + Migrator::up(&db, Some(at.try_into().unwrap())) + .await + .unwrap(); + db.execute_unprepared( + "INSERT INTO mst2_retention_node (node_id, kind, state, bytes, created_at) \ + VALUES ('parent', 'page', 'LIVE', 7, now()), ('child', 'page', 'LIVE', 7, now()); \ + INSERT INTO mst2_retention_edge (parent_id, child_id, created_at) VALUES ('parent', 'child', now())", + ) + .await + .unwrap(); + Migrator::up(&db, None).await.unwrap(); + let repository = PostgresRetentionRepository::new(db.clone()); + assert_eq!( + repository + .node("child") + .await + .unwrap() + .unwrap() + .incoming_refs, + 1 + ); + assert!( + db.execute_unprepared( + "INSERT INTO mst2_retention_edge (parent_id, child_id, created_at) VALUES ('missing', 'child', now())" + ) + .await + .is_err() + ); + assert!( + db.execute_unprepared( + "UPDATE mst2_retention_node SET incoming_refs = -1 WHERE node_id = 'child'" + ) + .await + .is_err() + ); + let completion_error = db + .execute_unprepared( + "INSERT INTO mst2_retention_gc_op (operation_id, node_id, operation, state, attempts, created_at) \ + VALUES ('bad', 'child', 'REMOVE', 'APPLIED', 0, now())", + ) + .await + .expect_err("APPLIED receipt without completed_at must violate its completion constraint"); + assert!( + completion_error + .to_string() + .contains("mst2_retention_gc_op_completion_check"), + "unexpected constraint failure: {completion_error}" + ); +} + +#[tokio::test] +async fn t06b_forward_migration_refuses_existing_cycles() { + let temp = tempfile::TempDir::new().unwrap(); + let db = test_db_connection(temp.path()).await; + let at = Migrator::migrations() + .iter() + .position(|migration| migration.name() == "m20261005_000200_harden_mst2_retention_graph") + .unwrap(); + Migrator::up(&db, Some(at.try_into().unwrap())) + .await + .unwrap(); + db.execute_unprepared( + "INSERT INTO mst2_retention_node (node_id, kind, state, bytes, created_at) \ + VALUES ('a', 'page', 'LIVE', 7, now()), ('b', 'page', 'LIVE', 7, now()); \ + INSERT INTO mst2_retention_edge (parent_id, child_id, created_at) VALUES ('a', 'b', now()), ('b', 'a', now())", + ) + .await + .unwrap(); + assert!(Migrator::up(&db, None).await.is_err()); + assert_eq!( + mst2_retention_edge::Entity::find() + .count(&db) + .await + .unwrap(), + 2 + ); +} diff --git a/src/jupiter/storage/native_chunk_map.rs b/src/jupiter/storage/native_chunk_map.rs new file mode 100644 index 00000000..1d5e5666 --- /dev/null +++ b/src/jupiter/storage/native_chunk_map.rs @@ -0,0 +1,748 @@ +//! Source-bound chunk-map receipts with indexed, authenticated selected pages. +//! +//! The object writer is trusted like the verified-object writer. A database +//! row, its checksum or a map's presence is never a full-source proof. Only +//! the opaque full-stream verifier can install a source receipt. Persistent +//! owners, reservations and physical generations bound its retention. + +use std::sync::Arc; + +use bytes::Bytes; +use futures::StreamExt; +use mst2_codec::chunkmap::{CHUNKS_PER_PAGE, ChunkLeaf, ChunkMap, ProofStep, verify_leaf}; +use sea_orm::{ + ConnectionTrait, DatabaseConnection, DatabaseTransaction, DbBackend, FromQueryResult, + IsolationLevel, QueryResult, Statement, TransactionTrait, +}; +use serde_json::json; +use sha2::{Digest, Sha256}; + +use super::object_storage::MegaObjectStorageWrapper; +use crate::{ + callisto::mst2_verified_object, + ceres::snapshot::{ + chunk_map_index::{ChunkMapNode, indexed_nodes, proof_intervals, selected_proof}, + chunks::{ChunkMapSource, VerifiedSourceChunkMap}, + content_budget::{MemoryBudget, MemoryLease, projection_budget}, + error::{SnapshotError, SnapshotErrorCode}, + }, + orbit_api::object_storage::{ObjectKey, ObjectMeta, ObjectNamespace}, +}; + +const DESCRIPTOR_CREDIT: usize = 32 * 1024; +const PAGE_CREDIT: usize = 64 * 1024; +const RECEIPT_DOMAIN: &[u8] = b"MST2-CHUNK-MAP-RECEIPT\0"; + +pub(crate) mod retention; + +#[derive(Clone)] +pub(crate) struct PostgresChunkMapRepository { + connection: DatabaseConnection, + primary_scope: Vec, + budget: Arc, + schema: String, + receipt_observation: Arc>>>, + cancelled_installs: Arc>>, + admission_maintenance: Arc>>, +} + +pub(crate) struct PersistedChunkMap { + pub map: ChunkMap, + pub map_id: [u8; 32], + source: ChunkMapSource, + source_id: [u8; 32], + reader: retention::ChunkMapReader, + _memory: MemoryLease, +} + +impl PersistedChunkMap { + pub(crate) fn source_id(&self) -> [u8; 32] { + self.source_id + } + + pub(crate) async fn ensure_live(&self) -> Result<(), SnapshotError> { + self.reader.ensure_live().await + } + + pub(crate) async fn record_progress(&self) -> Result<(), SnapshotError> { + self.reader.record_progress().await + } + + pub(crate) async fn await_backend( + &self, + future: F, + ) -> Result { + self.reader.await_backend(future).await + } + + #[cfg(test)] + pub(crate) async fn test_check_next_owner_operation(&self) { + self.reader.test_check_next_owner_operation().await; + } +} + +pub(crate) struct AuthenticatedChunkPage { + pub leaf: ChunkLeaf, + pub proof: Vec, + _memory: MemoryLease, +} + +impl AuthenticatedChunkPage { + pub(crate) fn verify_chunk( + &self, + map: &ChunkMap, + index: u64, + bytes: &[u8], + ) -> Result<(), SnapshotError> { + let length = map + .chunk_len(index) + .map_err(|_| integrity("invalid requested chunk index"))?; + if index / CHUNKS_PER_PAGE as u64 != self.leaf.page_index || bytes.len() as u64 != length { + return Err(integrity( + "exact range length or selected chunk page disagrees", + )); + } + let slot = (index % CHUNKS_PER_PAGE as u64) as usize; + let expected = self + .leaf + .chunk_sha256 + .get(slot) + .ok_or_else(|| integrity("selected chunk digest is missing"))?; + let digest: [u8; 32] = Sha256::digest(bytes).into(); + if &digest != expected { + return Err(integrity( + "exact range digest disagrees with the authenticated selected page", + )); + } + Ok(()) + } +} + +impl PostgresChunkMapRepository { + pub(crate) async fn new(connection: DatabaseConnection) -> Result { + let schema = capture_schema(&connection).await?; + let primary_scope = read_scope(&connection, &schema).await?; + Ok(Self { + connection, + primary_scope, + budget: projection_budget().clone(), + schema, + receipt_observation: Arc::default(), + cancelled_installs: Arc::default(), + admission_maintenance: Arc::default(), + }) + } + + pub(crate) async fn read( + &self, + source: &ChunkMapSource, + objects: &MegaObjectStorageWrapper, + ) -> Result>, SnapshotError> { + for pass in 0..2 { + let (map, pending) = self.read_current(source, objects).await?; + if !pending { + return Ok(map); + } + if pass == 0 { + self.maintain(objects, 8).await?; + } + } + Err(SnapshotError::new( + SnapshotErrorCode::TemporaryUnavailable, + "previous source receipt deletion is still pending", + )) + } + + async fn read_current( + &self, + source: &ChunkMapSource, + objects: &MegaObjectStorageWrapper, + ) -> Result<(Option>, bool), SnapshotError> { + let memory = self.budget.reserve(DESCRIPTOR_CREDIT)?; + let txn = self.transaction().await?; + let result = async { + self.require_scope(&txn).await?; + require_current_source(&txn, source, &self.schema).await?; + let Some(row) = source_row(&txn, source, &self.schema).await? else { + let pending = self.require_missing_source_retired(&txn, source).await?; + return Ok((None, pending)); + }; + let (map, source_id) = self.validate_source_row(source, &row)?; + let generation_state: Option = + row.try_get("", "generation_state").map_err(db_error)?; + if generation_state.as_deref() == Some("DELETING") { + // A legitimate retirement can still retain its source row + // until the physical delete completes. Replay it just like + // a missing retired row, after checking every source fact; + // damaged LIVE rows never enter this cold-rebuild path. + return Ok((None, true)); + } + let bytes = receipt(&self.primary_scope, source, &map)?; + let reader = self.admit_reader(&txn, source, &row, &map).await?; + let key = ObjectKey { + namespace: ObjectNamespace::ChunkMapReceipt, + key: row.try_get("", "receipt_key").map_err(db_error)?, + }; + Ok(( + Some(( + Arc::new(PersistedChunkMap { + map_id: map.map_id(), + map, + source: source.clone(), + source_id, + reader, + _memory: memory, + }), + key, + bytes, + )), + false, + )) + } + .await; + let (admitted, pending) = finish(txn, result).await?; + let Some((map, key, bytes)) = admitted else { + return Ok((None, pending)); + }; + // Admission is durable before I/O. No completion barrier or fact row + // lock is held while the independent backend receipt is consumed. + map.await_backend(read_receipt(objects, &key, &bytes)) + .await??; + map.reader.validate_after_receipt().await?; + Ok((Some(map), false)) + } + + /// Publication accepts no caller-provided map or "verified" flag. + pub(crate) async fn install( + &self, + verified: VerifiedSourceChunkMap, + objects: &MegaObjectStorageWrapper, + admission: &retention::ChunkMapInstall, + ) -> Result<(), SnapshotError> { + let result = self.install_reserved(verified, objects, admission).await; + if result.is_err() { + if let Err(error) = admission.retire().await { + tracing::warn!( + ?error, + "failed chunk-map installation will retire by database deadline" + ); + } else if let Err(error) = self.collect_maps(64).await { + tracing::warn!( + ?error, + "retired partial chunk-map indexes await bounded collection replay" + ); + } + } + result + } + + async fn install_reserved( + &self, + verified: VerifiedSourceChunkMap, + objects: &MegaObjectStorageWrapper, + admission: &retention::ChunkMapInstall, + ) -> Result<(), SnapshotError> { + let source = verified.source(); + let map = verified.map(); + let nodes = indexed_nodes(verified.leaf_hashes())?; + if nodes.last().map(|node| node.digest) != Some(map.pages_root) { + return Err(integrity( + "verified chunk map index disagrees with canonical root", + )); + } + let bytes = admission.prepare(&verified).await?; + let key = admission.key(); + admission + .await_backend(objects.inner.put_metadata_atomic_create( + key, + Bytes::copy_from_slice(&bytes), + ObjectMeta { + size: bytes.len() as i64, + ..Default::default() + }, + )) + .await? + .map_err(storage_error)?; + admission.acknowledge_create().await?; + admission + .await_backend(read_receipt(objects, key, &bytes)) + .await??; + let observation = retention::inventory(self, objects).await?; + let map_generation = self + .stage_map_start(&verified, admission, &observation) + .await?; + for batch in verified.leaves().chunks(64) { + self.stage_map_batch(&verified, admission, map_generation, Some(batch), None) + .await?; + } + for batch in nodes.chunks(1024) { + self.stage_map_batch(&verified, admission, map_generation, None, Some(batch)) + .await?; + } + self.compare_staged_map(&verified, &nodes, admission, map_generation) + .await?; + let txn = self.transaction().await?; + let result=async { + self.require_scope(&txn).await?; + require_current_source(&txn,source,&self.schema).await?; + retention::map_barrier(&txn,map.map_id()).await?; + retention::barrier(&txn,&self.schema).await?; + retention::ensure_quotas(&txn,&self.schema,&observation,0,0,0,Some(admission.owner())).await?; + let admitted=txn.query_one_raw(stmt(&self.schema,"UPDATE mst2_chunk_receipt_generation SET state='LIVE',owner=NULL,deadline=NULL,reserved_pages=0,reserved_nodes=0,reserved_bytes=0,last_progress=pg_catalog.clock_timestamp() WHERE receipt_key=$1 AND generation=$2 AND owner=$3::uuid AND state='CREATING' AND deadline>pg_catalog.clock_timestamp() AND create_completed AND map_id=$4 AND map_generation=$5 AND receipt_bytes=$6 AND EXISTS(SELECT 1 FROM mst2_chunk_map_lifetime WHERE map_id=$4 AND generation=$5 AND state='LIVE') RETURNING generation",[key.key.clone().into(),admission.generation().into(),admission.owner().to_string().into(),map.map_id().to_vec().into(),map_generation.into(),bytes.clone().into()])).await.map_err(db_error)?; + if admitted.is_none() {return Err(SnapshotError::new(SnapshotErrorCode::LeaseExpired,"chunk-map publication lost its exact live install reservation"));} + let fact=source.fact(); + let source_bytes=source.canonical_bytes()?; + let source_id=source_id(&self.primary_scope,&source_bytes); + txn.execute_raw(stmt(&self.schema,"INSERT INTO mst2_chunk_map_source(storage_domain,git_oid,object_kind,fact_id,source_id,source_bytes,primary_scope,map_id,receipt_digest,receipt_generation) VALUES($1,$2,$3,$4,$5,$6,$7,$8,$9,$10) ON CONFLICT(storage_domain,git_oid,object_kind) DO NOTHING",[fact.storage_domain.clone().into(),fact.git_oid.clone().into(),fact.object_kind.clone().into(),fact.id.into(),source_id.to_vec().into(),source_bytes.into(),self.primary_scope.clone().into(),map.map_id().to_vec().into(),Sha256::digest(&bytes).to_vec().into(),admission.generation().into()])).await.map_err(db_error)?; + let row=source_row(&txn,source,&self.schema).await?.ok_or_else(||integrity("installed source receipt is missing"))?; + let (stored,_)=self.validate_source_row(source,&row)?; + if stored!=*map || row.try_get::("","receipt_key").map_err(db_error)?!=key.key {return Err(integrity("concurrent source installation conflicts"));} + Ok(()) + }.await; + finish(txn, result).await + } + pub(crate) async fn selected_page( + &self, + map: &PersistedChunkMap, + page_index: u64, + ) -> Result, SnapshotError> { + map.ensure_live().await?; + let intervals = proof_intervals(map.map.page_count, page_index)?; + let memory = self.budget.reserve(PAGE_CREDIT)?; + let txn = self.transaction().await?; + let result = async { + self.require_scope(&txn).await?; + require_current_source(&txn, &map.source, &self.schema).await?; + let row = txn.query_one_raw(stmt(&self.schema, "SELECT pg_catalog.octet_length(payload) AS size,CASE WHEN pg_catalog.octet_length(payload) BETWEEN 48 AND 8208 THEN payload ELSE NULL END AS payload FROM mst2_chunk_map_leaf WHERE map_id=$1 AND page_index=$2", + [map.map_id.to_vec().into(), (page_index as i32).into()])).await.map_err(db_error)? + .ok_or_else(|| integrity("admitted chunk map selected leaf is missing"))?; + let bytes = row.try_get::>>("", "payload").map_err(db_error)? + .ok_or_else(|| integrity("admitted chunk map selected leaf exceeds its byte profile"))?; + let leaf = ChunkLeaf::decode(&bytes).map_err(|_| integrity("admitted chunk map selected leaf is not canonical MCL2"))?; + if leaf.page_index != page_index || leaf.chunk_sha256.len() as u64 != ChunkLeaf::expected_count(map.map.chunk_count, page_index) + || leaf.encode().map_err(|_| integrity("invalid selected leaf"))? != bytes { + return Err(integrity("selected chunk map leaf identity or chunk count disagrees")); + } + let requested: Vec<_> = intervals.iter().map(|&(_, start, pages)| json!({"first_page":start,"page_count":pages})).collect(); + let encoded = serde_json::to_string(&requested).map_err(db_error)?; + let rows = txn.query_all_raw(stmt(&self.schema, "SELECT n.first_page,n.page_count,CASE WHEN pg_catalog.octet_length(n.digest)=32 THEN n.digest ELSE NULL END AS digest FROM pg_catalog.jsonb_to_recordset($2::jsonb) AS p(first_page integer,page_count integer) JOIN mst2_chunk_map_node n ON n.map_id=$1 AND n.first_page=p.first_page AND n.page_count=p.page_count", + [map.map_id.to_vec().into(), encoded.into()])).await.map_err(db_error)?; + let mut nodes = Vec::with_capacity(rows.len()); + for row in rows { + let digest: Option> = row.try_get("", "digest").map_err(db_error)?; + nodes.push(ChunkMapNode { + start: row.try_get::("", "first_page").map_err(db_error)? as u64, + pages: row.try_get::("", "page_count").map_err(db_error)? as u64, + digest: digest.ok_or_else(|| integrity("admitted chunk map sibling has invalid digest size"))?.as_slice().try_into().map_err(|_| integrity("invalid persisted sibling digest"))?, + }); + } + let proof = selected_proof(&intervals, &nodes)?; + verify_leaf(map.map.page_count, page_index, leaf.leaf_hash().map_err(|_| integrity("invalid leaf digest"))?, &proof, map.map.pages_root) + .map_err(|_| integrity("selected chunk map leaf or proof authentication failed"))?; + tracing::debug!(target:"mst2::chunk_map", selected_leaf_bytes = bytes.len(), selected_sibling_rows = nodes.len(), page_count = map.map.page_count, "authenticated persisted chunk map selected page"); + Ok(Arc::new(AuthenticatedChunkPage { leaf, proof, _memory: memory })) + }.await; + let page = finish(txn, result).await?; + map.ensure_live().await?; + map.record_progress().await?; + Ok(page) + } + + async fn transaction(&self) -> Result { + let txn = self + .connection + .begin_with_config(Some(IsolationLevel::ReadCommitted), None) + .await + .map_err(db_error)?; + let search_path = format!("{},pg_catalog,pg_temp", quoted(&self.schema)); + txn.query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT pg_catalog.set_config('search_path',$1,true)", + [search_path.into()], + )) + .await + .map_err(db_error)?; + txn.execute_unprepared("SET LOCAL statement_timeout='5s'; SET LOCAL lock_timeout='5s'") + .await + .map_err(db_error)?; + Ok(txn) + } + + async fn require_scope(&self, connection: &C) -> Result<(), SnapshotError> { + if read_scope(connection, &self.schema).await? != self.primary_scope { + return Err(integrity( + "chunk map repository no longer targets its captured primary scope", + )); + } + Ok(()) + } + + pub(crate) fn memory_budget(&self) -> &Arc { + &self.budget + } + + pub(crate) fn source_identity( + &self, + source: &ChunkMapSource, + ) -> Result<[u8; 32], SnapshotError> { + Ok(source_id(&self.primary_scope, &source.canonical_bytes()?)) + } + + #[cfg(test)] + pub(crate) fn with_test_budget(mut self, budget: Arc) -> Self { + self.budget = budget; + self + } + + #[cfg(test)] + pub(crate) fn test_primary_scope(&self) -> &[u8] { + &self.primary_scope + } + + fn validate_source_row( + &self, + source: &ChunkMapSource, + row: &QueryResult, + ) -> Result<(ChunkMap, [u8; 32]), SnapshotError> { + let source_bytes = source.canonical_bytes()?; + let source_id = source_id(&self.primary_scope, &source_bytes); + let source_bytes_row = bounded_bytes(row, "source_bytes")?; + let primary_scope_row = bounded_bytes(row, "primary_scope")?; + if row.try_get::("", "fact_id").map_err(db_error)? != source.fact().id + || source_bytes_row != source_bytes + || primary_scope_row != self.primary_scope + || bounded_bytes(row, "source_id")? != source_id + { + return Err(integrity( + "persisted map admission disagrees with the exact current source fact or primary", + )); + } + let bytes = row + .try_get::>>("", "descriptor") + .map_err(db_error)? + .ok_or_else(|| { + integrity("admitted chunk map descriptor is missing or outside its byte profile") + })?; + let map = ChunkMap::decode(&bytes) + .map_err(|_| integrity("admitted chunk map descriptor is invalid MCM2"))?; + let receipt_digest: [u8; 32] = + Sha256::digest(receipt(&self.primary_scope, source, &map)?).into(); + if map.encode() != bytes + || map.map_id().as_slice() != bounded_bytes(row, "map_id")?.as_slice() + || map.file_content_id.as_slice() != source.fact().raw_sha256.as_slice() + || map.file_size != source.fact().size as u64 + || map.page_count != row.try_get::("", "page_count").map_err(db_error)? as u64 + || map.pages_root.as_slice() != bounded_bytes(row, "pages_root")?.as_slice() + || bounded_bytes(row, "receipt_digest")? != receipt_digest + { + return Err(integrity( + "admitted chunk map descriptor or receipt identity disagrees", + )); + } + Ok((map, source_id)) + } +} + +async fn capture_schema(connection: &C) -> Result { + let row = connection.query_one_raw(Statement::from_string(DbBackend::Postgres, + "SELECT n.nspname AS schema FROM pg_catalog.pg_namespace n JOIN pg_catalog.pg_class c ON c.relnamespace=n.oid AND c.relname='mst2_metadata_storage_scope' AND c.relkind='r' WHERE n.nspname=pg_catalog.current_schema() AND n.nspname NOT LIKE 'pg_temp_%'")) + .await.map_err(db_error)?.ok_or_else(|| integrity("chunk map actual primary storage relation is missing"))?; + row.try_get("", "schema").map_err(db_error) +} + +async fn read_scope( + connection: &C, + schema: &str, +) -> Result, SnapshotError> { + if connection.get_database_backend() != DbBackend::Postgres { + return Err(integrity("chunk map repository requires PostgreSQL")); + } + let row = connection.query_one_raw(stmt(schema, "SELECT pg_catalog.pg_is_in_recovery() AS replica,CASE WHEN pg_catalog.octet_length(s.storage_uuid)<=255 THEN s.storage_uuid ELSE NULL END AS storage_uuid,pg_catalog.current_database() AS database,d.oid::bigint AS database_oid,n.nspname AS schema,n.oid::bigint AS schema_oid,pg_catalog.inet_server_addr()::text AS address,pg_catalog.inet_server_port() AS port FROM mst2_metadata_storage_scope s JOIN pg_catalog.pg_database d ON d.datname=pg_catalog.current_database() JOIN pg_catalog.pg_namespace n ON n.nspname=$1 JOIN pg_catalog.pg_class c ON c.relnamespace=n.oid AND c.relname='mst2_metadata_storage_scope' AND c.relkind='r' WHERE s.singleton=1", [schema.into()])) + .await.map_err(db_error)?.ok_or_else(|| integrity("chunk map primary scope is missing"))?; + if row.try_get::("", "replica").map_err(db_error)? { + return Err(integrity("chunk map repository is on a replica")); + } + let scope = serde_json::to_vec(&( + row.try_get::>("", "storage_uuid") + .map_err(db_error)? + .ok_or_else(|| integrity("chunk map primary scope UUID exceeds its bounded profile"))?, + row.try_get::("", "database").map_err(db_error)?, + row.try_get::("", "database_oid").map_err(db_error)?, + row.try_get::("", "schema").map_err(db_error)?, + row.try_get::("", "schema_oid").map_err(db_error)?, + row.try_get::>("", "address") + .map_err(db_error)?, + row.try_get::>("", "port").map_err(db_error)?, + )) + .map_err(db_error)?; + if scope.len() > 1024 { + return Err(integrity( + "chunk map primary scope exceeds its admitted profile", + )); + } + Ok(scope) +} + +async fn require_current_source( + connection: &C, + source: &ChunkMapSource, + schema: &str, +) -> Result<(), SnapshotError> { + let row = connection + .query_one_raw(stmt(schema, + "SELECT id,CASE WHEN storage_domain='git' THEN storage_domain ELSE NULL END AS storage_domain,CASE WHEN git_oid=$2 AND pg_catalog.octet_length(git_oid) IN (40,64) THEN git_oid ELSE NULL END AS git_oid,CASE WHEN object_kind='blob' THEN object_kind ELSE NULL END AS object_kind,CASE WHEN pg_catalog.octet_length(raw_sha256)=32 THEN raw_sha256 ELSE NULL END AS raw_sha256,CASE WHEN size BETWEEN 1 AND 8796093022208 THEN size ELSE NULL END AS size,CASE WHEN verification_version=2 THEN verification_version ELSE NULL END AS verification_version,CASE WHEN state='VERIFIED' THEN state ELSE NULL END AS state,created_at FROM mst2_verified_object WHERE id=$1 FOR SHARE", + [source.fact().id.into(), source.fact().git_oid.clone().into()], + )) + .await + .map_err(db_error)? + .ok_or_else(|| { + SnapshotError::new( + SnapshotErrorCode::MetadataNotReady, + "chunk map source has no current verified fact", + ) + })?; + let current = mst2_verified_object::Model::from_query_result(&row, "").map_err(|_| { + integrity("chunk map current source fact is outside its bounded verified profile") + })?; + if ¤t != source.fact() { + return Err(integrity( + "chunk map source fact changed during request or verification", + )); + } + Ok(()) +} + +async fn source_row( + connection: &C, + source: &ChunkMapSource, + schema: &str, +) -> Result, SnapshotError> { + connection.query_one_raw(stmt(schema, "SELECT s.fact_id,s.receipt_key,s.receipt_generation,g.state AS generation_state,l.state AS map_state,l.generation AS map_generation,CASE WHEN pg_catalog.octet_length(g.receipt_bytes)<=4096 THEN g.receipt_bytes ELSE NULL END AS generation_receipt_bytes,CASE WHEN pg_catalog.octet_length(s.source_id)=32 THEN s.source_id ELSE NULL END AS source_id,CASE WHEN pg_catalog.octet_length(s.source_bytes)<=2048 THEN s.source_bytes ELSE NULL END AS source_bytes,CASE WHEN pg_catalog.octet_length(s.primary_scope)<=1024 THEN s.primary_scope ELSE NULL END AS primary_scope,CASE WHEN pg_catalog.octet_length(s.map_id)=32 THEN s.map_id ELSE NULL END AS map_id,CASE WHEN pg_catalog.octet_length(s.receipt_digest)=32 THEN s.receipt_digest ELSE NULL END AS receipt_digest,CASE WHEN pg_catalog.octet_length(m.descriptor)=100 THEN m.descriptor ELSE NULL END AS descriptor,m.page_count,CASE WHEN pg_catalog.octet_length(m.pages_root)=32 THEN m.pages_root ELSE NULL END AS pages_root FROM mst2_chunk_map_source s LEFT JOIN mst2_chunk_map m ON m.map_id=s.map_id LEFT JOIN mst2_chunk_receipt_generation g ON g.receipt_key=s.receipt_key AND g.generation=s.receipt_generation AND g.map_id=s.map_id LEFT JOIN mst2_chunk_map_lifetime l ON l.map_id=s.map_id AND l.generation=g.map_generation WHERE s.storage_domain=$1 AND s.git_oid=$2 AND s.object_kind=$3", + [source.fact().storage_domain.clone().into(), source.fact().git_oid.clone().into(), source.fact().object_kind.clone().into()])).await.map_err(db_error) +} + +fn source_id(scope: &[u8], source: &[u8]) -> [u8; 32] { + let mut hash = Sha256::new(); + hash.update(RECEIPT_DOMAIN); + hash.update((scope.len() as u32).to_be_bytes()); + hash.update(scope); + hash.update((source.len() as u32).to_be_bytes()); + hash.update(source); + hash.finalize().into() +} + +fn receipt( + scope: &[u8], + source: &ChunkMapSource, + map: &ChunkMap, +) -> Result, SnapshotError> { + let source = source.canonical_bytes()?; + if scope.len() > 1024 || source.len() > 2048 { + return Err(integrity( + "chunk map source receipt exceeds its fixed profile", + )); + } + let mut bytes = Vec::with_capacity(RECEIPT_DOMAIN.len() + 8 + scope.len() + source.len() + 100); + bytes.extend_from_slice(RECEIPT_DOMAIN); + bytes.extend_from_slice(&(scope.len() as u32).to_be_bytes()); + bytes.extend_from_slice(scope); + bytes.extend_from_slice(&(source.len() as u32).to_be_bytes()); + bytes.extend_from_slice(&source); + bytes.extend_from_slice(&map.encode()); + Ok(bytes) +} + +async fn read_receipt( + objects: &MegaObjectStorageWrapper, + key: &ObjectKey, + expected: &[u8], +) -> Result<(), SnapshotError> { + tokio::time::timeout(std::time::Duration::from_secs(5), async { + let (mut stream, meta) = objects + .inner + .get_stream(key) + .await + .map_err(|_| integrity("admitted chunk map trusted receipt is unavailable"))?; + if meta.size != expected.len() as i64 { + return Err(integrity( + "chunk map trusted receipt has invalid total size", + )); + } + let mut offset = 0; + while let Some(part) = stream.next().await { + let bytes = part.map_err(|_| integrity("chunk map trusted receipt stream failed"))?; + if bytes.len() > expected.len() - offset + || &expected[offset..offset + bytes.len()] != bytes.as_ref() + { + return Err(integrity( + "chunk map trusted receipt bytes disagree with source and descriptor", + )); + } + offset += bytes.len(); + } + if offset != expected.len() { + return Err(integrity("chunk map trusted receipt is truncated")); + } + Ok(()) + }) + .await + .map_err(|_| { + SnapshotError::new( + SnapshotErrorCode::TemporaryUnavailable, + "chunk map receipt read timed out", + ) + })? +} + +fn encode_leaves(leaves: &[ChunkLeaf]) -> Result { + let values: Result, SnapshotError> = leaves.iter().map(|l| Ok(json!({"page_index":l.page_index,"payload":hex::encode(l.encode().map_err(|_| integrity("invalid verified leaf"))?)}))).collect(); + serde_json::to_string(&values?).map_err(db_error) +} + +fn encode_nodes(nodes: &[ChunkMapNode]) -> Result { + serde_json::to_string(&nodes.iter().map(|n| json!({"first_page":n.start,"page_count":n.pages,"digest":hex::encode(n.digest)})).collect::>()).map_err(db_error) +} + +fn quoted(schema: &str) -> String { + format!("\"{}\"", schema.replace('"', "\"\"")) +} + +/// The static statements use these exact relation tokens. Catalog functions +/// are explicitly qualified; relations never resolve through pg_temp. +fn stmt(schema: &str, sql: &str, values: [sea_orm::Value; N]) -> Statement { + Statement::from_sql_and_values(DbBackend::Postgres, qualify_relations(schema, sql), values) +} + +fn qualify_relations(schema: &str, sql: &str) -> String { + let mut qualified = String::with_capacity(sql.len() + 256); + let prefix = quoted(schema); + let bytes = sql.as_bytes(); + let mut cursor = 0; + while cursor < bytes.len() { + let start = cursor; + let byte = bytes[cursor]; + if byte == b'\'' || byte == b'"' { + cursor += 1; + while cursor < bytes.len() { + if bytes[cursor] == byte { + cursor += 1; + if cursor < bytes.len() && bytes[cursor] == byte { + cursor += 1; + } else { + break; + } + } else { + cursor += 1; + } + } + } else if byte.is_ascii_alphanumeric() || byte == b'_' { + cursor += 1; + while cursor < bytes.len() + && (bytes[cursor].is_ascii_alphanumeric() || bytes[cursor] == b'_') + { + cursor += 1; + } + if [ + "mst2_metadata_storage_scope", + "mst2_verified_object", + "mst2_chunk_map", + "mst2_chunk_map_source", + "mst2_chunk_map_leaf", + "mst2_chunk_map_node", + "mst2_chunk_receipt_generation", + "mst2_chunk_map_lifetime", + "mst2_chunk_reader", + "mst2_chunk_map_gc", + "mst2_chunk_retention_barrier", + ] + .contains(&&sql[start..cursor]) + { + qualified.push_str(&prefix); + qualified.push('.'); + } + } else { + cursor += sql[cursor..].chars().next().map_or(1, char::len_utf8); + } + qualified.push_str(&sql[start..cursor]); + } + qualified +} + +fn bounded_bytes(row: &QueryResult, name: &str) -> Result, SnapshotError> { + row.try_get::>>("", name) + .map_err(db_error)? + .ok_or_else(|| { + integrity("persisted chunk map field is missing or outside its bounded profile") + }) +} +fn integrity(message: &str) -> SnapshotError { + SnapshotError::new(SnapshotErrorCode::IntegrityError, message) +} +fn db_error(error: impl std::fmt::Display) -> SnapshotError { + tracing::warn!(%error, "persisted chunk map storage operation failed"); + SnapshotError::new( + SnapshotErrorCode::TemporaryUnavailable, + "persisted chunk map storage operation failed", + ) +} +fn storage_error(error: impl std::fmt::Display) -> SnapshotError { + tracing::warn!(%error, "trusted chunk map receipt publication failed"); + SnapshotError::new( + SnapshotErrorCode::TemporaryUnavailable, + "trusted chunk map receipt publication failed", + ) +} +async fn finish( + txn: DatabaseTransaction, + result: Result, +) -> Result { + match result { + Ok(value) => { + txn.commit().await.map_err(db_error)?; + Ok(value) + } + Err(error) => { + let _ = txn.rollback().await; + Err(error) + } + } +} + +impl super::Storage { + pub(crate) async fn chunk_maps(&self) -> Result<&PostgresChunkMapRepository, SnapshotError> { + use super::base_storage::StorageConnector; + self.native_chunk_maps + .get_or_try_init(|| async { + PostgresChunkMapRepository::new(self.mono_storage().get_connection().clone()).await + }) + .await + } +} + +#[cfg(test)] +mod sql_tests { + use super::*; + + #[test] + fn authority_relations_are_qualified_without_rewriting_catalog_literals() { + let sql = "SELECT 'mst2_metadata_storage_scope','quoted ''mst2_chunk_map''',\"mst2_chunk_map\" FROM mst2_metadata_storage_scope s JOIN mst2_verified_object f ON s.singleton=1 JOIN mst2_chunk_map m ON true JOIN mst2_chunk_map_leaf l ON true JOIN mst2_chunk_map_node n ON true JOIN mst2_chunk_map_source r ON true WHERE c.relname='mst2_metadata_storage_scope'"; + let qualified = qualify_relations("schema\"name", sql); + assert!(qualified.contains(&format!( + "FROM {}.mst2_metadata_storage_scope", + quoted("schema\"name") + ))); + for name in [ + "mst2_verified_object", + "mst2_chunk_map", + "mst2_chunk_map_leaf", + "mst2_chunk_map_node", + "mst2_chunk_map_source", + ] { + assert!(qualified.contains(&format!("JOIN {}.{name}", quoted("schema\"name")))); + } + assert!(qualified.starts_with( + "SELECT 'mst2_metadata_storage_scope','quoted ''mst2_chunk_map''',\"mst2_chunk_map\"" + )); + assert!(qualified.ends_with("WHERE c.relname='mst2_metadata_storage_scope'")); + } +} diff --git a/src/jupiter/storage/native_chunk_map/retention.rs b/src/jupiter/storage/native_chunk_map/retention.rs new file mode 100644 index 00000000..235ec86f --- /dev/null +++ b/src/jupiter/storage/native_chunk_map/retention.rs @@ -0,0 +1,1220 @@ +//! Bounded primary owners and exact physical-generation reconciliation. + +use std::{ + sync::OnceLock, + time::{Duration, Instant}, +}; + +use tokio::sync::Mutex; +use uuid::Uuid; + +use super::*; +use crate::orbit_api::{ + error::IoOrbitError, + object_storage::{ChunkMapReceiptDeletion, ChunkMapReceiptInventory}, +}; + +const RENEW_INTERVAL: Duration = Duration::from_secs(10); +const HISTORY_CAP: i64 = 131_072; +const MAINTENANCE_MAX: usize = 64; +const OWNER_TTL: Duration = Duration::from_secs(59); + +struct OwnerDeadline(std::sync::Mutex); + +impl OwnerDeadline { + fn from_sql_start(started: Instant) -> Self { + // The SQL INSERT/renew sets clock_timestamp()+59s after this sample. + // Counting SQL and commit latency against the TTL is conservative. + Self(std::sync::Mutex::new(started + OWNER_TTL)) + } + + fn get(&self) -> Result { + let deadline = *self + .0 + .lock() + .map_err(|_| integrity("owner deadline is poisoned"))?; + if deadline <= Instant::now() { + return Err(expired("chunk-map ownership deadline elapsed")); + } + Ok(deadline) + } + + fn observe(&self, deadline: Instant, renewed: bool) -> Result<(), SnapshotError> { + let mut previous = self + .0 + .lock() + .map_err(|_| integrity("owner deadline is poisoned"))?; + *previous = if renewed { + deadline + } else { + (*previous).min(deadline) + }; + if *previous <= Instant::now() { + return Err(expired("chunk-map ownership deadline elapsed")); + } + Ok(()) + } + + async fn bound(&self, future: F) -> Result { + let deadline = self.get()?; + let result = tokio::time::timeout_at(tokio::time::Instant::from_std(deadline), future) + .await + .map_err(|_| expired("chunk-map ownership deadline elapsed"))?; + self.get()?; + Ok(result) + } +} + +fn remaining_deadline(started: Instant, row: &QueryResult) -> Result { + let remaining = row.try_get::("", "remaining_us").map_err(db_error)?; + let remaining = u64::try_from(remaining) + .ok() + .filter(|value| *value > 0) + .ok_or_else(|| expired("chunk-map database ownership deadline elapsed"))?; + started + .checked_add(Duration::from_micros(remaining)) + .ok_or_else(|| integrity("chunk-map ownership deadline is not representable")) +} + +fn retention_budget() -> &'static Arc { + // Inventory survives requests. Its separate process-wide credits bound + // independent repositories without retaining source or response credit. + static BUDGET: OnceLock> = OnceLock::new(); + BUDGET.get_or_init(|| MemoryBudget::new(128 * 1024 * 1024)) +} + +pub(super) struct BackingObservation { + inventory: ChunkMapReceiptInventory, + started_at: String, + captured: Instant, + backing: MegaObjectStorageWrapper, + _memory: MemoryLease, +} + +pub(crate) struct ChunkMapReader { + repository: PostgresChunkMapRepository, + source: ChunkMapSource, + owner: Uuid, + receipt_key: String, + generation: i64, + map_id: [u8; 32], + map_generation: i64, + confirmed: Mutex<(Instant, Instant)>, + deadline: OwnerDeadline, +} + +impl ChunkMapReader { + #[cfg(test)] + pub(crate) async fn test_check_next_owner_operation(&self) { + let before = Instant::now() - Duration::from_secs(11); + *self.confirmed.lock().await = (before, before); + } + pub(crate) async fn ensure_live(&self) -> Result<(), SnapshotError> { + self.deadline.get()?; + let mut confirmed = self.confirmed.lock().await; + self.deadline.get()?; + if confirmed.0.elapsed() < RENEW_INTERVAL { + return Ok(()); + } + let deadline = self.deadline.bound(self.check(false)).await??; + self.deadline.observe(deadline, false)?; + confirmed.0 = Instant::now(); + Ok(()) + } + + pub(super) async fn record_progress(&self) -> Result<(), SnapshotError> { + self.deadline.get()?; + let mut confirmed = self.confirmed.lock().await; + self.deadline.get()?; + if confirmed.1.elapsed() < RENEW_INTERVAL { + return Ok(()); + } + let deadline = self.deadline.bound(self.check(true)).await??; + self.deadline.observe(deadline, true)?; + *confirmed = (Instant::now(), Instant::now()); + Ok(()) + } + + pub(super) async fn validate_after_receipt(&self) -> Result<(), SnapshotError> { + let mut confirmed = self.confirmed.lock().await; + self.deadline.get()?; + let deadline = self.deadline.bound(self.check(true)).await??; + self.deadline.observe(deadline, true)?; + *confirmed = (Instant::now(), Instant::now()); + Ok(()) + } + + pub(crate) async fn await_backend( + &self, + future: F, + ) -> Result { + self.ensure_live().await?; + tokio::pin!(future); + loop { + let deadline = self.deadline.get()?; + let check_at = self.confirmed.lock().await.0 + RENEW_INTERVAL; + tokio::select! { + biased; + _ = tokio::time::sleep_until(tokio::time::Instant::from_std(deadline)) => { + return Err(expired("chunk-map reader backend wait outlived its ownership")); + } + _ = tokio::time::sleep_until(tokio::time::Instant::from_std(check_at)) => { + self.ensure_live().await?; + } + value = &mut future => { + self.deadline.get()?; + return Ok(value); + } + } + } + } + + async fn check(&self, renew: bool) -> Result { + let started = Instant::now(); + let txn = self.repository.transaction().await?; + let result = async { + self.repository.require_scope(&txn).await?; + require_current_source(&txn, &self.source, &self.repository.schema).await?; + barrier(&txn, &self.repository.schema).await?; + let row = if renew {txn.query_one_raw(stmt(&self.repository.schema, + "UPDATE mst2_chunk_reader SET deadline=pg_catalog.clock_timestamp()+interval '59 seconds',last_progress=pg_catalog.clock_timestamp() WHERE owner=$1::uuid AND receipt_key=$2 AND source_generation=$3 AND map_id=$4 AND map_generation=$5 AND deadline>pg_catalog.clock_timestamp() RETURNING owner", + [self.owner.to_string().into(),self.receipt_key.clone().into(),self.generation.into(),self.map_id.to_vec().into(),self.map_generation.into()])).await.map_err(db_error)?} else { + txn.query_one_raw(stmt(&self.repository.schema,"SELECT r.owner FROM mst2_chunk_reader r JOIN mst2_chunk_receipt_generation g ON g.receipt_key=r.receipt_key AND g.generation=r.source_generation JOIN mst2_chunk_map_lifetime l ON l.map_id=r.map_id AND l.generation=r.map_generation WHERE r.owner=$1::uuid AND r.receipt_key=$2 AND r.source_generation=$3 AND r.map_id=$4 AND r.map_generation=$5 AND r.deadline>pg_catalog.clock_timestamp() AND g.state='LIVE' AND l.state='LIVE'",[self.owner.to_string().into(),self.receipt_key.clone().into(),self.generation.into(),self.map_id.to_vec().into(),self.map_generation.into()])).await.map_err(db_error)? + }; + if row.is_none() { return Err(expired("chunk-map reader expired before progress resumed")); } + if renew {txn.execute_raw(stmt(&self.repository.schema, + "UPDATE mst2_chunk_map_lifetime SET last_used=pg_catalog.clock_timestamp() WHERE map_id=$1 AND generation=$2 AND state='LIVE'", + [self.map_id.to_vec().into(),self.map_generation.into()])).await.map_err(db_error)?; + txn.execute_raw(stmt(&self.repository.schema,"UPDATE mst2_chunk_receipt_generation SET last_progress=pg_catalog.clock_timestamp() WHERE receipt_key=$1 AND generation=$2 AND state='LIVE'",[self.receipt_key.clone().into(),self.generation.into()])).await.map_err(db_error)?; + } + let remaining = txn.query_one_raw(stmt(&self.repository.schema, + "SELECT pg_catalog.floor(EXTRACT(EPOCH FROM (deadline-pg_catalog.clock_timestamp()))*1000000)::bigint AS remaining_us FROM mst2_chunk_reader WHERE owner=$1::uuid", + [self.owner.to_string().into()])).await.map_err(db_error)? + .ok_or_else(|| expired("chunk-map reader deadline disappeared"))?; + remaining_deadline(started, &remaining) + }.await; + finish(txn, result).await + } +} + +impl Drop for ChunkMapReader { + fn drop(&mut self) { + let Ok(runtime) = tokio::runtime::Handle::try_current() else { + return; + }; + let repository = self.repository.clone(); + let owner = self.owner; + runtime.spawn(async move { + let result = async { + let txn = repository.transaction().await?; + let result = async { + repository.require_scope(&txn).await?; + barrier(&txn, &repository.schema).await?; + txn.execute_raw(stmt( + &repository.schema, + "DELETE FROM mst2_chunk_reader WHERE owner=$1::uuid", + [owner.to_string().into()], + )) + .await + .map_err(db_error)?; + Ok(()) + } + .await; + finish(txn, result).await + } + .await; + if let Err(error) = result { + tracing::warn!( + ?error, + "chunk-map reader release will expire by database deadline" + ); + } + }); + } +} + +/// A capacity reservation exists before the verifier opens the source body. +/// It cannot publish source trust; installation still consumes the opaque +/// complete-stream verifier and independently checks the backend receipt. +pub(crate) struct ChunkMapInstall { + repository: PostgresChunkMapRepository, + source: ChunkMapSource, + owner: Uuid, + key: ObjectKey, + generation: i64, + progress: Mutex<(Instant, Instant, u64)>, + deadline: OwnerDeadline, +} + +#[derive(Clone)] +pub(super) struct CancelledInstall { + owner: Uuid, + key: String, + generation: i64, +} + +impl ChunkMapInstall { + #[cfg(test)] + pub(crate) async fn test_check_next_owner_operation(&self) { + let mut progress = self.progress.lock().await; + let before = Instant::now() - Duration::from_secs(11); + progress.0 = before; + } + pub(crate) async fn ensure_live(&self) -> Result<(), SnapshotError> { + self.renew(false, 0).await + } + + pub(crate) async fn record_progress(&self, completed: u64) -> Result<(), SnapshotError> { + self.renew(true, completed).await + } + + pub(crate) async fn await_backend( + &self, + future: F, + ) -> Result { + self.ensure_live().await?; + tokio::pin!(future); + loop { + let deadline = self.deadline.get()?; + let check_at = self.progress.lock().await.0 + RENEW_INTERVAL; + tokio::select! { + biased; + _ = tokio::time::sleep_until(tokio::time::Instant::from_std(deadline)) => { + return Err(expired("chunk-map install backend wait outlived its ownership")); + } + _ = tokio::time::sleep_until(tokio::time::Instant::from_std(check_at)) => { + self.ensure_live().await?; + } + value = &mut future => { + self.deadline.get()?; + return Ok(value); + } + } + } + } + + async fn renew(&self, progress: bool, completed: u64) -> Result<(), SnapshotError> { + self.deadline.get()?; + let mut confirmed = self.progress.lock().await; + self.deadline.get()?; + if progress && completed <= confirmed.2 { + return Ok(()); + } + if (if progress { + confirmed.1.elapsed() + } else { + confirmed.0.elapsed() + }) < RENEW_INTERVAL + { + if progress { + confirmed.2 = completed; + } + return Ok(()); + } + // Calling ensure_live after a long stalled await checks the old + // deadline without extending it. Only new source progress renews it. + let started = Instant::now(); + let deadline = self.deadline.bound(async { + let txn = self.repository.transaction().await?; + let result=async { + self.repository.require_scope(&txn).await?; + require_current_source(&txn,&self.source,&self.repository.schema).await?; + barrier(&txn,&self.repository.schema).await?; + let row=if progress { + txn.query_one_raw(stmt(&self.repository.schema,"UPDATE mst2_chunk_receipt_generation SET deadline=pg_catalog.clock_timestamp()+interval '59 seconds',last_progress=pg_catalog.clock_timestamp() WHERE receipt_key=$1 AND generation=$2 AND owner=$3::uuid AND state IN ('RESERVED','CREATING') AND deadline>pg_catalog.clock_timestamp() RETURNING generation",[self.key.key.clone().into(),self.generation.into(),self.owner.to_string().into()])).await.map_err(db_error)? + } else { + txn.query_one_raw(stmt(&self.repository.schema,"SELECT generation FROM mst2_chunk_receipt_generation WHERE receipt_key=$1 AND generation=$2 AND owner=$3::uuid AND state IN ('RESERVED','CREATING') AND deadline>pg_catalog.clock_timestamp()",[self.key.key.clone().into(),self.generation.into(),self.owner.to_string().into()])).await.map_err(db_error)? + }; + if row.is_none() { return Err(expired("chunk-map install reservation expired or was retired")); } + let remaining = txn.query_one_raw(stmt(&self.repository.schema, + "SELECT pg_catalog.floor(EXTRACT(EPOCH FROM (deadline-pg_catalog.clock_timestamp()))*1000000)::bigint AS remaining_us FROM mst2_chunk_receipt_generation WHERE receipt_key=$1 AND generation=$2 AND owner=$3::uuid AND state IN ('RESERVED','CREATING')", + [self.key.key.clone().into(),self.generation.into(),self.owner.to_string().into()])).await.map_err(db_error)? + .ok_or_else(|| expired("chunk-map install deadline disappeared"))?; + remaining_deadline(started, &remaining) + }.await; + finish(txn, result).await + }).await??; + self.deadline.observe(deadline, progress)?; + if progress { + *confirmed = (Instant::now(), Instant::now(), completed); + } else { + confirmed.0 = Instant::now(); + } + Ok(()) + } + + pub(super) async fn prepare( + &self, + verified: &VerifiedSourceChunkMap, + ) -> Result, SnapshotError> { + if verified.source() != &self.source { + return Err(integrity( + "verified source differs from install reservation", + )); + } + let bytes = receipt(&self.repository.primary_scope, &self.source, verified.map())?; + let started = Instant::now(); + self.deadline.bound(async { + let txn = self.repository.transaction().await?; + let result=async { + self.repository.require_scope(&txn).await?; + require_current_source(&txn,&self.source,&self.repository.schema).await?; + barrier(&txn,&self.repository.schema).await?; + let row=txn.query_one_raw(stmt(&self.repository.schema,"UPDATE mst2_chunk_receipt_generation SET state='CREATING',map_id=$4,receipt_bytes=$5,deadline=pg_catalog.clock_timestamp()+interval '59 seconds',last_progress=pg_catalog.clock_timestamp() WHERE receipt_key=$1 AND generation=$2 AND owner=$3::uuid AND state='RESERVED' AND deadline>pg_catalog.clock_timestamp() RETURNING generation",[self.key.key.clone().into(),self.generation.into(),self.owner.to_string().into(),verified.map().map_id().to_vec().into(),bytes.clone().into()])).await.map_err(db_error)?; + if row.is_none() { return Err(expired("chunk-map install lost its exact reservation before receipt creation")); } + Ok(()) + }.await; + finish(txn, result).await?; + Ok::<_, SnapshotError>(()) + }).await??; + self.deadline.observe(started + OWNER_TTL, true)?; + Ok(bytes) + } + + pub(super) fn key(&self) -> &ObjectKey { + &self.key + } + pub(super) fn generation(&self) -> i64 { + self.generation + } + pub(super) fn owner(&self) -> Uuid { + self.owner + } + + pub(super) async fn acknowledge_create(&self) -> Result<(), SnapshotError> { + // This method is called only after the actual atomic create future + // returns success. Absence, a checksum or inventory cannot call it. + let started = Instant::now(); + let renewed = self.deadline.bound(async { + let txn = self.repository.transaction().await?; + let result=async { + self.repository.require_scope(&txn).await?; + barrier(&txn,&self.repository.schema).await?; + let row=txn.query_one_raw(stmt(&self.repository.schema,"UPDATE mst2_chunk_receipt_generation SET create_completed=true,create_completed_at=COALESCE(create_completed_at,pg_catalog.clock_timestamp()),deadline=CASE WHEN state='CREATING' AND owner=$3::uuid AND deadline>pg_catalog.clock_timestamp() THEN pg_catalog.clock_timestamp()+interval '59 seconds' ELSE deadline END WHERE receipt_key=$1 AND generation=$2 AND receipt_bytes IS NOT NULL RETURNING generation,CASE WHEN state='CREATING' AND owner=$3::uuid AND deadline>pg_catalog.clock_timestamp() THEN true ELSE false END AS deadline_renewed",[self.key.key.clone().into(),self.generation.into(),self.owner.to_string().into()])).await.map_err(db_error)? + .ok_or_else(|| integrity("completed create lost its persistent generation fence"))?; + row.try_get::("", "deadline_renewed").map_err(db_error) + }.await; + finish(txn, result).await + }).await??; + if renewed { + self.deadline.observe(started + OWNER_TTL, true)?; + } + Ok(()) + } + + pub(super) async fn retire(&self) -> Result<(), SnapshotError> { + retire_install(&self.repository, self.owner, &self.key.key, self.generation).await + } +} + +impl Drop for ChunkMapInstall { + fn drop(&mut self) { + if let Ok(mut cancelled) = self.repository.cancelled_installs.lock() + && cancelled.len() < 64 + { + cancelled.push(CancelledInstall { + owner: self.owner, + key: self.key.key.clone(), + generation: self.generation, + }); + } + let Ok(runtime) = tokio::runtime::Handle::try_current() else { + return; + }; + let repository = self.repository.clone(); + let owner = self.owner; + let key = self.key.key.clone(); + let generation = self.generation; + runtime.spawn(async move { + let result = retire_install(&repository, owner, &key, generation).await; + if result.is_ok() { + forget_cancelled_install(&repository, owner); + } + if let Err(error) = result { + tracing::warn!( + ?error, + "chunk-map install cancellation will reconcile by database deadline" + ); + } + }); + } +} + +fn forget_cancelled_install(repository: &PostgresChunkMapRepository, owner: Uuid) { + if let Ok(mut cancelled) = repository.cancelled_installs.lock() { + cancelled.retain(|install| install.owner != owner); + } +} + +async fn retire_install( + repository: &PostgresChunkMapRepository, + owner: Uuid, + key: &str, + generation: i64, +) -> Result<(), SnapshotError> { + let txn = repository.transaction().await?; + let result = async { + repository.require_scope(&txn).await?; + barrier(&txn, &repository.schema).await?; + txn.execute_raw(stmt(&repository.schema,"UPDATE mst2_chunk_receipt_generation SET state=CASE WHEN receipt_bytes IS NULL THEN 'APPLIED' ELSE 'DELETING' END,owner=NULL,deadline=NULL,reserved_pages=0,reserved_nodes=0,reserved_bytes=0 WHERE receipt_key=$1 AND generation=$2 AND owner=$3::uuid AND state IN ('RESERVED','CREATING')",[key.into(),generation.into(),owner.to_string().into()])).await.map_err(db_error)?; + Ok(()) + }.await; + finish(txn, result).await +} + +/// Constructor stays in the collector module. Its private fields carry a +/// real primary claim and the independently read exact backend body. +pub(crate) struct VerifiedReceiptDeletionClaim { + key: ObjectKey, + expected: Vec, + _generation: i64, +} +impl VerifiedReceiptDeletionClaim { + pub(crate) fn key(&self) -> &ObjectKey { + &self.key + } + pub(crate) fn expected_bytes(&self) -> &[u8] { + &self.expected + } +} + +impl PostgresChunkMapRepository { + async fn replay_cancelled_installs(&self) -> Result<(), SnapshotError> { + let cancelled = self + .cancelled_installs + .lock() + .map_err(|_| integrity("cancelled install replay ownership is poisoned"))? + .clone(); + for install in cancelled { + retire_install(self, install.owner, &install.key, install.generation).await?; + forget_cancelled_install(self, install.owner); + } + Ok(()) + } + + async fn maintain_before_admission( + &self, + objects: &MegaObjectStorageWrapper, + ) -> Result<(), SnapshotError> { + let mut completed = self.admission_maintenance.lock().await; + self.replay_cancelled_installs().await?; + if completed + .as_ref() + .is_some_and(|last| last.elapsed() < RENEW_INTERVAL) + { + let txn = self.transaction().await?; + self.require_scope(&txn).await?; + let pending=txn.query_one_raw(stmt(&self.schema,"SELECT EXISTS(SELECT 1 FROM mst2_chunk_receipt_generation WHERE state='DELETING' OR (state IN ('RESERVED','CREATING') AND deadline<=pg_catalog.clock_timestamp())) AS pending",[])).await.map_err(db_error)?.ok_or_else(||integrity("pending chunk-map admission replay observation missing"))?.try_get::("","pending").map_err(db_error)?; + txn.commit().await.map_err(db_error)?; + if !pending { + return Ok(()); + } + } + self.maintain(objects, 8).await?; + *completed = Some(Instant::now()); + Ok(()) + } + #[cfg(test)] + pub(crate) async fn test_new_install_capacity( + &self, + objects: &MegaObjectStorageWrapper, + ) -> Result<(), SnapshotError> { + let observation = inventory(self, objects).await?; + let txn = self.transaction().await?; + let result = async { + self.require_scope(&txn).await?; + barrier(&txn, &self.schema).await?; + ensure_quotas(&txn, &self.schema, &observation, 1, 1, 1024 * 1024, None).await + } + .await; + finish(txn, result).await + } + pub(super) async fn stage_map_start( + &self, + verified: &VerifiedSourceChunkMap, + admission: &ChunkMapInstall, + observation: &BackingObservation, + ) -> Result { + let map = verified.map(); + let txn = self.transaction().await?; + let result=async { + self.require_scope(&txn).await?; + require_current_source(&txn,verified.source(),&self.schema).await?; + map_barrier(&txn,map.map_id()).await?; + barrier(&txn,&self.schema).await?; + let inserted=txn.query_one_raw(stmt(&self.schema,"INSERT INTO mst2_chunk_map(map_id,descriptor,page_count,pages_root) VALUES($1,$2,$3,$4) ON CONFLICT(map_id) DO NOTHING RETURNING map_id",[map.map_id().to_vec().into(),map.encode().into(),(map.page_count as i32).into(),map.pages_root.to_vec().into()])).await.map_err(db_error)?.is_some(); + let lifetime=txn.query_one_raw(stmt(&self.schema,"SELECT generation,state FROM mst2_chunk_map_lifetime WHERE map_id=$1",[map.map_id().to_vec().into()])).await.map_err(db_error)?; + let generation=if let Some(row)=lifetime { + if row.try_get::("","state").map_err(db_error)?!="LIVE" {return Err(unavailable("shared chunk map collection is pending"));} + row.try_get::("","generation").map_err(db_error)? + } else if inserted { + let generation=next_generation(&txn,&self.schema).await?; + txn.execute_raw(stmt(&self.schema,"INSERT INTO mst2_chunk_map_lifetime(map_id,generation,state) VALUES($1,$2,'LIVE')",[map.map_id().to_vec().into(),generation.into()])).await.map_err(db_error)?; + generation + } else {return Err(integrity("existing chunk map has no exact lifetime; never repair it"));}; + let row=txn.query_one_raw(stmt(&self.schema,"SELECT CASE WHEN pg_catalog.octet_length(descriptor)=100 THEN descriptor ELSE NULL END AS descriptor,page_count,CASE WHEN pg_catalog.octet_length(pages_root)=32 THEN pages_root ELSE NULL END AS pages_root FROM mst2_chunk_map WHERE map_id=$1",[map.map_id().to_vec().into()])).await.map_err(db_error)?.ok_or_else(||integrity("staged map descriptor missing"))?; + if bounded_bytes(&row,"descriptor")?!=map.encode() || row.try_get::("","page_count").map_err(db_error)? as u64!=map.page_count || bounded_bytes(&row,"pages_root")?!=map.pages_root {return Err(integrity("immutable staged map descriptor conflicts with verified source"));} + let assigned=txn.query_one_raw(stmt(&self.schema,"UPDATE mst2_chunk_receipt_generation SET map_generation=$4,deadline=pg_catalog.clock_timestamp()+interval '59 seconds',last_progress=pg_catalog.clock_timestamp() WHERE receipt_key=$1 AND generation=$2 AND owner=$3::uuid AND state='CREATING' AND deadline>pg_catalog.clock_timestamp() AND map_id=$5 AND create_completed RETURNING generation",[admission.key.key.clone().into(),admission.generation.into(),admission.owner.to_string().into(),generation.into(),map.map_id().to_vec().into()])).await.map_err(db_error)?; + if assigned.is_none() {return Err(expired("map staging lost the exact completed-create reservation"));} + ensure_quotas(&txn,&self.schema,observation,0,0,0,Some(admission.owner)).await?; + Ok(generation) + }.await; + finish(txn, result).await + } + + pub(super) async fn stage_map_batch( + &self, + verified: &VerifiedSourceChunkMap, + admission: &ChunkMapInstall, + generation: i64, + leaves: Option<&[ChunkLeaf]>, + nodes: Option<&[ChunkMapNode]>, + ) -> Result<(), SnapshotError> { + let map_id = verified.map().map_id(); + let txn = self.transaction().await?; + let result=async { + self.require_scope(&txn).await?; + require_current_source(&txn,verified.source(),&self.schema).await?; + map_barrier(&txn,map_id).await?; + let sealed=txn.query_one_raw(stmt(&self.schema,"SELECT EXISTS(SELECT 1 FROM mst2_chunk_map_source WHERE map_id=$1) AS sealed",[map_id.to_vec().into()])).await.map_err(db_error)?.ok_or_else(||integrity("staged map seal observation missing"))?.try_get::("","sealed").map_err(db_error)?; + if !sealed { + if let Some(batch)=leaves { + txn.execute_raw(stmt(&self.schema,"INSERT INTO mst2_chunk_map_leaf(map_id,page_index,payload) SELECT $1,p.page_index,pg_catalog.decode(p.payload,'hex') FROM pg_catalog.jsonb_to_recordset($2::jsonb) AS p(page_index integer,payload text) ON CONFLICT(map_id,page_index) DO NOTHING",[map_id.to_vec().into(),encode_leaves(batch)?.into()])).await.map_err(db_error)?; + } + if let Some(batch)=nodes { + txn.execute_raw(stmt(&self.schema,"INSERT INTO mst2_chunk_map_node(map_id,first_page,page_count,digest) SELECT $1,p.first_page,p.page_count,pg_catalog.decode(p.digest,'hex') FROM pg_catalog.jsonb_to_recordset($2::jsonb) AS p(first_page integer,page_count integer,digest text) ON CONFLICT(map_id,first_page,page_count) DO NOTHING",[map_id.to_vec().into(),encode_nodes(batch)?.into()])).await.map_err(db_error)?; + } + } + self.checkpoint_staging(&txn,admission,map_id,generation).await + }.await; + finish(txn, result).await + } + + async fn checkpoint_staging( + &self, + txn: &DatabaseTransaction, + admission: &ChunkMapInstall, + map_id: [u8; 32], + generation: i64, + ) -> Result<(), SnapshotError> { + barrier(txn, &self.schema).await?; + let progress=txn.query_one_raw(stmt(&self.schema,"UPDATE mst2_chunk_receipt_generation SET deadline=pg_catalog.clock_timestamp()+interval '59 seconds',last_progress=pg_catalog.clock_timestamp() WHERE receipt_key=$1 AND generation=$2 AND owner=$3::uuid AND state='CREATING' AND deadline>pg_catalog.clock_timestamp() AND map_id=$4 AND map_generation=$5 AND EXISTS(SELECT 1 FROM mst2_chunk_map_lifetime WHERE map_id=$4 AND generation=$5 AND state='LIVE') RETURNING generation",[admission.key.key.clone().into(),admission.generation.into(),admission.owner.to_string().into(),map_id.to_vec().into(),generation.into()])).await.map_err(db_error)?; + if progress.is_none() { + return Err(expired( + "bounded map publication batch lost its exact install generation", + )); + } + Ok(()) + } + + pub(super) async fn compare_staged_map( + &self, + verified: &VerifiedSourceChunkMap, + nodes: &[ChunkMapNode], + admission: &ChunkMapInstall, + generation: i64, + ) -> Result<(), SnapshotError> { + let map = verified.map(); + let txn = self.transaction().await?; + let result=async { + self.require_scope(&txn).await?; + require_current_source(&txn,verified.source(),&self.schema).await?; + map_barrier(&txn,map.map_id()).await?; + let row=txn.query_one_raw(stmt(&self.schema,"SELECT CASE WHEN pg_catalog.octet_length(descriptor)=100 THEN descriptor ELSE NULL END AS descriptor,page_count,CASE WHEN pg_catalog.octet_length(pages_root)=32 THEN pages_root ELSE NULL END AS pages_root,(SELECT pg_catalog.count(*) FROM mst2_chunk_map_leaf WHERE map_id=$1) AS leaves,(SELECT pg_catalog.count(*) FROM mst2_chunk_map_node WHERE map_id=$1) AS nodes FROM mst2_chunk_map WHERE map_id=$1",[map.map_id().to_vec().into()])).await.map_err(db_error)?.ok_or_else(||integrity("staged map descriptor missing"))?; + if bounded_bytes(&row,"descriptor")?!=map.encode() || row.try_get::("","page_count").map_err(db_error)? as u64!=map.page_count || bounded_bytes(&row,"pages_root")?!=map.pages_root || row.try_get::("","leaves").map_err(db_error)? as u64!=map.page_count || row.try_get::("","nodes").map_err(db_error)? as usize!=nodes.len() {return Err(integrity("staged map has conflicting descriptor or incomplete coverage"));} + self.checkpoint_staging(&txn,admission,map.map_id(),generation).await + }.await; + finish(txn, result).await?; + for batch in verified.leaves().chunks(64) { + let txn = self.transaction().await?; + let result=async { + self.require_scope(&txn).await?; + require_current_source(&txn,verified.source(),&self.schema).await?; + map_barrier(&txn,map.map_id()).await?; + let bad=txn.query_one_raw(stmt(&self.schema,"SELECT p.page_index FROM pg_catalog.jsonb_to_recordset($2::jsonb) AS p(page_index integer,payload text) LEFT JOIN mst2_chunk_map_leaf l ON l.map_id=$1 AND l.page_index=p.page_index WHERE l.payload IS DISTINCT FROM pg_catalog.decode(p.payload,'hex') LIMIT 1",[map.map_id().to_vec().into(),encode_leaves(batch)?.into()])).await.map_err(db_error)?; + if bad.is_some() {return Err(integrity("immutable stored chunk map leaf conflicts with verified source"));} + self.checkpoint_staging(&txn,admission,map.map_id(),generation).await + }.await; + finish(txn, result).await?; + } + for batch in nodes.chunks(1024) { + let txn = self.transaction().await?; + let result=async { + self.require_scope(&txn).await?; + require_current_source(&txn,verified.source(),&self.schema).await?; + map_barrier(&txn,map.map_id()).await?; + let bad=txn.query_one_raw(stmt(&self.schema,"SELECT p.first_page FROM pg_catalog.jsonb_to_recordset($2::jsonb) AS p(first_page integer,page_count integer,digest text) LEFT JOIN mst2_chunk_map_node n ON n.map_id=$1 AND n.first_page=p.first_page AND n.page_count=p.page_count WHERE n.digest IS DISTINCT FROM pg_catalog.decode(p.digest,'hex') LIMIT 1",[map.map_id().to_vec().into(),encode_nodes(batch)?.into()])).await.map_err(db_error)?; + if bad.is_some() {return Err(integrity("immutable stored chunk map node conflicts with verified source"));} + self.checkpoint_staging(&txn,admission,map.map_id(),generation).await + }.await; + finish(txn, result).await?; + } + Ok(()) + } + + pub(crate) async fn maintain( + &self, + objects: &MegaObjectStorageWrapper, + limit: usize, + ) -> Result<(), SnapshotError> { + if !(1..=MAINTENANCE_MAX).contains(&limit) { + return Err(SnapshotError::new( + SnapshotErrorCode::InvalidRequest, + "chunk-map maintenance limit must be 1..64", + )); + } + // Unsupported enumeration must fail before changing any resident + // generation. Overquota stores may still replay known deletions. + match inventory(self, objects).await { + Ok(_) => {} + Err(error) if error.code == SnapshotErrorCode::LimitExceeded => {} + Err(error) => return Err(error), + } + self.replay_cancelled_installs().await?; + let _memory = retention_budget().reserve(8 * 1024 * 1024)?; + let txn = self.transaction().await?; + let result=async { + self.require_scope(&txn).await?; + barrier(&txn,&self.schema).await?; + txn.execute_raw(stmt(&self.schema,"DELETE FROM mst2_chunk_reader WHERE owner IN (SELECT owner FROM mst2_chunk_reader WHERE deadline<=pg_catalog.clock_timestamp() ORDER BY deadline,owner LIMIT $1)",[(limit as i64).into()])).await.map_err(db_error)?; + txn.execute_raw(stmt(&self.schema,"UPDATE mst2_chunk_receipt_generation SET state=CASE WHEN receipt_bytes IS NULL THEN 'APPLIED' ELSE 'DELETING' END,owner=NULL,deadline=NULL,reserved_pages=0,reserved_nodes=0,reserved_bytes=0 WHERE receipt_key IN (SELECT receipt_key FROM mst2_chunk_receipt_generation WHERE state IN ('RESERVED','CREATING') AND deadline<=pg_catalog.clock_timestamp() ORDER BY deadline,receipt_key LIMIT $1)",[(limit as i64).into()])).await.map_err(db_error)?; + // Replay unfinished receipt claims before admitting new victims. + let pending=txn.query_one_raw(stmt(&self.schema,"SELECT EXISTS(SELECT 1 FROM mst2_chunk_receipt_generation WHERE state='DELETING') AS pending",[])).await.map_err(db_error)?.ok_or_else(||integrity("chunk receipt replay observation missing"))?.try_get::("","pending").map_err(db_error)?; + if !pending { + txn.execute_raw(stmt(&self.schema,"UPDATE mst2_chunk_receipt_generation SET state='DELETING' WHERE receipt_key IN (SELECT g.receipt_key FROM mst2_chunk_receipt_generation g WHERE g.state='LIVE' AND g.last_progress<=pg_catalog.clock_timestamp()-interval '1 hour' AND NOT EXISTS(SELECT 1 FROM mst2_chunk_reader r WHERE r.receipt_key=g.receipt_key AND r.source_generation=g.generation AND r.deadline>pg_catalog.clock_timestamp()) ORDER BY g.last_progress,g.receipt_key LIMIT $1)",[(limit as i64).into()])).await.map_err(db_error)?; + } + // Rotate APPLIED claims too: absence is never a promise that a + // cancelled remote create cannot restore this retired key later. + txn.query_all_raw(stmt(&self.schema,"SELECT receipt_key,generation,CASE WHEN pg_catalog.octet_length(receipt_bytes)<=4096 THEN receipt_bytes ELSE NULL END AS receipt_bytes,CASE WHEN pg_catalog.octet_length(primary_scope)<=1024 THEN primary_scope ELSE NULL END AS primary_scope,map_id,map_generation,state FROM mst2_chunk_receipt_generation WHERE state IN ('DELETING','APPLIED') ORDER BY (state='DELETING') DESC,checked_at,receipt_key LIMIT $1",[(limit as i64).into()])).await.map_err(db_error) + }.await; + let claims = finish(txn, result).await?; + for row in claims { + self.reconcile_receipt(objects, row).await?; + } + self.collect_maps(limit).await?; + let observation = inventory(self, objects).await?; + let requested = serde_json::to_string( + &observation + .inventory + .objects + .iter() + .map(|(key, _)| &key.key) + .collect::>(), + ) + .map_err(db_error)?; + let txn = self.transaction().await?; + self.require_scope(&txn).await?; + let unknown=txn.query_all_raw(stmt(&self.schema,"SELECT p.key FROM pg_catalog.jsonb_array_elements_text($1::jsonb) AS p(key) LEFT JOIN mst2_chunk_receipt_generation g ON g.receipt_key=p.key WHERE g.receipt_key IS NULL ORDER BY p.key LIMIT $2",[requested.into(),(limit as i64).into()])).await.map_err(db_error)?; + txn.commit().await.map_err(db_error)?; + for row in unknown { + let key = ObjectKey { + namespace: ObjectNamespace::ChunkMapReceipt, + key: row.try_get("", "key").map_err(db_error)?, + }; + self.claim_orphan(objects, &key).await?; + } + tracing::debug!(target:"mst2::chunk_map",maintenance_limit=limit,actual_backing_receipts=observation.inventory.objects.len(),actual_backing_bytes=observation.inventory.bytes,"bounded chunk-map retention tick"); + Ok(()) + } + + async fn reconcile_receipt( + &self, + objects: &MegaObjectStorageWrapper, + row: QueryResult, + ) -> Result<(), SnapshotError> { + if bounded_bytes(&row, "primary_scope")? != self.primary_scope { + return Err(integrity( + "retired receipt belongs to a different primary scope", + )); + } + let key = ObjectKey { + namespace: ObjectNamespace::ChunkMapReceipt, + key: row.try_get("", "receipt_key").map_err(db_error)?, + }; + let generation: i64 = row.try_get("", "generation").map_err(db_error)?; + let expected: Option> = row.try_get("", "receipt_bytes").map_err(db_error)?; + let txn = self.transaction().await?; + let result=async { + self.require_scope(&txn).await?; + barrier(&txn,&self.schema).await?; + txn.query_one_raw(stmt(&self.schema,"SELECT generation FROM mst2_chunk_receipt_generation WHERE receipt_key=$1 AND generation=$2 AND receipt_bytes IS NOT DISTINCT FROM $3::bytea AND primary_scope=$4 AND state IN ('DELETING','APPLIED') AND NOT EXISTS(SELECT 1 FROM mst2_chunk_reader WHERE receipt_key=$1 AND source_generation=$2 AND deadline>pg_catalog.clock_timestamp())",[key.key.clone().into(),generation.into(),expected.clone().into(),self.primary_scope.clone().into()])).await.map_err(db_error) + }.await; + if finish(txn, result).await?.is_none() { + return Ok(()); + } + if let Some(bytes) = expected.as_ref() { + let (scope, _, map, physical_generation) = decode_orphan(&key, bytes)?; + let stored_map: Option> = row.try_get("", "map_id").map_err(db_error)?; + if scope != self.primary_scope + || physical_generation != generation + || stored_map.as_deref() != Some(map.map_id().as_slice()) + { + return Err(integrity( + "receipt deletion claim disagrees with the independent physical generation", + )); + } + } + if let Some(expected) = expected { + let present = tokio::time::timeout(Duration::from_secs(5), objects.inner.exists(&key)) + .await + .map_err(|_| unavailable("retired receipt existence check timed out"))? + .map_err(storage_error)?; + if present { + read_receipt(objects, &key, &expected).await?; + let authority = ChunkMapReceiptDeletion::from_claim(VerifiedReceiptDeletionClaim { + key: key.clone(), + expected, + _generation: generation, + }); + tokio::time::timeout( + Duration::from_secs(5), + objects.inner.delete_chunk_map_receipt(&authority), + ) + .await + .map_err(|_| unavailable("retired receipt deletion timed out"))? + .map_err(storage_error)?; + } + } else if tokio::time::timeout(Duration::from_secs(5), objects.inner.exists(&key)) + .await + .map_err(|_| unavailable("unused receipt check timed out"))? + .map_err(storage_error)? + { + return Err(integrity( + "receipt exists for a generation that never prepared a create", + )); + } + let txn = self.transaction().await?; + let result=async { + self.require_scope(&txn).await?; + barrier(&txn,&self.schema).await?; + let applied=txn.query_one_raw(stmt(&self.schema,"UPDATE mst2_chunk_receipt_generation SET state='APPLIED',owner=NULL,deadline=NULL,reserved_pages=0,reserved_nodes=0,reserved_bytes=0,checked_at=pg_catalog.clock_timestamp() WHERE receipt_key=$1 AND generation=$2 AND state IN ('DELETING','APPLIED') AND NOT EXISTS(SELECT 1 FROM mst2_chunk_reader r WHERE r.receipt_key=$1 AND r.source_generation=$2 AND r.deadline>pg_catalog.clock_timestamp()) RETURNING generation",[key.key.clone().into(),generation.into()])).await.map_err(db_error)?; + if applied.is_none() { return Err(integrity("retired receipt lost its exact reader-free completion claim")); } + txn.execute_raw(stmt(&self.schema,"DELETE FROM mst2_chunk_map_source WHERE receipt_key=$1 AND receipt_generation=$2",[key.key.into(),generation.into()])).await.map_err(db_error)?; + Ok(()) + }.await; + finish(txn, result).await + } + + async fn claim_orphan( + &self, + objects: &MegaObjectStorageWrapper, + key: &ObjectKey, + ) -> Result<(), SnapshotError> { + let actual = read_bounded_orphan(objects, key).await?; + let Some(actual) = actual else { return Ok(()) }; + let (scope, source_bytes, map, generation) = decode_orphan(key, &actual)?; + if scope != self.primary_scope { + return Err(integrity( + "orphan receipt belongs to a different captured primary", + )); + } + let identity = source_id(&scope, &source_bytes); + let txn = self.transaction().await?; + let result=async { + self.require_scope(&txn).await?; + barrier(&txn,&self.schema).await?; + let count=txn.query_one_raw(stmt(&self.schema,"SELECT pg_catalog.count(*) AS history FROM mst2_chunk_receipt_generation",[])).await.map_err(db_error)?.ok_or_else(||integrity("orphan receipt history count missing"))?.try_get::("","history").map_err(db_error)?; + if count>=HISTORY_CAP { return Err(SnapshotError::new(SnapshotErrorCode::LimitExceeded,"chunk receipt reconciliation history quota exhausted")); } + // An orphan never admits a source or map. Its actual body only + // identifies a retired physical key for collection. + txn.execute_raw(stmt(&self.schema,"INSERT INTO mst2_chunk_receipt_generation(receipt_key,source_id,generation,source_bytes,primary_scope,map_id,receipt_bytes,state,reserved_pages,reserved_nodes,reserved_bytes) VALUES($1,$2,$3,$4,$5,$6,$7,'DELETING',0,0,0) ON CONFLICT(receipt_key) DO NOTHING",[key.key.clone().into(),identity.to_vec().into(),generation.into(),source_bytes.into(),scope.into(),map.map_id().to_vec().into(),actual.into()])).await.map_err(db_error)?; + Ok(()) + }.await; + finish(txn, result).await + } + + pub(super) async fn collect_maps(&self, limit: usize) -> Result<(), SnapshotError> { + let txn = self.transaction().await?; + self.require_scope(&txn).await?; + let rows=txn.query_all_raw(stmt(&self.schema,"SELECT l.map_id,l.generation,l.state FROM mst2_chunk_map_lifetime l WHERE l.state='DELETING' OR (NOT EXISTS(SELECT 1 FROM mst2_chunk_map_lifetime WHERE state='DELETING') AND (l.last_used<=pg_catalog.clock_timestamp()-interval '1 hour' OR EXISTS(SELECT 1 FROM mst2_chunk_receipt_generation g WHERE g.map_id=l.map_id AND g.map_generation=l.generation AND g.state IN ('DELETING','APPLIED'))) AND NOT EXISTS(SELECT 1 FROM mst2_chunk_map_source s WHERE s.map_id=l.map_id) AND NOT EXISTS(SELECT 1 FROM mst2_chunk_receipt_generation g WHERE g.map_id=l.map_id AND g.map_generation=l.generation AND g.state='CREATING' AND g.deadline>pg_catalog.clock_timestamp())) ORDER BY (l.state='DELETING') DESC,l.last_used,l.map_id LIMIT $1",[(limit as i64).into()])).await.map_err(db_error)?; + txn.commit().await.map_err(db_error)?; + for row in rows { + let map_id: Vec = row.try_get("", "map_id").map_err(db_error)?; + let generation: i64 = row.try_get("", "generation").map_err(db_error)?; + let txn = self.transaction().await?; + let result=async { + self.require_scope(&txn).await?; + let acquired=txn.query_one_raw(Statement::from_sql_and_values(DbBackend::Postgres,"SELECT pg_catalog.pg_try_advisory_xact_lock(1296717364,pg_catalog.hashtext($1)) AS acquired",[hex::encode(&map_id).into()])).await.map_err(db_error)?.ok_or_else(||integrity("map collection lock observation missing"))?.try_get::("","acquired").map_err(db_error)?; + if !acquired { return Ok(()); } + barrier(&txn,&self.schema).await?; + let eligible=txn.query_one_raw(stmt(&self.schema,"SELECT state FROM mst2_chunk_map_lifetime l WHERE map_id=$1 AND generation=$2 AND (state='DELETING' OR last_used<=pg_catalog.clock_timestamp()-interval '1 hour' OR EXISTS(SELECT 1 FROM mst2_chunk_receipt_generation g WHERE g.map_id=l.map_id AND g.map_generation=l.generation AND g.state IN ('DELETING','APPLIED'))) AND NOT EXISTS(SELECT 1 FROM mst2_chunk_map_source s WHERE s.map_id=$1) AND NOT EXISTS(SELECT 1 FROM mst2_chunk_reader r WHERE r.map_id=$1 AND r.map_generation=$2 AND r.deadline>pg_catalog.clock_timestamp()) AND NOT EXISTS(SELECT 1 FROM mst2_chunk_receipt_generation g WHERE g.map_id=$1 AND g.map_generation=$2 AND g.state='CREATING' AND g.deadline>pg_catalog.clock_timestamp())",[map_id.clone().into(),generation.into()])).await.map_err(db_error)?; + let Some(eligible)=eligible else { return Ok(()) }; + if eligible.try_get::("","state").map_err(db_error)?=="LIVE" { + let count=txn.query_one_raw(stmt(&self.schema,"SELECT pg_catalog.count(*) AS history FROM mst2_chunk_map_gc",[])).await.map_err(db_error)?.ok_or_else(||integrity("map collection history count missing"))?.try_get::("","history").map_err(db_error)?; + if count>=HISTORY_CAP { return Err(SnapshotError::new(SnapshotErrorCode::LimitExceeded,"chunk map collection history quota exhausted")); } + txn.execute_raw(stmt(&self.schema,"INSERT INTO mst2_chunk_map_gc(map_id,generation,state) VALUES($1,$2,'PENDING') ON CONFLICT(map_id,generation) DO NOTHING",[map_id.clone().into(),generation.into()])).await.map_err(db_error)?; + txn.execute_raw(stmt(&self.schema,"UPDATE mst2_chunk_map_lifetime SET state='DELETING' WHERE map_id=$1 AND generation=$2 AND state='LIVE'",[map_id.clone().into(),generation.into()])).await.map_err(db_error)?; + } + // One bounded resident batch per exact map generation. Large + // maps stay PENDING and are replayed before new map victims. + txn.execute_raw(stmt(&self.schema,"DELETE FROM mst2_chunk_map_leaf WHERE map_id=$1 AND page_index IN (SELECT page_index FROM mst2_chunk_map_leaf WHERE map_id=$1 ORDER BY page_index LIMIT 1024)",[map_id.clone().into()])).await.map_err(db_error)?; + txn.execute_raw(stmt(&self.schema,"DELETE FROM mst2_chunk_map_node WHERE (map_id,first_page,page_count) IN (SELECT map_id,first_page,page_count FROM mst2_chunk_map_node WHERE map_id=$1 ORDER BY first_page,page_count LIMIT 1024)",[map_id.clone().into()])).await.map_err(db_error)?; + let empty=txn.query_one_raw(stmt(&self.schema,"SELECT NOT EXISTS(SELECT 1 FROM mst2_chunk_map_leaf WHERE map_id=$1) AND NOT EXISTS(SELECT 1 FROM mst2_chunk_map_node WHERE map_id=$1) AS empty",[map_id.clone().into()])).await.map_err(db_error)?.ok_or_else(||integrity("map collection completion observation missing"))?.try_get::("","empty").map_err(db_error)?; + if empty { + // Deferred lifetime FK permits map deletion while its + // exact PENDING generation is still available to guards. + txn.execute_raw(stmt(&self.schema,"DELETE FROM mst2_chunk_map WHERE map_id=$1",[map_id.clone().into()])).await.map_err(db_error)?; + txn.execute_raw(stmt(&self.schema,"DELETE FROM mst2_chunk_map_lifetime WHERE map_id=$1 AND generation=$2 AND state='DELETING'",[map_id.clone().into(),generation.into()])).await.map_err(db_error)?; + txn.execute_raw(stmt(&self.schema,"UPDATE mst2_chunk_map_gc SET state='APPLIED',applied_at=pg_catalog.clock_timestamp() WHERE map_id=$1 AND generation=$2 AND state='PENDING'",[map_id.clone().into(),generation.into()])).await.map_err(db_error)?; + } + Ok(()) + }.await; + finish(txn, result).await?; + } + Ok(()) + } + + pub(crate) async fn admit_install( + &self, + source: &ChunkMapSource, + objects: &MegaObjectStorageWrapper, + ) -> Result { + // Replay first. Every backend await happens outside the primary + // completion barrier; active reservations count concurrent creates. + self.maintain_before_admission(objects).await?; + let inventory = inventory(self, objects).await?; + let source_bytes = source.canonical_bytes()?; + let identity = source_id(&self.primary_scope, &source_bytes); + let pages = (source.fact().size as u64) + .div_ceil(mst2_codec::chunkmap::CHUNK_SIZE as u64) + .div_ceil(CHUNKS_PER_PAGE as u64) as i32; + let nodes = pages * 2 - 1; + let reserved_bytes = pages as i64 * 12288 + nodes as i64 * 512 + 1024 * 1024; + let owner = Uuid::new_v4(); + let txn = self.transaction().await?; + let result=async { + self.require_scope(&txn).await?; + require_current_source(&txn,source,&self.schema).await?; + barrier(&txn,&self.schema).await?; + ensure_quotas(&txn,&self.schema,&inventory,pages,nodes,reserved_bytes,None).await?; + let row=txn.query_one_raw(stmt(&self.schema,"SELECT EXISTS(SELECT 1 FROM mst2_chunk_receipt_generation WHERE source_id=$1) AS used,EXISTS(SELECT 1 FROM mst2_chunk_receipt_generation WHERE source_id=$1 AND state IN ('RESERVED','CREATING','LIVE','DELETING')) AS occupied",[identity.to_vec().into()])).await.map_err(db_error)?.ok_or_else(|| integrity("chunk-map reservation observation missing"))?; + if row.try_get::("","occupied").map_err(db_error)? { return Err(unavailable("exact source already has a live or pending chunk-map generation")); } + let generation=if row.try_get::("","used").map_err(db_error)? { next_generation(&txn,&self.schema).await? } else { 1 }; + let key=physical_key(identity,generation); + let owner_started=Instant::now(); + txn.execute_raw(stmt(&self.schema,"INSERT INTO mst2_chunk_receipt_generation(receipt_key,source_id,generation,source_bytes,primary_scope,owner,state,reserved_pages,reserved_nodes,reserved_bytes,deadline) VALUES($1,$2,$3,$4,$5,$6::uuid,'RESERVED',$7,$8,$9,pg_catalog.clock_timestamp()+interval '59 seconds')",[key.key.clone().into(),identity.to_vec().into(),generation.into(),source_bytes.clone().into(),self.primary_scope.clone().into(),owner.to_string().into(),pages.into(),nodes.into(),reserved_bytes.into()])).await.map_err(db_error)?; + Ok(ChunkMapInstall{repository:self.clone(),source:source.clone(),owner,key,generation,progress:Mutex::new((Instant::now(),Instant::now(),0)),deadline:OwnerDeadline::from_sql_start(owner_started)}) + }.await; + // An uncertain commit is an error. Its UUID is never reused to mint + // another reservation; the durable deadline resolves any lost reply. + finish(txn, result).await + } + + pub(super) async fn admit_reader( + &self, + txn: &DatabaseTransaction, + source: &ChunkMapSource, + row: &QueryResult, + map: &ChunkMap, + ) -> Result { + barrier(txn, &self.schema).await?; + let key: String = row.try_get("", "receipt_key").map_err(db_error)?; + let generation: i64 = row.try_get("", "receipt_generation").map_err(db_error)?; + let map_generation: i64 = row + .try_get::>("", "map_generation") + .map_err(db_error)? + .ok_or_else(|| integrity("admitted chunk map lifetime is missing"))?; + let state: Option = row.try_get("", "generation_state").map_err(db_error)?; + let map_state: Option = row.try_get("", "map_state").map_err(db_error)?; + if state.as_deref() == Some("DELETING") { + return Err(unavailable("admitted chunk map generation is retiring")); + } + if state.as_deref() != Some("LIVE") || map_state.as_deref() != Some("LIVE") { + return Err(integrity( + "admitted source has a missing or terminal map lifetime", + )); + } + let identity = self.source_identity(source)?; + if key != physical_key(identity, generation).key + || bounded_bytes(row, "generation_receipt_bytes")? + != receipt(&self.primary_scope, source, map)? + { + return Err(integrity( + "chunk map lifetime and independent receipt identity disagree", + )); + } + txn.execute_raw(stmt(&self.schema,"DELETE FROM mst2_chunk_reader WHERE owner IN (SELECT owner FROM mst2_chunk_reader WHERE deadline<=pg_catalog.clock_timestamp() ORDER BY deadline,owner LIMIT 64)",[])).await.map_err(db_error)?; + let count = txn + .query_one_raw(stmt( + &self.schema, + "SELECT pg_catalog.count(*) AS readers FROM mst2_chunk_reader", + [], + )) + .await + .map_err(db_error)? + .ok_or_else(|| integrity("chunk reader count missing"))? + .try_get::("", "readers") + .map_err(db_error)?; + if count >= 4096 { + return Err(unavailable("chunk-map reader quota is occupied")); + } + let owner = Uuid::new_v4(); + let owner_started = Instant::now(); + txn.execute_raw(stmt(&self.schema,"INSERT INTO mst2_chunk_reader(owner,receipt_key,source_generation,map_id,map_generation,deadline) VALUES($1::uuid,$2,$3,$4,$5,pg_catalog.clock_timestamp()+interval '59 seconds')",[owner.to_string().into(),key.clone().into(),generation.into(),map.map_id().to_vec().into(),map_generation.into()])).await.map_err(db_error)?; + txn.execute_raw(stmt(&self.schema,"UPDATE mst2_chunk_map_lifetime SET last_used=pg_catalog.clock_timestamp() WHERE map_id=$1 AND generation=$2 AND state='LIVE'",[map.map_id().to_vec().into(),map_generation.into()])).await.map_err(db_error)?; + txn.execute_raw(stmt(&self.schema,"UPDATE mst2_chunk_receipt_generation SET last_progress=pg_catalog.clock_timestamp() WHERE receipt_key=$1 AND generation=$2 AND state='LIVE'",[key.clone().into(),generation.into()])).await.map_err(db_error)?; + Ok(ChunkMapReader { + repository: self.clone(), + source: source.clone(), + owner, + receipt_key: key, + generation, + map_id: map.map_id(), + map_generation, + confirmed: Mutex::new((Instant::now(), Instant::now())), + deadline: OwnerDeadline::from_sql_start(owner_started), + }) + } + + pub(super) async fn require_missing_source_retired( + &self, + txn: &DatabaseTransaction, + source: &ChunkMapSource, + ) -> Result { + let identity = self.source_identity(source)?; + let row=txn.query_one_raw(stmt(&self.schema,"SELECT state FROM mst2_chunk_receipt_generation WHERE source_id=$1 AND state IN ('LIVE','DELETING') ORDER BY generation DESC LIMIT 1",[identity.to_vec().into()])).await.map_err(db_error)?; + if let Some(row) = row { + if row.try_get::("", "state").map_err(db_error)? == "LIVE" { + return Err(integrity( + "live chunk-map source row is missing; never fall back to the cold body", + )); + } + return Ok(true); + } + Ok(false) + } +} + +pub(super) async fn barrier(txn: &DatabaseTransaction, schema: &str) -> Result<(), SnapshotError> { + txn.query_one_raw(stmt(schema, "SELECT mst2_chunk_retention_barrier()", [])) + .await + .map_err(db_error)?; + Ok(()) +} + +pub(super) async fn map_barrier( + txn: &DatabaseTransaction, + map_id: [u8; 32], +) -> Result<(), SnapshotError> { + // Per-map lock precedes the short global completion barrier. Collection + // never holds the global barrier while waiting for a map writer. + txn.query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT pg_catalog.pg_advisory_xact_lock(1296717364,pg_catalog.hashtext($1))", + [hex::encode(map_id).into()], + )) + .await + .map_err(db_error)?; + Ok(()) +} + +pub(super) async fn next_generation( + txn: &DatabaseTransaction, + schema: &str, +) -> Result { + txn.query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT pg_catalog.nextval($1::regclass) AS generation", + [format!("{}.mst2_chunk_generation_sequence", quoted(schema)).into()], + )) + .await + .map_err(db_error)? + .ok_or_else(|| integrity("chunk generation sequence missing"))? + .try_get("", "generation") + .map_err(db_error) +} + +fn physical_key(source: [u8; 32], generation: i64) -> ObjectKey { + ObjectKey { + namespace: ObjectNamespace::ChunkMapReceipt, + key: if generation == 1 { + hex::encode(source) + } else { + format!("{}-{generation:016x}", hex::encode(source)) + }, + } +} + +pub(super) async fn inventory( + repository: &PostgresChunkMapRepository, + objects: &MegaObjectStorageWrapper, +) -> Result, SnapshotError> { + let mut cached = repository.receipt_observation.lock().await; + if let Some(observation) = cached.as_ref().filter(|observation| { + Arc::ptr_eq(&observation.backing.inner, &objects.inner) + && observation.captured.elapsed() < RENEW_INTERVAL + }) { + return Ok(observation.clone()); + } + let memory = retention_budget().reserve(4 * 1024 * 1024)?; + let txn = repository.transaction().await?; + repository.require_scope(&txn).await?; + let started_at = txn + .query_one_raw(Statement::from_string( + DbBackend::Postgres, + "SELECT pg_catalog.clock_timestamp()::text AS started_at", + )) + .await + .map_err(db_error)? + .ok_or_else(|| integrity("backing inventory database clock missing"))? + .try_get("", "started_at") + .map_err(db_error)?; + txn.commit().await.map_err(db_error)?; + let inventory = tokio::time::timeout( + Duration::from_secs(30), + objects.inner.chunk_map_receipt_inventory(), + ) + .await + .map_err(|_| unavailable("bounded chunk receipt inventory timed out"))? + .map_err(|error| match error { + IoOrbitError::ChunkMapRetentionUnsupported => SnapshotError::new( + SnapshotErrorCode::TemporaryUnavailable, + "backing store does not support bounded chunk-map retention", + ), + IoOrbitError::ChunkMapRetentionCapacityExceeded => SnapshotError::new( + SnapshotErrorCode::LimitExceeded, + "actual backing chunk-map receipts exceed retention quota", + ), + error => storage_error(error), + })?; + let observation = Arc::new(BackingObservation { + inventory, + started_at, + captured: Instant::now(), + backing: objects.clone(), + _memory: memory, + }); + *cached = Some(observation.clone()); + Ok(observation) +} + +pub(super) async fn ensure_quotas( + txn: &DatabaseTransaction, + schema: &str, + observation: &BackingObservation, + pages: i32, + nodes: i32, + bytes: i64, + exclude: Option, +) -> Result<(), SnapshotError> { + let inventory = &observation.inventory; + let row=txn.query_one_raw(stmt(schema,"SELECT (SELECT pg_catalog.count(*) FROM mst2_chunk_map) AS maps,(SELECT pg_catalog.count(*) FROM mst2_chunk_map_source) AS sources,(SELECT pg_catalog.count(*) FROM mst2_chunk_map_leaf) AS leaves,(SELECT pg_catalog.count(*) FROM mst2_chunk_map_node) AS nodes,(SELECT pg_catalog.count(*) FROM mst2_chunk_receipt_generation) AS history,(SELECT pg_catalog.count(*) FROM mst2_chunk_map_gc) AS map_history,(SELECT pg_catalog.count(*) FROM mst2_chunk_receipt_generation WHERE receipt_bytes IS NOT NULL AND NOT create_completed) AS uncertain,(SELECT pg_catalog.count(*) FROM mst2_chunk_receipt_generation WHERE receipt_bytes IS NOT NULL AND (created_at>=$2::timestamptz OR create_completed_at>=$2::timestamptz)) AS recent,pg_catalog.count(*) AS pending,COALESCE(pg_catalog.sum(reserved_pages),0)::bigint AS pending_pages,COALESCE(pg_catalog.sum(reserved_nodes),0)::bigint AS pending_nodes,COALESCE(pg_catalog.sum(reserved_bytes),0)::bigint AS pending_bytes FROM mst2_chunk_receipt_generation WHERE state IN ('RESERVED','CREATING') AND ($1::uuid IS NULL OR owner IS DISTINCT FROM $1::uuid)",[exclude.map(|id|id.to_string()).into(),observation.started_at.clone().into()])).await.map_err(db_error)?.ok_or_else(|| integrity("chunk-map actual quota observation missing"))?; + let get = |name: &str| row.try_get::("", name).map_err(db_error); + let pending = get("pending")?; + let relation_bytes=txn.query_one_raw(Statement::from_sql_and_values(DbBackend::Postgres,"SELECT COALESCE(pg_catalog.sum(pg_catalog.pg_total_relation_size(c.oid)),0)::bigint AS bytes FROM pg_catalog.pg_class c JOIN pg_catalog.pg_namespace n ON n.oid=c.relnamespace WHERE n.nspname=$1 AND c.relkind='r' AND c.relname IN ('mst2_chunk_map','mst2_chunk_map_source','mst2_chunk_map_leaf','mst2_chunk_map_node','mst2_chunk_receipt_generation','mst2_chunk_map_lifetime','mst2_chunk_reader','mst2_chunk_map_gc')",[schema.into()])).await.map_err(db_error)?.ok_or_else(||integrity("chunk relation byte observation missing"))?.try_get::("","bytes").map_err(db_error)?; + let extra = if exclude.is_none() { 1 } else { 0 }; + if get("maps")? + pending + extra > 4096 + || get("sources")? + pending + extra > 16384 + || get("leaves")? + get("pending_pages")? + pages as i64 > 65536 + || get("nodes")? + get("pending_nodes")? + nodes as i64 > 131072 + || relation_bytes + get("pending_bytes")? + bytes + 1024 * 1024 > 536870912 + || get("history")? + extra > HISTORY_CAP + || get("map_history")? > HISTORY_CAP + || pending + extra > 64 + || inventory.objects.len() as i64 + pending + get("uncertain")? + get("recent")? + extra + > 16384 + || inventory.bytes + ((pending + get("uncertain")? + get("recent")? + extra) as u64) * 4096 + > 64 * 1024 * 1024 + { + return Err(SnapshotError::new( + SnapshotErrorCode::LimitExceeded, + "actual chunk-map rows, history, backing or reserved capacity exceed retention quota", + )); + } + Ok(()) +} + +fn unavailable(message: &str) -> SnapshotError { + SnapshotError::new(SnapshotErrorCode::TemporaryUnavailable, message) +} +fn expired(message: &str) -> SnapshotError { + SnapshotError::new(SnapshotErrorCode::LeaseExpired, message) +} + +async fn read_bounded_orphan( + objects: &MegaObjectStorageWrapper, + key: &ObjectKey, +) -> Result>, SnapshotError> { + tokio::time::timeout(Duration::from_secs(5), async { + let (mut stream, meta) = match objects.inner.get_stream(key).await { + Ok(value) => value, + Err(error) if error.is_not_found() => return Ok(None), + Err(error) => return Err(storage_error(error)), + }; + if !(129..=4096).contains(&meta.size) { + return Err(integrity("orphan receipt exceeds its byte profile")); + } + let mut bytes = Vec::with_capacity(meta.size as usize); + while let Some(part) = stream.next().await { + let part = part.map_err(storage_error)?; + if part.len() > meta.size as usize - bytes.len() { + return Err(integrity( + "orphan receipt exceeds its advertised exact size", + )); + } + bytes.extend_from_slice(&part); + } + if bytes.len() != meta.size as usize { + return Err(integrity("orphan receipt is truncated")); + } + Ok(Some(bytes)) + }) + .await + .map_err(|_| unavailable("orphan receipt read timed out"))? +} + +fn decode_orphan( + key: &ObjectKey, + bytes: &[u8], +) -> Result<(Vec, Vec, ChunkMap, i64), SnapshotError> { + if !bytes.starts_with(RECEIPT_DOMAIN) { + return Err(integrity("orphan is not a canonical trusted receipt")); + } + let mut offset = RECEIPT_DOMAIN.len(); + let take = |offset: &mut usize, max: usize| -> Result, SnapshotError> { + let len_bytes = bytes + .get(*offset..*offset + 4) + .ok_or_else(|| integrity("orphan receipt length is truncated"))?; + let len = u32::from_be_bytes( + len_bytes + .try_into() + .map_err(|_| integrity("invalid orphan receipt length"))?, + ) as usize; + *offset += 4; + if len == 0 || len > max { + return Err(integrity("orphan receipt field exceeds its profile")); + } + let field = bytes + .get(*offset..*offset + len) + .ok_or_else(|| integrity("orphan receipt field is truncated"))? + .to_vec(); + *offset += len; + Ok(field) + }; + let scope = take(&mut offset, 1024)?; + let source = take(&mut offset, 2048)?; + let descriptor = bytes + .get(offset..) + .ok_or_else(|| integrity("orphan map descriptor is missing"))?; + let map = ChunkMap::decode(descriptor) + .map_err(|_| integrity("orphan map descriptor is not canonical MCM2"))?; + if map.encode() != descriptor { + return Err(integrity("orphan receipt has noncanonical map bytes")); + } + let generation = if key.key.len() == 64 { + 1 + } else if key.key.len() == 81 && key.key.as_bytes()[64] == b'-' { + i64::from_str_radix(&key.key[65..], 16) + .map_err(|_| integrity("orphan physical generation is invalid"))? + } else { + return Err(integrity( + "orphan receipt key has invalid generation profile", + )); + }; + if generation <= 0 || physical_key(source_id(&scope, &source), generation) != *key { + return Err(integrity( + "orphan physical receipt key disagrees with its exact body", + )); + } + Ok((scope, source, map, generation)) +} diff --git a/src/jupiter/storage/native_metadata_generation_tests.rs b/src/jupiter/storage/native_metadata_generation_tests.rs new file mode 100644 index 00000000..b565c091 --- /dev/null +++ b/src/jupiter/storage/native_metadata_generation_tests.rs @@ -0,0 +1,906 @@ +use std::sync::Arc; + +use mst2_codec::metapage::{Entry, EntryKind}; +use sea_orm::{Database, PaginatorTrait}; +use sea_orm_migration::MigratorTrait; + +use super::*; +use crate::{ + callisto::mst2_metadata_lifetime, + ceres::snapshot::retention_dag::MetadataDagBuilder, + jupiter::{ + migration::Migrator, + storage::mst2_retention::GcClaim, + tests::{TestSchemaGuard, test_db_config}, + }, +}; + +fn prepared(scope: &str) -> PreparedNativeMetadataRetention { + let entries = [Entry::file(EntryKind::Regular, b"file", 3, [42; 32])]; + let child = Page::build(&entries).unwrap(); + let root_entries = [ + Entry::dir(b"one", page_id(&child)), + Entry::dir(b"two", page_id(&child)), + ]; + let root = Page::build(&root_entries).unwrap(); + let mut builder = MetadataDagBuilder::new(MetadataDagLimits::default()); + builder.add_directory(&child, &entries).unwrap(); + builder.add_directory(&root, &root_entries).unwrap(); + PreparedNativeMetadataRetention::test_installation( + Arc::new(builder.finish(page_id(&root)).unwrap()), + scope, + ) +} + +async fn fixture() -> (DatabaseConnection, DatabaseConnection, TestSchemaGuard) { + let temp = tempfile::tempdir().unwrap(); + let (config, schema) = test_db_config(temp.path()).await; + let first = Database::connect(config.db_url.clone()).await.unwrap(); + Migrator::up(&first, None).await.unwrap(); + let second = Database::connect(config.db_url).await.unwrap(); + (first, second, schema) +} + +fn rejected(error: MetadataInstallError) -> SnapshotErrorCode { + match error { + MetadataInstallError::Rejected(error) => error.code, + _ => panic!("expected definitive rejection"), + } +} + +async fn scalar(db: &DatabaseConnection, sql: &str) -> i64 { + db.query_one_raw(statement(sql, [])) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap() +} + +async fn mappings(db: &DatabaseConnection, intent: &GenerationPrepareIntent) -> GenerationBindings { + let repository = PostgresMetadataGenerationRepository::new(db.clone()) + .await + .unwrap(); + repository + .require_fixed_plan(db, intent) + .await + .unwrap() + .bindings +} + +async fn wait_for_retention_waiter(txn: &DatabaseTransaction) { + let deadline = tokio::time::Instant::now() + Duration::from_secs(4); + loop { + let waiting = txn.query_one_raw(statement( + "SELECT EXISTS(SELECT 1 FROM pg_locks WHERE locktype='advisory' AND NOT granted + AND classid=$1::integer::oid AND objid=hashtext(current_schema())::oid + AND database=(SELECT oid FROM pg_database WHERE datname=current_database())) AS waiting", + [RETENTION_LOCK_KEY.into()], + )).await.unwrap().unwrap().try_get::("", "waiting").unwrap(); + if waiting { + break; + } + assert!( + tokio::time::Instant::now() < deadline, + "independent connection must wait on the schema retention barrier" + ); + tokio::task::yield_now().await; + } +} + +#[tokio::test] +async fn generation_begin_serializes_full_reserved_bindings_before_any_payload() { + let (first, second, _schema) = fixture().await; + let a = PostgresMetadataGenerationRepository::new(first.clone()) + .await + .unwrap(); + let b = PostgresMetadataGenerationRepository::new(second.clone()) + .await + .unwrap(); + let pages = prepared("/"); + let left = a.begin_intent("reserved-a", &pages).await.unwrap(); + let held = a.inner.transaction().await.unwrap(); + a.inner.barrier(&held).await.unwrap(); + let right_pages = prepared("/"); + let blocked = tokio::spawn(async move { b.begin_intent("reserved-b", &right_pages).await }); + wait_for_retention_waiter(&held).await; + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_payload").await, + 0 + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_retention_root").await, + 0 + ); + held.commit().await.unwrap(); + let right = blocked.await.unwrap().unwrap(); + assert_ne!(left.prepare_id(), right.prepare_id()); + assert_ne!(left.storage_seal, right.storage_seal); + assert_eq!(left.bindings_digest, right.bindings_digest); + assert_eq!( + mappings(&first, &left).await, + mappings(&second, &right).await + ); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_lifetime WHERE generation=1 AND state='RESERVED'" + ) + .await, + 2 + ); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_prepare_page WHERE generation=1" + ) + .await, + 4 + ); + let replay = a.begin_intent("reserved-a", &pages).await.unwrap(); + assert_eq!(left, replay); +} + +#[tokio::test] +async fn generation_partial_prepare_recovers_exact_complete_mapping_on_second_connection() { + let (first, second, _schema) = fixture().await; + let a = PostgresMetadataGenerationRepository::new(first.clone()) + .await + .unwrap(); + let pages = prepared("/"); + let intent = a.begin_intent("partial", &pages).await.unwrap(); + a.install_pages(&intent, &pages.dag().payloads()[..1]) + .await + .unwrap(); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_payload").await, + 1 + ); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_prepare_page WHERE generation=1" + ) + .await, + 2 + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_retention_node").await, + 0 + ); + drop(a); + let restarted = PostgresMetadataGenerationRepository::new(second.clone()) + .await + .unwrap(); + assert_eq!( + restarted + .inspect_prepare( + &second, + "partial", + intent.manifest_digest(), + MetadataCommitPhase::Payload + ) + .await + .unwrap(), + GenerationPrepareObservation::Preparing(intent.clone()) + ); + assert_eq!( + restarted + .observe_installed_dag(&intent) + .await + .unwrap_err() + .code, + SnapshotErrorCode::ObjectUnavailable + ); + restarted + .install_pages(&intent, pages.dag().payloads()) + .await + .unwrap(); + let receipt = restarted.finalize(&intent).await.unwrap(); + assert_eq!(receipt.intent(), &intent); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_lifetime WHERE state='LIVE' AND generation=1" + ) + .await, + 2 + ); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_retention_root WHERE root_kind='prepare'" + ) + .await, + 2 + ); +} + +#[tokio::test] +async fn generation_shared_live_prepare_pins_batch_before_generic_gc_can_claim() { + let (first, second, _schema) = fixture().await; + let a = PostgresMetadataGenerationRepository::new(first.clone()) + .await + .unwrap(); + let b = PostgresMetadataGenerationRepository::new(second.clone()) + .await + .unwrap(); + let pages = prepared("/"); + let old = a.begin_intent("live-old", &pages).await.unwrap(); + a.install_pages(&old, pages.dag().payloads()).await.unwrap(); + a.finalize(&old).await.unwrap(); + let shared = b.begin_intent("live-shared", &pages).await.unwrap(); + assert_eq!( + mappings(&first, &old).await, + mappings(&second, &shared).await + ); + let graph = PostgresRetentionRepository::new(first.clone()); + graph + .release_root(&RetentionRoot::Prepare(old.prepare_id().into())) + .await + .unwrap(); + assert_eq!( + graph + .mark_deleting("generic-cannot-claim-shared", &node_id(&pages.dag().root())) + .await + .unwrap(), + GcClaim::Unavailable + ); + assert_eq!( + scalar( + &second, + "SELECT count(*) FROM mst2_retention_root WHERE root_kind='prepare'" + ) + .await, + 2 + ); + b.install_pages(&shared, pages.dag().payloads()) + .await + .unwrap(); + b.finalize(&shared).await.unwrap(); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_payload").await, + 2 + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_retention_edge").await, + 1 + ); + assert_eq!( + scalar(&first, "SELECT max(generation) FROM mst2_metadata_lifetime").await, + 1 + ); +} + +#[tokio::test] +async fn generation_observation_is_bound_to_prepare_not_only_the_same_dag() { + let (first, second, _schema) = fixture().await; + let a = PostgresMetadataGenerationRepository::new(first.clone()) + .await + .unwrap(); + let b = PostgresMetadataGenerationRepository::new(second) + .await + .unwrap(); + let pages = prepared("/"); + let left = a.begin_intent("observation-left", &pages).await.unwrap(); + let right = b.begin_intent("observation-right", &pages).await.unwrap(); + a.install_pages(&left, pages.dag().payloads()) + .await + .unwrap(); + let observation = a.observe_installed_dag(&left).await.unwrap(); + assert_eq!( + rejected( + b.finalize_observation(&right, &observation) + .await + .unwrap_err() + ), + SnapshotErrorCode::IntegrityError + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_retention_node").await, + 0 + ); + let receipt = a.finalize_observation(&left, &observation).await.unwrap(); + assert_eq!(receipt, a.finalize(&left).await.unwrap()); +} + +#[tokio::test] +async fn generation_finalization_rechecks_changed_lifetime_after_lock_free_observation() { + let (first, second, _schema) = fixture().await; + let repository = PostgresMetadataGenerationRepository::new(first.clone()) + .await + .unwrap(); + let pages = prepared("/"); + let intent = repository + .begin_intent("state-change", &pages) + .await + .unwrap(); + repository + .install_pages(&intent, pages.dag().payloads()) + .await + .unwrap(); + let observed = repository.observe_installed_dag(&intent).await.unwrap(); + let txn = second.begin().await.unwrap(); + repository.inner.barrier(&txn).await.unwrap(); + // Fault injection only: G1 deliberately has no collector transition API. + txn.execute_unprepared( + "ALTER TABLE mst2_metadata_lifetime DISABLE TRIGGER mst2_metadata_lifetime_guard", + ) + .await + .unwrap(); + txn.execute_raw(statement( + "UPDATE mst2_metadata_lifetime SET state='DELETING' WHERE page_id=$1", + [pages.dag().root().to_vec().into()], + )) + .await + .unwrap(); + txn.execute_unprepared( + "ALTER TABLE mst2_metadata_lifetime ENABLE TRIGGER mst2_metadata_lifetime_guard", + ) + .await + .unwrap(); + txn.commit().await.unwrap(); + assert_eq!( + rejected( + repository + .finalize_observation(&intent, &observed) + .await + .unwrap_err() + ), + SnapshotErrorCode::ObjectUnavailable + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_retention_node").await, + 0 + ); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_prepare WHERE state='PREPARING'" + ) + .await, + 1 + ); + assert_eq!( + mst2_metadata_payload::Entity::find() + .count(&first) + .await + .unwrap(), + 2 + ); +} + +#[tokio::test] +async fn generation_deleting_removed_and_generic_tombstone_are_never_reopened() { + for state in ["DELETING", "REMOVED"] { + let (first, second, _schema) = fixture().await; + let repository = PostgresMetadataGenerationRepository::new(first.clone()) + .await + .unwrap(); + let pages = prepared("/"); + let old = repository.begin_intent("old", &pages).await.unwrap(); + let txn = second.begin().await.unwrap(); + repository.inner.barrier(&txn).await.unwrap(); + txn.execute_unprepared( + "ALTER TABLE mst2_metadata_lifetime DISABLE TRIGGER mst2_metadata_lifetime_guard", + ) + .await + .unwrap(); + txn.execute_raw(statement( + "UPDATE mst2_metadata_lifetime SET state=$1 WHERE page_id=$2", + [state.into(), pages.dag().root().to_vec().into()], + )) + .await + .unwrap(); + txn.execute_unprepared( + "ALTER TABLE mst2_metadata_lifetime ENABLE TRIGGER mst2_metadata_lifetime_guard", + ) + .await + .unwrap(); + txn.commit().await.unwrap(); + assert_eq!( + rejected( + repository + .begin_intent("fresh-disallowed", &pages) + .await + .unwrap_err() + ), + SnapshotErrorCode::ObjectUnavailable + ); + assert_eq!( + rejected( + repository + .install_pages(&old, pages.dag().payloads()) + .await + .unwrap_err() + ), + SnapshotErrorCode::ObjectUnavailable + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_prepare").await, + 1 + ); + assert_eq!( + scalar(&first, "SELECT max(generation) FROM mst2_metadata_lifetime").await, + 1 + ); + } + let (first, second, _schema) = fixture().await; + let pages = prepared("/"); + let graph = PostgresRetentionRepository::new(second.clone()); + graph + .retain_group(pages.dag().nodes(), pages.dag().edges(), &[]) + .await + .unwrap(); + assert_eq!( + graph + .mark_deleting("legacy-remove", &node_id(&pages.dag().root())) + .await + .unwrap(), + GcClaim::Marked + ); + graph.complete_gc("legacy-remove").await.unwrap(); + let repository = PostgresMetadataGenerationRepository::new(first.clone()) + .await + .unwrap(); + assert_eq!( + rejected( + repository + .begin_intent("no-resurrection", &pages) + .await + .unwrap_err() + ), + SnapshotErrorCode::ObjectUnavailable + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_prepare").await, + 0 + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_lifetime").await, + 0 + ); + assert_eq!( + graph + .retain_group(pages.dag().nodes(), pages.dag().edges(), &[]) + .await + .unwrap_err() + .code, + SnapshotErrorCode::ObjectUnavailable + ); +} + +#[tokio::test] +async fn generation_legacy_null_capability_remains_legacy_and_does_not_block_existing_installer() { + let (first, second, _schema) = fixture().await; + let legacy = PostgresMetadataInstallRepository::new(first.clone()) + .await + .unwrap(); + let fresh = PostgresMetadataGenerationRepository::new(second.clone()) + .await + .unwrap(); + let pages = prepared("/"); + let old = legacy + .begin_intent("legacy-operation", &pages) + .await + .unwrap(); + legacy + .install_pages(&old, pages.dag().payloads()) + .await + .unwrap(); + let original_receipt = legacy.finalize(&old).await.unwrap(); + assert_eq!( + rejected( + fresh + .begin_intent("legacy-operation", &pages) + .await + .unwrap_err() + ), + SnapshotErrorCode::ObjectUnavailable + ); + assert_eq!( + rejected( + fresh + .begin_intent("new-over-legacy-cas", &pages) + .await + .unwrap_err() + ), + SnapshotErrorCode::IntegrityError + ); + assert_eq!(legacy.finalize(&old).await.unwrap(), original_receipt); + let another = legacy + .begin_intent("legacy-new-resolve", &pages) + .await + .unwrap(); + legacy + .install_pages(&another, pages.dag().payloads()) + .await + .unwrap(); + legacy.finalize(&another).await.unwrap(); + assert_eq!( + scalar( + &second, + "SELECT count(*) FROM mst2_metadata_payload WHERE generation IS NULL" + ) + .await, + 2 + ); + assert_eq!( + scalar( + &second, + "SELECT count(*) FROM mst2_metadata_prepare WHERE storage_seal IS NULL" + ) + .await, + 2 + ); + assert_eq!( + mst2_metadata_lifetime::Entity::find() + .count(&first) + .await + .unwrap(), + 0 + ); +} + +#[tokio::test] +async fn generation_storage_seal_scope_and_complete_child_binding_cannot_be_substituted() { + let (first, second, _schema) = fixture().await; + let repository = PostgresMetadataGenerationRepository::new(first.clone()) + .await + .unwrap(); + let pages = prepared("/"); + let intent = repository + .begin_intent("seal-tamper", &pages) + .await + .unwrap(); + repository + .install_pages(&intent, pages.dag().payloads()) + .await + .unwrap(); + let mut wrong_bindings = intent.clone(); + wrong_bindings.bindings_digest[0] ^= 1; + let mut wrong_seal = intent.clone(); + wrong_seal.storage_seal[0] ^= 1; + let mut wrong_root = intent.clone(); + wrong_root.metadata_root[0] ^= 1; + for wrong in [wrong_bindings, wrong_seal, wrong_root] { + assert_eq!( + rejected( + repository + .install_pages(&wrong, pages.dag().payloads()) + .await + .unwrap_err() + ), + SnapshotErrorCode::IntegrityError + ); + } + let mut wrong_scope = intent.clone(); + wrong_scope.primary_scope[0] ^= 1; + assert_eq!( + repository + .observe_installed_dag(&wrong_scope) + .await + .unwrap_err() + .code, + SnapshotErrorCode::IntegrityError + ); + let child = pages + .dag() + .payloads() + .iter() + .find(|page| page.id != pages.dag().root()) + .unwrap(); + let txn = second.begin().await.unwrap(); + repository.inner.barrier(&txn).await.unwrap(); + txn.execute_unprepared("ALTER TABLE mst2_metadata_prepare_page DISABLE TRIGGER mst2_metadata_generation_mapping_guard").await.unwrap(); + txn.execute_raw(statement("UPDATE mst2_metadata_prepare_page SET expected_size=expected_size+1 WHERE prepare_id=$1 AND page_id=$2", [intent.prepare_id().into(), child.id.to_vec().into()])).await.unwrap(); + txn.execute_unprepared("ALTER TABLE mst2_metadata_prepare_page ENABLE TRIGGER mst2_metadata_generation_mapping_guard").await.unwrap(); + txn.commit().await.unwrap(); + assert_eq!( + repository + .observe_installed_dag(&intent) + .await + .unwrap_err() + .code, + SnapshotErrorCode::IntegrityError + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_retention_node").await, + 0 + ); +} + +#[tokio::test] +async fn generation_compact_receipt_checks_actual_seal_without_loading_the_dag() { + let (first, second, _schema) = fixture().await; + let repository = PostgresMetadataGenerationRepository::new(first.clone()) + .await + .unwrap(); + let pages = prepared("/"); + let intent = repository.begin_intent("compact", &pages).await.unwrap(); + repository + .install_pages(&intent, pages.dag().payloads()) + .await + .unwrap(); + let receipt = repository.finalize(&intent).await.unwrap(); + let source = receipt.legacy.identity.tagged_root_tree_oid.clone(); + let txn = second.begin().await.unwrap(); + repository + .verify_receipt_in_txn(&txn, &receipt, &source, "/") + .await + .unwrap(); + txn.rollback().await.unwrap(); + let txn = second.begin().await.unwrap(); + repository.inner.barrier(&txn).await.unwrap(); + txn.execute_unprepared( + "ALTER TABLE mst2_metadata_prepare DISABLE TRIGGER mst2_metadata_generation_seal_guard", + ) + .await + .unwrap(); + txn.execute_raw(statement( + "UPDATE mst2_metadata_prepare SET storage_seal=$1 WHERE prepare_id=$2", + [vec![0u8; 32].into(), intent.prepare_id().into()], + )) + .await + .unwrap(); + txn.execute_unprepared( + "ALTER TABLE mst2_metadata_prepare ENABLE TRIGGER mst2_metadata_generation_seal_guard", + ) + .await + .unwrap(); + txn.commit().await.unwrap(); + let txn = second.begin().await.unwrap(); + assert_eq!( + repository + .verify_receipt_in_txn(&txn, &receipt, &source, "/") + .await + .unwrap_err() + .code, + SnapshotErrorCode::IntegrityError + ); + txn.rollback().await.unwrap(); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_payload").await, + 2 + ); + let txn = second.begin().await.unwrap(); + repository.inner.barrier(&txn).await.unwrap(); + txn.execute_unprepared( + "ALTER TABLE mst2_metadata_prepare DISABLE TRIGGER mst2_metadata_generation_seal_guard", + ) + .await + .unwrap(); + txn.execute_raw(statement( + "UPDATE mst2_metadata_prepare SET storage_seal=$1, + canonical_bindings=set_byte(canonical_bindings,60,get_byte(canonical_bindings,60)#1) WHERE prepare_id=$2", + [intent.storage_seal.to_vec().into(),intent.prepare_id().into()], + )).await.unwrap(); + txn.execute_unprepared( + "ALTER TABLE mst2_metadata_prepare ENABLE TRIGGER mst2_metadata_generation_seal_guard", + ) + .await + .unwrap(); + txn.commit().await.unwrap(); + let txn = second.begin().await.unwrap(); + assert_eq!( + repository + .verify_receipt_in_txn(&txn, &receipt, &source, "/") + .await + .unwrap_err() + .code, + SnapshotErrorCode::IntegrityError + ); + txn.rollback().await.unwrap(); +} + +#[tokio::test] +async fn generation_db_guards_preserve_watermark_payload_scope_and_committed_mappings() { + let (first, second, _schema) = fixture().await; + let repository = PostgresMetadataGenerationRepository::new(first.clone()) + .await + .unwrap(); + let pages = prepared("/"); + let intent = repository.begin_intent("guards", &pages).await.unwrap(); + repository + .install_pages(&intent, pages.dag().payloads()) + .await + .unwrap(); + repository.finalize(&intent).await.unwrap(); + for sql in [ + "DELETE FROM mst2_metadata_lifetime", + "UPDATE mst2_metadata_lifetime SET generation=2", + "UPDATE mst2_metadata_lifetime SET state='REMOVED'", + "DELETE FROM mst2_metadata_payload", + "UPDATE mst2_metadata_payload SET generation=1", + "DELETE FROM mst2_metadata_storage_scope", + "UPDATE mst2_metadata_storage_scope SET storage_uuid=storage_uuid", + "DELETE FROM mst2_metadata_prepare_page", + "UPDATE mst2_metadata_prepare_page SET generation=1", + "UPDATE mst2_metadata_prepare SET storage_seal=NULL,canonical_bindings=NULL,bindings_digest=NULL,primary_scope=NULL", + ] { + assert!( + second.execute_unprepared(sql).await.is_err(), + "guard must reject {sql}" + ); + } + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_lifetime WHERE generation=1 AND state='LIVE'" + ) + .await, + 2 + ); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_prepare_page WHERE generation=1" + ) + .await, + 2 + ); + assert_eq!( + repository.finalize(&intent).await.unwrap().intent(), + &intent + ); + assert_eq!(scalar(&first, "SELECT count(*) FROM pg_indexes WHERE schemaname=current_schema() AND indexname='idx_mst2_metadata_prepare_page_lifetime'").await, 1); +} + +#[tokio::test] +async fn generation_finalize_row_lock_fences_a_late_mapping_insert_on_another_connection() { + let (first, second, _schema) = fixture().await; + let repository = PostgresMetadataGenerationRepository::new(first.clone()) + .await + .unwrap(); + let pages = prepared("/"); + let intent = repository + .begin_intent("late-mapping", &pages) + .await + .unwrap(); + repository + .install_pages(&intent, pages.dag().payloads()) + .await + .unwrap(); + let observation = repository.observe_installed_dag(&intent).await.unwrap(); + let held = repository.inner.transaction().await.unwrap(); + repository.inner.barrier(&held).await.unwrap(); + lock_prepare(&held, intent.prepare_id()).await.unwrap(); + let contender = second.begin().await.unwrap(); + let pid = contender + .query_one_raw(statement("SELECT pg_backend_pid() AS pid", [])) + .await + .unwrap() + .unwrap() + .try_get::("", "pid") + .unwrap(); + let prepare_id = intent.prepare_id().to_owned(); + let root = pages.dag().root(); + let size = pages.install_plan().unwrap().pages[&root] as i32; + let blocked = tokio::spawn(async move { + let result = contender.execute_raw(statement( + "INSERT INTO mst2_metadata_prepare_page(prepare_id,page_id,generation,expected_size) VALUES($1,$2,1,$3)", + [prepare_id.into(), root.to_vec().into(), size.into()], + )).await; + contender.rollback().await.unwrap(); + result + }); + let deadline = tokio::time::Instant::now() + Duration::from_secs(4); + loop { + let waiting = held + .query_one_raw(statement( + "SELECT EXISTS(SELECT 1 FROM pg_locks WHERE pid=$1 AND NOT granted) AS waiting", + [pid.into()], + )) + .await + .unwrap() + .unwrap() + .try_get::("", "waiting") + .unwrap(); + if waiting { + break; + } + assert!( + tokio::time::Instant::now() < deadline, + "late mapping writer must wait for the prepare row lock" + ); + tokio::task::yield_now().await; + } + repository + .finalize_in_txn(&held, &intent, &observation) + .await + .unwrap(); + held.commit().await.unwrap(); + let error = blocked.await.unwrap().unwrap_err(); + assert!( + error + .to_string() + .contains("metadata generation mappings are immutable"), + "{error}" + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_prepare_page").await, + 2 + ); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_prepare WHERE state='COMMITTED'" + ) + .await, + 1 + ); + repository.finalize(&intent).await.unwrap(); +} + +#[tokio::test] +async fn generation_wrong_primary_never_upgrades_intent_or_reports_false_absence() { + let (first, same_primary, _schema) = fixture().await; + let (other, _other_connection, _other_schema) = fixture().await; + let repository = PostgresMetadataGenerationRepository::new(first.clone()) + .await + .unwrap(); + let wrong = PostgresMetadataGenerationRepository::new(other.clone()) + .await + .unwrap(); + let pages = prepared("/"); + let intent = repository + .begin_intent("captured-primary", &pages) + .await + .unwrap(); + repository + .install_pages(&intent, pages.dag().payloads()) + .await + .unwrap(); + let receipt = repository.finalize(&intent).await.unwrap(); + assert_eq!( + rejected( + wrong + .install_pages(&intent, pages.dag().payloads()) + .await + .unwrap_err() + ), + SnapshotErrorCode::IntegrityError + ); + assert_eq!( + wrong.observe_installed_dag(&intent).await.unwrap_err().code, + SnapshotErrorCode::IntegrityError + ); + assert!(matches!( + repository + .inspect_prepare( + &other, + "captured-primary", + intent.manifest_digest(), + MetadataCommitPhase::Finalize + ) + .await, + Err(MetadataInstallError::CommitUncertain { + phase: MetadataCommitPhase::Finalize, + .. + }) + )); + assert_eq!( + repository + .inspect_prepare( + &same_primary, + "captured-primary", + intent.manifest_digest(), + MetadataCommitPhase::Finalize + ) + .await + .unwrap(), + GenerationPrepareObservation::Committed(Box::new(receipt)) + ); + assert_eq!( + scalar(&other, "SELECT count(*) FROM mst2_metadata_payload").await, + 0 + ); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_retention_root WHERE root_kind='prepare'" + ) + .await, + 2 + ); +} diff --git a/src/jupiter/storage/native_metadata_generations.rs b/src/jupiter/storage/native_metadata_generations.rs new file mode 100644 index 00000000..dde51fc2 --- /dev/null +++ b/src/jupiter/storage/native_metadata_generations.rs @@ -0,0 +1,1003 @@ +//! Generation-bound preparation repository, isolated from HTTP lease authority. + +use sha2::{Digest, Sha256}; + +use super::*; + +const MAX_BINDINGS_BYTES: usize = 12 + 48 * 4096; +const MAX_PRIMARY_SCOPE_BYTES: usize = 16384; +const GENERIC_GRAPH_DOMAIN: &str = "generic-v1"; + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +enum GraphDomain { + LegacyGeneric, + Generic, + Qualified, +} + +impl GraphDomain { + fn from_stored(domain: Option<&str>) -> Result { + match domain { + None => Ok(Self::LegacyGeneric), + Some(GENERIC_GRAPH_DOMAIN) => Ok(Self::Generic), + Some("qualified-v1") => Ok(Self::Qualified), + _ => Err(integrity("unknown stored metadata graph domain")), + } + } + + fn stored(self) -> Option<&'static str> { + match self { + Self::LegacyGeneric => None, + Self::Generic => Some(GENERIC_GRAPH_DOMAIN), + Self::Qualified => Some("qualified-v1"), + } + } +} + +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct GenerationPrepareIntent { + legacy: MetadataPrepareIntent, + metadata_root: [u8; 32], + root_generation: i64, + bindings_digest: [u8; 32], + primary_scope: Box<[u8]>, + storage_seal: [u8; 32], + graph_domain: GraphDomain, +} + +impl GenerationPrepareIntent { + pub fn prepare_id(&self) -> &str { + self.legacy.prepare_id() + } + + pub fn manifest_digest(&self) -> [u8; 32] { + self.legacy.manifest_digest() + } + + pub fn operation_id(&self) -> &str { + self.legacy.operation_id() + } +} + +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct GenerationMetadataReceipt { + legacy: PreparedMetadataReceipt, + intent: GenerationPrepareIntent, +} + +impl GenerationMetadataReceipt { + pub fn intent(&self) -> &GenerationPrepareIntent { + &self.intent + } + + pub fn metadata_root(&self) -> [u8; 32] { + self.intent.metadata_root + } + + pub fn payload_bytes(&self) -> u64 { + self.legacy.payload_bytes() + } +} + +#[derive(Debug)] +pub struct GenerationDagObservation { + intent: GenerationPrepareIntent, + dag: ValidatedMetadataDag, +} + +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum GenerationPrepareObservation { + Absent, + Preparing(GenerationPrepareIntent), + Committed(Box), +} + +#[derive(Debug, Clone, PartialEq, Eq)] +struct GenerationBindings(BTreeMap<[u8; 32], (i64, u64)>); + +impl GenerationBindings { + fn encode(&self) -> Result, SnapshotError> { + if self.0.is_empty() || self.0.len() > MetadataDagLimits::default().nodes { + return Err(integrity("invalid metadata generation binding count")); + } + let mut bytes = Vec::with_capacity(12 + 48 * self.0.len()); + bytes.extend_from_slice(b"MST2GEN1"); + bytes.extend_from_slice(&(self.0.len() as u32).to_be_bytes()); + for (page, (generation, size)) in &self.0 { + if *generation <= 0 || !(HEADER_LEN as u64..=PAGE_MAX_BYTES as u64).contains(size) { + return Err(integrity("invalid metadata lifetime generation or size")); + } + bytes.extend_from_slice(page); + bytes.extend_from_slice(&generation.to_be_bytes()); + bytes.extend_from_slice(&size.to_be_bytes()); + } + Ok(bytes) + } + + fn decode(bytes: &[u8], plan: &MetadataInstallPlan) -> Result { + if bytes.len() < 60 || bytes.len() > MAX_BINDINGS_BYTES || &bytes[..8] != b"MST2GEN1" { + return Err(integrity("invalid metadata generation binding encoding")); + } + let count = u32::from_be_bytes(bytes[8..12].try_into().map_err(internal)?) as usize; + if count != plan.pages.len() || bytes.len() != 12 + count * 48 { + return Err(integrity( + "metadata generation binding coverage differs from plan", + )); + } + let mut bindings = BTreeMap::new(); + for entry in bytes[12..].as_chunks::<48>().0 { + let page: [u8; 32] = entry[..32].try_into().map_err(internal)?; + let generation = i64::from_be_bytes(entry[32..40].try_into().map_err(internal)?); + let size = u64::from_be_bytes(entry[40..48].try_into().map_err(internal)?); + if generation <= 0 + || plan.pages.get(&page) != Some(&size) + || bindings.insert(page, (generation, size)).is_some() + { + return Err(integrity("invalid fixed metadata lifetime binding")); + } + } + let result = Self(bindings); + if result.encode()? != bytes { + return Err(integrity("noncanonical metadata generation bindings")); + } + Ok(result) + } +} + +struct FixedPlan { + stored: StoredPlan, + bindings: GenerationBindings, + intent: GenerationPrepareIntent, +} + +#[derive(Clone)] +pub struct PostgresMetadataGenerationRepository { + inner: PostgresMetadataInstallRepository, + graph_domain: &'static str, +} + +impl PostgresMetadataGenerationRepository { + pub async fn new(connection: DatabaseConnection) -> Result { + Ok(Self { + inner: PostgresMetadataInstallRepository::new(connection).await?, + graph_domain: GENERIC_GRAPH_DOMAIN, + }) + } + + pub async fn begin_intent( + &self, + operation_id: &str, + prepared: &PreparedNativeMetadataRetention, + ) -> Result { + self.begin_with_reopens(operation_id, prepared, &[]).await + } + + async fn begin_with_reopens( + &self, + operation_id: &str, + prepared: &PreparedNativeMetadataRetention, + reopens: &[qualified::MetadataGcReceipt], + ) -> Result { + validate_operation_id(operation_id)?; + let plan = prepared.install_plan()?; + let digest = plan.digest()?; + let primary_scope = self.primary_scope()?; + let txn = self.inner.transaction().await?; + let result = async { + self.inner.barrier(&txn).await?; + if let Some(fixed) = self.load_fixed_plan(&txn, operation_id, &digest).await? { + qualified::verify_reopen_replay(&txn, &fixed, reopens).await?; + return Ok(fixed.intent); + } + if !reopens.is_empty() { + qualified::reopen_in_txn(&txn, &plan, reopens).await?; + } + let bindings = allocate_lifetimes(&txn, &plan, self.graph_domain).await?; + let canonical_bindings = bindings.encode()?; + let bindings_digest = Sha256::digest(&canonical_bindings).into(); + let prepare_id = uuid::Uuid::new_v4().to_string(); + let intent = GenerationPrepareIntent { + legacy: MetadataPrepareIntent { prepare_id: prepare_id.clone(), operation_id: operation_id.into(), manifest_digest: digest }, + metadata_root: plan.root, + root_generation: bindings.0.get(&plan.root).ok_or_else(|| integrity("root lifetime is missing"))?.0, + bindings_digest, + storage_seal: seal(&prepare_id, &digest, &plan.root, &bindings_digest, &primary_scope,Some(self.graph_domain))?, + primary_scope:primary_scope.into_boxed_slice(), + graph_domain:GraphDomain::from_stored(Some(self.graph_domain))?, + }; + let identity = &plan.identity; + txn.execute_raw(statement( + "INSERT INTO mst2_metadata_prepare(prepare_id,operation_id,manifest_digest,canonical_plan, + source_domain,tagged_root_tree_oid,scope,schema_version,metadata_codec,materialization_policy, + fs_semantics,access_projection,verification_revision,projection_revision,metadata_root, + node_count,edge_count,total_bytes,state,canonical_bindings,bindings_digest,primary_scope,storage_seal,graph_domain) + VALUES($1,$2,$3,$4,$5,$6,$7,$8,$9,$10,$11,$12,$13,$14,$15,$16,$17,$18,'PREPARING',$19,$20,$21,$22,$23)", + [prepare_id.clone().into(),operation_id.into(),digest.to_vec().into(),plan.encode()?.into(), + identity.source_domain.clone().into(),identity.tagged_root_tree_oid.clone().into(),identity.scope.clone().into(), + (identity.schema_version as i16).into(),(identity.metadata_codec as i16).into(), + (identity.materialization_policy as i16).into(),(identity.fs_semantics as i16).into(), + (identity.access_projection as i16).into(),identity.verification_revision.into(), + (identity.projection_revision as i16).into(),plan.root.to_vec().into(),(plan.pages.len() as i32).into(), + (plan.edges.len() as i32).into(),(plan.total_bytes as i64).into(),canonical_bindings.into(), + intent.bindings_digest.to_vec().into(),intent.primary_scope.to_vec().into(),intent.storage_seal.to_vec().into(),self.graph_domain.into()], + )).await.map_err(internal)?; + let pages: Vec<_> = bindings.0.iter().map(|(page,(generation,size))| + json!({"page_id":hex::encode(page),"generation":generation,"size":size})).collect(); + txn.execute_raw(statement( + "INSERT INTO mst2_metadata_prepare_page(prepare_id,page_id,generation,expected_size) + SELECT $1,decode(p.page_id,'hex'),p.generation,p.size + FROM jsonb_to_recordset($2::jsonb) AS p(page_id text,generation bigint,size integer)", + [prepare_id.clone().into(),serde_json::to_string(&pages).map_err(internal)?.into()], + )).await.map_err(internal)?; + // Reuse only already LIVE graph rows. RESERVED pages without a graph + // are protected by the durable mappings, without inventing a DAG. + if self.graph_domain == "qualified-v1" { + qualified::retain_existing_roots(&txn, &intent).await?; + } else { + txn.execute_raw(statement( + "INSERT INTO mst2_retention_root(node_id,root_key,root_kind,created_at) + SELECT l.node_id,'prepare:'||p.prepare_id,'prepare',now() + FROM mst2_metadata_prepare_page p JOIN mst2_metadata_lifetime l + ON l.page_id=p.page_id AND l.generation=p.generation + JOIN mst2_retention_node n ON n.node_id=l.node_id AND n.state='LIVE' + WHERE p.prepare_id=$1 ON CONFLICT(node_id,root_key) DO NOTHING", + [prepare_id.into()], + )).await.map_err(internal)?; + } + Ok(intent) + }.await; + commit( + txn, + result, + operation_id, + digest, + MetadataCommitPhase::Intent, + ) + .await + } + + pub async fn install_page( + &self, + intent: &GenerationPrepareIntent, + payload: &MetadataPagePayload, + ) -> Result<(), MetadataInstallError> { + self.install_pages(intent, std::slice::from_ref(payload)) + .await + } + + pub async fn install_pages( + &self, + intent: &GenerationPrepareIntent, + payloads: &[MetadataPagePayload], + ) -> Result<(), MetadataInstallError> { + if payloads.is_empty() || payloads.len() > 64 { + return Err(SnapshotError::new( + SnapshotErrorCode::LimitExceeded, + "metadata installation batch must contain 1..=64 pages", + ) + .into()); + } + let mut ids = BTreeSet::new(); + for payload in payloads { + validate_payload(payload)?; + if !ids.insert(payload.id) { + return Err(integrity("duplicate page in metadata installation batch").into()); + } + } + let txn = self.inner.transaction().await?; + let result = async { + self.inner.barrier(&txn).await?; + let fixed = self.require_fixed_plan(&txn, intent).await?; + if fixed.stored.record.state == "COMMITTED" { + check_generation_payloads(&txn, &fixed).await?; + } + let mut pages = Vec::with_capacity(payloads.len()); + for page in payloads { + let &(generation,size) = fixed.bindings.0.get(&page.id) + .ok_or_else(|| integrity("metadata payload is not a member of its fixed generation installation"))?; + if size != page.size { + return Err(integrity("metadata payload size differs from its fixed lifetime")); + } + pages.push(json!({"page_id":hex::encode(page.id),"generation":generation,"size":size,"payload":hex::encode(&page.bytes)})); + } + let encoded = serde_json::to_string(&pages).map_err(internal)?; + txn.execute_raw(statement( + "INSERT INTO mst2_metadata_payload(page_id,generation,metadata_codec,byte_size,payload) + SELECT decode(p.page_id,'hex'),p.generation,$1,p.size,decode(p.payload,'hex') + FROM jsonb_to_recordset($2::jsonb) AS p(page_id text,generation bigint,size integer,payload text) + WHERE NOT EXISTS(SELECT 1 FROM mst2_metadata_payload b WHERE b.page_id=decode(p.page_id,'hex')) + ON CONFLICT(page_id) DO NOTHING", + [fixed.stored.record.metadata_codec.into(),encoded.clone().into()], + )).await.map_err(internal)?; + let bad=txn.query_one_raw(statement( + "SELECT p.page_id FROM jsonb_to_recordset($2::jsonb) AS p(page_id text,generation bigint,size integer,payload text) + LEFT JOIN mst2_metadata_payload b ON b.page_id=decode(p.page_id,'hex') + WHERE b.page_id IS NULL OR b.generation IS DISTINCT FROM p.generation OR b.metadata_codec<>$1 + OR b.byte_size<>p.size OR b.payload<>decode(p.payload,'hex') LIMIT 1", + [fixed.stored.record.metadata_codec.into(),encoded.into()], + )).await.map_err(internal)?; + if bad.is_some() { + return Err(integrity("immutable payload conflicts with its exact metadata lifetime")); + } + Ok(()) + }.await; + commit( + txn, + result, + &intent.legacy.operation_id, + intent.manifest_digest(), + MetadataCommitPhase::Payload, + ) + .await + } + + /// Read, hash and parse the actual fixed-generation bytes outside the lock. + pub async fn observe_installed_dag( + &self, + intent: &GenerationPrepareIntent, + ) -> Result { + self.inner + .verify_primary_connection(&self.inner.connection) + .await?; + let fixed = self + .require_fixed_plan(&self.inner.connection, intent) + .await?; + let dag = load_fixed_dag(&self.inner.connection, &fixed).await?; + Ok(GenerationDagObservation { + intent: intent.clone(), + dag, + }) + } + + pub async fn finalize( + &self, + intent: &GenerationPrepareIntent, + ) -> Result { + let observation = self.observe_installed_dag(intent).await?; + self.finalize_observation(intent, &observation).await + } + + pub async fn finalize_observation( + &self, + intent: &GenerationPrepareIntent, + observation: &GenerationDagObservation, + ) -> Result { + if &observation.intent != intent { + return Err( + integrity("metadata observation belongs to another fixed installation").into(), + ); + } + let txn = self.inner.transaction().await?; + let result = self.finalize_in_txn(&txn, intent, observation).await; + commit( + txn, + result, + &intent.legacy.operation_id, + intent.manifest_digest(), + MetadataCommitPhase::Finalize, + ) + .await + } + + // Only a successful outer commit or the recovery barrier exposes a receipt. + async fn finalize_in_txn( + &self, + txn: &DatabaseTransaction, + intent: &GenerationPrepareIntent, + observation: &GenerationDagObservation, + ) -> Result { + if &observation.intent != intent { + return Err(integrity( + "metadata observation belongs to another fixed installation", + )); + } + self.inner.barrier(txn).await?; + // Mapping INSERT guards take this same row lock. No late insertion + // may cross coverage validation and the COMMITTED transition. + lock_prepare(txn, intent.prepare_id()).await?; + let fixed = self.require_fixed_plan(txn, intent).await?; + check_generation_payloads(txn, &fixed).await?; + let legacy = if intent.graph_domain == GraphDomain::Qualified { + if self.inner.qualified_family.is_some() { + qualified::finalize_canonical_graph(txn, &fixed, &observation.dag).await? + } else { + qualified::finalize_graph(txn, &fixed, &observation.dag).await? + } + } else { + self.inner + .finalize_stored_plan_in_txn(txn, &intent.legacy, &observation.dag, fixed.stored) + .await? + }; + txn.execute_raw(statement( + "UPDATE mst2_metadata_lifetime l SET state='LIVE' + FROM mst2_metadata_prepare_page p WHERE p.prepare_id=$1 AND p.page_id=l.page_id + AND p.generation=l.generation AND l.state='RESERVED'", + [intent.prepare_id().into()], + )) + .await + .map_err(internal)?; + Ok(GenerationMetadataReceipt { + legacy, + intent: intent.clone(), + }) + } + + /// Compact proof for a future atomic handoff. No DAG/page scan under the + /// mono writer lock: hash <=196620 binding bytes and check the root lifetime. + pub(crate) async fn verify_receipt_in_txn( + &self, + txn: &DatabaseTransaction, + receipt: &GenerationMetadataReceipt, + tagged_root_tree_oid: &str, + scope: &str, + ) -> Result<(), SnapshotError> { + self.require_scope(&receipt.intent)?; + if receipt.intent.graph_domain == GraphDomain::Qualified { + return Err(unavailable("qualified production handoff is not admitted")); + } + self.inner + .verify_receipt_in_txn(txn, &receipt.legacy, tagged_root_tree_oid, scope) + .await?; + let row=txn.query_one_raw(statement( + "SELECT p.bindings_digest,p.primary_scope,p.storage_seal,p.graph_domain,p.coverage_retired_at IS NOT NULL AS retired, + CASE WHEN octet_length(p.canonical_bindings)<=$2 THEN sha256(p.canonical_bindings) END AS actual_bindings_digest, + l.generation,l.state,l.metadata_codec,l.expected_size,m.generation AS root_generation, + b.generation AS payload_generation,b.byte_size + FROM mst2_metadata_prepare p LEFT JOIN mst2_metadata_prepare_page m + ON m.prepare_id=p.prepare_id AND m.page_id=p.metadata_root + LEFT JOIN mst2_metadata_lifetime l ON l.page_id=m.page_id AND l.generation=m.generation + LEFT JOIN mst2_metadata_current c ON c.page_id=l.page_id AND c.generation=l.generation + LEFT JOIN mst2_metadata_payload b ON b.page_id=m.page_id + WHERE p.prepare_id=$1 AND c.page_id IS NOT NULL", + [receipt.intent.prepare_id().into(),(MAX_BINDINGS_BYTES as i32).into()], + )).await.map_err(internal)?.ok_or_else(|| unavailable("metadata generation receipt is missing"))?; + let domain = row + .try_get::>("", "graph_domain") + .map_err(internal)?; + if GraphDomain::from_stored(domain.as_deref())? != receipt.intent.graph_domain { + return Err(integrity( + "metadata generation receipt graph domain differs from storage", + )); + } + if row.try_get::("", "retired").map_err(internal)? { + return Err(unavailable( + "metadata generation receipt coverage was retired", + )); + } + for (column, expected) in [ + ("bindings_digest", receipt.intent.bindings_digest.as_slice()), + ( + "actual_bindings_digest", + receipt.intent.bindings_digest.as_slice(), + ), + ("primary_scope", receipt.intent.primary_scope.as_ref()), + ("storage_seal", receipt.intent.storage_seal.as_slice()), + ] { + if row + .try_get::>>("", column) + .map_err(internal)? + .as_deref() + != Some(expected) + { + return Err(integrity( + "metadata generation receipt seal differs from durable storage", + )); + } + } + let generation = row + .try_get::>("", "generation") + .map_err(internal)?; + if generation != Some(receipt.intent.root_generation) + || row + .try_get::>("", "root_generation") + .map_err(internal)? + != generation + || row + .try_get::>("", "payload_generation") + .map_err(internal)? + != generation + || row + .try_get::>("", "state") + .map_err(internal)? + .as_deref() + != Some("LIVE") + || row + .try_get::>("", "metadata_codec") + .map_err(internal)? + != Some(receipt.legacy.identity.metadata_codec as i16) + || row + .try_get::>("", "expected_size") + .map_err(internal)? + != Some(receipt.legacy.root_payload_bytes as i32) + || row + .try_get::>("", "byte_size") + .map_err(internal)? + != Some(receipt.legacy.root_payload_bytes as i32) + { + return Err(unavailable( + "metadata generation receipt root lifetime is unavailable", + )); + } + Ok(()) + } + + pub async fn inspect_prepare( + &self, + fresh_primary: &DatabaseConnection, + operation_id: &str, + digest: [u8; 32], + phase: MetadataCommitPhase, + ) -> Result { + validate_operation_id(operation_id)?; + let txn = fresh_primary + .begin_with_config(Some(IsolationLevel::ReadCommitted), None) + .await + .map_err(|_| uncertain(operation_id, digest, phase))?; + if self.inner.barrier(&txn).await.is_err() { + let _ = txn.rollback().await; + return Err(uncertain(operation_id, digest, phase)); + } + let result = async { + match self.load_fixed_plan(&txn, operation_id, &digest).await? { + None => Ok(GenerationPrepareObservation::Absent), + Some(fixed) if fixed.stored.record.state == "COMMITTED" => { + check_generation_payloads(&txn, &fixed).await?; + load_fixed_dag(&txn, &fixed).await?; + verify_fixed_graph(&txn, &fixed).await?; + Ok(GenerationPrepareObservation::Committed(Box::new( + GenerationMetadataReceipt { + legacy: fixed.stored.receipt()?, + intent: fixed.intent, + }, + ))) + } + Some(fixed) => Ok(GenerationPrepareObservation::Preparing(fixed.intent)), + } + } + .await; + txn.rollback() + .await + .map_err(|_| uncertain(operation_id, digest, phase))?; + result.map_err(|error: SnapshotError| { + if error.code == SnapshotErrorCode::Internal { + uncertain(operation_id, digest, phase) + } else { + MetadataInstallError::Rejected(error) + } + }) + } + + async fn require_fixed_plan( + &self, + connection: &C, + intent: &GenerationPrepareIntent, + ) -> Result { + self.require_scope(intent)?; + let fixed = self + .load_fixed_plan( + connection, + &intent.legacy.operation_id, + &intent.manifest_digest(), + ) + .await? + .ok_or_else(|| unavailable("fixed metadata preparation is missing"))?; + if &fixed.intent != intent { + return Err(integrity( + "metadata intent differs from its fixed generation seal", + )); + } + Ok(fixed) + } + + async fn load_fixed_plan( + &self, + connection: &C, + operation_id: &str, + digest: &[u8; 32], + ) -> Result, SnapshotError> { + let Some(stored) = load_plan(connection, operation_id, digest).await? else { + return Ok(None); + }; + if stored.record.coverage_retired_at.is_some() { + return Err(unavailable("metadata preparation coverage was retired")); + } + let canonical = stored + .record + .canonical_bindings + .as_deref() + .ok_or_else(|| unavailable("legacy preparation has no generation seal"))?; + let bindings = GenerationBindings::decode(canonical, &stored.plan)?; + let bindings_digest: [u8; 32] = Sha256::digest(canonical).into(); + let primary_scope = stored + .record + .primary_scope + .clone() + .ok_or_else(|| integrity("fixed preparation primary scope is missing"))?; + let storage_seal = seal( + &stored.record.prepare_id, + digest, + &stored.plan.root, + &bindings_digest, + &primary_scope, + stored.record.graph_domain.as_deref(), + )?; + if stored.record.bindings_digest.as_deref() != Some(bindings_digest.as_slice()) + || stored.record.storage_seal.as_deref() != Some(storage_seal.as_slice()) + { + return Err(integrity("metadata generation seal is corrupt")); + } + let intent = GenerationPrepareIntent { + legacy: stored.intent()?, + metadata_root: stored.plan.root, + root_generation: bindings + .0 + .get(&stored.plan.root) + .ok_or_else(|| integrity("root lifetime is missing"))? + .0, + bindings_digest, + primary_scope: primary_scope.into_boxed_slice(), + storage_seal, + graph_domain: GraphDomain::from_stored(stored.record.graph_domain.as_deref())?, + }; + self.require_scope(&intent)?; + let mut actual = BTreeMap::new(); + for row in &stored.prepare_pages { + let page: [u8; 32] = row.page_id.as_slice().try_into().map_err(internal)?; + let generation = row + .generation + .ok_or_else(|| integrity("fixed metadata page has no generation"))?; + actual.insert(page, (generation, row.expected_size as u64)); + } + if actual != bindings.0 { + return Err(integrity( + "metadata mappings differ from complete fixed generation bindings", + )); + } + check_lifetimes(connection, &stored).await?; + Ok(Some(FixedPlan { + stored, + bindings, + intent, + })) + } + + fn primary_scope(&self) -> Result, SnapshotError> { + let scope = &self.inner.storage_scope; + let bytes = serde_json::to_vec(&( + &scope.storage_uuid, + &scope.database, + scope.database_oid, + &scope.schema, + scope.schema_oid, + &scope.server_address, + scope.server_port, + )) + .map_err(internal)?; + if bytes.is_empty() || bytes.len() > MAX_PRIMARY_SCOPE_BYTES { + return Err(integrity("invalid captured metadata primary scope size")); + } + Ok(bytes) + } + + fn require_scope(&self, intent: &GenerationPrepareIntent) -> Result<(), SnapshotError> { + if intent.primary_scope.as_ref() != self.primary_scope()?.as_slice() { + return Err(integrity( + "metadata seal belongs to another captured primary storage scope", + )); + } + if intent.graph_domain.stored().unwrap_or(GENERIC_GRAPH_DOMAIN) != self.graph_domain { + return Err(integrity( + "metadata seal belongs to another immutable graph domain", + )); + } + Ok(()) + } +} + +fn seal( + prepare_id: &str, + manifest: &[u8; 32], + root: &[u8; 32], + bindings: &[u8; 32], + scope: &[u8], + graph_domain: Option<&str>, +) -> Result<[u8; 32], SnapshotError> { + if scope.is_empty() || scope.len() > MAX_PRIMARY_SCOPE_BYTES { + return Err(integrity("invalid metadata generation seal primary scope")); + } + let id = uuid::Uuid::parse_str(prepare_id).map_err(internal)?; + let mut hash = Sha256::new(); + match graph_domain { + None => hash.update(b"MST2-METADATA-STORAGE-SEAL-1\0"), + Some(domain) => { + if ![GENERIC_GRAPH_DOMAIN, "qualified-v1"].contains(&domain) { + return Err(integrity("unknown metadata graph domain")); + } + hash.update(b"MST2-METADATA-STORAGE-SEAL-2\0"); + hash.update((domain.len() as u32).to_be_bytes()); + hash.update(domain.as_bytes()); + } + } + hash.update(id.as_bytes()); + hash.update(manifest); + hash.update(root); + hash.update(bindings); + hash.update((scope.len() as u32).to_be_bytes()); + hash.update(scope); + Ok(hash.finalize().into()) +} + +async fn allocate_lifetimes( + txn: &DatabaseTransaction, + plan: &MetadataInstallPlan, + graph_domain: &str, +) -> Result { + if graph_domain == "qualified-v1" { + return qualified::allocate_lifetimes(txn, plan).await; + } + let pages: Vec<_> = plan + .pages + .iter() + .map(|(page, size)| json!({"page_id":hex::encode(page),"size":size})) + .collect(); + let encoded = serde_json::to_string(&pages).map_err(internal)?; + txn.execute_raw(statement( + "INSERT INTO mst2_metadata_lifetime(page_id,node_id,generation,state,metadata_codec,expected_size) + SELECT decode(p.page_id,'hex'),'page:sha256:'||p.page_id,1,'RESERVED',$1,p.size + FROM jsonb_to_recordset($2::jsonb) AS p(page_id text,size integer) + ON CONFLICT(page_id,generation) DO NOTHING", + [(plan.identity.metadata_codec as i16).into(),encoded.clone().into()], + )).await.map_err(internal)?; + txn.execute_raw(statement( + "INSERT INTO mst2_metadata_current(page_id,generation) + SELECT decode(p.page_id,'hex'),1 FROM jsonb_to_recordset($1::jsonb) AS p(page_id text,size integer) + ON CONFLICT(page_id) DO NOTHING", + [encoded.clone().into()], + )).await.map_err(internal)?; + if txn.query_one_raw(statement( + "SELECT p.page_id FROM jsonb_to_recordset($1::jsonb) AS p(page_id text,size integer) + JOIN mst2_metadata_prepare_page m ON m.page_id=decode(p.page_id,'hex') + JOIN mst2_metadata_current c ON c.page_id=m.page_id AND c.generation=m.generation + JOIN mst2_metadata_prepare q ON q.prepare_id=m.prepare_id + WHERE coalesce(q.graph_domain,'generic-v1')<>$2 + AND (q.state='PREPARING' OR (q.state='COMMITTED' AND q.coverage_retired_at IS NULL)) LIMIT 1", + [encoded.clone().into(),graph_domain.into()], + )).await.map_err(internal)?.is_some() { + return Err(unavailable("metadata lifetime is covered by another graph domain")); + } + let rows=txn.query_all_raw(statement( + "SELECT decode(p.page_id,'hex') AS page_id,p.size,l.generation,l.state,l.metadata_codec,l.expected_size, + n.state AS graph_state,n.kind AS graph_kind,n.bytes AS graph_bytes, + b.page_id IS NOT NULL AS payload_present,b.generation AS payload_generation, + b.metadata_codec AS payload_codec,b.byte_size AS payload_size, + EXISTS(SELECT 1 FROM mst2_retention_gc_op g WHERE g.node_id=l.node_id + AND g.operation='REMOVE' AND g.state IN ('PENDING','APPLIED')) AS tombstone + FROM jsonb_to_recordset($1::jsonb) AS p(page_id text,size integer) + JOIN mst2_metadata_current c ON c.page_id=decode(p.page_id,'hex') + JOIN mst2_metadata_lifetime l ON l.page_id=c.page_id AND l.generation=c.generation + LEFT JOIN mst2_retention_node n ON n.node_id=l.node_id + LEFT JOIN mst2_metadata_payload b ON b.page_id=l.page_id ORDER BY l.page_id", + [encoded.into()], + )).await.map_err(internal)?; + let mut bindings = BTreeMap::new(); + for row in rows { + let state: String = row.try_get("", "state").map_err(internal)?; + let generation: i64 = row.try_get("", "generation").map_err(internal)?; + let size: i32 = row.try_get("", "size").map_err(internal)?; + if state == "LIVE" + && (!row + .try_get::("", "payload_present") + .map_err(internal)? + || row + .try_get::>("", "graph_state") + .map_err(internal)? + .is_none()) + { + return Err(unavailable( + "LIVE metadata lifetime has lost its payload or graph", + )); + } + if !["RESERVED", "LIVE"].contains(&state.as_str()) + || row.try_get::("", "tombstone").map_err(internal)? + || row + .try_get::>("", "graph_state") + .map_err(internal)? + .is_some_and(|state| state != "LIVE") + { + return Err(unavailable("metadata lifetime is deleting or removed")); + } + if generation <= 0 + || row.try_get::("", "metadata_codec").map_err(internal)? + != plan.identity.metadata_codec as i16 + || row.try_get::("", "expected_size").map_err(internal)? != size + || row + .try_get::>("", "graph_kind") + .map_err(internal)? + .is_some_and(|kind| kind != "page") + || row + .try_get::>("", "graph_bytes") + .map_err(internal)? + .is_some_and(|bytes| bytes != size as i64) + { + return Err(integrity( + "metadata lifetime profile conflicts with fixed plan", + )); + } + if row + .try_get::("", "payload_present") + .map_err(internal)? + && (row + .try_get::>("", "payload_generation") + .map_err(internal)? + != Some(generation) + || row + .try_get::>("", "payload_codec") + .map_err(internal)? + != Some(plan.identity.metadata_codec as i16) + || row + .try_get::>("", "payload_size") + .map_err(internal)? + != Some(size)) + { + return Err(integrity( + "existing metadata payload has another or unbound lifetime", + )); + } + let page: Vec = row.try_get("", "page_id").map_err(internal)?; + bindings.insert( + page.as_slice().try_into().map_err(internal)?, + (generation, size as u64), + ); + } + if bindings.len() != plan.pages.len() { + return Err(integrity("incomplete metadata lifetime allocation")); + } + Ok(GenerationBindings(bindings)) +} + +async fn check_lifetimes( + connection: &C, + stored: &StoredPlan, +) -> Result<(), SnapshotError> { + if stored.record.graph_domain.as_deref() == Some("qualified-v1") { + return qualified::check_lifetimes(connection, stored).await; + } + let row=connection.query_one_raw(statement( + "SELECT p.page_id,l.generation,l.state,l.metadata_codec,l.expected_size,p.generation AS expected_generation, + n.state AS graph_state,n.kind AS graph_kind,n.bytes AS graph_bytes, + EXISTS(SELECT 1 FROM mst2_retention_gc_op g WHERE g.node_id=l.node_id + AND g.operation='REMOVE' AND g.state IN ('PENDING','APPLIED')) AS tombstone + FROM mst2_metadata_prepare_page p LEFT JOIN mst2_metadata_lifetime l ON l.page_id=p.page_id AND l.generation=p.generation + LEFT JOIN mst2_metadata_current c ON c.page_id=p.page_id AND c.generation=p.generation + LEFT JOIN mst2_retention_node n ON n.node_id=l.node_id WHERE p.prepare_id=$1 + AND (l.page_id IS NULL OR c.page_id IS NULL OR l.generation IS DISTINCT FROM p.generation OR l.metadata_codec<>$2 + OR l.expected_size<>p.expected_size OR l.state NOT IN ('RESERVED','LIVE') + OR ($3='COMMITTED' AND l.state<>'LIVE') + OR (l.state='LIVE' AND n.node_id IS NULL) + OR n.state<>'LIVE' OR n.kind<>'page' OR n.bytes<>p.expected_size + OR EXISTS(SELECT 1 FROM mst2_retention_gc_op g WHERE g.node_id=l.node_id + AND g.operation='REMOVE' AND g.state IN ('PENDING','APPLIED'))) LIMIT 1", + [stored.record.prepare_id.clone().into(),stored.record.metadata_codec.into(),stored.record.state.clone().into()], + )).await.map_err(internal)?; + if let Some(row) = row { + if row + .try_get::>("", "state") + .map_err(internal)? + .is_some_and(|state| !["RESERVED", "LIVE"].contains(&state.as_str())) + || row + .try_get::>("", "graph_state") + .map_err(internal)? + .is_some_and(|state| state != "LIVE") + || row.try_get::("", "tombstone").map_err(internal)? + { + return Err(unavailable( + "fixed metadata lifetime is no longer installable", + )); + } + return Err(integrity( + "fixed metadata lifetime registry conflicts with its bindings", + )); + } + Ok(()) +} + +async fn check_generation_payloads( + connection: &C, + fixed: &FixedPlan, +) -> Result<(), SnapshotError> { + let row=connection.query_one_raw(statement( + "SELECT p.page_id FROM mst2_metadata_prepare_page p LEFT JOIN mst2_metadata_payload b ON b.page_id=p.page_id + WHERE p.prepare_id=$1 AND (b.page_id IS NULL OR b.generation IS DISTINCT FROM p.generation + OR b.metadata_codec<>$2 OR b.byte_size<>p.expected_size OR octet_length(b.payload)<>p.expected_size) LIMIT 1", + [fixed.intent.prepare_id().into(),fixed.stored.record.metadata_codec.into()], + )).await.map_err(internal)?; + if row.is_some() { + return Err(unavailable( + "exact metadata generation payload coverage is incomplete", + )); + } + Ok(()) +} + +async fn load_fixed_dag( + connection: &C, + fixed: &FixedPlan, +) -> Result { + let rows = connection + .query_all_raw(statement( + "SELECT b.page_id,b.generation,b.metadata_codec,b.byte_size,b.payload + FROM mst2_metadata_prepare_page p JOIN mst2_metadata_payload b + ON b.page_id=p.page_id AND b.generation=p.generation WHERE p.prepare_id=$1", + [fixed.intent.prepare_id().into()], + )) + .await + .map_err(internal)?; + let mut pages = Vec::with_capacity(rows.len()); + for row in rows { + let id: Vec = row.try_get("", "page_id").map_err(internal)?; + let id: [u8; 32] = id.as_slice().try_into().map_err(internal)?; + let generation: i64 = row.try_get("", "generation").map_err(internal)?; + let size: i32 = row.try_get("", "byte_size").map_err(internal)?; + if fixed.bindings.0.get(&id) != Some(&(generation, size as u64)) + || row.try_get::("", "metadata_codec").map_err(internal)? + != fixed.stored.record.metadata_codec + { + return Err(integrity( + "installed bytes differ from their fixed lifetime binding", + )); + } + let payload = MetadataPagePayload { + id, + size: size as u64, + bytes: row.try_get("", "payload").map_err(internal)?, + }; + validate_payload(&payload)?; + pages.push(payload); + } + if pages.len() != fixed.bindings.0.len() { + return Err(unavailable( + "exact metadata generation payload coverage is incomplete", + )); + } + ValidatedMetadataDag::validate( + MetadataDagCandidate { + metadata_codec: fixed.stored.plan.identity.metadata_codec, + root: fixed.stored.plan.root, + pages, + edges: fixed.stored.plan.edges.iter().copied().collect(), + }, + MetadataDagLimits::default(), + ) +} + +async fn lock_prepare(txn: &DatabaseTransaction, prepare_id: &str) -> Result<(), SnapshotError> { + txn.query_one_raw(statement( + "SELECT prepare_id FROM mst2_metadata_prepare WHERE prepare_id=$1 FOR UPDATE", + [prepare_id.into()], + )) + .await + .map_err(internal)? + .ok_or_else(|| unavailable("fixed metadata prepare is missing"))?; + Ok(()) +} + +#[cfg(test)] +#[path = "native_metadata_generation_tests.rs"] +mod tests; + +#[path = "native_metadata_history.rs"] +pub mod history; + +#[path = "native_metadata_qualified.rs"] +pub mod qualified; + +async fn verify_fixed_graph( + connection: &C, + fixed: &FixedPlan, +) -> Result<(), SnapshotError> { + if fixed.intent.graph_domain == GraphDomain::Qualified { + qualified::verify_graph(connection, fixed).await + } else { + verify_graph(connection, &fixed.stored).await + } +} diff --git a/src/jupiter/storage/native_metadata_history.rs b/src/jupiter/storage/native_metadata_history.rs new file mode 100644 index 00000000..6ea7f821 --- /dev/null +++ b/src/jupiter/storage/native_metadata_history.rs @@ -0,0 +1,470 @@ +//! Explicit preparation termination. Payload deletion remains forbidden. + +use super::*; + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum MetadataTerminalAction { + Abort, + RetireCoverage, +} + +#[derive(Debug, thiserror::Error)] +pub enum MetadataTerminalError { + #[error(transparent)] + Rejected(#[from] SnapshotError), + #[error("metadata {action:?} commit outcome is unknown for operation {operation_id}")] + CommitUncertain { + operation_id: String, + manifest_digest: [u8; 32], + action: MetadataTerminalAction, + }, +} + +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct MetadataTerminalReceipt { + intent: GenerationPrepareIntent, + action: MetadataTerminalAction, + terminal_at: sea_orm::prelude::DateTimeWithTimeZone, +} + +impl MetadataTerminalReceipt { + pub fn intent(&self) -> &GenerationPrepareIntent { + &self.intent + } + + pub fn action(&self) -> MetadataTerminalAction { + self.action + } +} + +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum MetadataTerminalObservation { + /// The stored state matches this action; mutation still requires its full CAS proof. + Active, + Terminated(Box), +} + +impl PostgresMetadataGenerationRepository { + /// Recover an immutable sealed intent after restart, including retired/aborted history. + /// Current lifetimes and payloads are deliberately not consulted or revived. + pub async fn capture_terminal_intent( + &self, + operation_id: &str, + manifest_digest: [u8; 32], + ) -> Result { + validate_operation_id(operation_id)?; + let txn = self.inner.transaction().await?; + self.inner.barrier(&txn).await?; + let result = async { + let record = mst2_metadata_prepare::Entity::find() + .filter(mst2_metadata_prepare::Column::OperationId.eq(operation_id)) + .one(&txn) + .await + .map_err(internal)? + .ok_or_else(|| unavailable("terminal preparation history is missing"))?; + if record.manifest_digest.as_slice() != manifest_digest { + return Err(integrity( + "terminal recovery manifest differs from durable history", + )); + } + let plan = MetadataInstallPlan::decode(&record.canonical_plan, &manifest_digest)?; + let canonical = record.canonical_bindings.as_deref().ok_or_else(|| { + unavailable("legacy unsealed preparation has no fixed terminal capability") + })?; + let bindings = GenerationBindings::decode(canonical, &plan)?; + let digest: [u8; 32] = Sha256::digest(canonical).into(); + let primary = record + .primary_scope + .as_deref() + .ok_or_else(|| integrity("terminal history has no primary scope"))?; + let domain = GraphDomain::from_stored(record.graph_domain.as_deref())?; + let intent = GenerationPrepareIntent { + legacy: MetadataPrepareIntent { + prepare_id: record.prepare_id.clone(), + operation_id: operation_id.into(), + manifest_digest, + }, + metadata_root: plan.root, + root_generation: bindings + .0 + .get(&plan.root) + .ok_or_else(|| integrity("terminal root binding is missing"))? + .0, + bindings_digest: digest, + primary_scope: primary.into(), + graph_domain: domain, + storage_seal: seal( + &record.prepare_id, + &manifest_digest, + &plan.root, + &digest, + primary, + domain.stored(), + )?, + }; + self.require_terminal_record(&txn, &intent).await?; + Ok(intent) + } + .await; + txn.rollback().await.map_err(internal)?; + result + } + + pub async fn abort( + &self, + intent: &GenerationPrepareIntent, + ) -> Result { + self.terminate(intent, MetadataTerminalAction::Abort, None) + .await + } + + pub async fn retire_prepare_coverage( + &self, + receipt: &GenerationMetadataReceipt, + ) -> Result { + self.terminate( + &receipt.intent, + MetadataTerminalAction::RetireCoverage, + Some(receipt), + ) + .await + } + + async fn terminate( + &self, + intent: &GenerationPrepareIntent, + action: MetadataTerminalAction, + receipt: Option<&GenerationMetadataReceipt>, + ) -> Result { + self.require_scope(intent)?; + let txn = self.inner.transaction().await?; + let result = self.terminate_in_txn(&txn, intent, action, receipt).await; + match result { + Ok(receipt) => { + txn.commit().await.map_err(|_| uncertain(intent, action))?; + Ok(receipt) + } + Err(error) => { + txn.rollback().await.map_err(internal)?; + Err(error.into()) + } + } + } + + pub(super) async fn terminate_in_txn( + &self, + txn: &DatabaseTransaction, + intent: &GenerationPrepareIntent, + action: MetadataTerminalAction, + receipt: Option<&GenerationMetadataReceipt>, + ) -> Result { + self.inner.barrier(txn).await?; + lock_prepare(txn, intent.prepare_id()).await?; + let record = self.require_terminal_record(txn, intent).await?; + if let Some(terminal) = terminal_receipt(&record, intent, action)? { + verify_no_prepare_coverage(txn, intent).await?; + return Ok(terminal); + } + let expected = match action { + MetadataTerminalAction::Abort => "PREPARING", + MetadataTerminalAction::RetireCoverage => "COMMITTED", + }; + if record.state != expected { + return Err(SnapshotError::new( + SnapshotErrorCode::Conflict, + "metadata prepare is in another terminal state", + )); + } + let fixed = self.require_fixed_plan(txn, intent).await?; + if action == MetadataTerminalAction::RetireCoverage { + let receipt = receipt + .ok_or_else(|| integrity("coverage retirement requires a definitive receipt"))?; + if receipt.legacy != fixed.stored.receipt()? || receipt.intent != fixed.intent { + return Err(integrity( + "coverage retirement receipt differs from its fixed preparation", + )); + } + check_generation_payloads(txn, &fixed).await?; + verify_fixed_graph(txn, &fixed).await?; + } + if intent.graph_domain == GraphDomain::Qualified { + qualified::verify_roots( + txn, + &fixed, + action == MetadataTerminalAction::RetireCoverage, + ) + .await?; + txn.execute_raw(statement( + "DELETE FROM mst2_metadata_graph_root WHERE prepare_id=$1 AND storage_seal=$2", + [ + intent.prepare_id().into(), + intent.storage_seal.to_vec().into(), + ], + )) + .await + .map_err(internal)?; + } else { + let roots = mst2_retention_root::Entity::find() + .filter( + mst2_retention_root::Column::RootKey + .eq(format!("prepare:{}", intent.prepare_id())), + ) + .limit((MetadataDagLimits::default().nodes + 1) as u64) + .all(txn) + .await + .map_err(internal)?; + let expected_nodes: BTreeSet<_> = fixed.bindings.0.keys().map(node_id).collect(); + if roots + .iter() + .any(|root| root.root_kind != "prepare" || !expected_nodes.contains(&root.node_id)) + { + return Err(integrity( + "terminal prepare coverage differs from its fixed lifetime mappings", + )); + } + if action == MetadataTerminalAction::RetireCoverage + && roots + .iter() + .map(|root| root.node_id.clone()) + .collect::>() + != expected_nodes + { + return Err(unavailable( + "coverage retirement cannot recover missing or transferred prepare roots", + )); + } + txn.execute_raw(statement( + "DELETE FROM mst2_retention_root r USING mst2_metadata_prepare_page p + WHERE p.prepare_id=$1 AND r.root_key='prepare:'||p.prepare_id AND r.root_kind='prepare' + AND r.node_id='page:sha256:'||encode(p.page_id,'hex')", + [intent.prepare_id().into()], + )) + .await + .map_err(internal)?; + } + let sql=match action { + MetadataTerminalAction::Abort=> + "UPDATE mst2_metadata_prepare SET state='ABORTED',aborted_at=clock_timestamp() + WHERE prepare_id=$1 AND state='PREPARING' AND storage_seal=$2 RETURNING *", + MetadataTerminalAction::RetireCoverage=> + "UPDATE mst2_metadata_prepare SET coverage_retired_at=clock_timestamp() + WHERE prepare_id=$1 AND state='COMMITTED' AND coverage_retired_at IS NULL AND storage_seal=$2 RETURNING *", + }; + let row = txn + .query_one_raw(statement( + sql, + [ + intent.prepare_id().into(), + intent.storage_seal.to_vec().into(), + ], + )) + .await + .map_err(internal)? + .ok_or_else(|| integrity("metadata terminal CAS changed during its barrier"))?; + let column = match action { + MetadataTerminalAction::Abort => "aborted_at", + MetadataTerminalAction::RetireCoverage => "coverage_retired_at", + }; + let terminal_at = row.try_get("", column).map_err(internal)?; + Ok(MetadataTerminalReceipt { + intent: intent.clone(), + action, + terminal_at, + }) + } + + /// A fresh same-primary barrier resolves a lost terminal commit response. + /// An old receipt is replayed without consulting or changing a newer lifetime. + pub async fn inspect_terminal( + &self, + fresh_primary: &DatabaseConnection, + intent: &GenerationPrepareIntent, + action: MetadataTerminalAction, + ) -> Result { + self.require_scope(intent)?; + let txn = fresh_primary + .begin_with_config(Some(IsolationLevel::ReadCommitted), None) + .await + .map_err(|_| uncertain(intent, action))?; + if self.inner.barrier(&txn).await.is_err() { + let _ = txn.rollback().await; + return Err(uncertain(intent, action)); + } + let result = async { + let record = self.require_terminal_record(&txn, intent).await?; + match terminal_receipt(&record, intent, action)? { + Some(receipt) => { + verify_no_prepare_coverage(&txn, intent).await?; + Ok(MetadataTerminalObservation::Terminated(Box::new(receipt))) + } + None => { + let expected = match action { + MetadataTerminalAction::Abort => "PREPARING", + MetadataTerminalAction::RetireCoverage => "COMMITTED", + }; + if record.state != expected { + return Err(SnapshotError::new( + SnapshotErrorCode::Conflict, + "metadata prepare is in another terminal state", + )); + } + Ok(MetadataTerminalObservation::Active) + } + } + } + .await; + txn.rollback() + .await + .map_err(|_| uncertain(intent, action))?; + result.map_err(|error: SnapshotError| { + if error.code == SnapshotErrorCode::Internal { + uncertain(intent, action) + } else { + MetadataTerminalError::Rejected(error) + } + }) + } + + async fn require_terminal_record( + &self, + connection: &C, + intent: &GenerationPrepareIntent, + ) -> Result { + self.require_scope(intent)?; + let record = mst2_metadata_prepare::Entity::find() + .filter(mst2_metadata_prepare::Column::OperationId.eq(intent.operation_id())) + .one(connection) + .await + .map_err(internal)? + .ok_or_else(|| unavailable("sealed metadata preparation is missing"))?; + let canonical = record + .canonical_bindings + .as_deref() + .ok_or_else(|| unavailable("legacy preparation has no generation seal"))?; + let plan = MetadataInstallPlan::decode(&record.canonical_plan, &intent.manifest_digest())?; + let bindings = GenerationBindings::decode(canonical, &plan)?; + let actual_digest: [u8; 32] = Sha256::digest(canonical).into(); + let domain = GraphDomain::from_stored(record.graph_domain.as_deref())?; + if record.prepare_id != intent.prepare_id() + || record.manifest_digest.as_slice() != intent.manifest_digest() + || record.metadata_root.as_slice() != intent.metadata_root + || record.bindings_digest.as_deref() != Some(intent.bindings_digest.as_slice()) + || actual_digest != intent.bindings_digest + || record.primary_scope.as_deref() != Some(intent.primary_scope.as_ref()) + || record.storage_seal.as_deref() != Some(intent.storage_seal.as_slice()) + || domain != intent.graph_domain + || bindings.0.get(&intent.metadata_root).map(|pair| pair.0) + != Some(intent.root_generation) + || seal( + intent.prepare_id(), + &intent.manifest_digest(), + &intent.metadata_root, + &intent.bindings_digest, + &intent.primary_scope, + domain.stored(), + )? != intent.storage_seal + { + return Err(integrity( + "terminal metadata action differs from its immutable generation seal", + )); + } + let pages = mst2_metadata_prepare_page::Entity::find() + .filter(mst2_metadata_prepare_page::Column::PrepareId.eq(intent.prepare_id())) + .limit((MetadataDagLimits::default().nodes + 1) as u64) + .all(connection) + .await + .map_err(internal)?; + let mut actual = BTreeMap::new(); + for page in pages { + let id: [u8; 32] = page.page_id.as_slice().try_into().map_err(internal)?; + actual.insert( + id, + ( + page.generation + .ok_or_else(|| integrity("terminal preparation has an unbound mapping"))?, + page.expected_size as u64, + ), + ); + } + if actual != bindings.0 { + return Err(integrity( + "terminal metadata mappings differ from the fixed seal", + )); + } + Ok(record) + } +} + +fn terminal_receipt( + record: &mst2_metadata_prepare::Model, + intent: &GenerationPrepareIntent, + action: MetadataTerminalAction, +) -> Result, SnapshotError> { + let terminal_at = match action { + MetadataTerminalAction::Abort if record.state == "ABORTED" => Some( + record + .aborted_at + .ok_or_else(|| integrity("aborted metadata preparation has no terminal receipt"))?, + ), + MetadataTerminalAction::RetireCoverage if record.state == "COMMITTED" => { + record.coverage_retired_at + } + _ => None, + }; + Ok(terminal_at.map(|terminal_at| MetadataTerminalReceipt { + intent: intent.clone(), + action, + terminal_at, + })) +} + +async fn verify_no_prepare_coverage( + connection: &C, + intent: &GenerationPrepareIntent, +) -> Result<(), SnapshotError> { + if intent.graph_domain == GraphDomain::Qualified { + if connection + .query_one_raw(statement( + "SELECT prepare_id FROM mst2_metadata_graph_root WHERE prepare_id=$1 LIMIT 1", + [intent.prepare_id().into()], + )) + .await + .map_err(internal)? + .is_some() + { + return Err(integrity( + "terminal qualified preparation still has coverage", + )); + } + return Ok(()); + } + if connection + .query_one_raw(statement( + "SELECT node_id FROM mst2_retention_root WHERE root_key='prepare:'||$1 LIMIT 1", + [intent.prepare_id().into()], + )) + .await + .map_err(internal)? + .is_some() + { + return Err(integrity( + "terminal metadata preparation still has prepare coverage", + )); + } + Ok(()) +} + +fn uncertain( + intent: &GenerationPrepareIntent, + action: MetadataTerminalAction, +) -> MetadataTerminalError { + MetadataTerminalError::CommitUncertain { + operation_id: intent.operation_id().into(), + manifest_digest: intent.manifest_digest(), + action, + } +} + +#[cfg(test)] +#[path = "native_metadata_history_tests.rs"] +mod tests; diff --git a/src/jupiter/storage/native_metadata_history_tests.rs b/src/jupiter/storage/native_metadata_history_tests.rs new file mode 100644 index 00000000..250d99a6 --- /dev/null +++ b/src/jupiter/storage/native_metadata_history_tests.rs @@ -0,0 +1,634 @@ +use sea_orm::{Database, PaginatorTrait}; +use sea_orm_migration::MigratorTrait; + +use super::*; +use crate::{ + callisto::{mst2_metadata_current, mst2_metadata_lifetime}, + jupiter::{ + migration::Migrator, + tests::{TestSchemaGuard, test_db_config}, + }, +}; + +async fn fixture() -> (DatabaseConnection, DatabaseConnection, TestSchemaGuard) { + let temp = tempfile::tempdir().unwrap(); + let (config, schema) = test_db_config(temp.path()).await; + let first = Database::connect(config.db_url.clone()).await.unwrap(); + Migrator::up(&first, None).await.unwrap(); + let second = Database::connect(config.db_url).await.unwrap(); + (first, second, schema) +} + +fn prepared() -> PreparedNativeMetadataRetention { + let entries = [mst2_codec::metapage::Entry::file( + mst2_codec::metapage::EntryKind::Regular, + b"file", + 3, + [42; 32], + )]; + let child = Page::build(&entries).unwrap(); + let root_entries = [mst2_codec::metapage::Entry::dir(b"one", page_id(&child))]; + let root = Page::build(&root_entries).unwrap(); + let mut builder = crate::ceres::snapshot::retention_dag::MetadataDagBuilder::new( + MetadataDagLimits::default(), + ); + builder.add_directory(&child, &entries).unwrap(); + builder.add_directory(&root, &root_entries).unwrap(); + PreparedNativeMetadataRetention::test_installation( + std::sync::Arc::new(builder.finish(page_id(&root)).unwrap()), + "/", + ) +} + +async fn scalar(db: &DatabaseConnection, sql: &str) -> i64 { + db.query_one_raw(statement(sql, [])) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap() +} + +fn rejected(error: MetadataTerminalError) -> SnapshotErrorCode { + match error { + MetadataTerminalError::Rejected(error) => error.code, + _ => panic!("expected definite terminal rejection"), + } +} + +#[tokio::test] +async fn history_keeps_old_composite_fk_bindings_while_current_is_an_independent_watermark() { + let (first, second, _schema) = fixture().await; + let repository = PostgresMetadataGenerationRepository::new(first.clone()) + .await + .unwrap(); + let pages = prepared(); + let intent = repository.begin_intent("history", &pages).await.unwrap(); + repository + .install_pages(&intent, pages.dag().payloads()) + .await + .unwrap(); + let page = &pages.dag().payloads()[0]; + second.execute_raw(statement( + "INSERT INTO mst2_metadata_lifetime(page_id,node_id,generation,state,metadata_codec,expected_size) + VALUES($1,$2,2,'RESERVED',1,$3)", + [page.id.to_vec().into(),node_id(&page.id).into(),(page.size as i32).into()], + )).await.unwrap(); + assert!( + mst2_metadata_lifetime::Entity::find_by_id((page.id.to_vec(), 1)) + .one(&second) + .await + .unwrap() + .is_some() + ); + assert!( + mst2_metadata_lifetime::Entity::find_by_id((page.id.to_vec(), 2)) + .one(&second) + .await + .unwrap() + .is_some() + ); + assert_eq!( + mst2_metadata_current::Entity::find_by_id(page.id.to_vec()) + .one(&second) + .await + .unwrap() + .unwrap() + .generation, + 1 + ); + assert!( + second + .execute_raw(statement( + "UPDATE mst2_metadata_current SET generation=2 WHERE page_id=$1", + [page.id.to_vec().into()] + )) + .await + .is_err() + ); + assert_eq!(scalar(&first,"SELECT count(*) FROM mst2_metadata_prepare_page p JOIN mst2_metadata_lifetime l ON l.page_id=p.page_id AND l.generation=p.generation WHERE p.generation=1").await,2); + assert_eq!(scalar(&first,"SELECT count(*) FROM mst2_metadata_payload b JOIN mst2_metadata_lifetime l ON l.page_id=b.page_id AND l.generation=b.generation WHERE b.generation=1").await,2); + repository.finalize(&intent).await.unwrap(); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_lifetime").await, + 3 + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_current").await, + 2 + ); +} + +#[tokio::test] +async fn history_abort_fences_a_late_installer_waiting_on_the_real_retention_barrier() { + let (first, second, _schema) = fixture().await; + let repository = PostgresMetadataGenerationRepository::new(first.clone()) + .await + .unwrap(); + let other = PostgresMetadataGenerationRepository::new(second.clone()) + .await + .unwrap(); + let pages = prepared(); + let intent = repository + .begin_intent("abort-late-install", &pages) + .await + .unwrap(); + repository + .install_page(&intent, &pages.dag().payloads()[0]) + .await + .unwrap(); + let held = repository.inner.transaction().await.unwrap(); + repository.inner.barrier(&held).await.unwrap(); + let late_intent = intent.clone(); + let late_pages = pages.dag().payloads().to_vec(); + let late = tokio::spawn(async move { other.install_pages(&late_intent, &late_pages).await }); + let deadline = tokio::time::Instant::now() + Duration::from_secs(4); + loop { + let waiting=held.query_one_raw(statement( + "SELECT EXISTS(SELECT 1 FROM pg_locks WHERE locktype='advisory' AND NOT granted + AND classid=$1::integer::oid AND objid=hashtext(current_schema())::oid + AND database=(SELECT oid FROM pg_database WHERE datname=current_database())) AS waiting", + [RETENTION_LOCK_KEY.into()], + )).await.unwrap().unwrap().try_get::("","waiting").unwrap(); + if waiting { + break; + } + assert!( + tokio::time::Instant::now() < deadline, + "late installer must wait on the captured schema barrier" + ); + tokio::task::yield_now().await; + } + let terminal = repository + .terminate_in_txn(&held, &intent, MetadataTerminalAction::Abort, None) + .await + .unwrap(); + held.commit().await.unwrap(); + assert!(matches!( + late.await.unwrap(), + Err(MetadataInstallError::Rejected(SnapshotError { + code: SnapshotErrorCode::IntegrityError, + .. + })) + )); + assert_eq!(repository.abort(&intent).await.unwrap(), terminal); + assert_eq!( + repository + .inspect_terminal(&second, &intent, MetadataTerminalAction::Abort) + .await + .unwrap(), + MetadataTerminalObservation::Terminated(Box::new(terminal)) + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_payload").await, + 1 + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_prepare_page").await, + 2 + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_retention_node").await, + 0 + ); + assert_eq!(scalar(&first,"SELECT count(*) FROM mst2_metadata_prepare WHERE state='ABORTED' AND aborted_at IS NOT NULL").await,1); +} + +#[tokio::test] +async fn history_abort_after_observation_rejects_old_finalize_and_preserves_other_prepares() { + let (first, second, _schema) = fixture().await; + let repository = PostgresMetadataGenerationRepository::new(first.clone()) + .await + .unwrap(); + let other = PostgresMetadataGenerationRepository::new(second.clone()) + .await + .unwrap(); + let pages = prepared(); + let owner = repository + .begin_intent("committed-owner", &pages) + .await + .unwrap(); + repository + .install_pages(&owner, pages.dag().payloads()) + .await + .unwrap(); + let owner_receipt = repository.finalize(&owner).await.unwrap(); + let abandoned = other + .begin_intent("abandoned-shared", &pages) + .await + .unwrap(); + let observation = other.observe_installed_dag(&abandoned).await.unwrap(); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_retention_root").await, + 4 + ); + other.abort(&abandoned).await.unwrap(); + assert!(matches!( + other.finalize_observation(&abandoned, &observation).await, + Err(MetadataInstallError::Rejected(_)) + )); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_retention_root").await, + 2 + ); + assert_eq!(repository.finalize(&owner).await.unwrap(), owner_receipt); + assert_eq!( + rejected(repository.abort(&owner).await.unwrap_err()), + SnapshotErrorCode::Conflict + ); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_lifetime WHERE state='LIVE' AND generation=1" + ) + .await, + 2 + ); + assert_eq!( + scalar(&second, "SELECT count(*) FROM mst2_metadata_payload").await, + 2 + ); +} + +#[tokio::test] +async fn history_terminal_outer_rollback_keeps_preparing_protection_and_durable_recovery_active() { + let (first, second, _schema) = fixture().await; + let repository = PostgresMetadataGenerationRepository::new(first.clone()) + .await + .unwrap(); + let pages = prepared(); + let intent = repository + .begin_intent("abort-rollback", &pages) + .await + .unwrap(); + let held = repository.inner.transaction().await.unwrap(); + repository + .terminate_in_txn(&held, &intent, MetadataTerminalAction::Abort, None) + .await + .unwrap(); + held.rollback().await.unwrap(); + assert_eq!( + repository + .inspect_terminal(&second, &intent, MetadataTerminalAction::Abort) + .await + .unwrap(), + MetadataTerminalObservation::Active + ); + assert_eq!(scalar(&first,"SELECT count(*) FROM mst2_metadata_prepare WHERE state='PREPARING' AND aborted_at IS NULL").await,1); + repository + .install_pages(&intent, pages.dag().payloads()) + .await + .unwrap(); + repository.finalize(&intent).await.unwrap(); +} + +#[tokio::test] +async fn history_retirement_replays_original_receipt_and_never_recreates_lost_prepare_coverage() { + let (first, second, _schema) = fixture().await; + let repository = PostgresMetadataGenerationRepository::new(first.clone()) + .await + .unwrap(); + let pages = prepared(); + let intent = repository.begin_intent("retire", &pages).await.unwrap(); + repository + .install_pages(&intent, pages.dag().payloads()) + .await + .unwrap(); + let receipt = repository.finalize(&intent).await.unwrap(); + let terminal = repository.retire_prepare_coverage(&receipt).await.unwrap(); + assert_eq!( + repository.retire_prepare_coverage(&receipt).await.unwrap(), + terminal + ); + assert_eq!( + repository + .inspect_terminal(&second, &intent, MetadataTerminalAction::RetireCoverage) + .await + .unwrap(), + MetadataTerminalObservation::Terminated(Box::new(terminal)) + ); + assert!(matches!( + repository + .install_pages(&intent, pages.dag().payloads()) + .await, + Err(MetadataInstallError::Rejected(SnapshotError { + code: SnapshotErrorCode::ObjectUnavailable, + .. + })) + )); + assert!(repository.finalize(&intent).await.is_err()); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_retention_root").await, + 0 + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_payload").await, + 2 + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_prepare_page").await, + 2 + ); + assert_eq!(scalar(&first,"SELECT count(*) FROM mst2_metadata_prepare WHERE state='COMMITTED' AND coverage_retired_at IS NOT NULL").await,1); + assert!( + second + .execute_unprepared("UPDATE mst2_metadata_prepare SET coverage_retired_at=NULL") + .await + .is_err() + ); +} + +#[tokio::test] +async fn history_committed_missing_payload_rejects_a_late_installer_without_repairing_corruption() { + let (first, second, _schema) = fixture().await; + let repository = PostgresMetadataGenerationRepository::new(first.clone()) + .await + .unwrap(); + let pages = prepared(); + let intent = repository + .begin_intent("committed-missing", &pages) + .await + .unwrap(); + repository + .install_page(&intent, &pages.dag().payloads()[0]) + .await + .unwrap(); + // Persist a corrupt COMMITTED fixture without deleting bytes or disabling + // any guard. A late installer must refuse to fill its missing page. + let txn = second.begin().await.unwrap(); + repository.inner.barrier(&txn).await.unwrap(); + PostgresRetentionRepository::retain_group_in_txn( + &txn, + pages.dag().nodes(), + pages.dag().edges(), + &[RetentionRoot::Prepare(intent.prepare_id().into())], + ) + .await + .unwrap(); + txn.execute_unprepared("UPDATE mst2_metadata_lifetime SET state='LIVE'") + .await + .unwrap(); + txn.execute_raw(statement("UPDATE mst2_metadata_prepare SET state='COMMITTED',committed_at=clock_timestamp() WHERE prepare_id=$1",[intent.prepare_id().into()])).await.unwrap(); + txn.commit().await.unwrap(); + assert!(matches!( + repository + .install_pages(&intent, pages.dag().payloads()) + .await, + Err(MetadataInstallError::Rejected(SnapshotError { + code: SnapshotErrorCode::ObjectUnavailable, + .. + })) + )); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_payload").await, + 1 + ); + assert_eq!( + scalar(&second, "SELECT count(*) FROM mst2_retention_root").await, + 2 + ); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_prepare WHERE state='COMMITTED'" + ) + .await, + 1 + ); +} + +#[tokio::test] +async fn history_graph_domain_is_durable_sealed_and_cannot_be_changed_by_a_new_repository() { + let (first, second, _schema) = fixture().await; + let repository = PostgresMetadataGenerationRepository::new(first.clone()) + .await + .unwrap(); + let mut qualified = PostgresMetadataGenerationRepository::new(second.clone()) + .await + .unwrap(); + qualified.graph_domain = "qualified-v1"; + let pages = prepared(); + let intent = repository + .begin_intent("generic-domain", &pages) + .await + .unwrap(); + assert_eq!(intent.graph_domain, GraphDomain::Generic); + assert!(matches!( + qualified.begin_intent("generic-domain", &pages).await, + Err(MetadataInstallError::Rejected(SnapshotError { + code: SnapshotErrorCode::IntegrityError, + .. + })) + )); + assert!(matches!( + qualified.begin_intent("cross-domain-page", &pages).await, + Err(MetadataInstallError::Rejected(SnapshotError { + code: SnapshotErrorCode::ObjectUnavailable, + .. + })) + )); + assert!(matches!( + qualified + .install_pages(&intent, pages.dag().payloads()) + .await, + Err(MetadataInstallError::Rejected(SnapshotError { + code: SnapshotErrorCode::IntegrityError, + .. + })) + )); + assert!( + second + .execute_unprepared("UPDATE mst2_metadata_prepare SET graph_domain='qualified-v1'") + .await + .is_err() + ); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_prepare WHERE graph_domain='generic-v1'" + ) + .await, + 1 + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_payload").await, + 0 + ); + assert_eq!( + scalar(&second, "SELECT count(*) FROM mst2_metadata_current").await, + 2 + ); +} + +#[tokio::test] +async fn history_terminal_wrong_primary_is_unknown_and_current_delete_payload_update_stay_forbidden() + { + let (first, second, _schema) = fixture().await; + let (other, _other_connection, _other_schema) = fixture().await; + let repository = PostgresMetadataGenerationRepository::new(first.clone()) + .await + .unwrap(); + let pages = prepared(); + let intent = repository + .begin_intent("terminal-primary", &pages) + .await + .unwrap(); + repository + .install_page(&intent, &pages.dag().payloads()[0]) + .await + .unwrap(); + assert!(matches!( + repository + .inspect_terminal(&other, &intent, MetadataTerminalAction::Abort) + .await, + Err(MetadataTerminalError::CommitUncertain { .. }) + )); + assert_eq!( + repository + .inspect_terminal(&second, &intent, MetadataTerminalAction::Abort) + .await + .unwrap(), + MetadataTerminalObservation::Active + ); + for sql in [ + "DELETE FROM mst2_metadata_current", + "DELETE FROM mst2_metadata_lifetime", + "DELETE FROM mst2_metadata_payload", + "UPDATE mst2_metadata_payload SET payload=payload", + "DELETE FROM mst2_metadata_storage_scope", + ] { + assert!(second.execute_unprepared(sql).await.is_err(), "{sql}"); + } + repository.abort(&intent).await.unwrap(); + assert!( + second + .execute_unprepared( + "UPDATE mst2_metadata_prepare SET state='PREPARING',aborted_at=NULL" + ) + .await + .is_err() + ); + assert_eq!( + mst2_metadata_current::Entity::find() + .count(&first) + .await + .unwrap(), + 2 + ); +} + +#[tokio::test] +async fn history_migration_preserves_preexisting_g1_null_domain_seal_and_composite_mappings() { + let temp = tempfile::tempdir().unwrap(); + let (config, _schema) = test_db_config(temp.path()).await; + let first = Database::connect(config.db_url.clone()).await.unwrap(); + let second = Database::connect(config.db_url).await.unwrap(); + let at = Migrator::migrations() + .iter() + .position(|migration| { + migration.name() == "m20261007_000300_add_mst2_metadata_lifetime_history" + }) + .unwrap(); + Migrator::up(&first, Some(at.try_into().unwrap())) + .await + .unwrap(); + let inner = PostgresMetadataInstallRepository::new(first.clone()) + .await + .unwrap(); + let pages = prepared(); + let plan = pages.install_plan().unwrap(); + let manifest = plan.digest().unwrap(); + let bindings = GenerationBindings( + plan.pages + .iter() + .map(|(page, size)| (*page, (1, *size))) + .collect(), + ); + let canonical = bindings.encode().unwrap(); + let digest: [u8; 32] = Sha256::digest(&canonical).into(); + let scope = serde_json::to_vec(&( + &inner.storage_scope.storage_uuid, + &inner.storage_scope.database, + inner.storage_scope.database_oid, + &inner.storage_scope.schema, + inner.storage_scope.schema_oid, + &inner.storage_scope.server_address, + inner.storage_scope.server_port, + )) + .unwrap(); + let prepare_id = uuid::Uuid::new_v4().to_string(); + let old_seal = seal(&prepare_id, &manifest, &plan.root, &digest, &scope, None).unwrap(); + let txn = first.begin().await.unwrap(); + inner.barrier(&txn).await.unwrap(); + for page in pages.dag().payloads() { + txn.execute_raw(statement("INSERT INTO mst2_metadata_lifetime(page_id,node_id,generation,state,metadata_codec,expected_size) VALUES($1,$2,1,'RESERVED',1,$3)", + [page.id.to_vec().into(),node_id(&page.id).into(),(page.size as i32).into()])).await.unwrap(); + txn.execute_raw(statement("INSERT INTO mst2_metadata_payload(page_id,generation,metadata_codec,byte_size,payload) VALUES($1,1,1,$2,$3)", + [page.id.to_vec().into(),(page.size as i32).into(),page.bytes.clone().into()])).await.unwrap(); + } + let identity = &plan.identity; + txn.execute_raw(statement( + "INSERT INTO mst2_metadata_prepare(prepare_id,operation_id,manifest_digest,canonical_plan, + source_domain,tagged_root_tree_oid,scope,schema_version,metadata_codec,materialization_policy, + fs_semantics,access_projection,verification_revision,projection_revision,metadata_root, + node_count,edge_count,total_bytes,state,canonical_bindings,bindings_digest,primary_scope,storage_seal) + VALUES($1,'g1-before-history',$2,$3,$4,$5,$6,$7,$8,$9,$10,$11,$12,$13,$14,$15,$16,$17,'PREPARING',$18,$19,$20,$21)", + [prepare_id.clone().into(),manifest.to_vec().into(),plan.encode().unwrap().into(),identity.source_domain.clone().into(), + identity.tagged_root_tree_oid.clone().into(),identity.scope.clone().into(),(identity.schema_version as i16).into(), + (identity.metadata_codec as i16).into(),(identity.materialization_policy as i16).into(),(identity.fs_semantics as i16).into(), + (identity.access_projection as i16).into(),identity.verification_revision.into(),(identity.projection_revision as i16).into(), + plan.root.to_vec().into(),(plan.pages.len() as i32).into(),(plan.edges.len() as i32).into(),(plan.total_bytes as i64).into(), + canonical.into(),digest.to_vec().into(),scope.into(),old_seal.to_vec().into()], + )).await.unwrap(); + for (page, size) in &plan.pages { + txn.execute_raw(statement("INSERT INTO mst2_metadata_prepare_page(prepare_id,page_id,generation,expected_size) VALUES($1,$2,1,$3)", + [prepare_id.clone().into(),page.to_vec().into(),(*size as i32).into()])).await.unwrap(); + } + PostgresRetentionRepository::retain_group_in_txn( + &txn, + pages.dag().nodes(), + pages.dag().edges(), + &[RetentionRoot::Prepare(prepare_id.clone())], + ) + .await + .unwrap(); + txn.execute_unprepared("UPDATE mst2_metadata_lifetime SET state='LIVE'") + .await + .unwrap(); + txn.execute_raw(statement("UPDATE mst2_metadata_prepare SET state='COMMITTED',committed_at=clock_timestamp() WHERE prepare_id=$1",[prepare_id.clone().into()])).await.unwrap(); + txn.commit().await.unwrap(); + Migrator::up(&first, Some(1)).await.unwrap(); + let repository = PostgresMetadataGenerationRepository::new(second.clone()) + .await + .unwrap(); + let restored = repository + .begin_intent("g1-before-history", &pages) + .await + .unwrap(); + assert_eq!(restored.graph_domain, GraphDomain::LegacyGeneric); + assert_eq!(restored.storage_seal, old_seal); + assert_eq!(restored.prepare_id(), prepare_id); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_current WHERE generation=1" + ) + .await, + 2 + ); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_prepare WHERE graph_domain IS NULL" + ) + .await, + 1 + ); + let receipt = repository.finalize(&restored).await.unwrap(); + repository.retire_prepare_coverage(&receipt).await.unwrap(); + assert_eq!(scalar(&first,"SELECT count(*) FROM mst2_metadata_prepare_page p JOIN mst2_metadata_lifetime l ON l.page_id=p.page_id AND l.generation=p.generation").await,2); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_payload").await, + 2 + ); +} diff --git a/src/jupiter/storage/native_metadata_install.rs b/src/jupiter/storage/native_metadata_install.rs new file mode 100644 index 00000000..7d978fa0 --- /dev/null +++ b/src/jupiter/storage/native_metadata_install.rs @@ -0,0 +1,1114 @@ +//! Immutable metadata payload CAS and durable native preparation receipts. +//! The native HTTP session authority installs batches and transfers prepare pins. + +use std::{ + collections::{BTreeMap, BTreeSet}, + time::Duration, +}; + +use mst2_codec::metapage::{HEADER_LEN, PAGE_MAX_BYTES, Page, page_id}; +use sea_orm::{ + ColumnTrait, ConnectionTrait, DatabaseConnection, DatabaseTransaction, DbBackend, EntityTrait, + IsolationLevel, QueryFilter, QueryResult, QuerySelect, Statement, TransactionTrait, +}; +use serde_json::json; + +use super::mst2_retention::{PostgresRetentionRepository, RETENTION_LOCK_KEY}; +use crate::{ + callisto::{ + mst2_metadata_payload, mst2_metadata_prepare, mst2_metadata_prepare_page, + mst2_retention_edge, mst2_retention_node, mst2_retention_root, + }, + ceres::snapshot::{ + error::{SnapshotError, SnapshotErrorCode}, + metadata_install::{MAX_PLAN_BYTES, MetadataInstallIdentity, MetadataInstallPlan}, + pages::PreparedNativeMetadataRetention, + retention::RetentionRoot, + retention_dag::{ + MetadataDagCandidate, MetadataDagLimits, MetadataPagePayload, ValidatedMetadataDag, + }, + }, +}; + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum MetadataCommitPhase { + Intent, + Payload, + Finalize, +} + +#[derive(Debug, thiserror::Error)] +pub enum MetadataInstallError { + #[error(transparent)] + Rejected(#[from] SnapshotError), + #[error("native metadata {phase:?} commit outcome is unknown for operation {operation_id}")] + CommitUncertain { + operation_id: String, + manifest_digest: [u8; 32], + phase: MetadataCommitPhase, + }, +} + +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct MetadataPrepareIntent { + prepare_id: String, + operation_id: String, + manifest_digest: [u8; 32], +} + +impl MetadataPrepareIntent { + pub fn prepare_id(&self) -> &str { + &self.prepare_id + } + pub fn operation_id(&self) -> &str { + &self.operation_id + } + pub fn manifest_digest(&self) -> [u8; 32] { + self.manifest_digest + } +} + +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct PreparedMetadataReceipt { + intent: MetadataPrepareIntent, + identity: MetadataInstallIdentity, + metadata_root: [u8; 32], + payload_bytes: u64, + root_payload_bytes: u64, + node_count: usize, + edge_count: usize, +} + +impl PreparedMetadataReceipt { + pub fn intent(&self) -> &MetadataPrepareIntent { + &self.intent + } + pub fn metadata_root(&self) -> [u8; 32] { + self.metadata_root + } + pub fn payload_bytes(&self) -> u64 { + self.payload_bytes + } +} + +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum MetadataPrepareObservation { + Absent, + Preparing(MetadataPrepareIntent), + Committed(PreparedMetadataReceipt), +} + +struct StoredPlan { + record: mst2_metadata_prepare::Model, + plan: MetadataInstallPlan, + prepare_pages: Vec, +} + +impl StoredPlan { + fn intent(&self) -> Result { + Ok(MetadataPrepareIntent { + prepare_id: self.record.prepare_id.clone(), + operation_id: self.record.operation_id.clone(), + manifest_digest: self + .record + .manifest_digest + .as_slice() + .try_into() + .map_err(|_| integrity("invalid stored metadata manifest digest"))?, + }) + } + fn receipt(&self) -> Result { + Ok(PreparedMetadataReceipt { + intent: self.intent()?, + identity: self.plan.identity.clone(), + metadata_root: self.plan.root, + payload_bytes: self.plan.total_bytes, + root_payload_bytes: *self + .plan + .pages + .get(&self.plan.root) + .ok_or_else(|| integrity("metadata preparation root page is missing"))?, + node_count: self.plan.pages.len(), + edge_count: self.plan.edges.len(), + }) + } +} + +#[derive(Clone)] +pub struct PostgresMetadataInstallRepository { + connection: DatabaseConnection, + barrier_timeout: Duration, + storage_scope: PrimaryStorageScope, + qualified_family: Option, +} + +#[derive(Debug, Clone, PartialEq, Eq)] +struct PrimaryStorageScope { + storage_uuid: String, + database: String, + database_oid: i64, + schema: String, + schema_oid: i64, + server_address: Option, + server_port: Option, +} + +impl PostgresMetadataInstallRepository { + /// The connection must target the deployment's primary PostgreSQL database. + pub async fn new(connection: DatabaseConnection) -> Result { + let schema = capture_storage_schema(&connection).await?; + let storage_scope = read_storage_scope(&connection, &schema).await?; + if storage_scope.schema != schema { + return Err(internal( + "metadata primary schema changed during storage scope capture", + )); + } + Ok(Self { + connection, + barrier_timeout: Duration::from_secs(5), + storage_scope, + qualified_family: None, + }) + } + + pub(crate) fn captured_schema(&self) -> &str { + &self.storage_scope.schema + } + + /// These columns are returned by the same query that authorizes a lease, + /// so a warm cache cannot authorize a stale replica or another schema. + pub(crate) fn verify_primary_scope_row(&self, row: &QueryResult) -> Result<(), SnapshotError> { + let actual = PrimaryStorageScope { + storage_uuid: row + .try_get::>("", "authority_storage_uuid") + .map_err(internal)? + .ok_or_else(|| internal("session authority storage scope is missing"))?, + database: row.try_get("", "authority_database").map_err(internal)?, + database_oid: row + .try_get("", "authority_database_oid") + .map_err(internal)?, + schema: row.try_get("", "authority_schema").map_err(internal)?, + schema_oid: row.try_get("", "authority_schema_oid").map_err(internal)?, + server_address: row + .try_get("", "authority_server_address") + .map_err(internal)?, + server_port: row.try_get("", "authority_server_port").map_err(internal)?, + }; + if row + .try_get::("", "authority_replica") + .map_err(internal)? + || actual != self.storage_scope + { + return Err(internal( + "session authority no longer targets its captured primary storage scope", + )); + } + Ok(()) + } + + pub(crate) async fn verify_primary_connection( + &self, + connection: &C, + ) -> Result<(), SnapshotError> { + if read_storage_scope(connection, self.captured_schema()).await? != self.storage_scope { + return Err(internal( + "session mutation no longer targets its captured primary storage scope", + )); + } + Ok(()) + } + + pub async fn begin_intent( + &self, + operation_id: &str, + prepared: &PreparedNativeMetadataRetention, + ) -> Result { + validate_operation_id(operation_id)?; + let plan = prepared.install_plan()?; + let digest = plan.digest()?; + let txn = self.transaction().await?; + let result = async { + self.barrier(&txn).await?; + if let Some(stored) = load_plan(&txn, operation_id, &digest).await? { + return stored.intent(); + } + let id = uuid::Uuid::new_v4().to_string(); + let identity = &plan.identity; + txn.execute_raw(statement( + "INSERT INTO mst2_metadata_prepare (prepare_id, operation_id, manifest_digest, canonical_plan, + source_domain, tagged_root_tree_oid, scope, schema_version, metadata_codec, materialization_policy, + fs_semantics, access_projection, verification_revision, projection_revision, metadata_root, + node_count, edge_count, total_bytes, state) + VALUES ($1,$2,$3,$4,$5,$6,$7,$8,$9,$10,$11,$12,$13,$14,$15,$16,$17,$18,'PREPARING')", + [id.clone().into(), operation_id.into(), digest.to_vec().into(), plan.encode()?.into(), + identity.source_domain.clone().into(), identity.tagged_root_tree_oid.clone().into(), identity.scope.clone().into(), + (identity.schema_version as i16).into(), (identity.metadata_codec as i16).into(), + (identity.materialization_policy as i16).into(), (identity.fs_semantics as i16).into(), + (identity.access_projection as i16).into(), identity.verification_revision.into(), + (identity.projection_revision as i16).into(), plan.root.to_vec().into(), (plan.pages.len() as i32).into(), + (plan.edges.len() as i32).into(), (plan.total_bytes as i64).into()], + )).await.map_err(internal)?; + let pages: Vec<_> = plan.pages.iter().map(|(id,size)| json!({"page_id":hex::encode(id),"size":size})).collect(); + txn.execute_raw(statement( + "INSERT INTO mst2_metadata_prepare_page (prepare_id,page_id,expected_size) + SELECT $1,decode(p.page_id,'hex'),p.size FROM jsonb_to_recordset($2::jsonb) AS p(page_id text,size integer)", + [id.clone().into(), serde_json::to_string(&pages).map_err(internal)?.into()], + )).await.map_err(internal)?; + Ok(MetadataPrepareIntent { prepare_id:id, operation_id:operation_id.into(), manifest_digest:digest }) + }.await; + commit( + txn, + result, + operation_id, + digest, + MetadataCommitPhase::Intent, + ) + .await + } + + pub async fn install_page( + &self, + intent: &MetadataPrepareIntent, + payload: &MetadataPagePayload, + ) -> Result<(), MetadataInstallError> { + self.install_pages(intent, std::slice::from_ref(payload)) + .await + } + + /// At most 64 pages (1 MiB of canonical payload) per transaction. + pub async fn install_pages( + &self, + intent: &MetadataPrepareIntent, + payloads: &[MetadataPagePayload], + ) -> Result<(), MetadataInstallError> { + if payloads.is_empty() || payloads.len() > 64 { + return Err(SnapshotError::new( + SnapshotErrorCode::LimitExceeded, + "metadata installation batch must contain 1..=64 pages", + ) + .into()); + } + let mut ids = BTreeSet::new(); + for payload in payloads { + validate_payload(payload)?; + if !ids.insert(payload.id) { + return Err(integrity("duplicate page in metadata installation batch").into()); + } + } + let txn = self.transaction().await?; + let result = async { + self.barrier(&txn).await?; + let stored = require_plan(&txn, intent).await?; + for payload in payloads { + if stored.plan.pages.get(&payload.id) != Some(&payload.size) { + return Err(integrity("metadata payload is not a member of this fixed installation")); + } + } + let pages: Vec<_> = payloads.iter().map(|p| json!({ + "page_id":hex::encode(p.id), "size":p.size, "payload":hex::encode(&p.bytes) + })).collect(); + let encoded = serde_json::to_string(&pages).map_err(internal)?; + txn.execute_raw(statement( + "INSERT INTO mst2_metadata_payload(page_id,metadata_codec,byte_size,payload) + SELECT decode(p.page_id,'hex'),$1,p.size,decode(p.payload,'hex') + FROM jsonb_to_recordset($2::jsonb) AS p(page_id text,size integer,payload text) + ON CONFLICT(page_id) DO NOTHING", + [(stored.plan.identity.metadata_codec as i16).into(),encoded.clone().into()], + )).await.map_err(internal)?; + let bad = txn.query_one_raw(statement( + "SELECT p.page_id FROM jsonb_to_recordset($2::jsonb) AS p(page_id text,size integer,payload text) + LEFT JOIN mst2_metadata_payload b ON b.page_id=decode(p.page_id,'hex') + WHERE b.page_id IS NULL OR b.metadata_codec<>$1 OR b.byte_size<>p.size + OR b.payload<>decode(p.payload,'hex') LIMIT 1", + [(stored.plan.identity.metadata_codec as i16).into(),encoded.into()], + )).await.map_err(internal)?; + if bad.is_some() { + return Err(integrity("immutable metadata payload identity conflicts with stored bytes")); + } + Ok(()) + }.await; + commit( + txn, + result, + &intent.operation_id, + intent.manifest_digest, + MetadataCommitPhase::Payload, + ) + .await + } + + pub(crate) async fn verify_receipt_in_txn( + &self, + txn: &DatabaseTransaction, + receipt: &PreparedMetadataReceipt, + tagged_root_tree_oid: &str, + scope: &str, + ) -> Result<(), SnapshotError> { + self.barrier(txn).await?; + let identity = &receipt.identity; + if identity.tagged_root_tree_oid != tagged_root_tree_oid || identity.scope != scope { + return Err(integrity( + "metadata receipt belongs to another fixed source", + )); + } + // The private receipt can only be issued after full validation and a + // definitive commit. Hash the bounded canonical plan in PostgreSQL, + // bind all redundant identity/summary columns to that receipt, and + // check its still-protected LIVE root under the retention barrier. + // This hashes <=2 MiB; it does not transfer or scan pages/edges here. + let row = txn.query_one_raw(statement( + "SELECT p.operation_id,p.manifest_digest, + CASE WHEN octet_length(p.canonical_plan)<=$2 THEN sha256(p.canonical_plan) END AS plan_digest, + p.source_domain,p.tagged_root_tree_oid,p.scope,p.schema_version,p.metadata_codec, + p.materialization_policy,p.fs_semantics,p.access_projection,p.verification_revision, + p.projection_revision,p.metadata_root,p.node_count,p.edge_count,p.total_bytes, + p.state,p.committed_at IS NOT NULL AS committed, + n.kind AS root_kind,n.state AS root_state,n.bytes AS root_bytes, + EXISTS(SELECT 1 FROM mst2_retention_root r WHERE r.node_id=n.node_id + AND r.root_key='prepare:'||p.prepare_id AND r.root_kind='prepare') AS prepare_covered + FROM mst2_metadata_prepare p LEFT JOIN mst2_retention_node n + ON n.node_id='page:sha256:'||encode(p.metadata_root,'hex') WHERE p.prepare_id=$1", + [receipt.intent.prepare_id.clone().into(), (MAX_PLAN_BYTES as i32).into()], + )).await.map_err(internal)?.ok_or_else(|| unavailable("metadata preparation is missing"))?; + if row + .try_get::("", "operation_id") + .map_err(internal)? + != receipt.intent.operation_id + || row + .try_get::>("", "manifest_digest") + .map_err(internal)? + != receipt.intent.manifest_digest + || row + .try_get::>>("", "plan_digest") + .map_err(internal)? + .as_deref() + != Some(receipt.intent.manifest_digest.as_slice()) + || row + .try_get::("", "source_domain") + .map_err(internal)? + != identity.source_domain + || row + .try_get::("", "tagged_root_tree_oid") + .map_err(internal)? + != identity.tagged_root_tree_oid + || row.try_get::("", "scope").map_err(internal)? != identity.scope + || row.try_get::("", "schema_version").map_err(internal)? + != identity.schema_version as i16 + || row.try_get::("", "metadata_codec").map_err(internal)? + != identity.metadata_codec as i16 + || row + .try_get::("", "materialization_policy") + .map_err(internal)? + != identity.materialization_policy as i16 + || row.try_get::("", "fs_semantics").map_err(internal)? + != identity.fs_semantics as i16 + || row + .try_get::("", "access_projection") + .map_err(internal)? + != identity.access_projection as i16 + || row + .try_get::("", "verification_revision") + .map_err(internal)? + != identity.verification_revision + || row + .try_get::("", "projection_revision") + .map_err(internal)? + != identity.projection_revision as i16 + || row + .try_get::>("", "metadata_root") + .map_err(internal)? + != receipt.metadata_root + || row.try_get::("", "node_count").map_err(internal)? != receipt.node_count as i32 + || row.try_get::("", "edge_count").map_err(internal)? != receipt.edge_count as i32 + || row.try_get::("", "total_bytes").map_err(internal)? + != receipt.payload_bytes as i64 + || row.try_get::("", "state").map_err(internal)? != "COMMITTED" + || !row.try_get::("", "committed").map_err(internal)? + { + return Err(integrity( + "metadata preparation differs from its definitive receipt", + )); + } + if row + .try_get::>("", "root_kind") + .map_err(internal)? + .as_deref() + != Some("page") + || row + .try_get::>("", "root_state") + .map_err(internal)? + .as_deref() + != Some("LIVE") + || row + .try_get::>("", "root_bytes") + .map_err(internal)? + != Some(receipt.root_payload_bytes as i64) + || !row + .try_get::("", "prepare_covered") + .map_err(internal)? + { + return Err(unavailable( + "metadata receipt root is not LIVE and protected by its prepare", + )); + } + Ok(()) + } + + pub(crate) async fn restore_session_dag(&self, prepare_id: &str) -> Result<(), SnapshotError> { + let record = mst2_metadata_prepare::Entity::find_by_id(prepare_id.to_owned()) + .one(&self.connection) + .await + .map_err(internal)? + .ok_or_else(|| unavailable("snapshot preparation is missing"))?; + let digest = record + .manifest_digest + .as_slice() + .try_into() + .map_err(|_| integrity("invalid metadata preparation digest"))?; + let stored = load_plan(&self.connection, &record.operation_id, &digest) + .await? + .ok_or_else(|| unavailable("snapshot preparation disappeared"))?; + if stored.record.state != "COMMITTED" { + return Err(unavailable("snapshot preparation is not committed")); + } + load_installed_dag(&self.connection, &stored).await?; + let txn = self.transaction().await?; + let result = async { + self.barrier(&txn).await?; + verify_graph(&txn, &stored).await + } + .await; + match result { + Ok(()) => txn.commit().await.map_err(internal), + Err(error) => { + txn.rollback().await.map_err(internal)?; + Err(error) + } + } + } + + /// Verify all stored bytes outside the graph transaction. Immutable payloads + /// cannot change; LIVE state and coverage are checked again under the lock. + pub async fn load_installed_dag( + &self, + intent: &MetadataPrepareIntent, + ) -> Result { + let stored = require_plan(&self.connection, intent).await?; + load_installed_dag(&self.connection, &stored).await + } + + pub async fn finalize( + &self, + intent: &MetadataPrepareIntent, + ) -> Result { + let dag = self.load_installed_dag(intent).await?; + let txn = self.transaction().await?; + let result = self.finalize_in_txn(&txn, intent, &dag).await; + commit( + txn, + result, + &intent.operation_id, + intent.manifest_digest, + MetadataCommitPhase::Finalize, + ) + .await + } + + // This provisional result is private: only finalize's successful outer + // commit or the recovery barrier can issue a durable receipt to a caller. + async fn finalize_in_txn( + &self, + txn: &DatabaseTransaction, + intent: &MetadataPrepareIntent, + dag: &ValidatedMetadataDag, + ) -> Result { + self.barrier(txn).await?; + let stored = require_plan(txn, intent).await?; + self.finalize_stored_plan_in_txn(txn, intent, dag, stored) + .await + } + + async fn finalize_stored_plan_in_txn( + &self, + txn: &DatabaseTransaction, + intent: &MetadataPrepareIntent, + dag: &ValidatedMetadataDag, + stored: StoredPlan, + ) -> Result { + let expected: BTreeSet<_> = dag + .payloads() + .iter() + .map(|page| (page.id, page.size)) + .collect(); + if dag.root() != stored.plan.root + || expected + != stored + .plan + .pages + .iter() + .map(|(id, size)| (*id, *size)) + .collect() + || dag + .edges() + .iter() + .map(|edge| (edge.parent.clone(), edge.child.clone())) + .collect::>() + != stored + .plan + .edges + .iter() + .map(|(parent, child)| (node_id(parent), node_id(child))) + .collect() + { + return Err(integrity( + "validated installed DAG differs from durable preparation plan", + )); + } + check_payload_coverage(txn, &stored).await?; + if stored.record.state == "COMMITTED" { + verify_graph(txn, &stored).await?; + return stored.receipt(); + } + PostgresRetentionRepository::retain_group_in_txn( + txn, + dag.nodes(), + dag.edges(), + &[RetentionRoot::Prepare(intent.prepare_id.clone())], + ) + .await?; + verify_graph(txn, &stored).await?; + let result = txn + .execute_raw(statement( + "UPDATE mst2_metadata_prepare SET state='COMMITTED',committed_at=now() + WHERE prepare_id=$1 AND manifest_digest=$2 AND state='PREPARING'", + [ + intent.prepare_id.clone().into(), + intent.manifest_digest.to_vec().into(), + ], + )) + .await + .map_err(internal)?; + if result.rows_affected() != 1 { + return Err(integrity("metadata prepare state changed during finalize")); + } + stored.receipt() + } + + /// Call with a fresh connection to the same primary after a commit error. + /// Lock acquisition is the completion barrier; absence before it proves nothing. + pub async fn inspect_prepare( + &self, + fresh_primary: &DatabaseConnection, + operation_id: &str, + digest: [u8; 32], + phase: MetadataCommitPhase, + ) -> Result { + validate_operation_id(operation_id)?; + let txn = fresh_primary + .begin_with_config(Some(IsolationLevel::ReadCommitted), None) + .await + .map_err(|_| uncertain(operation_id, digest, phase))?; + if self.barrier(&txn).await.is_err() { + let _ = txn.rollback().await; + return Err(uncertain(operation_id, digest, phase)); + } + let result = async { + match load_plan(&txn, operation_id, &digest).await? { + None => Ok(MetadataPrepareObservation::Absent), + Some(stored) if stored.record.state == "COMMITTED" => { + // Recovery rechecks the bounded bytes/DAG under the barrier, + // so corruption cannot turn a stored state into a receipt. + // This may read 64 MiB; normal finalization verifies outside + // its short graph transaction instead. + load_installed_dag(&txn, &stored).await?; + check_payload_coverage(&txn, &stored).await?; + verify_graph(&txn, &stored).await?; + Ok(MetadataPrepareObservation::Committed(stored.receipt()?)) + } + Some(stored) => Ok(MetadataPrepareObservation::Preparing(stored.intent()?)), + } + } + .await; + txn.rollback() + .await + .map_err(|_| uncertain(operation_id, digest, phase))?; + result.map_err(|error: SnapshotError| { + if error.code == SnapshotErrorCode::Internal { + uncertain(operation_id, digest, phase) + } else { + MetadataInstallError::Rejected(error) + } + }) + } + + async fn transaction(&self) -> Result { + if self.connection.get_database_backend() != DbBackend::Postgres { + return Err(internal("metadata installation requires PostgreSQL")); + } + self.connection + .begin_with_config(Some(IsolationLevel::ReadCommitted), None) + .await + .map_err(internal) + } + + pub(super) async fn metadata_read_barrier( + &self, + txn: &DatabaseTransaction, + ) -> Result<(), SnapshotError> { + self.capability_barrier(txn).await + } + + async fn barrier(&self, txn: &DatabaseTransaction) -> Result<(), SnapshotError> { + if txn.get_database_backend() != DbBackend::Postgres { + return Err(internal("metadata installation requires PostgreSQL")); + } + if let Some(family) = &self.qualified_family { + family.enter(txn).await?; + } + if read_storage_scope(txn, self.captured_schema()).await? != self.storage_scope { + return Err(internal( + "metadata recovery connection is outside the captured primary storage scope", + )); + } + let isolation = txn + .query_one_raw(statement("SHOW transaction_isolation", [])) + .await + .map_err(internal)? + .ok_or_else(|| internal("missing transaction isolation"))? + .try_get_by_index::(0) + .map_err(internal)?; + if isolation != "read committed" { + return Err(internal("metadata installation requires READ COMMITTED")); + } + let recovery = txn + .query_one_raw(statement("SELECT pg_is_in_recovery()", [])) + .await + .map_err(internal)? + .ok_or_else(|| internal("missing primary status"))? + .try_get_by_index::(0) + .map_err(internal)?; + if recovery { + return Err(internal( + "metadata installation recovery requires the primary", + )); + } + let timeout = format!("{}ms", self.barrier_timeout.as_millis().clamp(1, 5000)); + txn.execute_raw(statement( + "SELECT set_config('lock_timeout',$1,true)", + [timeout.into()], + )) + .await + .map_err(internal)?; + txn.execute_raw(statement( + "SELECT pg_advisory_xact_lock($1,hashtext(current_schema()))", + [RETENTION_LOCK_KEY.into()], + )) + .await + .map_err(internal)?; + Ok(()) + } +} + +async fn capture_storage_schema( + connection: &C, +) -> Result { + if connection.get_database_backend() != DbBackend::Postgres { + return Err(internal("metadata storage scope requires PostgreSQL")); + } + let row = connection + .query_one_raw(statement( + "SELECT n.nspname AS schema FROM pg_catalog.pg_namespace n + JOIN pg_catalog.pg_class c ON c.relnamespace=n.oid + AND c.relname='mst2_metadata_storage_scope' AND c.relkind='r' + WHERE n.nspname=pg_catalog.current_schema() AND n.nspname NOT LIKE 'pg_temp_%'", + [], + )) + .await + .map_err(internal)? + .ok_or_else(|| internal("metadata actual primary storage relation is missing"))?; + row.try_get("", "schema").map_err(internal) +} + +async fn read_storage_scope( + connection: &C, + schema: &str, +) -> Result { + if connection.get_database_backend() != DbBackend::Postgres { + return Err(internal("metadata storage scope requires PostgreSQL")); + } + let row = connection + .query_one_raw(statement( + &format!( + "SELECT s.storage_uuid, pg_catalog.current_database() AS database, d.oid::bigint AS database_oid, + pg_catalog.current_schema() AS schema, n.oid::bigint AS schema_oid, + pg_catalog.inet_server_addr()::text AS server_address, pg_catalog.inet_server_port() AS server_port, + pg_catalog.pg_is_in_recovery() AS replica FROM \"{}\".mst2_metadata_storage_scope s + JOIN pg_catalog.pg_database d ON d.datname=pg_catalog.current_database() + JOIN pg_catalog.pg_namespace n ON n.nspname=pg_catalog.current_schema() WHERE s.singleton=1", + schema.replace('"', "\"\"") + ), + [], + )) + .await + .map_err(internal)? + .ok_or_else(|| internal("metadata primary storage scope is missing"))?; + if row.try_get::("", "replica").map_err(internal)? { + return Err(internal("metadata storage scope requires the primary")); + } + let storage_uuid: String = row.try_get("", "storage_uuid").map_err(internal)?; + let id = uuid::Uuid::parse_str(&storage_uuid) + .map_err(|_| internal("invalid metadata storage UUID"))?; + if id.is_nil() || id.to_string() != storage_uuid { + return Err(internal("noncanonical metadata storage UUID")); + } + Ok(PrimaryStorageScope { + storage_uuid, + database: row.try_get("", "database").map_err(internal)?, + database_oid: row.try_get("", "database_oid").map_err(internal)?, + schema: row.try_get("", "schema").map_err(internal)?, + schema_oid: row.try_get("", "schema_oid").map_err(internal)?, + server_address: row.try_get("", "server_address").map_err(internal)?, + server_port: row.try_get("", "server_port").map_err(internal)?, + }) +} + +async fn load_plan( + connection: &C, + operation_id: &str, + digest: &[u8; 32], +) -> Result, SnapshotError> { + let Some(record) = mst2_metadata_prepare::Entity::find() + .filter(mst2_metadata_prepare::Column::OperationId.eq(operation_id)) + .one(connection) + .await + .map_err(internal)? + else { + return Ok(None); + }; + if record.manifest_digest.as_slice() != digest { + return Err(SnapshotError::new( + SnapshotErrorCode::Conflict, + "metadata operation ID is bound to a different manifest", + )); + } + let plan = MetadataInstallPlan::decode(&record.canonical_plan, digest)?; + let identity = &plan.identity; + if record.source_domain != identity.source_domain + || record.tagged_root_tree_oid != identity.tagged_root_tree_oid + || record.scope != identity.scope + || record.schema_version != identity.schema_version as i16 + || record.metadata_codec != identity.metadata_codec as i16 + || record.materialization_policy != identity.materialization_policy as i16 + || record.fs_semantics != identity.fs_semantics as i16 + || record.access_projection != identity.access_projection as i16 + || record.verification_revision != identity.verification_revision + || record.projection_revision != identity.projection_revision as i16 + || record.metadata_root.as_slice() != plan.root + || record.node_count as usize != plan.pages.len() + || record.edge_count as usize != plan.edges.len() + || record.total_bytes as u64 != plan.total_bytes + || !["PREPARING", "COMMITTED"].contains(&record.state.as_str()) + || (record.state == "COMMITTED") != record.committed_at.is_some() + { + return Err(integrity( + "stored metadata preparation fields disagree with their canonical plan", + )); + } + let id = uuid::Uuid::parse_str(&record.prepare_id) + .map_err(|_| integrity("invalid stored preparation identity"))?; + if id.is_nil() || id.to_string() != record.prepare_id { + return Err(integrity("noncanonical stored preparation identity")); + } + let rows = mst2_metadata_prepare_page::Entity::find() + .filter(mst2_metadata_prepare_page::Column::PrepareId.eq(&record.prepare_id)) + .limit((MetadataDagLimits::default().nodes + 1) as u64) + .all(connection) + .await + .map_err(internal)?; + let mut coverage = BTreeSet::new(); + for row in &rows { + let page: [u8; 32] = row + .page_id + .as_slice() + .try_into() + .map_err(|_| integrity("invalid stored preparation page ID"))?; + coverage.insert((page, row.expected_size as u64)); + } + if coverage != plan.pages.iter().map(|(id, size)| (*id, *size)).collect() { + return Err(integrity( + "stored metadata preparation coverage differs from its canonical plan", + )); + } + Ok(Some(StoredPlan { + record, + plan, + prepare_pages: rows, + })) +} + +async fn require_plan( + connection: &C, + intent: &MetadataPrepareIntent, +) -> Result { + validate_operation_id(&intent.operation_id)?; + let stored = load_plan(connection, &intent.operation_id, &intent.manifest_digest) + .await? + .ok_or_else(|| unavailable("metadata preparation intent is not durable"))?; + if stored.record.prepare_id != intent.prepare_id { + return Err(integrity("metadata intent identity mismatch")); + } + Ok(stored) +} + +async fn load_installed_dag( + connection: &C, + stored: &StoredPlan, +) -> Result { + let rows = mst2_metadata_payload::Entity::find() + .filter( + mst2_metadata_payload::Column::PageId + .is_in(stored.plan.pages.keys().map(|id| id.to_vec())), + ) + .all(connection) + .await + .map_err(internal)?; + let mut pages = Vec::with_capacity(rows.len()); + for row in rows { + let id: [u8; 32] = row + .page_id + .as_slice() + .try_into() + .map_err(|_| integrity("invalid stored page ID"))?; + if row.metadata_codec != stored.plan.identity.metadata_codec as i16 + || stored.plan.pages.get(&id) != Some(&(row.byte_size as u64)) + { + return Err(integrity( + "stored page profile or size disagrees with its manifest", + )); + } + let page = MetadataPagePayload { + id, + size: row.byte_size as u64, + bytes: row.payload, + }; + validate_payload(&page)?; + pages.push(page); + } + if pages.len() != stored.plan.pages.len() { + return Err(unavailable( + "native metadata installation has missing payloads", + )); + } + ValidatedMetadataDag::validate( + MetadataDagCandidate { + metadata_codec: stored.plan.identity.metadata_codec, + root: stored.plan.root, + pages, + edges: stored.plan.edges.iter().copied().collect(), + }, + MetadataDagLimits::default(), + ) +} + +async fn check_payload_coverage( + connection: &C, + stored: &StoredPlan, +) -> Result<(), SnapshotError> { + let missing=connection.query_one_raw(statement( + "SELECT p.page_id FROM mst2_metadata_prepare_page p LEFT JOIN mst2_metadata_payload b ON b.page_id=p.page_id + WHERE p.prepare_id=$1 AND (b.page_id IS NULL OR b.byte_size<>p.expected_size OR b.metadata_codec<>$2 + OR octet_length(b.payload)<>p.expected_size) LIMIT 1", + [stored.record.prepare_id.clone().into(),stored.record.metadata_codec.into()], + )).await.map_err(internal)?; + if missing.is_some() { + return Err(unavailable( + "durable metadata payload coverage is incomplete", + )); + } + Ok(()) +} + +async fn verify_graph( + connection: &C, + stored: &StoredPlan, +) -> Result<(), SnapshotError> { + let sizes: BTreeMap<_, _> = stored + .plan + .pages + .iter() + .map(|(id, size)| (node_id(id), *size)) + .collect(); + let ids: Vec<_> = sizes.keys().cloned().collect(); + let nodes = mst2_retention_node::Entity::find() + .filter(mst2_retention_node::Column::NodeId.is_in(ids.clone())) + .all(connection) + .await + .map_err(internal)?; + if nodes.len() != ids.len() { + return Err(unavailable("prepared metadata graph has missing nodes")); + } + for node in nodes { + if node.state != "LIVE" || node.kind != "page" { + return Err(unavailable("prepared metadata graph is not LIVE")); + } + if sizes.get(&node.node_id) != Some(&(node.bytes as u64)) { + return Err(integrity( + "prepared metadata graph bytes differ from its plan", + )); + } + } + let edges = mst2_retention_edge::Entity::find() + .filter(mst2_retention_edge::Column::ParentId.is_in(ids.clone())) + .limit((MetadataDagLimits::default().edges + 1) as u64) + .all(connection) + .await + .map_err(internal)?; + let actual: BTreeSet<_> = edges + .into_iter() + .map(|edge| (edge.parent_id, edge.child_id)) + .collect(); + let expected = stored + .plan + .edges + .iter() + .map(|(parent, child)| (node_id(parent), node_id(child))) + .collect(); + if actual != expected { + return Err(integrity( + "prepared metadata retention edges differ from its plan", + )); + } + let roots = mst2_retention_root::Entity::find() + .filter( + mst2_retention_root::Column::RootKey + .eq(format!("prepare:{}", stored.record.prepare_id)), + ) + .limit((MetadataDagLimits::default().nodes + 1) as u64) + .all(connection) + .await + .map_err(internal)?; + if roots.is_empty() { + let transferred = connection + .query_one_raw(statement( + "SELECT s.snapshot_id FROM mst2_snapshot_context s + JOIN mst2_retention_root r ON r.root_key='pin:session:'||s.snapshot_id + AND r.node_id='page:sha256:'||encode(s.metadata_root,'hex') AND r.root_kind='pin' + WHERE s.prepare_id=$1 LIMIT 1", + [stored.record.prepare_id.clone().into()], + )) + .await + .map_err(internal)?; + if transferred.is_none() { + return Err(unavailable("metadata protection handoff is incomplete")); + } + } else if roots.iter().any(|root| root.root_kind != "prepare") + || roots + .into_iter() + .map(|root| root.node_id) + .collect::>() + != ids.iter().cloned().collect() + { + return Err(unavailable("prepared metadata pin coverage is incomplete")); + } + let bad=connection.query_one_raw(statement( + "SELECT n.node_id FROM mst2_retention_node n WHERE n.node_id IN (SELECT jsonb_array_elements_text($1::jsonb)) + AND n.incoming_refs<>(SELECT count(*) FROM mst2_retention_edge e WHERE e.child_id=n.node_id) LIMIT 1", + [serde_json::to_string(&ids).map_err(internal)?.into()], + )).await.map_err(internal)?; + if bad.is_some() { + return Err(integrity("prepared metadata graph counter audit failed")); + } + Ok(()) +} + +fn validate_payload(payload: &MetadataPagePayload) -> Result<(), SnapshotError> { + if !(HEADER_LEN..=PAGE_MAX_BYTES).contains(&payload.bytes.len()) { + return Err(SnapshotError::new( + SnapshotErrorCode::LimitExceeded, + "metadata page exceeds protocol length bounds", + )); + } + if payload.size != payload.bytes.len() as u64 || page_id(&payload.bytes) != payload.id { + return Err(SnapshotError::new( + SnapshotErrorCode::DigestMismatch, + "metadata payload size or protocol page ID mismatch", + )); + } + Page::decode(&payload.bytes) + .map_err(|error| integrity(&format!("invalid canonical metadata page: {error}")))?; + Ok(()) +} + +fn validate_operation_id(operation: &str) -> Result<(), SnapshotError> { + if operation.is_empty() || operation.len() > 255 || operation.contains('\0') { + return Err(SnapshotError::new( + SnapshotErrorCode::InvalidRequest, + "metadata operation ID must be 1..=255 UTF8 bytes", + )); + } + Ok(()) +} + +async fn commit( + txn: DatabaseTransaction, + result: Result, + operation: &str, + digest: [u8; 32], + phase: MetadataCommitPhase, +) -> Result { + match result { + Ok(value) => { + txn.commit() + .await + .map_err(|_| uncertain(operation, digest, phase))?; + Ok(value) + } + Err(error) => { + txn.rollback().await.map_err(internal)?; + Err(error.into()) + } + } +} +fn uncertain( + operation: &str, + digest: [u8; 32], + phase: MetadataCommitPhase, +) -> MetadataInstallError { + MetadataInstallError::CommitUncertain { + operation_id: operation.into(), + manifest_digest: digest, + phase, + } +} +fn node_id(id: &[u8; 32]) -> String { + format!("page:sha256:{}", hex::encode(id)) +} +fn statement(sql: &str, values: [sea_orm::Value; N]) -> Statement { + Statement::from_sql_and_values(DbBackend::Postgres, sql, values) +} +fn internal(error: impl std::fmt::Display) -> SnapshotError { + SnapshotError::new(SnapshotErrorCode::Internal, error.to_string()) +} +fn integrity(message: &str) -> SnapshotError { + SnapshotError::new(SnapshotErrorCode::IntegrityError, message) +} +fn unavailable(message: &str) -> SnapshotError { + SnapshotError::new(SnapshotErrorCode::ObjectUnavailable, message) +} + +#[path = "native_metadata_install_capability.rs"] +mod capability; +pub(crate) use capability::{LegacyPayloadInstallWork, ValidatedLegacyInstallCapability}; + +#[cfg(test)] +#[path = "native_metadata_install_capability_tests.rs"] +mod capability_tests; + +// Additive repository only. HTTP sessions continue using the existing installer +// until session anchors and lease adoption bind the generation seal. +#[path = "native_metadata_generations.rs"] +pub mod generations; + +#[cfg(test)] +#[path = "native_metadata_install_tests.rs"] +mod tests; diff --git a/src/jupiter/storage/native_metadata_install_capability.rs b/src/jupiter/storage/native_metadata_install_capability.rs new file mode 100644 index 00000000..7b912095 --- /dev/null +++ b/src/jupiter/storage/native_metadata_install_capability.rs @@ -0,0 +1,708 @@ +//! Reuse pure plan validation only after its durable membership is frozen. + +use sha2::{Digest, Sha256}; + +use super::{ + BTreeMap, BTreeSet, ConnectionTrait, DatabaseTransaction, MetadataCommitPhase, + MetadataInstallError, MetadataInstallIdentity, MetadataPagePayload, MetadataPrepareIntent, + PostgresMetadataInstallRepository, PrimaryStorageScope, QueryResult, SnapshotError, + SnapshotErrorCode, check_payload_coverage, commit, integrity, internal, json, + load_installed_dag, require_plan, statement, unavailable, validate_payload, verify_graph, +}; + +const SEAL_DOMAIN: &[u8] = b"MST2-LEGACY-INSTALL-CAPABILITY-1\0"; + +#[derive(Debug)] +pub(crate) struct ValidatedLegacyInstallCapability { + intent: MetadataPrepareIntent, + identity: MetadataInstallIdentity, + root: [u8; 32], + edge_count: usize, + total_bytes: u64, + members: BTreeMap<[u8; 32], (i32, Option)>, + members_digest: [u8; 32], + scope: PrimaryStorageScope, + primary_scope: Vec, + install_seal: [u8; 32], +} + +/// Payload batch work only, excluding barriers, intent/capability setup, +/// finalize and the complete COMMITTED replay oracle. Classification reuses +/// the requested-member query; its batch count is not an extra SQL trip. +#[derive(Debug, Clone, Default, PartialEq, Eq)] +pub(crate) struct LegacyPayloadInstallWork { + pub requested_pages: u64, + pub payload_pages_omitted: u64, + pub payload_pages_encoded: u64, + pub requested_payload_bytes_validated: u64, + pub payload_bytes_omitted: u64, + pub payload_bytes_encoded: u64, + pub metadata_parameter_bytes: u64, + pub payload_parameter_bytes: u64, + pub transactions: u64, + pub registration_queries: u64, + pub requested_member_queries: u64, + pub classification_batches: u64, + pub insert_statements: u64, + pub byte_comparison_queries: u64, + pub committed_replay_pages: u64, + pub elapsed_micros: u64, +} + +impl LegacyPayloadInstallWork { + pub(crate) fn record(&mut self, batch: Self) { + self.requested_pages += batch.requested_pages; + self.payload_pages_omitted += batch.payload_pages_omitted; + self.payload_pages_encoded += batch.payload_pages_encoded; + self.requested_payload_bytes_validated += batch.requested_payload_bytes_validated; + self.payload_bytes_omitted += batch.payload_bytes_omitted; + self.payload_bytes_encoded += batch.payload_bytes_encoded; + self.metadata_parameter_bytes += batch.metadata_parameter_bytes; + self.payload_parameter_bytes += batch.payload_parameter_bytes; + self.transactions += batch.transactions; + self.registration_queries += batch.registration_queries; + self.requested_member_queries += batch.requested_member_queries; + self.classification_batches += batch.classification_batches; + self.insert_statements += batch.insert_statements; + self.byte_comparison_queries += batch.byte_comparison_queries; + self.committed_replay_pages += batch.committed_replay_pages; + self.elapsed_micros += batch.elapsed_micros; + } +} + +impl PostgresMetadataInstallRepository { + pub(crate) async fn mint_legacy_install_capability( + &self, + intent: &MetadataPrepareIntent, + ) -> Result { + let txn = self.transaction().await?; + let result = async { + self.capability_barrier(&txn).await?; + let stored = require_plan(&txn, intent).await?; + let record = &stored.record; + if record.storage_seal.is_some() + || record.canonical_bindings.is_some() + || record.bindings_digest.is_some() + || record.primary_scope.is_some() + || record.graph_domain.is_some() + || record.coverage_retired_at.is_some() + || record.aborted_at.is_some() + { + return Err(unavailable("install capability requires an active unbound legacy preparation")); + } + let mut members = BTreeMap::new(); + for member in &stored.prepare_pages { + if member.generation.is_some() { + return Err(integrity("legacy install capability cannot adopt generation bindings")); + } + let id = member.page_id.as_slice().try_into().map_err(internal)?; + members.insert(id, (member.expected_size, member.generation)); + } + if record.state == "COMMITTED" { + load_installed_dag(&txn, &stored).await?; + check_payload_coverage(&txn, &stored).await?; + verify_graph(&txn, &stored).await?; + } + let members_digest = member_digest(&members); + let primary_scope = scope_bytes(&self.storage_scope)?; + let install_seal = seal(intent, &members_digest, &primary_scope)?; + txn.execute_raw(statement( + "INSERT INTO mst2_metadata_install_seal(prepare_id,operation_id,manifest_digest,members_digest,primary_scope,install_seal) + VALUES($1,$2,$3,$4,$5,$6) ON CONFLICT(prepare_id) DO NOTHING", + [intent.prepare_id.clone().into(),intent.operation_id.clone().into(), + intent.manifest_digest.to_vec().into(),members_digest.to_vec().into(), + primary_scope.clone().into(),install_seal.to_vec().into()], + )).await.map_err(internal)?; + let capability = ValidatedLegacyInstallCapability { + intent: intent.clone(), identity: stored.plan.identity, root: stored.plan.root, + edge_count: stored.plan.edges.len(), total_bytes: stored.plan.total_bytes, + members,members_digest,scope:self.storage_scope.clone(),primary_scope,install_seal, + }; + read_registered_prepare(&txn, &capability).await?; + Ok(capability) + }.await; + commit( + txn, + result, + &intent.operation_id, + intent.manifest_digest, + MetadataCommitPhase::Intent, + ) + .await + } + + pub(crate) async fn install_pages_validated( + &self, + capability: &ValidatedLegacyInstallCapability, + payloads: &[MetadataPagePayload], + ) -> Result<(), MetadataInstallError> { + if capability.scope != self.storage_scope { + return Err( + integrity("install capability belongs to another captured primary scope").into(), + ); + } + if payloads.is_empty() || payloads.len() > 64 { + return Err(SnapshotError::new( + SnapshotErrorCode::LimitExceeded, + "metadata installation batch must contain 1..=64 pages", + ) + .into()); + } + let mut ids = BTreeSet::new(); + for payload in payloads { + validate_payload(payload)?; + if !ids.insert(payload.id) { + return Err(integrity("duplicate page in metadata installation batch").into()); + } + if capability + .members + .get(&payload.id) + .map(|member| member.0 as u64) + != Some(payload.size) + { + return Err(integrity( + "metadata payload is not a member of its validated installation", + ) + .into()); + } + } + let pages: Vec<_> = payloads + .iter() + .map(|p| { + json!({ + "page_id":hex::encode(p.id),"size":p.size,"payload":hex::encode(&p.bytes) + }) + }) + .collect(); + let encoded = serde_json::to_string(&pages).map_err(internal)?; + let txn = self.transaction().await?; + let result = async { + self.capability_barrier(&txn).await?; + let state = read_registered_prepare(&txn,capability).await?; + check_requested_members(&txn,capability,&encoded,payloads.len()).await?; + if state == "COMMITTED" { + // Committed replay retains the complete receipt oracle and never repairs bytes. + let stored = require_plan(&txn,&capability.intent).await?; + load_installed_dag(&txn,&stored).await?; + check_payload_coverage(&txn,&stored).await?; + verify_graph(&txn,&stored).await?; + } else { + txn.execute_raw(statement( + "INSERT INTO mst2_metadata_payload(page_id,metadata_codec,byte_size,payload) + SELECT decode(p.page_id,'hex'),$1,p.size,decode(p.payload,'hex') + FROM jsonb_to_recordset($2::jsonb) AS p(page_id text,size integer,payload text) + ON CONFLICT(page_id) DO NOTHING", + [(capability.identity.metadata_codec as i16).into(),encoded.clone().into()], + )).await.map_err(internal)?; + } + let bad = txn.query_one_raw(statement( + "SELECT p.page_id FROM jsonb_to_recordset($2::jsonb) AS p(page_id text,size integer,payload text) + LEFT JOIN mst2_metadata_payload b ON b.page_id=decode(p.page_id,'hex') + WHERE b.page_id IS NULL OR b.metadata_codec<>$1 OR b.byte_size<>p.size + OR b.payload<>decode(p.payload,'hex') LIMIT 1", + [(capability.identity.metadata_codec as i16).into(),encoded.into()], + )).await.map_err(internal)?; + if bad.is_some() { + return Err(integrity("immutable metadata payload identity conflicts with stored bytes")); + } + Ok(()) + }.await; + commit( + txn, + result, + &capability.intent.operation_id, + capability.intent.manifest_digest, + MetadataCommitPhase::Payload, + ) + .await + } + + /// Classify and install in one existing payload transaction. Presence is + /// only a write hint; finalize still reads and validates every stored byte. + pub(crate) async fn install_missing_pages_validated( + &self, + capability: &ValidatedLegacyInstallCapability, + payloads: &[MetadataPagePayload], + ) -> Result { + let started = std::time::Instant::now(); + if capability.scope != self.storage_scope { + return Err( + integrity("install capability belongs to another captured primary scope").into(), + ); + } + if payloads.is_empty() || payloads.len() > 64 { + return Err(SnapshotError::new( + SnapshotErrorCode::LimitExceeded, + "metadata installation batch must contain 1..=64 pages", + ) + .into()); + } + let mut ids = BTreeSet::new(); + for payload in payloads { + validate_payload(payload)?; + if !ids.insert(payload.id) { + return Err(integrity("duplicate page in metadata installation batch").into()); + } + if capability + .members + .get(&payload.id) + .map(|member| member.0 as u64) + != Some(payload.size) + { + return Err(integrity( + "metadata payload is not a member of its validated installation", + ) + .into()); + } + } + let metadata = serde_json::to_string( + &payloads + .iter() + .map(|page| json!({"page_id":hex::encode(page.id),"size":page.size})) + .collect::>(), + ) + .map_err(internal)?; + let txn = self.transaction().await?; + let result = async { + self.capability_barrier(&txn).await?; + let state = read_registered_prepare(&txn, capability).await?; + let rows = read_requested_members(&txn, capability, &metadata, payloads.len()).await?; + let mut present = BTreeSet::new(); + for row in rows { + if !row.try_get::("", "payload_present").map_err(internal)? { + continue; + } + if row.try_get::>("", "payload_codec").map_err(internal)? + != Some(capability.identity.metadata_codec as i16) + || row.try_get::>("", "payload_size").map_err(internal)? + != row.try_get::>("", "expected_size").map_err(internal)? + { + return Err(integrity("immutable metadata payload profile conflicts with its fixed member")); + } + let page: Vec = row.try_get("", "page_id").map_err(internal)?; + present.insert(<[u8; 32]>::try_from(page.as_slice()).map_err(internal)?); + } + let committed = state == "COMMITTED"; + if committed { + // A committed replay never repairs missing or corrupt bytes. + let stored = require_plan(&txn, &capability.intent).await?; + load_installed_dag(&txn, &stored).await?; + check_payload_coverage(&txn, &stored).await?; + verify_graph(&txn, &stored).await?; + } + let submitted: Vec<_> = payloads + .iter() + .filter(|page| committed || !present.contains(&page.id)) + .collect(); + let requested_bytes = payloads.iter().map(|page| page.size).sum::(); + let submitted_bytes = submitted.iter().map(|page| page.size).sum::(); + let mut work = LegacyPayloadInstallWork { + requested_pages: payloads.len() as u64, + payload_pages_omitted: (payloads.len() - submitted.len()) as u64, + payload_pages_encoded: submitted.len() as u64, + requested_payload_bytes_validated: requested_bytes, + payload_bytes_omitted: requested_bytes - submitted_bytes, + payload_bytes_encoded: submitted_bytes, + metadata_parameter_bytes: metadata.len() as u64, + transactions: 1, + registration_queries: 1, + requested_member_queries: 1, + classification_batches: u64::from(!committed), + committed_replay_pages: if committed { payloads.len() as u64 } else { 0 }, + ..LegacyPayloadInstallWork::default() + }; + if !submitted.is_empty() { + let pages: Vec<_> = submitted.iter().map(|page| json!({ + "page_id":hex::encode(page.id),"size":page.size,"payload":hex::encode(&page.bytes) + })).collect(); + let encoded = serde_json::to_string(&pages).map_err(internal)?; + let encoded_len = encoded.len() as u64; + if !committed { + txn.execute_raw(statement( + "INSERT INTO mst2_metadata_payload(page_id,metadata_codec,byte_size,payload) + SELECT decode(p.page_id,'hex'),$1,p.size,decode(p.payload,'hex') + FROM jsonb_to_recordset($2::jsonb) AS p(page_id text,size integer,payload text) + ON CONFLICT(page_id) DO NOTHING", + [(capability.identity.metadata_codec as i16).into(),encoded.clone().into()], + )).await.map_err(internal)?; + work.insert_statements = 1; + work.payload_parameter_bytes += encoded_len; + } + let bad = txn.query_one_raw(statement( + "SELECT p.page_id FROM jsonb_to_recordset($2::jsonb) AS p(page_id text,size integer,payload text) + LEFT JOIN mst2_metadata_payload b ON b.page_id=decode(p.page_id,'hex') + WHERE b.page_id IS NULL OR b.metadata_codec<>$1 OR b.byte_size<>p.size + OR b.payload<>decode(p.payload,'hex') LIMIT 1", + [(capability.identity.metadata_codec as i16).into(),encoded.into()], + )).await.map_err(internal)?; + work.byte_comparison_queries = 1; + work.payload_parameter_bytes += encoded_len; + if bad.is_some() { + return Err(integrity("immutable metadata payload identity conflicts with stored bytes")); + } + } + Ok(work) + }.await; + let mut work = commit( + txn, + result, + &capability.intent.operation_id, + capability.intent.manifest_digest, + MetadataCommitPhase::Payload, + ) + .await?; + work.elapsed_micros = started.elapsed().as_micros().min(u64::MAX as u128) as u64; + Ok(work) + } + + pub(super) async fn capability_barrier( + &self, + txn: &DatabaseTransaction, + ) -> Result<(), SnapshotError> { + let schema = txn + .query_one_raw(statement( + "SELECT pg_catalog.current_schema() AS schema", + [], + )) + .await + .map_err(internal)? + .ok_or_else(|| internal("primary schema is missing"))? + .try_get::("", "schema") + .map_err(internal)?; + if schema != self.storage_scope.schema { + return Err(integrity( + "install capability transaction is outside its captured schema", + )); + } + txn.execute_raw(statement( + "SELECT pg_catalog.set_config('search_path',pg_catalog.quote_ident($1)||',pg_catalog,pg_temp',true)", + [schema.into()], + )).await.map_err(internal)?; + self.barrier(txn).await + } +} + +async fn read_registered_prepare( + txn: &DatabaseTransaction, + capability: &ValidatedLegacyInstallCapability, +) -> Result { + let row=txn.query_one_raw(statement( + "SELECT p.prepare_id,p.operation_id,p.manifest_digest,p.source_domain,p.tagged_root_tree_oid,p.scope, + p.schema_version,p.metadata_codec,p.materialization_policy,p.fs_semantics,p.access_projection, + p.verification_revision,p.projection_revision,p.metadata_root,p.node_count,p.edge_count,p.total_bytes, + p.state,p.committed_at IS NOT NULL AS committed,p.aborted_at IS NOT NULL AS aborted, + p.coverage_retired_at IS NOT NULL AS retired, + p.storage_seal IS NOT NULL OR p.canonical_bindings IS NOT NULL OR p.bindings_digest IS NOT NULL + OR p.primary_scope IS NOT NULL OR p.graph_domain IS NOT NULL AS generation_bound, + s.operation_id AS sealed_operation,s.manifest_digest AS sealed_manifest,s.members_digest, + s.primary_scope AS sealed_scope,s.install_seal + FROM mst2_metadata_prepare p JOIN mst2_metadata_install_seal s USING(prepare_id) + WHERE p.prepare_id=$1", + [capability.intent.prepare_id.clone().into()], + )).await.map_err(internal)?.ok_or_else(||unavailable("validated install registration is missing"))?; + let identity = &capability.identity; + for (column, expected) in [ + ("prepare_id", &capability.intent.prepare_id), + ("operation_id", &capability.intent.operation_id), + ("sealed_operation", &capability.intent.operation_id), + ("source_domain", &identity.source_domain), + ("tagged_root_tree_oid", &identity.tagged_root_tree_oid), + ("scope", &identity.scope), + ] { + if row.try_get::("", column).map_err(internal)? != *expected { + return Err(integrity( + "validated install identity differs from durable registration", + )); + } + } + for (column, expected) in [ + ("schema_version", identity.schema_version), + ("metadata_codec", identity.metadata_codec), + ("materialization_policy", identity.materialization_policy), + ("fs_semantics", identity.fs_semantics), + ("access_projection", identity.access_projection), + ("projection_revision", identity.projection_revision), + ] { + if row.try_get::("", column).map_err(internal)? != expected as i16 { + return Err(integrity( + "validated install profile differs from durable registration", + )); + } + } + for (column, expected) in [ + ( + "manifest_digest", + capability.intent.manifest_digest.as_slice(), + ), + ( + "sealed_manifest", + capability.intent.manifest_digest.as_slice(), + ), + ("metadata_root", capability.root.as_slice()), + ("members_digest", capability.members_digest.as_slice()), + ("sealed_scope", capability.primary_scope.as_slice()), + ("install_seal", capability.install_seal.as_slice()), + ] { + if row + .try_get::>("", column) + .map_err(internal)? + .as_slice() + != expected + { + return Err(integrity( + "validated install proof differs from durable registration", + )); + } + } + let state = row.try_get::("", "state").map_err(internal)?; + if row + .try_get::("", "verification_revision") + .map_err(internal)? + != identity.verification_revision + || row.try_get::("", "node_count").map_err(internal)? as usize + != capability.members.len() + || row.try_get::("", "edge_count").map_err(internal)? as usize != capability.edge_count + || row.try_get::("", "total_bytes").map_err(internal)? as u64 != capability.total_bytes + || !["PREPARING", "COMMITTED"].contains(&state.as_str()) + || row.try_get::("", "committed").map_err(internal)? != (state == "COMMITTED") + || row.try_get::("", "aborted").map_err(internal)? + || row.try_get::("", "retired").map_err(internal)? + || row + .try_get::("", "generation_bound") + .map_err(internal)? + { + return Err(unavailable( + "validated install preparation is no longer active with its fixed profile", + )); + } + Ok(state) +} + +async fn check_requested_members( + txn: &DatabaseTransaction, + capability: &ValidatedLegacyInstallCapability, + encoded: &str, + expected_count: usize, +) -> Result<(), SnapshotError> { + read_requested_members(txn, capability, encoded, expected_count) + .await + .map(|_| ()) +} + +async fn read_requested_members( + txn: &DatabaseTransaction, + capability: &ValidatedLegacyInstallCapability, + encoded: &str, + expected_count: usize, +) -> Result, SnapshotError> { + let rows=txn.query_all_raw(statement( + "SELECT decode(p.page_id,'hex') AS page_id,m.expected_size,m.generation AS member_generation, + b.page_id IS NOT NULL AS payload_present,b.generation AS payload_generation, + b.metadata_codec AS payload_codec,b.byte_size AS payload_size, + c.generation AS current_generation,l.generation AS lifetime_generation,l.graph_domain, + l.state AS lifetime_state,l.metadata_codec AS lifetime_codec,l.expected_size AS lifetime_size, + n.state AS graph_state,n.kind AS graph_kind,n.bytes AS graph_bytes, + EXISTS(SELECT 1 FROM mst2_retention_gc_op g WHERE g.node_id='page:sha256:'||p.page_id + AND g.operation='REMOVE' AND g.state IN ('PENDING','APPLIED')) AS tombstone + FROM jsonb_to_recordset($2::jsonb) AS p(page_id text,size integer,payload text) + LEFT JOIN mst2_metadata_prepare_page m ON m.prepare_id=$1 AND m.page_id=decode(p.page_id,'hex') + LEFT JOIN mst2_metadata_payload b ON b.page_id=m.page_id + LEFT JOIN mst2_metadata_current c ON c.page_id=m.page_id + LEFT JOIN mst2_metadata_lifetime l ON l.page_id=c.page_id AND l.generation=c.generation + LEFT JOIN mst2_retention_node n ON n.node_id='page:sha256:'||p.page_id", + [capability.intent.prepare_id.clone().into(),encoded.into()], + )).await.map_err(internal)?; + if rows.len() != expected_count { + return Err(integrity("requested install membership is incomplete")); + } + for row in &rows { + let page: Vec = row.try_get("", "page_id").map_err(internal)?; + let page: [u8; 32] = page.as_slice().try_into().map_err(internal)?; + let size = row + .try_get::>("", "expected_size") + .map_err(internal)?; + let generation = row + .try_get::>("", "member_generation") + .map_err(internal)?; + if capability.members.get(&page).copied() != size.map(|size| (size, generation)) { + return Err(integrity( + "requested member differs from validated installation", + )); + } + check_physical_member( + row, + capability.identity.metadata_codec as i16, + size.ok_or_else(|| integrity("requested member is missing"))?, + )?; + } + Ok(rows) +} + +fn check_physical_member(row: &QueryResult, codec: i16, size: i32) -> Result<(), SnapshotError> { + let current = row + .try_get::>("", "current_generation") + .map_err(internal)?; + let lifetime = row + .try_get::>("", "lifetime_generation") + .map_err(internal)?; + let payload = row + .try_get::>("", "payload_generation") + .map_err(internal)?; + let graph = row + .try_get::>("", "graph_state") + .map_err(internal)?; + if row.try_get::("", "tombstone").map_err(internal)? + || graph.as_deref().is_some_and(|state| state != "LIVE") + || row + .try_get::>("", "graph_kind") + .map_err(internal)? + .is_some_and(|kind| kind != "page") + || row + .try_get::>("", "graph_bytes") + .map_err(internal)? + .is_some_and(|bytes| bytes != size as i64) + { + return Err(unavailable( + "requested generic metadata graph is unavailable", + )); + } + if payload.is_some() && (payload != current || payload != lifetime) { + return Err(integrity( + "generic payload is not its exact current physical lifetime", + )); + } + if current.is_some() { + let state = row + .try_get::>("", "lifetime_state") + .map_err(internal)?; + if current != lifetime + || current.is_some_and(|generation| generation <= 0) + || row + .try_get::>("", "graph_domain") + .map_err(internal)? + .as_deref() + != Some("generic-v1") + || !matches!(state.as_deref(), Some("RESERVED" | "LIVE")) + || row + .try_get::>("", "lifetime_codec") + .map_err(internal)? + != Some(codec) + || row + .try_get::>("", "lifetime_size") + .map_err(internal)? + != Some(size) + || (state.as_deref() == Some("LIVE") + && (graph.is_none() + || payload != current + || !row + .try_get::("", "payload_present") + .map_err(internal)?)) + { + return Err(unavailable( + "requested physical lifetime cannot be shared by legacy generic installation", + )); + } + } + Ok(()) +} + +fn member_digest(members: &BTreeMap<[u8; 32], (i32, Option)>) -> [u8; 32] { + let mut hash = Sha256::new(); + for (id, (size, generation)) in members { + hash.update(id); + hash.update(size.to_be_bytes()); + match generation { + None => hash.update([0]), + Some(generation) => { + hash.update([1]); + hash.update(generation.to_be_bytes()); + } + } + } + hash.finalize().into() +} + +fn scope_bytes(scope: &PrimaryStorageScope) -> Result, SnapshotError> { + let bytes = serde_json::to_vec(&( + &scope.storage_uuid, + &scope.database, + scope.database_oid, + &scope.schema, + scope.schema_oid, + &scope.server_address, + scope.server_port, + )) + .map_err(internal)?; + if bytes.is_empty() || bytes.len() > 16384 { + return Err(integrity("invalid primary install scope size")); + } + Ok(bytes) +} + +fn seal( + intent: &MetadataPrepareIntent, + members: &[u8; 32], + scope: &[u8], +) -> Result<[u8; 32], SnapshotError> { + let id = uuid::Uuid::parse_str(&intent.prepare_id).map_err(internal)?; + let mut hash = Sha256::new(); + hash.update(SEAL_DOMAIN); + hash.update(id.as_bytes()); + hash.update((intent.operation_id.len() as u32).to_be_bytes()); + hash.update(intent.operation_id.as_bytes()); + hash.update(intent.manifest_digest); + hash.update(members); + hash.update((scope.len() as u32).to_be_bytes()); + hash.update(scope); + Ok(hash.finalize().into()) +} + +#[cfg(test)] +#[tokio::test] +async fn install_capability_physical_checker_rejects_real_qualified_owner() { + let (db, _other, _schema) = super::capability_tests::fixture().await; + let pages = super::capability_tests::prepared(2); + let qualified = + super::generations::qualified::PostgresQualifiedMetadataRepository::new(db.clone()) + .await + .unwrap(); + let bound = qualified + .begin_intent("physical-qualified", &pages) + .await + .unwrap(); + qualified + .install_pages(&bound, pages.dag().payloads()) + .await + .unwrap(); + let legacy = PostgresMetadataInstallRepository::new(db).await.unwrap(); + let txn = legacy.transaction().await.unwrap(); + legacy.capability_barrier(&txn).await.unwrap(); + for page in pages.dag().payloads() { + let row = txn.query_one_raw(statement( + "SELECT b.page_id IS NOT NULL AS payload_present,b.generation AS payload_generation, + c.generation AS current_generation,l.generation AS lifetime_generation,l.graph_domain, + l.state AS lifetime_state,l.metadata_codec AS lifetime_codec,l.expected_size AS lifetime_size, + n.state AS graph_state,n.kind AS graph_kind,n.bytes AS graph_bytes, + EXISTS(SELECT 1 FROM mst2_retention_gc_op g WHERE g.node_id=l.node_id + AND g.operation='REMOVE' AND g.state IN ('PENDING','APPLIED')) AS tombstone + FROM mst2_metadata_current c JOIN mst2_metadata_lifetime l USING(page_id,generation) + JOIN mst2_metadata_payload b USING(page_id,generation) + LEFT JOIN mst2_retention_node n ON n.node_id=l.node_id WHERE c.page_id=$1", + [page.id.to_vec().into()], + )).await.unwrap().unwrap(); + assert_eq!( + row.try_get::("", "graph_domain").unwrap(), + "qualified-v1" + ); + assert_eq!(row.try_get::("", "current_generation").unwrap(), 1); + assert_eq!(row.try_get::("", "payload_generation").unwrap(), 1); + assert!(row.try_get::("", "payload_present").unwrap()); + let error = check_physical_member(&row, 1, page.size as i32).unwrap_err(); + assert_eq!(error.code, SnapshotErrorCode::ObjectUnavailable); + assert_eq!( + error.message, + "requested physical lifetime cannot be shared by legacy generic installation" + ); + } + txn.rollback().await.unwrap(); +} diff --git a/src/jupiter/storage/native_metadata_install_capability_fault_tests.rs b/src/jupiter/storage/native_metadata_install_capability_fault_tests.rs new file mode 100644 index 00000000..9903085e --- /dev/null +++ b/src/jupiter/storage/native_metadata_install_capability_fault_tests.rs @@ -0,0 +1,217 @@ +use super::*; + +async fn count(db: &DatabaseConnection, sql: &str) -> i64 { + db.query_one_raw(statement(sql, [])) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap() +} + +async fn mint_fault(fault: Fault) { + let (direct, recovery, _schema, url) = fixture().await; + let proxy = PgCommitFaultProxy::start(&url, fault).await; + let mut options = sea_orm::ConnectOptions::new(proxy.url.clone()); + options.max_connections(1).min_connections(1); + let repository = + PostgresMetadataInstallRepository::new(Database::connect(options).await.unwrap()) + .await + .unwrap(); + let pages = prepared("/"); + let intent = repository + .begin_intent("cap-mint-fault", &pages) + .await + .unwrap(); + proxy.armed.store(true, Ordering::SeqCst); + let error = tokio::time::timeout( + Duration::from_secs(15), + repository.mint_legacy_install_capability(&intent), + ) + .await + .unwrap() + .unwrap_err(); + assert!(matches!( + error, + MetadataInstallError::CommitUncertain { + phase: MetadataCommitPhase::Intent, + .. + } + )); + proxy.wait_for_fault().await; + assert_eq!( + count(&direct, "SELECT count(*) FROM mst2_metadata_payload").await, + 0 + ); + assert_eq!( + count(&direct, "SELECT count(*) FROM mst2_metadata_install_seal").await, + if matches!(fault, Fault::AfterCommit) { + 1 + } else { + 0 + } + ); + assert_eq!( + proxy.commit_observed.load(Ordering::SeqCst), + matches!(fault, Fault::AfterCommit) + ); + let restarted = PostgresMetadataInstallRepository::new(recovery.clone()) + .await + .unwrap(); + let cap = restarted + .mint_legacy_install_capability(&intent) + .await + .unwrap(); + assert_eq!( + count(&direct, "SELECT count(*) FROM mst2_metadata_install_seal").await, + 1 + ); + restarted + .install_pages_validated(&cap, pages.dag().payloads()) + .await + .unwrap(); + restarted.finalize(&intent).await.unwrap(); +} + +async fn batch_fault(fault: Fault) { + let (direct, recovery, _schema, url) = fixture().await; + let proxy = PgCommitFaultProxy::start(&url, fault).await; + let mut options = sea_orm::ConnectOptions::new(proxy.url.clone()); + options.max_connections(1).min_connections(1); + let repository = + PostgresMetadataInstallRepository::new(Database::connect(options).await.unwrap()) + .await + .unwrap(); + let pages = prepared("/"); + let intent = repository + .begin_intent("cap-batch-fault", &pages) + .await + .unwrap(); + let cap = repository + .mint_legacy_install_capability(&intent) + .await + .unwrap(); + proxy.armed.store(true, Ordering::SeqCst); + let error = tokio::time::timeout( + Duration::from_secs(15), + repository.install_pages_validated(&cap, pages.dag().payloads()), + ) + .await + .unwrap() + .unwrap_err(); + assert!(matches!( + error, + MetadataInstallError::CommitUncertain { + phase: MetadataCommitPhase::Payload, + .. + } + )); + proxy.wait_for_fault().await; + assert_eq!( + count(&direct, "SELECT count(*) FROM mst2_metadata_install_seal").await, + 1 + ); + assert_eq!( + count(&direct, "SELECT count(*) FROM mst2_metadata_payload").await, + if matches!(fault, Fault::AfterCommit) { + pages.dag().payloads().len() as i64 + } else { + 0 + } + ); + let restarted = PostgresMetadataInstallRepository::new(recovery.clone()) + .await + .unwrap(); + let fresh = restarted + .mint_legacy_install_capability(&intent) + .await + .unwrap(); + restarted + .install_pages_validated(&fresh, pages.dag().payloads()) + .await + .unwrap(); + let receipt = restarted.finalize(&intent).await.unwrap(); + assert_eq!(receipt.metadata_root(), pages.dag().root()); + assert_eq!( + count( + &direct, + "SELECT count(*) FROM mst2_retention_root WHERE root_kind='prepare'" + ) + .await, + pages.dag().payloads().len() as i64 + ); + restarted + .install_pages_validated(&fresh, pages.dag().payloads()) + .await + .unwrap(); + assert_eq!(restarted.finalize(&intent).await.unwrap(), receipt); +} + +#[tokio::test] +async fn install_capability_real_registration_commit_response_loss_recovers_fresh_proof() { + mint_fault(Fault::AfterCommit).await; +} + +#[tokio::test] +async fn install_capability_real_registration_rollback_leaves_no_freeze_or_capability() { + mint_fault(Fault::BeforeCommit).await; +} + +#[tokio::test] +async fn install_capability_real_batch_commit_response_loss_preserves_exact_bytes() { + batch_fault(Fault::AfterCommit).await; +} + +#[tokio::test] +async fn install_capability_real_batch_connection_loss_rolls_back_without_false_success() { + batch_fault(Fault::BeforeCommit).await; +} + +#[tokio::test] +async fn install_capability_failed_registration_trigger_rolls_back_freeze_atomically() { + let (direct, recovery, _schema, _url) = fixture().await; + let repository = PostgresMetadataInstallRepository::new(direct.clone()) + .await + .unwrap(); + let pages = prepared("/"); + let intent = repository + .begin_intent("cap-register-rollback", &pages) + .await + .unwrap(); + direct.execute_unprepared("CREATE FUNCTION test_cap_registration_fault() RETURNS trigger LANGUAGE plpgsql AS $$ + BEGIN RAISE EXCEPTION 'registration fault after production guard'; END $$; + CREATE TRIGGER test_cap_registration_fault AFTER INSERT ON mst2_metadata_install_seal FOR EACH ROW + EXECUTE FUNCTION test_cap_registration_fault()").await.unwrap(); + assert!( + repository + .mint_legacy_install_capability(&intent) + .await + .is_err() + ); + assert_eq!( + count(&direct, "SELECT count(*) FROM mst2_metadata_install_seal").await, + 0 + ); + recovery + .execute_unprepared("UPDATE mst2_metadata_prepare_page SET expected_size=expected_size") + .await + .unwrap(); + direct + .execute_unprepared( + "DROP TRIGGER test_cap_registration_fault ON mst2_metadata_install_seal; + DROP FUNCTION test_cap_registration_fault()", + ) + .await + .unwrap(); + let restarted = PostgresMetadataInstallRepository::new(recovery) + .await + .unwrap(); + let cap = restarted + .mint_legacy_install_capability(&intent) + .await + .unwrap(); + restarted + .install_pages_validated(&cap, pages.dag().payloads()) + .await + .unwrap(); +} diff --git a/src/jupiter/storage/native_metadata_install_capability_tests.rs b/src/jupiter/storage/native_metadata_install_capability_tests.rs new file mode 100644 index 00000000..bb760c08 --- /dev/null +++ b/src/jupiter/storage/native_metadata_install_capability_tests.rs @@ -0,0 +1,966 @@ +use std::sync::Arc; + +use mst2_codec::metapage::{Entry, EntryKind}; +use sea_orm::{Database, PaginatorTrait}; +use sea_orm_migration::MigratorTrait; + +use super::*; +use crate::{ + ceres::snapshot::retention_dag::MetadataDagBuilder, + jupiter::{ + migration::Migrator, + tests::{TestSchemaGuard, test_db_config}, + }, +}; + +pub(super) fn prepared(children: usize) -> PreparedNativeMetadataRetention { + let mut builder = MetadataDagBuilder::new(MetadataDagLimits::default()); + let mut entries = Vec::new(); + for index in 0..children { + let file = [Entry::file( + EntryKind::Regular, + b"file", + 3, + [index as u8; 32], + )]; + let child = Page::build(&file).unwrap(); + builder.add_directory(&child, &file).unwrap(); + entries.push(Entry::dir( + format!("dir-{index:03}").as_bytes(), + page_id(&child), + )); + } + let root = Page::build(&entries).unwrap(); + builder.add_directory(&root, &entries).unwrap(); + PreparedNativeMetadataRetention::test_installation( + Arc::new(builder.finish(page_id(&root)).unwrap()), + "/", + ) +} + +pub(super) async fn fixture() -> (DatabaseConnection, DatabaseConnection, TestSchemaGuard) { + let temp = tempfile::tempdir().unwrap(); + let (config, schema) = test_db_config(temp.path()).await; + let first = Database::connect(config.db_url.clone()).await.unwrap(); + Migrator::up(&first, None).await.unwrap(); + let second = Database::connect(config.db_url).await.unwrap(); + (first, second, schema) +} + +async fn scalar(db: &C, sql: &str) -> i64 { + db.query_one_raw(statement(sql, [])) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap() +} + +async fn install( + repository: &PostgresMetadataInstallRepository, + capability: &ValidatedLegacyInstallCapability, + prepared: &PreparedNativeMetadataRetention, +) { + for batch in prepared.dag().payloads().chunks(64) { + repository + .install_pages_validated(capability, batch) + .await + .unwrap(); + } +} + +async fn assert_guard(db: &DatabaseConnection, sql: &str) { + let error = db.execute_unprepared(sql).await.expect_err(sql); + assert!( + error.to_string().contains("install capability") + || error.to_string().contains("registered metadata"), + "{error}" + ); +} + +#[tokio::test] +async fn install_capability_many_batches_concurrent_mint_and_fresh_repository_finalize() { + let (first, second, _schema) = fixture().await; + let a = PostgresMetadataInstallRepository::new(first.clone()) + .await + .unwrap(); + let b = PostgresMetadataInstallRepository::new(second.clone()) + .await + .unwrap(); + let pages = prepared(130); + assert_eq!(pages.dag().payloads().len(), 133); + assert_eq!(pages.dag().edges().len(), 132); + let root = pages + .dag() + .payloads() + .iter() + .find(|page| page.id == pages.dag().root()) + .unwrap(); + let ( + Page::Branch { + prefix, + terminal, + children, + }, + count, + ) = Page::decode(&root.bytes).unwrap() + else { + panic!("130 directory entries must use a canonical radix branch"); + }; + assert_eq!(prefix, b"dir-"); + assert!(terminal.is_none()); + assert_eq!(count, 130); + assert_eq!( + children + .iter() + .map(|child| (child.label, child.subtree_entries)) + .collect::>(), + [(b'0', 100), (b'1', 30)] + ); + for child in children { + let payload = pages + .dag() + .payloads() + .iter() + .find(|page| page.id == child.child_page_id) + .unwrap(); + let (Page::Leaf { entries }, count) = Page::decode(&payload.bytes).unwrap() else { + panic!("each root partition must be one canonical radix leaf"); + }; + assert_eq!(count, child.subtree_entries); + assert_eq!(entries.len() as u64, count); + for entry in entries { + assert_eq!(entry.kind, EntryKind::Directory); + assert_eq!(entry.name[4], child.label); + assert!( + pages + .dag() + .payloads() + .iter() + .any(|page| page.id == entry.child_root) + ); + } + } + assert_eq!( + pages + .dag() + .payloads() + .chunks(64) + .map(|batch| batch.len()) + .collect::>(), + [64, 64, 5] + ); + let intent = a.begin_intent("many", &pages).await.unwrap(); + let (left, right) = tokio::join!( + a.mint_legacy_install_capability(&intent), + b.mint_legacy_install_capability(&intent) + ); + let left = left.unwrap(); + let right = right.unwrap(); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_install_seal").await, + 1 + ); + for (index, batch) in pages.dag().payloads().chunks(64).enumerate() { + if index == 0 { + a.install_pages_validated(&left, batch).await.unwrap(); + } else { + b.install_pages_validated(&right, batch).await.unwrap(); + } + } + drop(left); + drop(right); + drop(a); + drop(b); + let restarted = PostgresMetadataInstallRepository::new(second.clone()) + .await + .unwrap(); + let fresh = restarted + .mint_legacy_install_capability(&intent) + .await + .unwrap(); + install(&restarted, &fresh, &pages).await; + let receipt = restarted.finalize(&intent).await.unwrap(); + assert_eq!(receipt.metadata_root(), pages.dag().root()); + assert_eq!(receipt.payload_bytes(), pages.dag().payload_bytes()); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_payload WHERE generation IS NULL" + ) + .await, + 133 + ); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_retention_node WHERE state='LIVE'" + ) + .await, + 133 + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_retention_edge").await, + 132 + ); + let before = scalar( + &first, + "SELECT sum(incoming_refs)::bigint FROM mst2_retention_node", + ) + .await; + install(&restarted, &fresh, &pages).await; + assert_eq!(restarted.finalize(&intent).await.unwrap(), receipt); + assert_eq!( + scalar( + &first, + "SELECT sum(incoming_refs)::bigint FROM mst2_retention_node" + ) + .await, + before + ); +} + +#[tokio::test] +async fn install_capability_freezes_every_identity_member_anchor_and_truncate_path() { + let (first, _second, _schema) = fixture().await; + let repository = PostgresMetadataInstallRepository::new(first.clone()) + .await + .unwrap(); + let pages = prepared(2); + let intent = repository.begin_intent("freeze", &pages).await.unwrap(); + let other = repository + .begin_intent("unregistered", &pages) + .await + .unwrap(); + let cap = repository + .mint_legacy_install_capability(&intent) + .await + .unwrap(); + let before = mst2_metadata_prepare::Entity::find_by_id(intent.prepare_id().to_owned()) + .one(&first) + .await + .unwrap() + .unwrap(); + let member_count_before = + scalar(&first, "SELECT count(*) FROM mst2_metadata_prepare_page").await; + let seal_count_before = scalar(&first, "SELECT count(*) FROM mst2_metadata_install_seal").await; + let restrict_error = first + .execute_unprepared("TRUNCATE mst2_metadata_prepare_page") + .await + .expect_err("plain truncate must preserve foreign-key protection"); + assert!( + restrict_error + .to_string() + .contains("cannot truncate a table referenced in a foreign key constraint"), + "{restrict_error}" + ); + assert_eq!( + first.execute_raw(statement( + "UPDATE mst2_metadata_prepare SET verification_revision=verification_revision WHERE prepare_id=$1", + [intent.prepare_id().into()], + )).await.unwrap().rows_affected(), + 1 + ); + for mutation in [ + "operation_id='changed'", + "manifest_digest=decode(repeat('01',32),'hex')", + "canonical_plan=canonical_plan||decode('ff','hex')", + "source_domain='other'", + "tagged_root_tree_oid='sha1:bad'", + "scope='/changed'", + "schema_version=3", + "metadata_codec=2", + "materialization_policy=2", + "fs_semantics=2", + "access_projection=2", + "verification_revision=verification_revision+1", + "projection_revision=2", + "metadata_root=decode(repeat('01',32),'hex')", + "node_count=node_count+1", + "edge_count=edge_count+1", + "total_bytes=total_bytes+1", + "created_at=created_at+interval '1 second'", + ] { + assert_guard( + &first, + &format!( + "UPDATE mst2_metadata_prepare SET {mutation} WHERE prepare_id='{}'", + intent.prepare_id() + ), + ) + .await; + } + for sql in [ + format!( + "DELETE FROM mst2_metadata_prepare WHERE prepare_id='{}'", + intent.prepare_id() + ), + format!( + "UPDATE mst2_metadata_prepare_page SET expected_size=expected_size+1 WHERE prepare_id='{}'", + intent.prepare_id() + ), + format!( + "UPDATE mst2_metadata_prepare_page SET generation=1 WHERE prepare_id='{}'", + intent.prepare_id() + ), + format!( + "UPDATE mst2_metadata_prepare_page SET prepare_id='{}' WHERE prepare_id='{}'", + other.prepare_id(), + intent.prepare_id() + ), + format!( + "UPDATE mst2_metadata_prepare_page SET prepare_id='{}' WHERE prepare_id='{}'", + intent.prepare_id(), + other.prepare_id() + ), + format!( + "DELETE FROM mst2_metadata_prepare_page WHERE prepare_id='{}'", + intent.prepare_id() + ), + format!( + "INSERT INTO mst2_metadata_prepare_page(prepare_id,page_id,expected_size) VALUES('{}',decode(repeat('ff',32),'hex'),64)", + intent.prepare_id() + ), + "UPDATE mst2_metadata_install_seal SET members_digest=decode(repeat('ff',32),'hex')".into(), + "DELETE FROM mst2_metadata_install_seal".into(), + "TRUNCATE mst2_metadata_install_seal".into(), + "TRUNCATE mst2_metadata_prepare_page CASCADE".into(), + "TRUNCATE mst2_metadata_prepare CASCADE".into(), + ] { + assert_guard(&first, &sql).await; + } + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_prepare_page").await, + member_count_before + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_install_seal").await, + seal_count_before + ); + assert_eq!( + mst2_metadata_prepare::Entity::find_by_id(intent.prepare_id().to_owned()) + .one(&first) + .await + .unwrap() + .unwrap(), + before + ); + assert_eq!( + mst2_metadata_prepare_page::Entity::find() + .filter(mst2_metadata_prepare_page::Column::PrepareId.eq(intent.prepare_id())) + .count(&first) + .await + .unwrap(), + 3 + ); + install(&repository, &cap, &pages).await; + repository.finalize(&intent).await.unwrap(); +} + +#[tokio::test] +async fn install_capability_unregistered_corruption_and_seal_conflict_cannot_mint() { + let (first, _second, _schema) = fixture().await; + let repository = PostgresMetadataInstallRepository::new(first.clone()) + .await + .unwrap(); + let pages = prepared(2); + let intent = repository.begin_intent("corrupt", &pages).await.unwrap(); + first + .execute_unprepared("UPDATE mst2_metadata_prepare SET projection_revision=2") + .await + .unwrap(); + assert!( + repository + .mint_legacy_install_capability(&intent) + .await + .is_err() + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_install_seal").await, + 0 + ); + first.execute_unprepared("UPDATE mst2_metadata_prepare SET projection_revision=1,canonical_plan=canonical_plan||decode('ff','hex')").await.unwrap(); + assert!( + repository + .mint_legacy_install_capability(&intent) + .await + .is_err() + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_install_seal").await, + 0 + ); + let mapping = repository + .begin_intent("mapping-corrupt", &pages) + .await + .unwrap(); + first.execute_raw(statement( + "UPDATE mst2_metadata_prepare_page SET expected_size=expected_size+1 WHERE prepare_id=$1", + [mapping.prepare_id().into()], + )).await.unwrap(); + assert!( + repository + .mint_legacy_install_capability(&mapping) + .await + .is_err() + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_install_seal").await, + 0 + ); + first.execute_raw(statement( + "UPDATE mst2_metadata_prepare_page SET expected_size=expected_size-1 WHERE prepare_id=$1", + [mapping.prepare_id().into()], + )).await.unwrap(); + let intent = repository.begin_intent("conflict", &pages).await.unwrap(); + let sql=format!("INSERT INTO mst2_metadata_install_seal(prepare_id,operation_id,manifest_digest,members_digest,primary_scope,install_seal) + SELECT prepare_id,operation_id,manifest_digest,decode(repeat('00',32),'hex'),convert_to('[]','UTF8'),decode(repeat('00',32),'hex') + FROM mst2_metadata_prepare WHERE prepare_id='{}'",intent.prepare_id()); + assert_guard(&first, &sql).await; + let cap = repository + .mint_legacy_install_capability(&intent) + .await + .unwrap(); + install(&repository, &cap, &pages).await; +} + +#[tokio::test] +async fn install_capability_bounded_invalid_batches_are_atomic() { + let (first, _second, _schema) = fixture().await; + let repository = PostgresMetadataInstallRepository::new(first.clone()) + .await + .unwrap(); + let pages = prepared(70); + let intent = repository.begin_intent("invalid", &pages).await.unwrap(); + let cap = repository + .mint_legacy_install_capability(&intent) + .await + .unwrap(); + let payload = &pages.dag().payloads()[0]; + assert!(repository.install_pages_validated(&cap, &[]).await.is_err()); + assert!( + repository + .install_pages_validated(&cap, &pages.dag().payloads()[..65]) + .await + .is_err() + ); + assert!( + repository + .install_pages_validated(&cap, &[payload.clone(), payload.clone()]) + .await + .is_err() + ); + let mut wrong = payload.clone(); + wrong.size += 1; + assert!( + repository + .install_pages_validated(&cap, &[payload.clone(), wrong]) + .await + .is_err() + ); + let mut wrong = payload.clone(); + wrong.id = [255; 32]; + assert!( + repository + .install_pages_validated(&cap, &[wrong]) + .await + .is_err() + ); + let mut wrong = payload.clone(); + wrong.bytes = vec![0; PAGE_MAX_BYTES + 1]; + wrong.size = wrong.bytes.len() as u64; + wrong.id = page_id(&wrong.bytes); + assert!( + repository + .install_pages_validated(&cap, &[wrong]) + .await + .is_err() + ); + let mut wrong = payload.clone(); + wrong.bytes = vec![0; HEADER_LEN]; + wrong.size = wrong.bytes.len() as u64; + wrong.id = page_id(&wrong.bytes); + assert!( + repository + .install_pages_validated(&cap, &[wrong]) + .await + .is_err() + ); + let stranger = prepared(71); + let stranger = stranger + .dag() + .payloads() + .iter() + .find(|p| { + !pages + .dag() + .payloads() + .iter() + .any(|member| member.id == p.id) + }) + .unwrap(); + assert!( + repository + .install_pages_validated(&cap, std::slice::from_ref(stranger)) + .await + .is_err() + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_payload").await, + 0 + ); + let mut wrong = payload.bytes.clone(); + let last = wrong.len() - 1; + wrong[last] ^= 1; + first.execute_raw(statement("INSERT INTO mst2_metadata_payload(page_id,metadata_codec,byte_size,payload) VALUES($1,1,$2,$3)", + [payload.id.to_vec().into(),(payload.size as i32).into(),wrong.into()])).await.unwrap(); + assert!( + repository + .install_pages_validated(&cap, &pages.dag().payloads()[..64]) + .await + .is_err() + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_payload").await, + 1 + ); +} + +#[tokio::test] +async fn install_capability_raw_committed_missing_bytes_does_not_repair_and_retirement_rejects() { + let (first, second, _schema) = fixture().await; + let repository = PostgresMetadataInstallRepository::new(first.clone()) + .await + .unwrap(); + let pages = prepared(2); + let intent = repository.begin_intent("promoted", &pages).await.unwrap(); + let cap = repository + .mint_legacy_install_capability(&intent) + .await + .unwrap(); + second.execute_raw(statement("UPDATE mst2_metadata_prepare SET state='COMMITTED',committed_at=clock_timestamp() WHERE prepare_id=$1", + [intent.prepare_id().into()])).await.unwrap(); + assert!( + repository + .install_pages_validated(&cap, pages.dag().payloads()) + .await + .is_err() + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_payload").await, + 0 + ); + assert!( + repository + .mint_legacy_install_capability(&intent) + .await + .is_err() + ); + let intent = repository.begin_intent("retired", &pages).await.unwrap(); + let cap = repository + .mint_legacy_install_capability(&intent) + .await + .unwrap(); + install(&repository, &cap, &pages).await; + repository.finalize(&intent).await.unwrap(); + second.execute_raw(statement("UPDATE mst2_metadata_prepare SET coverage_retired_at=clock_timestamp() WHERE prepare_id=$1", + [intent.prepare_id().into()])).await.unwrap(); + assert!( + repository + .install_pages_validated(&cap, pages.dag().payloads()) + .await + .is_err() + ); + assert!( + repository + .mint_legacy_install_capability(&intent) + .await + .is_err() + ); +} + +#[tokio::test] +async fn install_capability_cross_primary_schema_repeatable_read_and_temp_spoof_refuse() { + let (first, second, _schema) = fixture().await; + let (alien, _unused, _alien_schema) = fixture().await; + let a = PostgresMetadataInstallRepository::new(first.clone()) + .await + .unwrap(); + let b = PostgresMetadataInstallRepository::new(alien.clone()) + .await + .unwrap(); + let pages = prepared(2); + let intent = a.begin_intent("scope", &pages).await.unwrap(); + let cap = a.mint_legacy_install_capability(&intent).await.unwrap(); + let alien_intent = b.begin_intent("scope", &pages).await.unwrap(); + assert_ne!(alien_intent.prepare_id(), intent.prepare_id()); + let copied_prepare = first.query_one_raw(statement( + "SELECT row_to_json(p)::text AS copied FROM mst2_metadata_prepare p WHERE prepare_id=$1", + [intent.prepare_id().into()], + )).await.unwrap().unwrap().try_get::("", "copied").unwrap(); + let copied_members = first.query_one_raw(statement( + "SELECT json_agg(m)::text AS copied FROM mst2_metadata_prepare_page m WHERE prepare_id=$1", + [intent.prepare_id().into()], + )).await.unwrap().unwrap().try_get::("", "copied").unwrap(); + alien + .execute_raw(statement( + "DELETE FROM mst2_metadata_prepare_page WHERE prepare_id=$1", + [alien_intent.prepare_id().into()], + )) + .await + .unwrap(); + alien + .execute_raw(statement( + "DELETE FROM mst2_metadata_prepare WHERE prepare_id=$1", + [alien_intent.prepare_id().into()], + )) + .await + .unwrap(); + alien.execute_raw(statement("INSERT INTO mst2_metadata_prepare SELECT * FROM json_populate_record(NULL::mst2_metadata_prepare,$1::json)", + [copied_prepare.into()])).await.unwrap(); + alien.execute_raw(statement("INSERT INTO mst2_metadata_prepare_page SELECT * FROM json_populate_recordset(NULL::mst2_metadata_prepare_page,$1::json)", + [copied_members.into()])).await.unwrap(); + let own_scope_cap = b.mint_legacy_install_capability(&intent).await.unwrap(); + assert!( + b.install_pages_validated(&cap, pages.dag().payloads()) + .await + .is_err() + ); + assert_eq!( + scalar(&alien, "SELECT count(*) FROM mst2_metadata_payload").await, + 0 + ); + install(&b, &own_scope_cap, &pages).await; + let txn = second + .begin_with_config(Some(IsolationLevel::RepeatableRead), None) + .await + .unwrap(); + assert!(a.capability_barrier(&txn).await.is_err()); + assert!( + txn.execute_unprepared("UPDATE mst2_metadata_prepare SET scope='/rr'") + .await + .is_err() + ); + txn.rollback().await.unwrap(); + let txn = second.begin().await.unwrap(); + let schema = a.storage_scope.schema.clone(); + txn.execute_unprepared("SET LOCAL search_path=pg_catalog") + .await + .unwrap(); + assert!(a.capability_barrier(&txn).await.is_err()); + let sql = format!( + "UPDATE \"{}\".mst2_metadata_prepare SET scope='/wrongpath'", + schema.replace('"', "\"\"") + ); + assert!(txn.execute_unprepared(&sql).await.is_err()); + txn.rollback().await.unwrap(); + let txn = second.begin().await.unwrap(); + txn.execute_unprepared( + "CREATE TEMP TABLE mst2_metadata_install_seal(prepare_id text); + CREATE TEMP TABLE mst2_metadata_storage_scope(singleton integer,storage_uuid text)", + ) + .await + .unwrap(); + let raw_sql = format!( + "UPDATE \"{}\".mst2_metadata_prepare SET scope='/raw-temp-spoof'", + schema.replace('"', "\"\"") + ); + assert!( + txn.execute_unprepared(&raw_sql) + .await + .unwrap_err() + .to_string() + .contains("registered metadata") + ); + txn.rollback().await.unwrap(); + let txn = second.begin().await.unwrap(); + txn.execute_unprepared( + "CREATE TEMP TABLE mst2_metadata_install_seal(prepare_id text); + CREATE TEMP TABLE mst2_metadata_storage_scope(singleton integer,storage_uuid text)", + ) + .await + .unwrap(); + a.capability_barrier(&txn).await.unwrap(); + assert_eq!( + scalar(&txn, "SELECT count(*) FROM mst2_metadata_install_seal").await, + 1 + ); + assert!( + txn.execute_unprepared("UPDATE mst2_metadata_prepare SET scope='/spoof'") + .await + .is_err() + ); + txn.rollback().await.unwrap(); + install(&a, &cap, &pages).await; +} + +#[tokio::test] +async fn install_capability_reuses_bound_generic_bytes_without_adopting_qualified_generation() { + let (first, _second, _schema) = fixture().await; + let generic = generations::PostgresMetadataGenerationRepository::new(first.clone()) + .await + .unwrap(); + let legacy = PostgresMetadataInstallRepository::new(first.clone()) + .await + .unwrap(); + let pages = prepared(2); + let bound = generic.begin_intent("bound-generic", &pages).await.unwrap(); + generic + .install_pages(&bound, pages.dag().payloads()) + .await + .unwrap(); + generic.finalize(&bound).await.unwrap(); + let intent = legacy.begin_intent("legacy-share", &pages).await.unwrap(); + let cap = legacy + .mint_legacy_install_capability(&intent) + .await + .unwrap(); + install(&legacy, &cap, &pages).await; + legacy.finalize(&intent).await.unwrap(); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_payload WHERE generation IS NOT NULL" + ) + .await, + 3 + ); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_prepare_page WHERE generation IS NULL" + ) + .await, + 3 + ); + let bound_intent = MetadataPrepareIntent { + prepare_id: bound.prepare_id().into(), + operation_id: bound.operation_id().into(), + manifest_digest: bound.manifest_digest(), + }; + assert!( + legacy + .mint_legacy_install_capability(&bound_intent) + .await + .is_err() + ); + let (qualified_db, _other, _q_schema) = fixture().await; + let qualified = + generations::qualified::PostgresQualifiedMetadataRepository::new(qualified_db.clone()) + .await + .unwrap(); + let q_legacy = PostgresMetadataInstallRepository::new(qualified_db.clone()) + .await + .unwrap(); + let bound = qualified.begin_intent("qualified", &pages).await.unwrap(); + qualified + .install_pages(&bound, pages.dag().payloads()) + .await + .unwrap(); + let q_state = domain_state_for_test(&qualified_db).await; + let error = q_legacy + .begin_intent("generic-refusal", &pages) + .await + .unwrap_err(); + assert!( + error + .to_string() + .contains("qualified mapping cannot cross domain/current/state fence"), + "{error}" + ); + assert_eq!(domain_state_for_test(&qualified_db).await, q_state); + assert_eq!( + scalar(&qualified_db, "SELECT count(*) FROM mst2_metadata_payload").await, + 3 + ); + assert_eq!( + scalar(&qualified_db, "SELECT count(*) FROM mst2_retention_node").await, + 0 + ); + + let (legacy_db, _other, _legacy_schema) = fixture().await; + let legacy = PostgresMetadataInstallRepository::new(legacy_db.clone()) + .await + .unwrap(); + let qualified = + generations::qualified::PostgresQualifiedMetadataRepository::new(legacy_db.clone()) + .await + .unwrap(); + let intent = legacy.begin_intent("generic-first", &pages).await.unwrap(); + let cap = legacy + .mint_legacy_install_capability(&intent) + .await + .unwrap(); + let generic_state = domain_state_for_test(&legacy_db).await; + let error = qualified + .begin_intent("qualified-refusal", &pages) + .await + .unwrap_err(); + assert!( + error + .to_string() + .contains("metadata lifetime has durable or ambiguous coverage"), + "{error}" + ); + assert_eq!(domain_state_for_test(&legacy_db).await, generic_state); + install(&legacy, &cap, &pages).await; + assert_eq!( + legacy.finalize(&intent).await.unwrap().metadata_root(), + pages.dag().root() + ); + assert_eq!( + scalar( + &legacy_db, + "SELECT count(*) FROM mst2_metadata_payload WHERE generation IS NULL" + ) + .await, + 3 + ); + assert_eq!( + scalar( + &legacy_db, + "SELECT count(*) FROM mst2_retention_node WHERE state='LIVE'" + ) + .await, + 3 + ); + assert_eq!( + scalar(&legacy_db, "SELECT count(*) FROM mst2_metadata_lifetime").await, + 0 + ); + assert_eq!( + scalar(&legacy_db, "SELECT count(*) FROM mst2_metadata_current").await, + 0 + ); + assert_eq!( + scalar(&legacy_db, "SELECT count(*) FROM mst2_metadata_graph_node").await, + 0 + ); +} + +async fn domain_state_for_test(db: &DatabaseConnection) -> Vec<(String, Vec)> { + let mut state = Vec::new(); + for table in [ + "mst2_metadata_prepare", + "mst2_metadata_prepare_page", + "mst2_metadata_install_seal", + "mst2_metadata_payload", + "mst2_metadata_current", + "mst2_metadata_lifetime", + "mst2_metadata_graph_node", + "mst2_metadata_graph_edge", + "mst2_metadata_graph_root", + "mst2_metadata_gc_op", + "mst2_retention_node", + "mst2_retention_edge", + "mst2_retention_root", + "mst2_retention_gc_op", + "mst2_snapshot_context", + "mst2_snapshot_lease", + ] { + let rows = db + .query_all_raw(statement( + &format!("SELECT to_jsonb(t)::text FROM {table} t ORDER BY 1"), + [], + )) + .await + .unwrap() + .into_iter() + .map(|row| row.try_get_by_index(0).unwrap()) + .collect(); + state.push((table.into(), rows)); + } + state +} + +#[tokio::test] +async fn install_capability_batch_fault_rolls_back_and_raw_writer_waits_before_row_lock() { + let (first, second, _schema) = fixture().await; + let repository = PostgresMetadataInstallRepository::new(first.clone()) + .await + .unwrap(); + let pages = prepared(2); + let intent = repository + .begin_intent("batch-fault", &pages) + .await + .unwrap(); + let cap = repository + .mint_legacy_install_capability(&intent) + .await + .unwrap(); + let failed = hex::encode(pages.dag().payloads()[1].id); + first.execute_unprepared(&format!("CREATE FUNCTION test_install_batch_fault() RETURNS trigger LANGUAGE plpgsql AS $$ + BEGIN IF NEW.page_id=decode('{failed}','hex') THEN RAISE EXCEPTION 'injected batch fault'; END IF; RETURN NEW; END $$; + CREATE TRIGGER test_install_batch_fault AFTER INSERT ON mst2_metadata_payload FOR EACH ROW EXECUTE FUNCTION test_install_batch_fault()")).await.unwrap(); + assert!( + repository + .install_pages_validated(&cap, pages.dag().payloads()) + .await + .is_err() + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_payload").await, + 0 + ); + first.execute_unprepared("DROP TRIGGER test_install_batch_fault ON mst2_metadata_payload;DROP FUNCTION test_install_batch_fault()").await.unwrap(); + let held = first + .begin_with_config(Some(IsolationLevel::ReadCommitted), None) + .await + .unwrap(); + repository.capability_barrier(&held).await.unwrap(); + let worker = second + .begin_with_config(Some(IsolationLevel::ReadCommitted), None) + .await + .unwrap(); + let pid = scalar(&worker, "SELECT pg_backend_pid()::bigint").await; + let id = intent.prepare_id().to_owned(); + let blocked = tokio::spawn(async move { + let result = worker + .execute_raw(statement( + "UPDATE mst2_metadata_prepare SET scope='/blocked' WHERE prepare_id=$1", + [id.into()], + )) + .await; + worker.rollback().await.unwrap(); + result + }); + let deadline = tokio::time::Instant::now() + Duration::from_secs(4); + loop { + let waiting=held.query_one_raw(statement("SELECT EXISTS(SELECT 1 FROM pg_locks WHERE locktype='advisory' AND NOT granted + AND classid=$1::integer::oid AND objid=hashtext(current_schema())::oid + AND database=(SELECT oid FROM pg_database WHERE datname=current_database())) AS waiting",[RETENTION_LOCK_KEY.into()])).await.unwrap().unwrap().try_get::("","waiting").unwrap(); + if waiting { + break; + } + assert!( + tokio::time::Instant::now() < deadline, + "real writer must wait before frozen-row proof" + ); + tokio::task::yield_now().await; + } + let row_locked=held.query_one_raw(statement("SELECT EXISTS(SELECT 1 FROM pg_locks WHERE pid=$1 AND locktype='tuple' AND relation='mst2_metadata_prepare'::regclass) AS row_locked",[(pid as i32).into()])).await.unwrap().unwrap().try_get::("","row_locked").unwrap(); + assert!( + !row_locked, + "statement retention barrier precedes prepare row locking" + ); + held.query_one_raw(statement( + "SELECT prepare_id FROM mst2_metadata_prepare WHERE prepare_id=$1 FOR UPDATE", + [intent.prepare_id().into()], + )) + .await + .unwrap() + .unwrap(); + held.commit().await.unwrap(); + assert!( + blocked + .await + .unwrap() + .unwrap_err() + .to_string() + .contains("registered metadata") + ); + install(&repository, &cap, &pages).await; +} + +#[path = "native_metadata_install_missing_tests.rs"] +mod missing; diff --git a/src/jupiter/storage/native_metadata_install_missing_tests.rs b/src/jupiter/storage/native_metadata_install_missing_tests.rs new file mode 100644 index 00000000..eb7b7885 --- /dev/null +++ b/src/jupiter/storage/native_metadata_install_missing_tests.rs @@ -0,0 +1,902 @@ +use mst2_codec::descriptor::ServingDescriptor; + +use super::*; +use crate::{ + ceres::snapshot::descriptor::BuiltDescriptor, + jupiter::storage::{ + mst2_retention::GcClaim, native_snapshot_session::PostgresNativeSessionRepository, + }, +}; + +fn descriptor(pages: &PreparedNativeMetadataRetention, view: u8) -> BuiltDescriptor { + let descriptor = ServingDescriptor { + instance_uuid: [7; 16], + namespace_view_id: [view; 32], + scope: pages.scope().into(), + metadata_root: pages.dag().root(), + }; + BuiltDescriptor { + instance_id: uuid::Uuid::from_bytes(descriptor.instance_uuid).to_string(), + snapshot_id: format!("sha256:{}", hex::encode(descriptor.snapshot_id().unwrap())), + metadata_root: format!("sha256:{}", hex::encode(descriptor.metadata_root)), + descriptor, + } +} + +fn assert_preparing_work( + work: &LegacyPayloadInstallWork, + pages: &[MetadataPagePayload], + existing: &BTreeSet<[u8; 32]>, +) { + let omitted: Vec<_> = pages + .iter() + .filter(|page| existing.contains(&page.id)) + .collect(); + let requested_bytes = pages.iter().map(|page| page.size).sum::(); + let omitted_bytes = omitted.iter().map(|page| page.size).sum::(); + assert_eq!(work.requested_pages, pages.len() as u64); + assert_eq!(work.payload_pages_omitted, omitted.len() as u64); + assert_eq!( + work.payload_pages_encoded, + (pages.len() - omitted.len()) as u64 + ); + assert_eq!(work.requested_payload_bytes_validated, requested_bytes); + assert_eq!(work.payload_bytes_omitted, omitted_bytes); + assert_eq!(work.payload_bytes_encoded, requested_bytes - omitted_bytes); + assert_eq!(work.committed_replay_pages, 0); + assert_eq!(work.transactions, pages.chunks(64).len() as u64); + assert_eq!(work.registration_queries, work.transactions); + assert_eq!(work.requested_member_queries, work.transactions); + assert_eq!(work.classification_batches, work.transactions); + assert!(work.metadata_parameter_bytes > 0); + let missing_batches = pages + .chunks(64) + .filter(|batch| batch.iter().any(|page| !existing.contains(&page.id))) + .count() as u64; + assert_eq!(work.insert_statements, missing_batches); + assert_eq!(work.byte_comparison_queries, missing_batches); + if missing_batches == 0 { + assert_eq!(work.payload_parameter_bytes, 0); + } else { + assert!(work.payload_parameter_bytes >= 4 * work.payload_bytes_encoded); + } +} + +async fn install_missing( + repository: &PostgresMetadataInstallRepository, + cap: &ValidatedLegacyInstallCapability, + pages: &PreparedNativeMetadataRetention, +) -> LegacyPayloadInstallWork { + let mut work = LegacyPayloadInstallWork::default(); + for batch in pages.dag().payloads().chunks(64) { + work.record( + repository + .install_missing_pages_validated(cap, batch) + .await + .unwrap(), + ); + } + work +} + +async fn payload_modes(db: &DatabaseConnection) -> Vec<(String, String)> { + db.query_all_raw(statement( + "SELECT tgname,tgenabled::text AS mode FROM pg_catalog.pg_trigger + WHERE tgrelid='mst2_metadata_payload'::regclass AND NOT tgisinternal ORDER BY tgname", + [], + )) + .await + .unwrap() + .into_iter() + .map(|row| { + ( + row.try_get("", "tgname").unwrap(), + row.try_get("", "mode").unwrap(), + ) + }) + .collect() +} + +enum PayloadDamage { + Bytes(Vec), + Remove, + UnbindGeneration, +} + +async fn damage_payload( + db: &DatabaseConnection, + page: &MetadataPagePayload, + damage: PayloadDamage, +) { + let modes = payload_modes(db).await; + assert!(modes.iter().all(|(_, mode)| mode == "O")); + assert!( + modes + .iter() + .any(|(name, _)| name == "mst2_metadata_payload_fenced") + ); + assert!( + modes + .iter() + .any(|(name, _)| name == "mst2_metadata_payload_removed") + ); + assert!( + db.execute_raw(statement( + "UPDATE mst2_metadata_payload SET payload=payload WHERE page_id=$1", + [page.id.to_vec().into()], + )) + .await + .is_err() + ); + let txn = db.begin().await.unwrap(); + txn.execute_raw(statement( + "SELECT pg_advisory_xact_lock($1,hashtext(current_schema()))", + [RETENTION_LOCK_KEY.into()], + )) + .await + .unwrap(); + txn.execute_unprepared( + "ALTER TABLE mst2_metadata_payload DISABLE TRIGGER mst2_metadata_payload_fenced", + ) + .await + .unwrap(); + let removed = matches!(&damage, PayloadDamage::Remove); + if removed { + txn.execute_unprepared( + "ALTER TABLE mst2_metadata_payload DISABLE TRIGGER mst2_metadata_payload_removed", + ) + .await + .unwrap(); + } + let result = match damage { + PayloadDamage::Bytes(bytes) => { + assert_eq!(bytes.len(), page.bytes.len()); + txn.execute_raw(statement( + "UPDATE mst2_metadata_payload SET payload=$2 WHERE page_id=$1", + [page.id.to_vec().into(), bytes.into()], + )) + .await + .unwrap() + } + PayloadDamage::Remove => txn + .execute_raw(statement( + "DELETE FROM mst2_metadata_payload WHERE page_id=$1", + [page.id.to_vec().into()], + )) + .await + .unwrap(), + PayloadDamage::UnbindGeneration => txn + .execute_raw(statement( + "UPDATE mst2_metadata_payload SET generation=NULL WHERE page_id=$1", + [page.id.to_vec().into()], + )) + .await + .unwrap(), + }; + assert_eq!(result.rows_affected(), 1); + if removed { + txn.execute_unprepared( + "ALTER TABLE mst2_metadata_payload ENABLE TRIGGER mst2_metadata_payload_removed", + ) + .await + .unwrap(); + } + txn.execute_unprepared( + "ALTER TABLE mst2_metadata_payload ENABLE TRIGGER mst2_metadata_payload_fenced", + ) + .await + .unwrap(); + txn.commit().await.unwrap(); + assert_eq!(payload_modes(db).await, modes); + assert!( + db.execute_unprepared("UPDATE mst2_metadata_payload SET payload=payload",) + .await + .is_err(), + "fault injection must restore the production UPDATE fence" + ); +} + +fn corrupt(page: &MetadataPagePayload) -> Vec { + let mut bytes = page.bytes.clone(); + let last = bytes.len() - 1; + bytes[last] ^= 1; + bytes +} + +#[tokio::test] +async fn install_missing_actual_session_new_sid_mixed_reuse_and_restart_keep_complete_oracles() { + let (first, second, _schema) = fixture().await; + let sessions = PostgresNativeSessionRepository::new(first.clone()); + let old = prepared(130); + let (old_receipt, cold) = sessions + .install_with_work(&descriptor(&old, 1), &old) + .await + .unwrap(); + assert_preparing_work(&cold, old.dag().payloads(), &BTreeSet::new()); + assert_eq!(old_receipt.payload_bytes(), old.dag().payload_bytes()); + let existing: BTreeSet<_> = old.dag().payloads().iter().map(|page| page.id).collect(); + let new = prepared(131); + assert_ne!(old.dag().root(), new.dag().root()); + assert!( + new.dag() + .payloads() + .iter() + .any(|page| existing.contains(&page.id)) + ); + assert!( + new.dag() + .payloads() + .iter() + .any(|page| !existing.contains(&page.id)) + ); + first.execute_unprepared( + "CREATE FUNCTION test_missing_mixed_existing_trap() RETURNS trigger LANGUAGE plpgsql AS $$ + BEGIN IF EXISTS(SELECT 1 FROM mst2_metadata_payload WHERE page_id=NEW.page_id) THEN + RAISE EXCEPTION 'mixed reuse attempted existing payload INSERT'; END IF; RETURN NEW; END $$; + CREATE TRIGGER test_missing_mixed_existing_trap BEFORE INSERT ON mst2_metadata_payload + FOR EACH ROW EXECUTE FUNCTION test_missing_mixed_existing_trap()", + ).await.unwrap(); + let old_state = domain_state_for_test(&first).await; + assert!(first.execute_unprepared( + "INSERT INTO mst2_metadata_payload(page_id,metadata_codec,byte_size,payload) + SELECT page_id,metadata_codec,byte_size,payload FROM mst2_metadata_payload ORDER BY page_id LIMIT 1 + ON CONFLICT(page_id) DO NOTHING", + ).await.unwrap_err().to_string().contains("mixed reuse attempted existing payload INSERT")); + assert_eq!(domain_state_for_test(&first).await, old_state); + let (new_receipt, mixed) = sessions + .install_with_work(&descriptor(&new, 2), &new) + .await + .unwrap(); + assert_preparing_work(&mixed, new.dag().payloads(), &existing); + assert_ne!( + old_receipt.intent().prepare_id(), + new_receipt.intent().prepare_id() + ); + let union: BTreeSet<_> = existing + .iter() + .copied() + .chain(new.dag().payloads().iter().map(|page| page.id)) + .collect(); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_payload").await, + union.len() as i64 + ); + first + .execute_unprepared( + "DROP TRIGGER test_missing_mixed_existing_trap ON mst2_metadata_payload; + DROP FUNCTION test_missing_mixed_existing_trap()", + ) + .await + .unwrap(); + first + .execute_unprepared( + "CREATE FUNCTION test_missing_no_insert() RETURNS trigger LANGUAGE plpgsql AS $$ + BEGIN RAISE EXCEPTION 'full reuse attempted payload INSERT'; END $$; + CREATE TRIGGER test_missing_no_insert BEFORE INSERT ON mst2_metadata_payload + FOR EACH STATEMENT EXECUTE FUNCTION test_missing_no_insert()", + ) + .await + .unwrap(); + drop(sessions); + let restarted = PostgresNativeSessionRepository::new(second.clone()); + let (reused_receipt, reused) = restarted + .install_with_work(&descriptor(&new, 3), &new) + .await + .unwrap(); + assert_preparing_work(&reused, new.dag().payloads(), &union); + assert_ne!( + new_receipt.intent().prepare_id(), + reused_receipt.intent().prepare_id() + ); + assert_eq!(reused_receipt.metadata_root(), new.dag().root()); + let installer = PostgresMetadataInstallRepository::new(second) + .await + .unwrap(); + installer + .restore_session_dag(old_receipt.intent().prepare_id()) + .await + .unwrap(); + installer + .restore_session_dag(new_receipt.intent().prepare_id()) + .await + .unwrap(); + installer + .restore_session_dag(reused_receipt.intent().prepare_id()) + .await + .unwrap(); + let cap = installer + .mint_legacy_install_capability(reused_receipt.intent()) + .await + .unwrap(); + installer + .install_missing_pages_validated(&cap, &new.dag().payloads()[..64]) + .await + .unwrap(); + let wrapper_receipt = restarted.install(&descriptor(&new, 4), &new).await.unwrap(); + assert_ne!( + wrapper_receipt.intent().prepare_id(), + reused_receipt.intent().prepare_id() + ); + assert_eq!(wrapper_receipt.metadata_root(), new.dag().root()); + installer + .restore_session_dag(wrapper_receipt.intent().prepare_id()) + .await + .unwrap(); + first.execute_unprepared( + "DROP TRIGGER test_missing_no_insert ON mst2_metadata_payload; DROP FUNCTION test_missing_no_insert()", + ).await.unwrap(); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_prepare WHERE state='COMMITTED'" + ) + .await, + 4 + ); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_retention_node WHERE state='LIVE'" + ) + .await, + union.len() as i64 + ); +} + +#[tokio::test] +async fn install_missing_sql_statement_trap_proves_full_reuse_skips_conflict_insert() { + let (first, _second, _schema) = fixture().await; + let repository = PostgresMetadataInstallRepository::new(first.clone()) + .await + .unwrap(); + let pages = prepared(2); + let intent = repository + .begin_intent("reuse-statement", &pages) + .await + .unwrap(); + let cap = repository + .mint_legacy_install_capability(&intent) + .await + .unwrap(); + install(&repository, &cap, &pages).await; + first + .execute_unprepared( + "CREATE FUNCTION test_missing_insert_trap() RETURNS trigger LANGUAGE plpgsql AS $$ + BEGIN RAISE EXCEPTION 'payload statement trap fired'; END $$; + CREATE TRIGGER test_missing_insert_trap BEFORE INSERT ON mst2_metadata_payload + FOR EACH STATEMENT EXECUTE FUNCTION test_missing_insert_trap()", + ) + .await + .unwrap(); + let before = domain_state_for_test(&first).await; + assert!( + repository + .install_pages_validated(&cap, pages.dag().payloads()) + .await + .unwrap_err() + .to_string() + .contains("payload statement trap fired") + ); + assert_eq!(domain_state_for_test(&first).await, before); + let work = install_missing(&repository, &cap, &pages).await; + let existing = pages.dag().payloads().iter().map(|page| page.id).collect(); + assert_preparing_work(&work, pages.dag().payloads(), &existing); + assert_eq!(domain_state_for_test(&first).await, before); + let receipt = repository.finalize(&intent).await.unwrap(); + let replay = repository + .install_missing_pages_validated(&cap, pages.dag().payloads()) + .await + .unwrap(); + assert_eq!( + replay.committed_replay_pages, + pages.dag().payloads().len() as u64 + ); + assert_eq!(replay.payload_pages_omitted, 0); + assert_eq!(replay.payload_pages_encoded, replay.requested_pages); + assert_eq!(replay.payload_bytes_encoded, pages.dag().payload_bytes()); + assert_eq!(replay.payload_bytes_omitted, 0); + assert_eq!(replay.classification_batches, 0); + assert_eq!(replay.insert_statements, 0); + assert_eq!(replay.byte_comparison_queries, 1); + assert!(replay.payload_parameter_bytes >= 2 * replay.payload_bytes_encoded); + assert_eq!(repository.finalize(&intent).await.unwrap(), receipt); + first.execute_unprepared( + "DROP TRIGGER test_missing_insert_trap ON mst2_metadata_payload; DROP FUNCTION test_missing_insert_trap()", + ).await.unwrap(); +} + +#[tokio::test] +async fn install_missing_invalid_batch_and_foreign_scope_cannot_write_any_member() { + let (first, _second, _schema) = fixture().await; + let (alien, _other, _alien_schema) = fixture().await; + let repository = PostgresMetadataInstallRepository::new(first.clone()) + .await + .unwrap(); + let foreign = PostgresMetadataInstallRepository::new(alien.clone()) + .await + .unwrap(); + let pages = prepared(70); + let intent = repository + .begin_intent("missing-bounds", &pages) + .await + .unwrap(); + let cap = repository + .mint_legacy_install_capability(&intent) + .await + .unwrap(); + let member = pages.dag().payloads()[0].clone(); + let mut bad_size = member.clone(); + bad_size.size += 1; + let mut bad_digest = member.clone(); + bad_digest.bytes = corrupt(&member); + let malformed = MetadataPagePayload { + id: page_id(&[0; HEADER_LEN]), + size: HEADER_LEN as u64, + bytes: vec![0; HEADER_LEN], + }; + let too_big = MetadataPagePayload { + id: page_id(&vec![0; PAGE_MAX_BYTES + 1]), + size: (PAGE_MAX_BYTES + 1) as u64, + bytes: vec![0; PAGE_MAX_BYTES + 1], + }; + let other = prepared(71); + let stranger = other + .dag() + .payloads() + .iter() + .find(|page| { + !pages + .dag() + .payloads() + .iter() + .any(|member| member.id == page.id) + }) + .unwrap() + .clone(); + let before = domain_state_for_test(&first).await; + for batch in [ + vec![], + pages.dag().payloads()[..65].to_vec(), + vec![member.clone(), member.clone()], + vec![member.clone(), bad_size], + vec![member.clone(), bad_digest], + vec![member.clone(), malformed], + vec![member.clone(), too_big], + vec![member, stranger], + ] { + assert!( + repository + .install_missing_pages_validated(&cap, &batch) + .await + .is_err() + ); + assert_eq!(domain_state_for_test(&first).await, before); + } + let foreign_before = domain_state_for_test(&alien).await; + assert!( + foreign + .install_missing_pages_validated(&cap, &pages.dag().payloads()[..64]) + .await + .is_err() + ); + assert_eq!(domain_state_for_test(&alien).await, foreign_before); + assert_eq!(domain_state_for_test(&first).await, before); +} + +#[tokio::test] +async fn install_missing_actual_size_conflict_rejects_before_inserting_missing_siblings() { + let (first, _second, _schema) = fixture().await; + let repository = PostgresMetadataInstallRepository::new(first.clone()) + .await + .unwrap(); + let pages = prepared(2); + let intent = repository + .begin_intent("actual-size", &pages) + .await + .unwrap(); + let cap = repository + .mint_legacy_install_capability(&intent) + .await + .unwrap(); + let page = &pages.dag().payloads()[0]; + let mut wrong = page.bytes.clone(); + wrong.push(0); + assert!(wrong.len() <= PAGE_MAX_BYTES); + first.execute_raw(statement( + "INSERT INTO mst2_metadata_payload(page_id,metadata_codec,byte_size,payload) VALUES($1,1,$2,$3)", + [page.id.to_vec().into(), (wrong.len() as i32).into(), wrong.into()], + )).await.unwrap(); + let before = domain_state_for_test(&first).await; + assert!( + repository + .install_missing_pages_validated(&cap, pages.dag().payloads()) + .await + .unwrap_err() + .to_string() + .contains("payload profile conflicts") + ); + assert_eq!(domain_state_for_test(&first).await, before); + assert!(repository.finalize(&intent).await.is_err()); + assert_eq!(domain_state_for_test(&first).await, before); +} + +#[tokio::test] +async fn install_missing_actual_codec_check_and_after_insert_fault_roll_back_entire_batch() { + let (first, _second, _schema) = fixture().await; + let repository = PostgresMetadataInstallRepository::new(first.clone()) + .await + .unwrap(); + let pages = prepared(2); + let intent = repository + .begin_intent("missing-atomic", &pages) + .await + .unwrap(); + let cap = repository + .mint_legacy_install_capability(&intent) + .await + .unwrap(); + repository + .install_pages_validated(&cap, &pages.dag().payloads()[..1]) + .await + .unwrap(); + let original_modes = payload_modes(&first).await; + let before = domain_state_for_test(&first).await; + let failed = hex::encode(pages.dag().payloads()[1].id); + first.execute_unprepared(&format!( + "CREATE FUNCTION test_missing_bad_codec() RETURNS trigger LANGUAGE plpgsql AS $$ + BEGIN IF NEW.page_id=decode('{failed}','hex') THEN NEW.metadata_codec:=2; END IF; RETURN NEW; END $$; + CREATE TRIGGER test_missing_bad_codec BEFORE INSERT ON mst2_metadata_payload + FOR EACH ROW EXECUTE FUNCTION test_missing_bad_codec()" + )).await.unwrap(); + assert!( + repository + .install_missing_pages_validated(&cap, pages.dag().payloads()) + .await + .unwrap_err() + .to_string() + .contains("mst2_metadata_payload_metadata_codec_check") + ); + assert_eq!(domain_state_for_test(&first).await, before); + first.execute_unprepared( + "DROP TRIGGER test_missing_bad_codec ON mst2_metadata_payload; DROP FUNCTION test_missing_bad_codec(); + CREATE FUNCTION test_missing_after_insert_fault() RETURNS trigger LANGUAGE plpgsql AS $$ + BEGIN RAISE EXCEPTION 'missing insert fault after production guards'; END $$; + CREATE TRIGGER test_missing_after_insert_fault AFTER INSERT ON mst2_metadata_payload + FOR EACH STATEMENT EXECUTE FUNCTION test_missing_after_insert_fault()", + ).await.unwrap(); + assert!( + repository + .install_missing_pages_validated(&cap, pages.dag().payloads()) + .await + .unwrap_err() + .to_string() + .contains("missing insert fault after production guards") + ); + assert_eq!(domain_state_for_test(&first).await, before); + first.execute_unprepared( + "DROP TRIGGER test_missing_after_insert_fault ON mst2_metadata_payload; DROP FUNCTION test_missing_after_insert_fault()", + ).await.unwrap(); + assert_eq!(payload_modes(&first).await, original_modes); + let existing = pages.dag().payloads()[..1] + .iter() + .map(|page| page.id) + .collect(); + let work = install_missing(&repository, &cap, &pages).await; + assert_preparing_work(&work, pages.dag().payloads(), &existing); + repository.finalize(&intent).await.unwrap(); +} + +#[tokio::test] +async fn install_missing_presence_hint_never_certifies_metadata_consistent_corrupt_body() { + let (first, second, _schema) = fixture().await; + let repository = PostgresMetadataInstallRepository::new(first.clone()) + .await + .unwrap(); + let pages = prepared(2); + let intent = repository + .begin_intent("hint-not-proof", &pages) + .await + .unwrap(); + let cap = repository + .mint_legacy_install_capability(&intent) + .await + .unwrap(); + let page = &pages.dag().payloads()[0]; + first.execute_raw(statement( + "INSERT INTO mst2_metadata_payload(page_id,metadata_codec,byte_size,payload) VALUES($1,1,$2,$3)", + [page.id.to_vec().into(), (page.size as i32).into(), corrupt(page).into()], + )).await.unwrap(); + let poisoned = domain_state_for_test(&first).await; + assert!( + repository + .install_pages_validated(&cap, pages.dag().payloads()) + .await + .unwrap_err() + .to_string() + .contains("payload identity conflicts with stored bytes") + ); + assert_eq!(domain_state_for_test(&first).await, poisoned); + let work = install_missing(&repository, &cap, &pages).await; + assert_preparing_work(&work, pages.dag().payloads(), &BTreeSet::from([page.id])); + let before = domain_state_for_test(&first).await; + assert!(repository.finalize(&intent).await.is_err()); + let restarted = PostgresMetadataInstallRepository::new(second) + .await + .unwrap(); + assert!(restarted.finalize(&intent).await.is_err()); + assert!( + restarted + .restore_session_dag(intent.prepare_id()) + .await + .is_err() + ); + assert_eq!(domain_state_for_test(&first).await, before); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_prepare WHERE state='COMMITTED'" + ) + .await, + 0 + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_retention_root").await, + 0 + ); +} + +#[tokio::test] +async fn install_missing_committed_replay_rejects_corrupt_and_missing_bytes_without_repair() { + for remove in [false, true] { + let (first, second, _schema) = fixture().await; + let repository = PostgresMetadataInstallRepository::new(first.clone()) + .await + .unwrap(); + let pages = prepared(2); + let intent = repository + .begin_intent("committed-no-repair", &pages) + .await + .unwrap(); + let cap = repository + .mint_legacy_install_capability(&intent) + .await + .unwrap(); + install_missing(&repository, &cap, &pages).await; + repository.finalize(&intent).await.unwrap(); + let page = &pages.dag().payloads()[0]; + let damage = if remove { + PayloadDamage::Remove + } else { + PayloadDamage::Bytes(corrupt(page)) + }; + damage_payload(&first, page, damage).await; + let before = domain_state_for_test(&first).await; + assert!( + repository + .install_missing_pages_validated(&cap, pages.dag().payloads()) + .await + .is_err() + ); + assert!(repository.finalize(&intent).await.is_err()); + let restarted = PostgresMetadataInstallRepository::new(second) + .await + .unwrap(); + assert!( + restarted + .mint_legacy_install_capability(&intent) + .await + .is_err() + ); + assert!( + restarted + .restore_session_dag(intent.prepare_id()) + .await + .is_err() + ); + assert_eq!(domain_state_for_test(&first).await, before); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_retention_root").await, + pages.dag().payloads().len() as i64 + ); + } +} + +#[tokio::test] +async fn install_missing_generic_generation_reuse_requires_exact_live_physical_binding() { + let (first, second, _schema) = fixture().await; + let generic = generations::PostgresMetadataGenerationRepository::new(first.clone()) + .await + .unwrap(); + let repository = PostgresMetadataInstallRepository::new(first.clone()) + .await + .unwrap(); + let pages = prepared(2); + let bound = generic + .begin_intent("missing-bound-generic", &pages) + .await + .unwrap(); + generic + .install_pages(&bound, pages.dag().payloads()) + .await + .unwrap(); + generic.finalize(&bound).await.unwrap(); + let intent = repository + .begin_intent("missing-share-generic", &pages) + .await + .unwrap(); + let cap = repository + .mint_legacy_install_capability(&intent) + .await + .unwrap(); + let existing = pages.dag().payloads().iter().map(|page| page.id).collect(); + let work = install_missing(&repository, &cap, &pages).await; + assert_preparing_work(&work, pages.dag().payloads(), &existing); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_payload WHERE generation IS NOT NULL" + ) + .await, + pages.dag().payloads().len() as i64 + ); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_prepare_page WHERE generation IS NULL" + ) + .await, + pages.dag().payloads().len() as i64 + ); + let page = &pages.dag().payloads()[0]; + damage_payload(&first, page, PayloadDamage::UnbindGeneration).await; + let before = domain_state_for_test(&first).await; + let restarted = PostgresMetadataInstallRepository::new(second) + .await + .unwrap(); + assert!( + restarted + .install_missing_pages_validated(&cap, pages.dag().payloads()) + .await + .is_err() + ); + assert_eq!(domain_state_for_test(&first).await, before); +} + +#[tokio::test] +async fn install_missing_qualified_owner_cannot_be_adopted_by_generic_preparation() { + let (first, _second, _schema) = fixture().await; + let qualified = generations::qualified::PostgresQualifiedMetadataRepository::new(first.clone()) + .await + .unwrap(); + let legacy = PostgresMetadataInstallRepository::new(first.clone()) + .await + .unwrap(); + let pages = prepared(2); + let bound = qualified + .begin_intent("missing-qualified", &pages) + .await + .unwrap(); + qualified + .install_pages(&bound, pages.dag().payloads()) + .await + .unwrap(); + let before = domain_state_for_test(&first).await; + assert!( + legacy + .begin_intent("missing-wrong-domain", &pages) + .await + .unwrap_err() + .to_string() + .contains("qualified mapping cannot cross domain/current/state fence") + ); + let bound_intent = MetadataPrepareIntent { + prepare_id: bound.prepare_id().into(), + operation_id: bound.operation_id().into(), + manifest_digest: bound.manifest_digest(), + }; + assert!( + legacy + .mint_legacy_install_capability(&bound_intent) + .await + .is_err() + ); + assert_eq!(domain_state_for_test(&first).await, before); +} + +#[tokio::test] +async fn install_missing_postclassification_release_gc_and_tombstone_keep_finalization_fenced() { + let (first, second, _schema) = fixture().await; + let repository = PostgresMetadataInstallRepository::new(first.clone()) + .await + .unwrap(); + let graph = PostgresRetentionRepository::new(second); + let pages = prepared(2); + let old = repository + .begin_intent("old-covered", &pages) + .await + .unwrap(); + let old_cap = repository + .mint_legacy_install_capability(&old) + .await + .unwrap(); + install_missing(&repository, &old_cap, &pages).await; + repository.finalize(&old).await.unwrap(); + let next = repository + .begin_intent("after-classification", &pages) + .await + .unwrap(); + let cap = repository + .mint_legacy_install_capability(&next) + .await + .unwrap(); + let work = install_missing(&repository, &cap, &pages).await; + let existing = pages.dag().payloads().iter().map(|page| page.id).collect(); + assert_preparing_work(&work, pages.dag().payloads(), &existing); + let root = node_id(&pages.dag().root()); + let lease = RetentionRoot::Lease("missing-classification-lease-root".into()); + let txn = first.begin().await.unwrap(); + PostgresRetentionRepository::acquire_existing_roots_in_txn( + &txn, + &root, + std::slice::from_ref(&lease), + ) + .await + .unwrap(); + txn.commit().await.unwrap(); + graph + .release_root(&RetentionRoot::Prepare(old.prepare_id().into())) + .await + .unwrap(); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_retention_root WHERE root_kind='lease'" + ) + .await, + 1 + ); + let protected = domain_state_for_test(&first).await; + assert_eq!( + graph + .mark_deleting("missing-gc-protected", &root) + .await + .unwrap(), + GcClaim::Unavailable + ); + assert_eq!(domain_state_for_test(&first).await, protected); + graph.release_root(&lease).await.unwrap(); + assert_eq!( + graph.mark_deleting("missing-gc-root", &root).await.unwrap(), + GcClaim::Marked + ); + let before = domain_state_for_test(&first).await; + assert!(repository.finalize(&next).await.is_err()); + assert!( + repository + .install_missing_pages_validated(&cap, pages.dag().payloads()) + .await + .is_err() + ); + assert_eq!(domain_state_for_test(&first).await, before); + assert_eq!(graph.node(&root).await.unwrap().unwrap().state, "DELETING"); + let completed = graph.complete_gc("missing-gc-root").await.unwrap(); + assert!(!completed.replayed); + assert!(graph.node(&root).await.unwrap().is_none()); + let tombstoned = domain_state_for_test(&first).await; + assert!( + repository + .install_missing_pages_validated(&cap, pages.dag().payloads()) + .await + .is_err() + ); + assert!(repository.finalize(&next).await.is_err()); + assert_eq!(domain_state_for_test(&first).await, tombstoned); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_retention_root").await, + 0 + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_payload").await, + pages.dag().payloads().len() as i64 + ); +} diff --git a/src/jupiter/storage/native_metadata_install_tests.rs b/src/jupiter/storage/native_metadata_install_tests.rs new file mode 100644 index 00000000..4cb387dc --- /dev/null +++ b/src/jupiter/storage/native_metadata_install_tests.rs @@ -0,0 +1,1385 @@ +use std::sync::{ + Arc, + atomic::{AtomicBool, Ordering}, +}; + +use mst2_codec::metapage::{Entry, EntryKind}; +use sea_orm::{Database, PaginatorTrait}; +use sea_orm_migration::MigratorTrait; +use tokio::{ + io::{AsyncReadExt, AsyncWriteExt}, + net::{TcpListener, TcpStream}, + sync::{Notify, watch}, + task::{JoinHandle, JoinSet}, +}; + +use super::*; +use crate::{ + ceres::snapshot::retention_dag::MetadataDagBuilder, + jupiter::{ + migration::Migrator, + tests::{TestSchemaGuard, test_db_config}, + }, +}; + +fn prepared(scope: &str) -> PreparedNativeMetadataRetention { + let child_entries = [Entry::file(EntryKind::Regular, b"file", 3, [42; 32])]; + let child = Page::build(&child_entries).unwrap(); + let entries = [ + Entry::dir(b"one", page_id(&child)), + Entry::dir(b"two", page_id(&child)), + ]; + let root = Page::build(&entries).unwrap(); + let mut builder = MetadataDagBuilder::new(MetadataDagLimits::default()); + builder.add_directory(&child, &child_entries).unwrap(); + builder.add_directory(&root, &entries).unwrap(); + PreparedNativeMetadataRetention::test_installation( + Arc::new(builder.finish(page_id(&root)).unwrap()), + scope, + ) +} + +async fn fixture() -> ( + DatabaseConnection, + DatabaseConnection, + TestSchemaGuard, + String, +) { + let temp = tempfile::TempDir::new().unwrap(); + let (config, schema) = test_db_config(temp.path()).await; + let first = Database::connect(config.db_url.clone()).await.unwrap(); + Migrator::up(&first, None).await.unwrap(); + let second = Database::connect(config.db_url.clone()).await.unwrap(); + (first, second, schema, config.db_url) +} + +async fn install( + repository: &PostgresMetadataInstallRepository, + intent: &MetadataPrepareIntent, + prepared: &PreparedNativeMetadataRetention, +) { + for page in prepared.dag().payloads() { + repository.install_page(intent, page).await.unwrap(); + } +} + +fn rejected(error: MetadataInstallError) -> SnapshotErrorCode { + match error { + MetadataInstallError::Rejected(error) => error.code, + _ => panic!("expected definite rejection"), + } +} + +#[tokio::test] +async fn native_metadata_scope_ignores_poisoned_temp_table_and_rejects_changed_primary_schema() { + let (first, _second, _schema, url) = fixture().await; + let (_other, _unused, other_schema, _other_url) = fixture().await; + let expected = PostgresMetadataInstallRepository::new(first) + .await + .unwrap() + .storage_scope; + let mut options = sea_orm::ConnectOptions::new(url); + options.max_connections(1).min_connections(1); + let connection = Database::connect(options).await.unwrap(); + connection + .execute_unprepared( + "CREATE TEMP TABLE mst2_metadata_storage_scope(singleton integer,storage_uuid text); + INSERT INTO pg_temp.mst2_metadata_storage_scope VALUES(1,'temp-poison')", + ) + .await + .unwrap(); + let repository = PostgresMetadataInstallRepository::new(connection.clone()) + .await + .unwrap(); + assert_eq!(repository.storage_scope, expected); + assert_eq!(repository.captured_schema(), _schema.schema()); + repository + .verify_primary_connection(&connection) + .await + .unwrap(); + let txn = connection.begin().await.unwrap(); + repository.barrier(&txn).await.unwrap(); + txn.rollback().await.unwrap(); + connection + .execute_raw(statement( + "UPDATE pg_temp.mst2_metadata_storage_scope SET storage_uuid=$1", + [repository.storage_scope.storage_uuid.clone().into()], + )) + .await + .unwrap(); + let txn = connection.begin().await.unwrap(); + txn.execute_unprepared(&format!( + "SET LOCAL search_path=\"{}\",pg_catalog,pg_temp", + other_schema.schema().replace('"', "\"\"") + )) + .await + .unwrap(); + let error = repository + .verify_primary_connection(&txn) + .await + .unwrap_err(); + assert_eq!(error.code, SnapshotErrorCode::Internal); + assert!( + error + .message + .contains("session mutation no longer targets its captured primary storage scope") + ); + let error = repository.barrier(&txn).await.unwrap_err(); + assert_eq!(error.code, SnapshotErrorCode::Internal); + assert!( + error + .message + .contains("metadata recovery connection is outside the captured primary storage scope") + ); + txn.rollback().await.unwrap(); + connection + .execute_unprepared("SET search_path=pg_temp,pg_catalog") + .await + .unwrap(); + let error = PostgresMetadataInstallRepository::new(connection.clone()) + .await + .err() + .unwrap(); + assert_eq!(error.code, SnapshotErrorCode::Internal); + assert_eq!( + error.message, + "metadata actual primary storage relation is missing" + ); + connection + .execute_unprepared(&format!( + "SET search_path=\"{}\",pg_catalog,pg_temp", + repository.captured_schema().replace('"', "\"\"") + )) + .await + .unwrap(); + repository + .verify_primary_connection(&connection) + .await + .unwrap(); + assert_eq!( + connection + .query_one_raw(statement( + "SELECT storage_uuid FROM pg_temp.mst2_metadata_storage_scope WHERE singleton=1", + [], + )) + .await + .unwrap() + .unwrap() + .try_get::("", "storage_uuid") + .unwrap(), + expected.storage_uuid + ); +} + +#[tokio::test] +async fn native_metadata_two_primary_connections_replay_one_receipt_without_recounting() { + let (first, second, _schema, _url) = fixture().await; + let a = PostgresMetadataInstallRepository::new(first.clone()) + .await + .unwrap(); + let b = PostgresMetadataInstallRepository::new(second.clone()) + .await + .unwrap(); + let prepared = prepared("/"); + let (ia, ib) = tokio::join!( + a.begin_intent("same", &prepared), + b.begin_intent("same", &prepared) + ); + let ia = ia.unwrap(); + assert_eq!(ia, ib.unwrap()); + for page in prepared.dag().payloads() { + let (left, right) = tokio::join!(a.install_page(&ia, page), b.install_page(&ia, page)); + left.unwrap(); + right.unwrap(); + } + let (ra, rb) = tokio::join!(a.finalize(&ia), b.finalize(&ia)); + assert_eq!(ra.unwrap(), rb.unwrap()); + assert_eq!( + mst2_metadata_prepare::Entity::find() + .count(&first) + .await + .unwrap(), + 1 + ); + assert_eq!( + mst2_metadata_payload::Entity::find() + .count(&first) + .await + .unwrap(), + 2 + ); + assert_eq!( + mst2_retention_edge::Entity::find() + .count(&first) + .await + .unwrap(), + 1 + ); + assert_eq!( + mst2_retention_root::Entity::find() + .count(&first) + .await + .unwrap(), + 2 + ); + let child = prepared.dag().edges()[0].child.clone(); + assert_eq!( + PostgresRetentionRepository::new(first) + .node(&child) + .await + .unwrap() + .unwrap() + .incoming_refs, + 1 + ); +} + +#[tokio::test] +async fn native_metadata_operation_conflict_and_multibyte_byte_limit_reject_without_writes() { + let (first, second, _schema, _url) = fixture().await; + let a = PostgresMetadataInstallRepository::new(first.clone()) + .await + .unwrap(); + let b = PostgresMetadataInstallRepository::new(second) + .await + .unwrap(); + a.begin_intent("bound", &prepared("/")).await.unwrap(); + assert_eq!( + rejected( + b.begin_intent("bound", &prepared("/scope")) + .await + .unwrap_err() + ), + SnapshotErrorCode::Conflict + ); + assert_eq!( + rejected( + a.begin_intent(&"é".repeat(128), &prepared("/")) + .await + .unwrap_err() + ), + SnapshotErrorCode::InvalidRequest + ); + assert_eq!( + mst2_metadata_prepare::Entity::find() + .count(&first) + .await + .unwrap(), + 1 + ); + assert_eq!( + mst2_metadata_payload::Entity::find() + .count(&first) + .await + .unwrap(), + 0 + ); + assert_eq!( + mst2_retention_root::Entity::find() + .count(&first) + .await + .unwrap(), + 0 + ); +} + +#[tokio::test] +async fn native_metadata_protocol_hash_and_immutable_bytes_conflicts_cannot_be_overwritten() { + let (first, _second, _schema, _url) = fixture().await; + let repository = PostgresMetadataInstallRepository::new(first.clone()) + .await + .unwrap(); + let prepared = prepared("/"); + let intent = repository.begin_intent("bytes", &prepared).await.unwrap(); + let original = &prepared.dag().payloads()[0]; + let mut invalid = original.clone(); + invalid.id[0] ^= 1; + assert_eq!( + rejected( + repository + .install_page(&intent, &invalid) + .await + .unwrap_err() + ), + SnapshotErrorCode::DigestMismatch + ); + let mut oversized = original.clone(); + oversized.bytes.resize(PAGE_MAX_BYTES + 1, 0); + oversized.size = oversized.bytes.len() as u64; + oversized.id = page_id(&oversized.bytes); + assert_eq!( + rejected( + repository + .install_page(&intent, &oversized) + .await + .unwrap_err() + ), + SnapshotErrorCode::LimitExceeded + ); + // A corrupt existing row is never replaced by a good upload. + let mut corrupt = original.bytes.clone(); + corrupt[0] ^= 1; + first.execute_raw(statement("INSERT INTO mst2_metadata_payload(page_id,metadata_codec,byte_size,payload) VALUES($1,1,$2,$3)", + [original.id.to_vec().into(),(original.size as i32).into(),corrupt.clone().into()])).await.unwrap(); + assert_eq!( + rejected( + repository + .install_page(&intent, original) + .await + .unwrap_err() + ), + SnapshotErrorCode::IntegrityError + ); + assert_eq!( + mst2_metadata_payload::Entity::find_by_id(original.id.to_vec()) + .one(&first) + .await + .unwrap() + .unwrap() + .payload, + corrupt + ); + assert!( + first + .execute_unprepared("UPDATE mst2_metadata_payload SET payload=payload") + .await + .is_err() + ); + assert!( + first + .execute_unprepared("DELETE FROM mst2_metadata_payload") + .await + .is_err() + ); + assert_eq!( + mst2_retention_root::Entity::find() + .count(&first) + .await + .unwrap(), + 0 + ); +} + +#[tokio::test] +async fn native_metadata_partial_installation_survives_repository_and_connection_restart() { + let (first, second, _schema, _url) = fixture().await; + let repository = PostgresMetadataInstallRepository::new(first.clone()) + .await + .unwrap(); + let prepared = prepared("/"); + let intent = repository.begin_intent("restart", &prepared).await.unwrap(); + repository + .install_page(&intent, &prepared.dag().payloads()[0]) + .await + .unwrap(); + assert_eq!( + mst2_retention_node::Entity::find() + .count(&first) + .await + .unwrap(), + 0 + ); + drop(repository); + first.close().await.unwrap(); + let restarted = PostgresMetadataInstallRepository::new(second.clone()) + .await + .unwrap(); + assert!(matches!( + restarted + .inspect_prepare( + &second, + intent.operation_id(), + intent.manifest_digest(), + MetadataCommitPhase::Payload + ) + .await + .unwrap(), + MetadataPrepareObservation::Preparing(_) + )); + assert_eq!( + restarted + .load_installed_dag(&intent) + .await + .unwrap_err() + .code, + SnapshotErrorCode::ObjectUnavailable + ); + install(&restarted, &intent, &prepared).await; + let receipt = restarted.finalize(&intent).await.unwrap(); + assert_eq!(receipt.metadata_root(), prepared.dag().root()); + assert_eq!(receipt.payload_bytes(), prepared.dag().payload_bytes()); +} + +#[tokio::test] +async fn native_metadata_half_installation_survives_real_process_kill() { + const CHILD_DB: &str = "MEGA_MST2_METADATA_CRASH_CHILD_DB"; + const CHECKPOINT: &str = "MEGA_MST2_METADATA_CRASH_CHECKPOINT"; + const OPERATION: &str = "process-crash"; + if let Ok(url) = std::env::var(CHILD_DB) { + let database = Database::connect(url).await.unwrap(); + let repository = PostgresMetadataInstallRepository::new(database) + .await + .unwrap(); + let prepared = prepared("/"); + let intent = repository.begin_intent(OPERATION, &prepared).await.unwrap(); + repository + .install_page(&intent, &prepared.dag().payloads()[0]) + .await + .unwrap(); + let checkpoint = std::path::PathBuf::from(std::env::var(CHECKPOINT).unwrap()); + let staging = checkpoint.with_extension("staging"); + std::fs::write( + &staging, + format!( + "{}\n{}", + intent.prepare_id(), + hex::encode(intent.manifest_digest()) + ), + ) + .unwrap(); + std::fs::rename(staging, checkpoint).unwrap(); + // Keep repository/connection alive until the parent kills this process. + std::future::pending::<()>().await; + unreachable!(); + } + let (first, second, _schema, url) = fixture().await; + let checkpoint_dir = tempfile::tempdir().unwrap(); + let checkpoint = checkpoint_dir.path().join("partial-installation"); + let full_name = concat!( + module_path!(), + "::native_metadata_half_installation_survives_real_process_kill" + ); + let test_name = full_name.split_once("::").unwrap().1; + let mut worker = tokio::process::Command::new(std::env::current_exe().unwrap()) + .args(["--exact", test_name, "--nocapture"]) + .env(CHILD_DB, url) + .env(CHECKPOINT, &checkpoint) + .stdout(std::process::Stdio::null()) + .stderr(std::process::Stdio::inherit()) + .kill_on_drop(true) + .spawn() + .unwrap(); + let marker = tokio::time::timeout(Duration::from_secs(60), async { + loop { + if checkpoint.exists() { + break std::fs::read_to_string(&checkpoint).unwrap(); + } + if let Some(status) = worker.try_wait().unwrap() { + panic!("metadata crash child exited before partial durable commit: {status}"); + } + tokio::time::sleep(Duration::from_millis(20)).await; + } + }) + .await + .expect("metadata crash child did not reach its durable checkpoint"); + worker.kill().await.unwrap(); + let status = worker.wait().await.unwrap(); + assert!(!status.success()); + #[cfg(unix)] + { + use std::os::unix::process::ExitStatusExt; + assert_eq!(status.signal(), Some(libc::SIGKILL)); + } + first.close().await.unwrap(); + let restarted = PostgresMetadataInstallRepository::new(second.clone()) + .await + .unwrap(); + let prepared = prepared("/"); + let digest = prepared.install_plan().unwrap().digest().unwrap(); + let original = match restarted + .inspect_prepare(&second, OPERATION, digest, MetadataCommitPhase::Payload) + .await + .unwrap() + { + MetadataPrepareObservation::Preparing(intent) => intent, + other => panic!("half installation exposed a false committed receipt: {other:?}"), + }; + let mut marker = marker.lines(); + assert_eq!(marker.next(), Some(original.prepare_id())); + assert_eq!(marker.next(), Some(hex::encode(digest).as_str())); + assert_eq!( + restarted.begin_intent(OPERATION, &prepared).await.unwrap(), + original + ); + assert_eq!( + mst2_metadata_payload::Entity::find() + .count(&second) + .await + .unwrap(), + 1 + ); + assert_eq!( + mst2_metadata_prepare_page::Entity::find() + .count(&second) + .await + .unwrap(), + prepared.dag().payloads().len() as u64 + ); + assert_eq!( + rejected(restarted.finalize(&original).await.unwrap_err()), + SnapshotErrorCode::ObjectUnavailable + ); + assert_eq!( + mst2_retention_root::Entity::find() + .count(&second) + .await + .unwrap(), + 0 + ); + assert_eq!( + mst2_retention_node::Entity::find() + .count(&second) + .await + .unwrap(), + 0 + ); + install(&restarted, &original, &prepared).await; + let receipt = restarted.finalize(&original).await.unwrap(); + assert_eq!(receipt.intent(), &original); + assert_eq!(restarted.finalize(&original).await.unwrap(), receipt); + assert_eq!( + mst2_retention_root::Entity::find() + .count(&second) + .await + .unwrap(), + prepared.dag().payloads().len() as u64 + ); + assert_eq!( + mst2_retention_edge::Entity::find() + .count(&second) + .await + .unwrap(), + 1 + ); +} + +async fn overwrite_installed_payload_for_test( + connection: &DatabaseConnection, + page: &MetadataPagePayload, + bytes: Vec, +) { + assert_eq!(bytes.len(), page.bytes.len()); + let guard_modes = payload_trigger_modes_for_test(connection).await; + assert!( + guard_modes + .iter() + .any(|(name, _)| name == "mst2_metadata_payload_fenced") + ); + assert!(guard_modes.iter().all(|(_, mode)| mode == "O")); + assert!( + connection + .execute_raw(statement( + "UPDATE mst2_metadata_payload SET payload=payload WHERE page_id=$1", + [page.id.to_vec().into()], + )) + .await + .is_err(), + "production payload UPDATE fence must still reject writes" + ); + let txn = connection.begin().await.unwrap(); + txn.execute_unprepared("SELECT pg_advisory_xact_lock(1296717362,hashtext(current_schema()))") + .await + .unwrap(); + txn.execute_unprepared( + "ALTER TABLE mst2_metadata_payload DISABLE TRIGGER mst2_metadata_payload_fenced", + ) + .await + .unwrap(); + let result = txn + .execute_raw(statement( + "UPDATE mst2_metadata_payload SET payload=$2 WHERE page_id=$1", + [page.id.to_vec().into(), bytes.into()], + )) + .await + .unwrap(); + assert_eq!(result.rows_affected(), 1); + txn.execute_unprepared( + "ALTER TABLE mst2_metadata_payload ENABLE TRIGGER mst2_metadata_payload_fenced", + ) + .await + .unwrap(); + txn.commit().await.unwrap(); + assert_eq!( + payload_trigger_modes_for_test(connection).await, + guard_modes + ); + assert!( + connection + .execute_raw(statement( + "UPDATE mst2_metadata_payload SET payload=payload WHERE page_id=$1", + [page.id.to_vec().into()], + )) + .await + .is_err(), + "corruption injection must restore the production UPDATE fence" + ); +} + +async fn payload_trigger_modes_for_test(connection: &DatabaseConnection) -> Vec<(String, String)> { + connection + .query_all_raw(statement( + "SELECT tgname,tgenabled::text AS mode FROM pg_trigger \ + WHERE tgrelid='mst2_metadata_payload'::regclass AND NOT tgisinternal ORDER BY tgname", + [], + )) + .await + .unwrap() + .into_iter() + .map(|row| { + ( + row.try_get("", "tgname").unwrap(), + row.try_get("", "mode").unwrap(), + ) + }) + .collect() +} + +#[tokio::test] +async fn native_metadata_committed_recovery_rejects_same_length_corruption_and_preserves_pins() { + let (first, second, _schema, _url) = fixture().await; + let repository = PostgresMetadataInstallRepository::new(first.clone()) + .await + .unwrap(); + let prepared = prepared("/"); + let intent = repository + .begin_intent("corruption", &prepared) + .await + .unwrap(); + install(&repository, &intent, &prepared).await; + let receipt = repository.finalize(&intent).await.unwrap(); + let page = &prepared.dag().payloads()[0]; + let counters = mst2_retention_node::Entity::find() + .all(&first) + .await + .unwrap() + .into_iter() + .map(|node| (node.node_id, node.incoming_refs)) + .collect::>(); + let mut corrupt = page.bytes.clone(); + corrupt[0] ^= 1; + overwrite_installed_payload_for_test(&first, page, corrupt).await; + assert_eq!( + rejected( + repository + .inspect_prepare( + &second, + intent.operation_id(), + intent.manifest_digest(), + MetadataCommitPhase::Finalize + ) + .await + .unwrap_err() + ), + SnapshotErrorCode::DigestMismatch + ); + assert_eq!( + rejected(repository.finalize(&intent).await.unwrap_err()), + SnapshotErrorCode::DigestMismatch + ); + assert_eq!( + mst2_retention_root::Entity::find() + .count(&second) + .await + .unwrap(), + prepared.dag().payloads().len() as u64 + ); + assert_eq!( + mst2_retention_node::Entity::find() + .all(&second) + .await + .unwrap() + .into_iter() + .map(|node| (node.node_id, node.incoming_refs)) + .collect::>(), + counters + ); + assert_eq!( + mst2_metadata_prepare::Entity::find_by_id(intent.prepare_id().to_owned()) + .one(&second) + .await + .unwrap() + .unwrap() + .state, + "COMMITTED" + ); + overwrite_installed_payload_for_test(&first, page, page.bytes.clone()).await; + assert_eq!( + repository + .inspect_prepare( + &second, + intent.operation_id(), + intent.manifest_digest(), + MetadataCommitPhase::Finalize + ) + .await + .unwrap(), + MetadataPrepareObservation::Committed(receipt) + ); +} + +#[tokio::test] +async fn native_metadata_outer_rollback_preserves_intent_but_exposes_no_graph_or_receipt() { + let (first, second, _schema, _url) = fixture().await; + let repository = PostgresMetadataInstallRepository::new(first.clone()) + .await + .unwrap(); + let prepared = prepared("/"); + let intent = repository + .begin_intent("rollback", &prepared) + .await + .unwrap(); + install(&repository, &intent, &prepared).await; + let dag = repository.load_installed_dag(&intent).await.unwrap(); + let txn = repository.transaction().await.unwrap(); + repository + .finalize_in_txn(&txn, &intent, &dag) + .await + .unwrap(); + assert_eq!( + mst2_retention_root::Entity::find() + .count(&second) + .await + .unwrap(), + 0 + ); + assert_eq!( + mst2_metadata_prepare::Entity::find_by_id(intent.prepare_id().to_owned()) + .one(&second) + .await + .unwrap() + .unwrap() + .state, + "PREPARING" + ); + txn.rollback().await.unwrap(); + assert!(matches!( + repository + .inspect_prepare( + &second, + intent.operation_id(), + intent.manifest_digest(), + MetadataCommitPhase::Finalize + ) + .await + .unwrap(), + MetadataPrepareObservation::Preparing(_) + )); + assert_eq!( + mst2_retention_node::Entity::find() + .count(&second) + .await + .unwrap(), + 0 + ); + repository.finalize(&intent).await.unwrap(); +} + +#[tokio::test] +async fn native_metadata_final_lock_rechecks_deleting_after_payload_verification() { + let (first, second, _schema, _url) = fixture().await; + let repository = PostgresMetadataInstallRepository::new(first.clone()) + .await + .unwrap(); + let prepared = prepared("/"); + let intent = repository + .begin_intent("deleting", &prepared) + .await + .unwrap(); + install(&repository, &intent, &prepared).await; + let dag = repository.load_installed_dag(&intent).await.unwrap(); + let graph = PostgresRetentionRepository::new(second.clone()); + graph + .retain_group(dag.nodes(), dag.edges(), &[]) + .await + .unwrap(); + let root = node_id(&dag.root()); + assert_eq!( + graph.mark_deleting("gc-root", &root).await.unwrap(), + super::super::mst2_retention::GcClaim::Marked + ); + let txn = repository.transaction().await.unwrap(); + assert_eq!( + repository + .finalize_in_txn(&txn, &intent, &dag) + .await + .unwrap_err() + .code, + SnapshotErrorCode::ObjectUnavailable + ); + txn.rollback().await.unwrap(); + assert_eq!( + mst2_retention_root::Entity::find() + .count(&second) + .await + .unwrap(), + 0 + ); + assert_eq!( + mst2_metadata_prepare::Entity::find_by_id(intent.prepare_id().to_owned()) + .one(&second) + .await + .unwrap() + .unwrap() + .state, + "PREPARING" + ); + assert_eq!(graph.node(&root).await.unwrap().unwrap().state, "DELETING"); +} + +#[tokio::test] +async fn native_metadata_receipt_replay_rejects_lost_pin_without_recreating_it() { + let (first, second, _schema, _url) = fixture().await; + let repository = PostgresMetadataInstallRepository::new(first.clone()) + .await + .unwrap(); + let prepared = prepared("/"); + let intent = repository + .begin_intent("lost-pin", &prepared) + .await + .unwrap(); + install(&repository, &intent, &prepared).await; + repository.finalize(&intent).await.unwrap(); + PostgresRetentionRepository::new(first) + .release_root(&RetentionRoot::Prepare(intent.prepare_id().into())) + .await + .unwrap(); + assert_eq!( + rejected( + repository + .inspect_prepare( + &second, + intent.operation_id(), + intent.manifest_digest(), + MetadataCommitPhase::Finalize + ) + .await + .unwrap_err() + ), + SnapshotErrorCode::ObjectUnavailable + ); + assert_eq!( + rejected(repository.finalize(&intent).await.unwrap_err()), + SnapshotErrorCode::ObjectUnavailable + ); + assert_eq!( + mst2_retention_root::Entity::find() + .count(&second) + .await + .unwrap(), + 0 + ); + assert_eq!( + mst2_metadata_payload::Entity::find() + .count(&second) + .await + .unwrap(), + 2 + ); +} + +#[tokio::test] +async fn native_metadata_recovery_lock_timeout_is_unknown_and_never_releases_pins() { + let (first, second, _schema, _url) = fixture().await; + let repository = PostgresMetadataInstallRepository::new(first.clone()) + .await + .unwrap(); + let prepared = prepared("/"); + let intent = repository.begin_intent("timeout", &prepared).await.unwrap(); + install(&repository, &intent, &prepared).await; + repository.finalize(&intent).await.unwrap(); + let blocker = repository.transaction().await.unwrap(); + repository.barrier(&blocker).await.unwrap(); + let mut recovery = PostgresMetadataInstallRepository::new(second.clone()) + .await + .unwrap(); + recovery.barrier_timeout = Duration::from_millis(25); + assert!(matches!( + recovery + .inspect_prepare( + &second, + intent.operation_id(), + intent.manifest_digest(), + MetadataCommitPhase::Finalize + ) + .await + .unwrap_err(), + MetadataInstallError::CommitUncertain { + phase: MetadataCommitPhase::Finalize, + .. + } + )); + assert_eq!( + mst2_retention_root::Entity::find() + .count(&second) + .await + .unwrap(), + 2 + ); + blocker.rollback().await.unwrap(); + assert!(matches!( + recovery + .inspect_prepare( + &second, + intent.operation_id(), + intent.manifest_digest(), + MetadataCommitPhase::Finalize + ) + .await + .unwrap(), + MetadataPrepareObservation::Committed(_) + )); +} + +#[tokio::test] +async fn native_metadata_stored_identity_and_plan_tampering_fail_closed_on_restart() { + let (first, second, _schema, _url) = fixture().await; + let repository = PostgresMetadataInstallRepository::new(first.clone()) + .await + .unwrap(); + let intent = repository + .begin_intent("profile", &prepared("/")) + .await + .unwrap(); + first + .execute_unprepared("UPDATE mst2_metadata_prepare SET projection_revision=2") + .await + .unwrap(); + assert_eq!( + rejected( + repository + .inspect_prepare( + &second, + intent.operation_id(), + intent.manifest_digest(), + MetadataCommitPhase::Intent + ) + .await + .unwrap_err() + ), + SnapshotErrorCode::IntegrityError + ); + first.execute_unprepared("UPDATE mst2_metadata_prepare SET projection_revision=1,canonical_plan=canonical_plan || decode('ff','hex')").await.unwrap(); + assert_eq!( + rejected( + repository + .inspect_prepare( + &second, + intent.operation_id(), + intent.manifest_digest(), + MetadataCommitPhase::Intent + ) + .await + .unwrap_err() + ), + SnapshotErrorCode::IntegrityError + ); + assert_eq!( + mst2_retention_root::Entity::find() + .count(&second) + .await + .unwrap(), + 0 + ); +} + +#[derive(Clone, Copy)] +enum Fault { + BeforeCommit, + AfterCommit, + PrepareRead, +} + +struct PgCommitFaultProxy { + url: String, + armed: Arc, + fired: Arc, + commit_observed: Arc, + changed: Arc, + task: JoinHandle<()>, +} + +impl PgCommitFaultProxy { + async fn start(original: &str, fault: Fault) -> Self { + let mut url = url::Url::parse(original).unwrap(); + let host = url.host_str().unwrap().to_owned(); + let port = url.port().unwrap_or(5432); + let listener = TcpListener::bind("127.0.0.1:0").await.unwrap(); + let address = listener.local_addr().unwrap(); + url.set_host(Some("127.0.0.1")).unwrap(); + url.set_port(Some(address.port())).unwrap(); + let pairs: Vec<_> = url + .query_pairs() + .filter(|(key, _)| key != "sslmode") + .map(|(key, value)| (key.into_owned(), value.into_owned())) + .collect(); + url.set_query(None); + url.query_pairs_mut() + .extend_pairs(pairs) + .append_pair("sslmode", "disable"); + let armed = Arc::new(AtomicBool::new(false)); + let fired = Arc::new(AtomicBool::new(false)); + let commit_observed = Arc::new(AtomicBool::new(false)); + let changed = Arc::new(Notify::new()); + let a = armed.clone(); + let f = fired.clone(); + let c = commit_observed.clone(); + let n = changed.clone(); + let task = tokio::spawn(async move { + let mut connections = JoinSet::new(); + loop { + tokio::select! { + accepted=listener.accept()=>{ + let (client,_)=accepted.unwrap();let host=host.clone(); + let a=a.clone();let f=f.clone();let c=c.clone();let n=n.clone(); + connections.spawn(async move { + let server=TcpStream::connect((host.as_str(),port)).await?; + forward_connection(client,server,fault,a,f,c,n).await + }); + } + _=connections.join_next(),if !connections.is_empty()=>{} + } + } + }); + Self { + url: url.to_string(), + armed, + fired, + commit_observed, + changed, + task, + } + } + async fn wait_for_fault(&self) { + tokio::time::timeout(Duration::from_secs(10), async { + loop { + let changed = self.changed.notified(); + if self.fired.load(Ordering::SeqCst) { + break; + } + changed.await; + } + }) + .await + .expect("proxy never observed the real COMMIT fault"); + } +} +impl Drop for PgCommitFaultProxy { + fn drop(&mut self) { + self.task.abort(); + } +} + +async fn read_frame( + reader: &mut R, +) -> std::io::Result<(u8, Vec)> { + let kind = reader.read_u8().await?; + let length = reader.read_u32().await?; + if !(4..=4 * 1024 * 1024).contains(&length) { + return Err(std::io::Error::other("invalid PostgreSQL frame length")); + } + let mut body = vec![0; length as usize - 4]; + reader.read_exact(&mut body).await?; + Ok((kind, body)) +} +async fn write_frame( + writer: &mut W, + kind: u8, + body: &[u8], +) -> std::io::Result<()> { + writer.write_u8(kind).await?; + writer.write_u32(body.len() as u32 + 4).await?; + writer.write_all(body).await?; + writer.flush().await +} + +async fn forward_connection( + mut client: TcpStream, + mut server: TcpStream, + fault: Fault, + armed: Arc, + fired: Arc, + commit_observed: Arc, + changed: Arc, +) -> std::io::Result<()> { + // sslmode=disable makes the first message a bounded StartupMessage. + let length = client.read_u32().await?; + if !(8..=65536).contains(&length) { + return Err(std::io::Error::other("invalid PostgreSQL startup length")); + } + let mut startup = vec![0; length as usize - 4]; + client.read_exact(&mut startup).await?; + server.write_u32(length).await?; + server.write_all(&startup).await?; + server.flush().await?; + let (mut cr, mut cw) = client.into_split(); + let (mut sr, mut sw) = server.into_split(); + let suppress = Arc::new(AtomicBool::new(false)); + let sender_suppress = suppress.clone(); + let (stop, mut stopped) = watch::channel(false); + let sender_stop = stop.clone(); + let sender_fired = fired.clone(); + let sender_changed = changed.clone(); + let to_server = async move { + loop { + let frame = tokio::select! {result=read_frame(&mut cr)=>result?, _=stopped.changed()=>return Ok::<_,std::io::Error>(())}; + let is_commit = frame.0 == b'Q' && frame.1.as_slice() == b"COMMIT\0"; + let fault_target = match fault { + Fault::PrepareRead => { + b"PQ".contains(&frame.0) + && frame + .1 + .windows(b"mst2_metadata_prepare".len()) + .any(|bytes| bytes == b"mst2_metadata_prepare") + } + _ => is_commit, + }; + if fault_target && armed.swap(false, Ordering::SeqCst) { + match fault { + Fault::BeforeCommit | Fault::PrepareRead => { + sender_fired.store(true, Ordering::SeqCst); + sender_changed.notify_one(); + sw.shutdown().await?; + sender_stop.send_replace(true); + return Ok(()); + } + Fault::AfterCommit => sender_suppress.store(true, Ordering::SeqCst), + } + } + write_frame(&mut sw, frame.0, &frame.1).await?; + } + }; + let mut stopped = stop.subscribe(); + let to_client = async move { + loop { + let frame = tokio::select! {result=read_frame(&mut sr)=>result?, _=stopped.changed()=>{cw.shutdown().await?;return Ok::<_,std::io::Error>(());}}; + if suppress.load(Ordering::SeqCst) { + if frame.0 == b'C' && frame.1.as_slice() == b"COMMIT\0" { + commit_observed.store(true, Ordering::SeqCst); + fired.store(true, Ordering::SeqCst); + changed.notify_one(); + cw.shutdown().await?; + stop.send_replace(true); + return Ok(()); + } + } else { + write_frame(&mut cw, frame.0, &frame.1).await?; + } + } + }; + tokio::try_join!(to_server, to_client)?; + Ok(()) +} + +async fn commit_fault_case(fault: Fault) { + let (direct, recovery, _schema, url) = fixture().await; + let proxy = PgCommitFaultProxy::start(&url, fault).await; + let mut options = sea_orm::ConnectOptions::new(proxy.url.clone()); + options.max_connections(1).min_connections(1); + let proxied = Database::connect(options).await.unwrap(); + let repository = PostgresMetadataInstallRepository::new(proxied) + .await + .unwrap(); + let prepared = prepared("/"); + let intent = repository + .begin_intent("commit-fault", &prepared) + .await + .unwrap(); + install(&repository, &intent, &prepared).await; + proxy.armed.store(true, Ordering::SeqCst); + let error = tokio::time::timeout(Duration::from_secs(15), repository.finalize(&intent)) + .await + .unwrap() + .unwrap_err(); + assert!(matches!( + error, + MetadataInstallError::CommitUncertain { + phase: MetadataCommitPhase::Finalize, + .. + } + )); + proxy.wait_for_fault().await; + let observed = repository + .inspect_prepare( + &recovery, + intent.operation_id(), + intent.manifest_digest(), + MetadataCommitPhase::Finalize, + ) + .await + .unwrap(); + match fault { + Fault::AfterCommit => { + assert!(proxy.commit_observed.load(Ordering::SeqCst)); + assert!(matches!(observed, MetadataPrepareObservation::Committed(_))); + assert_eq!( + mst2_retention_root::Entity::find() + .count(&direct) + .await + .unwrap(), + 2 + ); + assert_eq!( + mst2_retention_edge::Entity::find() + .count(&direct) + .await + .unwrap(), + 1 + ); + let restarted = PostgresMetadataInstallRepository::new(direct.clone()) + .await + .unwrap(); + restarted.finalize(&intent).await.unwrap(); + assert_eq!( + mst2_retention_edge::Entity::find() + .count(&direct) + .await + .unwrap(), + 1 + ); + } + Fault::BeforeCommit => { + assert!(!proxy.commit_observed.load(Ordering::SeqCst)); + assert!(matches!(observed, MetadataPrepareObservation::Preparing(_))); + assert_eq!( + mst2_retention_root::Entity::find() + .count(&direct) + .await + .unwrap(), + 0 + ); + assert_eq!( + mst2_retention_node::Entity::find() + .count(&direct) + .await + .unwrap(), + 0 + ); + PostgresMetadataInstallRepository::new(direct) + .await + .unwrap() + .finalize(&intent) + .await + .unwrap(); + } + Fault::PrepareRead => panic!("query fault uses its independent recovery test"), + } +} + +#[tokio::test] +async fn native_metadata_real_commit_response_loss_recovers_receipt_without_releasing_pin() { + commit_fault_case(Fault::AfterCommit).await; +} + +#[tokio::test] +async fn native_metadata_real_connection_loss_before_commit_recovers_preparing_without_false_success() + { + commit_fault_case(Fault::BeforeCommit).await; +} + +#[tokio::test] +async fn native_metadata_wrong_schema_primary_cannot_report_absent_for_another_committed_operation() +{ + let (first, second, _schema, _url) = fixture().await; + let (wrong_primary, _other, _wrong_schema, _wrong_url) = fixture().await; + let repository = PostgresMetadataInstallRepository::new(first.clone()) + .await + .unwrap(); + let prepared = prepared("/"); + let intent = repository.begin_intent("scope", &prepared).await.unwrap(); + install(&repository, &intent, &prepared).await; + repository.finalize(&intent).await.unwrap(); + assert!(matches!( + repository + .inspect_prepare( + &wrong_primary, + intent.operation_id(), + intent.manifest_digest(), + MetadataCommitPhase::Finalize + ) + .await + .unwrap_err(), + MetadataInstallError::CommitUncertain { .. } + )); + assert_eq!( + mst2_retention_root::Entity::find() + .count(&second) + .await + .unwrap(), + 2 + ); + assert_eq!( + mst2_metadata_prepare::Entity::find() + .count(&wrong_primary) + .await + .unwrap(), + 0 + ); + assert!(matches!( + repository + .inspect_prepare( + &second, + intent.operation_id(), + intent.manifest_digest(), + MetadataCommitPhase::Finalize + ) + .await + .unwrap(), + MetadataPrepareObservation::Committed(_) + )); +} + +#[tokio::test] +async fn native_metadata_query_loss_after_recovery_barrier_preserves_typed_unknown() { + let (first, second, _schema, url) = fixture().await; + let repository = PostgresMetadataInstallRepository::new(first.clone()) + .await + .unwrap(); + let prepared = prepared("/"); + let intent = repository + .begin_intent("read-fault", &prepared) + .await + .unwrap(); + install(&repository, &intent, &prepared).await; + repository.finalize(&intent).await.unwrap(); + let proxy = PgCommitFaultProxy::start(&url, Fault::PrepareRead).await; + let mut options = sea_orm::ConnectOptions::new(proxy.url.clone()); + options.max_connections(1).min_connections(1); + let fresh = Database::connect(options).await.unwrap(); + proxy.armed.store(true, Ordering::SeqCst); + assert!(matches!( + repository + .inspect_prepare( + &fresh, + intent.operation_id(), + intent.manifest_digest(), + MetadataCommitPhase::Finalize + ) + .await + .unwrap_err(), + MetadataInstallError::CommitUncertain { .. } + )); + proxy.wait_for_fault().await; + assert!(!proxy.commit_observed.load(Ordering::SeqCst)); + assert_eq!( + mst2_retention_root::Entity::find() + .count(&second) + .await + .unwrap(), + 2 + ); + assert!(matches!( + repository + .inspect_prepare( + &second, + intent.operation_id(), + intent.manifest_digest(), + MetadataCommitPhase::Finalize + ) + .await + .unwrap(), + MetadataPrepareObservation::Committed(_) + )); +} + +#[path = "native_metadata_install_capability_fault_tests.rs"] +mod capability_fault_tests; diff --git a/src/jupiter/storage/native_metadata_qualified.rs b/src/jupiter/storage/native_metadata_qualified.rs new file mode 100644 index 00000000..221f9c98 --- /dev/null +++ b/src/jupiter/storage/native_metadata_qualified.rs @@ -0,0 +1,1035 @@ +//! Qualified metadata incarnations. Runtime serving adoption is intentionally closed. + +use std::ops::Deref; + +use super::*; + +pub struct PostgresQualifiedMetadataRepository { + inner: PostgresMetadataGenerationRepository, +} + +impl Deref for PostgresQualifiedMetadataRepository { + type Target = PostgresMetadataGenerationRepository; + + fn deref(&self) -> &Self::Target { + &self.inner + } +} + +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct MetadataLifetime { + page_id: [u8; 32], + generation: i64, + metadata_codec: i16, + expected_size: i32, + primary_scope: Box<[u8]>, +} + +impl MetadataLifetime { + pub fn page_id(&self) -> [u8; 32] { + self.page_id + } + + pub fn generation(&self) -> i64 { + self.generation + } +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum MetadataGcPhase { + Claim, + Apply, +} + +#[derive(Debug, thiserror::Error)] +pub enum MetadataGcError { + #[error(transparent)] + Rejected(#[from] SnapshotError), + #[error("metadata GC {phase:?} commit outcome is unknown for {operation_id}")] + CommitUncertain { + operation_id: String, + page_id: [u8; 32], + generation: i64, + phase: MetadataGcPhase, + }, +} + +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct MetadataGcClaim { + operation_id: String, + lifetime: MetadataLifetime, + graph_present: bool, + had_payload: bool, +} + +impl MetadataGcClaim { + pub fn operation_id(&self) -> &str { + &self.operation_id + } + + pub fn lifetime(&self) -> &MetadataLifetime { + &self.lifetime + } +} + +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct MetadataGcReceipt { + claim: MetadataGcClaim, + created_at: sea_orm::prelude::DateTimeWithTimeZone, + completed_at: sea_orm::prelude::DateTimeWithTimeZone, +} + +impl MetadataGcReceipt { + pub fn claim(&self) -> &MetadataGcClaim { + &self.claim + } +} + +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum MetadataGcObservation { + Absent, + Pending(Box), + Applied(Box), +} + +struct GcRecord { + claim: MetadataGcClaim, + state: String, + created_at: sea_orm::prelude::DateTimeWithTimeZone, + completed_at: Option, +} + +impl GcRecord { + fn receipt(self) -> Result { + if self.state != "APPLIED" { + return Err(unavailable("metadata GC operation is not APPLIED")); + } + Ok(MetadataGcReceipt { + claim: self.claim, + created_at: self.created_at, + completed_at: self + .completed_at + .ok_or_else(|| integrity("APPLIED metadata GC receipt has no timestamp"))?, + }) + } +} + +impl PostgresQualifiedMetadataRepository { + #[cfg(test)] + pub(crate) async fn registered_shadow( + connection: DatabaseConnection, + family: crate::jupiter::storage::qualified_metadata_family::VerifiedQualifiedNamespace, + ) -> Result { + let mut inner = PostgresMetadataInstallRepository::new(connection).await?; + inner.qualified_family = Some(family); + Ok(Self { + inner: PostgresMetadataGenerationRepository { + inner, + graph_domain: "qualified-v1", + }, + }) + } + + #[cfg(test)] + pub async fn new(connection: DatabaseConnection) -> Result { + Ok(Self { + inner: PostgresMetadataGenerationRepository { + inner: PostgresMetadataInstallRepository::new(connection).await?, + graph_domain: "qualified-v1", + }, + }) + } + + pub async fn lifetime(&self, page_id: [u8; 32]) -> Result { + self.inner + .inner + .verify_primary_connection(&self.inner.inner.connection) + .await?; + let row = self + .inner + .inner + .connection + .query_one_raw(statement( + "SELECT l.generation,l.metadata_codec,l.expected_size FROM mst2_metadata_current c + JOIN mst2_metadata_lifetime l USING(page_id,generation) + WHERE c.page_id=$1 AND l.graph_domain='qualified-v1'", + [page_id.to_vec().into()], + )) + .await + .map_err(internal)? + .ok_or_else(|| unavailable("qualified lifetime is missing"))?; + Ok(MetadataLifetime { + page_id, + generation: row.try_get("", "generation").map_err(internal)?, + metadata_codec: row.try_get("", "metadata_codec").map_err(internal)?, + expected_size: row.try_get("", "expected_size").map_err(internal)?, + primary_scope: self.inner.primary_scope()?.into_boxed_slice(), + }) + } + + /// Read-only recovery selector; claim still requires this tuple to be current. + pub async fn historical_lifetime( + &self, + page_id: [u8; 32], + generation: i64, + ) -> Result { + if generation <= 0 { + return Err(integrity("historical generation must be positive")); + } + self.inner + .inner + .verify_primary_connection(&self.inner.inner.connection) + .await?; + let row = self + .inner + .inner + .connection + .query_one_raw(statement( + "SELECT metadata_codec,expected_size FROM mst2_metadata_lifetime + WHERE page_id=$1 AND generation=$2 AND graph_domain='qualified-v1'", + [page_id.to_vec().into(), generation.into()], + )) + .await + .map_err(internal)? + .ok_or_else(|| unavailable("qualified historical incarnation is missing"))?; + Ok(MetadataLifetime { + page_id, + generation, + metadata_codec: row.try_get("", "metadata_codec").map_err(internal)?, + expected_size: row.try_get("", "expected_size").map_err(internal)?, + primary_scope: self.inner.primary_scope()?.into_boxed_slice(), + }) + } + + fn require_lifetime_scope(&self, lifetime: &MetadataLifetime) -> Result<(), SnapshotError> { + if lifetime.primary_scope.as_ref() != self.inner.primary_scope()?.as_slice() { + return Err(integrity( + "metadata GC belongs to another captured primary scope", + )); + } + Ok(()) + } + + /// Bounded indexed discovery only. Every returned tuple still needs a full claim proof. + pub async fn scan_current_lifetimes( + &self, + after: Option<([u8; 32], i64)>, + limit: u16, + ) -> Result, SnapshotError> { + if !(1..=256).contains(&limit) { + return Err(integrity("metadata scan limit must be 1..=256")); + } + self.inner + .inner + .verify_primary_connection(&self.inner.inner.connection) + .await?; + let (page, generation) = after.map(|(p, g)| (p.to_vec(), g)).unwrap_or_default(); + let rows=self.inner.inner.connection.query_all_raw(statement( + "SELECT l.page_id,l.generation,l.metadata_codec,l.expected_size FROM mst2_metadata_lifetime l + JOIN mst2_metadata_current c USING(page_id,generation) + WHERE l.graph_domain='qualified-v1' AND l.state IN ('RESERVED','LIVE') AND (l.page_id,l.generation)>($1,$2) + ORDER BY l.page_id,l.generation LIMIT $3", + [page.into(),generation.into(),i64::from(limit).into()], + )).await.map_err(internal)?; + let scope = self.inner.primary_scope()?.into_boxed_slice(); + rows.into_iter() + .map(|row| { + let page: Vec = row.try_get("", "page_id").map_err(internal)?; + Ok(MetadataLifetime { + page_id: page.as_slice().try_into().map_err(internal)?, + generation: row.try_get("", "generation").map_err(internal)?, + metadata_codec: row.try_get("", "metadata_codec").map_err(internal)?, + expected_size: row.try_get("", "expected_size").map_err(internal)?, + primary_scope: scope.clone(), + }) + }) + .collect() + } + + pub async fn claim( + &self, + operation_id: &str, + lifetime: &MetadataLifetime, + ) -> Result { + validate_gc_operation(operation_id)?; + self.require_lifetime_scope(lifetime)?; + let txn = self.inner.inner.transaction().await?; + let result = self.claim_in_txn(&txn, operation_id, lifetime).await; + finish_gc_transaction(txn, result, operation_id, lifetime, MetadataGcPhase::Claim).await + } + + async fn claim_in_txn( + &self, + txn: &DatabaseTransaction, + operation_id: &str, + lifetime: &MetadataLifetime, + ) -> Result { + self.require_lifetime_scope(lifetime)?; + self.inner.inner.barrier(txn).await?; + if let Some(record) = load_gc(txn, operation_id).await? { + if &record.claim.lifetime != lifetime { + return Err(integrity( + "GC operation cannot be retargeted to another lifetime", + )); + } + return Ok(record.claim); + } + let row=txn.query_one_raw(statement( + "SELECT EXISTS(SELECT 1 FROM mst2_metadata_graph_node n WHERE n.page_id=l.page_id AND n.generation=l.generation) AS graph_present, + EXISTS(SELECT 1 FROM mst2_metadata_payload b WHERE b.page_id=l.page_id) AS had_payload + FROM mst2_metadata_current c JOIN mst2_metadata_lifetime l USING(page_id,generation) + WHERE l.page_id=$1 AND l.generation=$2 AND l.graph_domain='qualified-v1' + AND l.metadata_codec=$3 AND l.expected_size=$4 FOR UPDATE OF c,l", + [lifetime.page_id.to_vec().into(),lifetime.generation.into(),lifetime.metadata_codec.into(),lifetime.expected_size.into()], + )).await.map_err(internal)?.ok_or_else(|| unavailable("GC candidate no longer names its captured current incarnation"))?; + let graph_present: bool = row.try_get("", "graph_present").map_err(internal)?; + let had_payload: bool = row.try_get("", "had_payload").map_err(internal)?; + txn.execute_raw(statement( + "INSERT INTO mst2_metadata_gc_op(operation_id,page_id,generation,primary_scope,graph_domain, + metadata_codec,expected_size,graph_present,had_payload,state) + VALUES($1::uuid,$2,$3,$4,'qualified-v1',$5,$6,$7,$8,'PENDING')", + [operation_id.into(),lifetime.page_id.to_vec().into(),lifetime.generation.into(), + lifetime.primary_scope.to_vec().into(),lifetime.metadata_codec.into(),lifetime.expected_size.into(), + graph_present.into(),had_payload.into()], + )).await.map_err(internal)?; + Ok(load_gc(txn, operation_id) + .await? + .ok_or_else(|| integrity("claimed GC operation disappeared"))? + .claim) + } + + pub async fn apply( + &self, + claim: &MetadataGcClaim, + ) -> Result { + self.require_lifetime_scope(&claim.lifetime)?; + let txn = self.inner.inner.transaction().await?; + let result = self.apply_in_txn(&txn, claim).await; + finish_gc_transaction( + txn, + result, + &claim.operation_id, + &claim.lifetime, + MetadataGcPhase::Apply, + ) + .await + } + + async fn apply_in_txn( + &self, + txn: &DatabaseTransaction, + claim: &MetadataGcClaim, + ) -> Result { + self.require_lifetime_scope(&claim.lifetime)?; + self.inner.inner.barrier(txn).await?; + let record = load_gc(txn, &claim.operation_id) + .await? + .ok_or_else(|| unavailable("metadata GC operation is missing"))?; + if &record.claim != claim { + return Err(integrity( + "metadata GC claim differs from immutable operation", + )); + } + if record.state == "APPLIED" { + return record.receipt(); + } + txn.query_one_raw(statement( + "SELECT mst2_metadata_gc_apply($1::uuid)", + [claim.operation_id.clone().into()], + )) + .await + .map_err(internal)?; + load_gc(txn, &claim.operation_id) + .await? + .ok_or_else(|| integrity("applied metadata GC operation disappeared"))? + .receipt() + } + + /// Absence is definitive only after a fresh same-primary completion barrier. + pub async fn inspect_gc( + &self, + fresh_primary: &DatabaseConnection, + operation_id: &str, + lifetime: &MetadataLifetime, + phase: MetadataGcPhase, + ) -> Result { + validate_gc_operation(operation_id)?; + self.require_lifetime_scope(lifetime)?; + let txn = fresh_primary + .begin_with_config(Some(IsolationLevel::ReadCommitted), None) + .await + .map_err(|_| gc_uncertain(operation_id, lifetime, phase))?; + if self.inner.inner.barrier(&txn).await.is_err() { + let _ = txn.rollback().await; + return Err(gc_uncertain(operation_id, lifetime, phase)); + } + let result = async { + let Some(record) = load_gc(&txn, operation_id).await? else { + return Ok(MetadataGcObservation::Absent); + }; + if &record.claim.lifetime != lifetime { + return Err(integrity( + "GC inspection differs from its captured lifetime", + )); + } + match record.state.as_str() { + "PENDING" => Ok(MetadataGcObservation::Pending(Box::new(record.claim))), + "APPLIED" => Ok(MetadataGcObservation::Applied(Box::new(record.receipt()?))), + _ => Err(integrity("unknown metadata GC operation state")), + } + } + .await; + txn.rollback() + .await + .map_err(|_| gc_uncertain(operation_id, lifetime, phase))?; + result.map_err(|error: SnapshotError| { + if error.code == SnapshotErrorCode::Internal { + gc_uncertain(operation_id, lifetime, phase) + } else { + MetadataGcError::Rejected(error) + } + }) + } + + pub async fn pending_gc(&self, limit: u16) -> Result, SnapshotError> { + if !(1..=256).contains(&limit) { + return Err(integrity("metadata pending GC limit must be 1..=256")); + } + self.inner + .inner + .verify_primary_connection(&self.inner.inner.connection) + .await?; + let rows=self.inner.inner.connection.query_all_raw(statement( + "SELECT operation_id::text,page_id,generation,primary_scope,metadata_codec,expected_size, + graph_present,had_payload,state,created_at,completed_at FROM mst2_metadata_gc_op WHERE state='PENDING' + ORDER BY created_at,operation_id LIMIT $1", + [i64::from(limit).into()], + )).await.map_err(internal)?; + let mut claims = Vec::with_capacity(rows.len()); + for row in rows { + let record = gc_record(row)?; + self.require_lifetime_scope(&record.claim.lifetime)?; + claims.push(record.claim); + } + Ok(claims) + } + + pub async fn begin_fresh_intent( + &self, + operation_id: &str, + prepared: &PreparedNativeMetadataRetention, + applied: &[MetadataGcReceipt], + ) -> Result { + if applied.is_empty() || applied.len() > MetadataDagLimits::default().nodes { + return Err(integrity("fresh begin requires bounded exact APPLIED receipts").into()); + } + for receipt in applied { + self.require_lifetime_scope(&receipt.claim.lifetime)?; + } + self.inner + .begin_with_reopens(operation_id, prepared, applied) + .await + } +} + +fn validate_gc_operation(operation: &str) -> Result<(), SnapshotError> { + let uuid = uuid::Uuid::parse_str(operation).map_err(internal)?; + if uuid.get_version_num() != 4 || uuid.to_string() != operation { + return Err(integrity("metadata GC operation must be canonical UUID v4")); + } + Ok(()) +} + +fn gc_uncertain( + operation: &str, + lifetime: &MetadataLifetime, + phase: MetadataGcPhase, +) -> MetadataGcError { + MetadataGcError::CommitUncertain { + operation_id: operation.into(), + page_id: lifetime.page_id, + generation: lifetime.generation, + phase, + } +} + +async fn finish_gc_transaction( + txn: DatabaseTransaction, + result: Result, + operation: &str, + lifetime: &MetadataLifetime, + phase: MetadataGcPhase, +) -> Result { + match result { + Ok(value) => { + txn.commit() + .await + .map_err(|_| gc_uncertain(operation, lifetime, phase))?; + Ok(value) + } + Err(error) => { + txn.rollback().await.map_err(internal)?; + Err(error.into()) + } + } +} + +async fn load_gc( + connection: &C, + operation: &str, +) -> Result, SnapshotError> { + let Some(row)=connection.query_one_raw(statement( + "SELECT operation_id::text,page_id,generation,primary_scope,metadata_codec,expected_size, + graph_present,had_payload,state,created_at,completed_at FROM mst2_metadata_gc_op WHERE operation_id=$1::uuid", + [operation.into()], + )).await.map_err(internal)? else { return Ok(None); }; + Ok(Some(gc_record(row)?)) +} + +fn gc_record(row: QueryResult) -> Result { + let page: Vec = row.try_get("", "page_id").map_err(internal)?; + Ok(GcRecord { + claim: MetadataGcClaim { + operation_id: row.try_get("", "operation_id").map_err(internal)?, + lifetime: MetadataLifetime { + page_id: page.as_slice().try_into().map_err(internal)?, + generation: row.try_get("", "generation").map_err(internal)?, + metadata_codec: row.try_get("", "metadata_codec").map_err(internal)?, + expected_size: row.try_get("", "expected_size").map_err(internal)?, + primary_scope: row + .try_get::>("", "primary_scope") + .map_err(internal)? + .into_boxed_slice(), + }, + graph_present: row.try_get("", "graph_present").map_err(internal)?, + had_payload: row.try_get("", "had_payload").map_err(internal)?, + }, + state: row.try_get("", "state").map_err(internal)?, + created_at: row.try_get("", "created_at").map_err(internal)?, + completed_at: row.try_get("", "completed_at").map_err(internal)?, + }) +} + +async fn validate_applied_receipts( + connection: &C, + receipts: &[MetadataGcReceipt], +) -> Result<(), SnapshotError> { + if receipts.is_empty() { + return Ok(()); + } + let operations: Vec<_> = receipts + .iter() + .map(|r| r.claim.operation_id.as_str()) + .collect(); + let rows=connection.query_all_raw(statement( + "SELECT operation_id::text,page_id,generation,primary_scope,metadata_codec,expected_size, + graph_present,had_payload,state,created_at,completed_at FROM mst2_metadata_gc_op + WHERE operation_id IN (SELECT jsonb_array_elements_text($1::jsonb)::uuid) LIMIT 4097", + [serde_json::to_string(&operations).map_err(internal)?.into()], + )).await.map_err(internal)?; + let mut actual = BTreeMap::new(); + for row in rows { + let record = gc_record(row)?; + actual.insert(record.claim.operation_id.clone(), record.receipt()?); + } + if actual.len() != receipts.len() + || receipts + .iter() + .any(|r| actual.get(&r.claim.operation_id) != Some(r)) + { + return Err(integrity( + "fresh proof differs from exact immutable APPLIED operation and timestamps", + )); + } + Ok(()) +} + +pub(super) async fn reopen_in_txn( + txn: &DatabaseTransaction, + plan: &MetadataInstallPlan, + receipts: &[MetadataGcReceipt], +) -> Result<(), SnapshotError> { + let mut pages = BTreeSet::new(); + validate_applied_receipts(txn, receipts).await?; + for receipt in receipts { + let lifetime = &receipt.claim.lifetime; + if !pages.insert(lifetime.page_id) + || plan.pages.get(&lifetime.page_id).copied() != Some(lifetime.expected_size as u64) + || plan.identity.metadata_codec as i16 != lifetime.metadata_codec + { + return Err(integrity( + "fresh receipts are not unique exact members of the new plan", + )); + } + lifetime + .generation + .checked_add(1) + .ok_or_else(|| unavailable("metadata generation watermark exhausted"))?; + } + let operations: Vec<_> = receipts + .iter() + .map(|r| r.claim.operation_id.as_str()) + .collect(); + let encoded = serde_json::to_string(&operations).map_err(internal)?; + txn.execute_raw(statement( + "INSERT INTO mst2_metadata_lifetime(page_id,node_id,generation,state,metadata_codec,expected_size,graph_domain) + SELECT o.page_id,'page:sha256:'||encode(o.page_id,'hex'),o.generation+1,'RESERVED',o.metadata_codec,o.expected_size,'qualified-v1' + FROM mst2_metadata_gc_op o WHERE operation_id IN (SELECT jsonb_array_elements_text($1::jsonb)::uuid) + ORDER BY o.page_id ON CONFLICT(page_id,generation) DO NOTHING", + [encoded.clone().into()], + )).await.map_err(internal)?; + let changed=txn.execute_raw(statement( + "UPDATE mst2_metadata_current c SET generation=o.generation+1 FROM mst2_metadata_gc_op o + WHERE o.operation_id IN (SELECT jsonb_array_elements_text($1::jsonb)::uuid) + AND c.page_id=o.page_id AND c.generation=o.generation", + [encoded.into()], + )).await.map_err(internal)?; + if changed.rows_affected() != receipts.len() as u64 { + return Err(unavailable( + "fresh receipts do not name all current removed incarnations", + )); + } + Ok(()) +} + +pub(super) async fn verify_reopen_replay( + connection: &C, + fixed: &FixedPlan, + receipts: &[MetadataGcReceipt], +) -> Result<(), SnapshotError> { + validate_applied_receipts(connection, receipts).await?; + let mut pages = BTreeSet::new(); + for receipt in receipts { + let old = &receipt.claim.lifetime; + if !pages.insert(old.page_id) + || old.primary_scope != fixed.intent.primary_scope + || old.generation.checked_add(1) != fixed.bindings.0.get(&old.page_id).map(|b| b.0) + || Some(old.expected_size as u64) != fixed.bindings.0.get(&old.page_id).map(|b| b.1) + { + return Err(integrity( + "fresh replay receipts differ from the fixed new incarnation bindings", + )); + } + } + Ok(()) +} + +pub(super) async fn allocate_lifetimes( + txn: &DatabaseTransaction, + plan: &MetadataInstallPlan, +) -> Result { + let pages: Vec<_> = plan + .pages + .iter() + .map(|(id, size)| json!({"id":hex::encode(id),"size":size})) + .collect(); + let encoded = serde_json::to_string(&pages).map_err(internal)?; + if txn.query_one_raw(statement( + "SELECT l.page_id FROM jsonb_to_recordset($1::jsonb) p(id text,size integer) + JOIN mst2_metadata_current c ON c.page_id=decode(p.id,'hex') + JOIN mst2_metadata_lifetime l USING(page_id,generation) WHERE l.graph_domain<>'qualified-v1' LIMIT 1", + [encoded.clone().into()], + )).await.map_err(internal)?.is_some() { + return Err(unavailable("metadata incarnation permanently belongs to another graph domain")); + } + txn.execute_raw(statement( + "INSERT INTO mst2_metadata_lifetime(page_id,node_id,generation,state,metadata_codec,expected_size,graph_domain) + SELECT decode(p.id,'hex'),'page:sha256:'||p.id,1,'RESERVED',$1,p.size,'qualified-v1' + FROM jsonb_to_recordset($2::jsonb) p(id text,size integer) + WHERE NOT EXISTS(SELECT 1 FROM mst2_metadata_current c WHERE c.page_id=decode(p.id,'hex')) + ON CONFLICT(page_id,generation) DO NOTHING", + [(plan.identity.metadata_codec as i16).into(),encoded.clone().into()], + )).await.map_err(internal)?; + txn.execute_raw(statement( + "INSERT INTO mst2_metadata_current(page_id,generation) SELECT decode(p.id,'hex'),1 + FROM jsonb_to_recordset($1::jsonb) p(id text,size integer) + WHERE NOT EXISTS(SELECT 1 FROM mst2_metadata_current c WHERE c.page_id=decode(p.id,'hex'))", + [encoded.clone().into()], + )).await.map_err(internal)?; + let rows=txn.query_all_raw(statement( + "SELECT l.page_id,l.generation,l.state,l.graph_domain,l.metadata_codec,l.expected_size,p.size, + n.state AS graph_state,n.metadata_codec AS graph_codec,n.bytes AS graph_bytes, + b.page_id IS NOT NULL AS payload_present,b.generation AS payload_generation,b.metadata_codec AS payload_codec,b.byte_size AS payload_size, + EXISTS(SELECT 1 FROM mst2_metadata_gc_op o WHERE o.page_id=l.page_id AND o.generation=l.generation) AS tombstone, + mst2_metadata_has_generic_overlap(l.page_id) AS generic_graph + FROM jsonb_to_recordset($1::jsonb) p(id text,size integer) + JOIN mst2_metadata_current c ON c.page_id=decode(p.id,'hex') + JOIN mst2_metadata_lifetime l USING(page_id,generation) + LEFT JOIN mst2_metadata_graph_node n USING(page_id,generation) + LEFT JOIN mst2_metadata_payload b ON b.page_id=l.page_id ORDER BY l.page_id", + [encoded.into()], + )).await.map_err(internal)?; + let mut bindings = BTreeMap::new(); + for row in rows { + let state: String = row.try_get("", "state").map_err(internal)?; + let generation: i64 = row.try_get("", "generation").map_err(internal)?; + let size: i32 = row.try_get("", "size").map_err(internal)?; + let graph: Option = row.try_get("", "graph_state").map_err(internal)?; + let payload: bool = row.try_get("", "payload_present").map_err(internal)?; + if !["RESERVED", "LIVE"].contains(&state.as_str()) + || row.try_get::("", "tombstone").map_err(internal)? + || row.try_get::("", "generic_graph").map_err(internal)? + || graph.as_deref().is_some_and(|s| s != "LIVE") + { + return Err(unavailable( + "qualified incarnation is removed, deleting, or has generic evidence", + )); + } + if state == "LIVE" && (!payload || graph.is_none()) { + return Err(unavailable( + "LIVE qualified incarnation lost bytes or graph", + )); + } + if row + .try_get::("", "graph_domain") + .map_err(internal)? + != "qualified-v1" + || row.try_get::("", "metadata_codec").map_err(internal)? + != plan.identity.metadata_codec as i16 + || row.try_get::("", "expected_size").map_err(internal)? != size + || (graph.is_some() + && (row + .try_get::>("", "graph_codec") + .map_err(internal)? + != Some(plan.identity.metadata_codec as i16) + || row + .try_get::>("", "graph_bytes") + .map_err(internal)? + != Some(i64::from(size)))) + || (payload + && (row + .try_get::>("", "payload_generation") + .map_err(internal)? + != Some(generation) + || row + .try_get::>("", "payload_codec") + .map_err(internal)? + != Some(plan.identity.metadata_codec as i16) + || row + .try_get::>("", "payload_size") + .map_err(internal)? + != Some(size))) + { + return Err(integrity( + "qualified incarnation immutable profile mismatch", + )); + } + let page: Vec = row.try_get("", "page_id").map_err(internal)?; + bindings.insert( + page.as_slice().try_into().map_err(internal)?, + (generation, size as u64), + ); + } + if bindings.len() != plan.pages.len() { + return Err(integrity("qualified lifetime allocation is incomplete")); + } + Ok(GenerationBindings(bindings)) +} + +pub(super) async fn retain_existing_roots( + txn: &DatabaseTransaction, + intent: &GenerationPrepareIntent, +) -> Result<(), SnapshotError> { + txn.execute_raw(statement( + "INSERT INTO mst2_metadata_graph_root(prepare_id,storage_seal,page_id,generation) + SELECT q.prepare_id,q.storage_seal,m.page_id,m.generation FROM mst2_metadata_prepare q + JOIN mst2_metadata_prepare_page m USING(prepare_id) + JOIN mst2_metadata_graph_node n USING(page_id,generation) WHERE q.prepare_id=$1 AND n.state='LIVE' + ON CONFLICT(prepare_id,page_id,generation) DO NOTHING", + [intent.prepare_id().into()], + )).await.map_err(internal)?; + Ok(()) +} + +pub(super) async fn check_lifetimes( + connection: &C, + stored: &StoredPlan, +) -> Result<(), SnapshotError> { + if connection.query_one_raw(statement( + "SELECT m.page_id FROM mst2_metadata_prepare_page m + LEFT JOIN mst2_metadata_current c ON c.page_id=m.page_id AND c.generation=m.generation + LEFT JOIN mst2_metadata_lifetime l ON l.page_id=m.page_id AND l.generation=m.generation + LEFT JOIN mst2_metadata_graph_node n ON n.page_id=m.page_id AND n.generation=m.generation + WHERE m.prepare_id=$1 AND (c.page_id IS NULL OR l.page_id IS NULL OR l.graph_domain<>'qualified-v1' + OR l.state NOT IN ('RESERVED','LIVE') OR l.metadata_codec<>$2 OR l.expected_size<>m.expected_size + OR ($3='COMMITTED' AND l.state<>'LIVE') OR (l.state='LIVE' AND n.page_id IS NULL) + OR n.state<>'LIVE' OR n.metadata_codec<>$2 OR n.bytes<>m.expected_size + OR EXISTS(SELECT 1 FROM mst2_metadata_gc_op o WHERE o.page_id=m.page_id AND o.generation=m.generation) + OR mst2_metadata_has_generic_overlap(l.page_id)) LIMIT 1", + [stored.record.prepare_id.clone().into(),stored.record.metadata_codec.into(),stored.record.state.clone().into()], + )).await.map_err(internal)?.is_some() { return Err(unavailable("fixed qualified incarnation is no longer installable")); } + Ok(()) +} + +fn validate_fixed_observation( + fixed: &FixedPlan, + dag: &ValidatedMetadataDag, +) -> Result<(), SnapshotError> { + let actual_pages: BTreeSet<_> = dag.payloads().iter().map(|p| (p.id, p.size)).collect(); + let actual_edges: BTreeSet<_> = dag + .edges() + .iter() + .map(|e| (e.parent.clone(), e.child.clone())) + .collect(); + if dag.root() != fixed.stored.plan.root + || actual_pages + != fixed + .stored + .plan + .pages + .iter() + .map(|(p, s)| (*p, *s)) + .collect() + || actual_edges + != fixed + .stored + .plan + .edges + .iter() + .map(|(p, c)| (node_id(p), node_id(c))) + .collect() + { + return Err(integrity( + "qualified DAG observation differs from immutable plan", + )); + } + Ok(()) +} + +pub(super) async fn finalize_canonical_graph( + txn: &DatabaseTransaction, + fixed: &FixedPlan, + dag: &ValidatedMetadataDag, +) -> Result { + validate_fixed_observation(fixed, dag)?; + if fixed.stored.record.state == "COMMITTED" { + verify_graph(txn, fixed).await?; + return fixed.stored.receipt(); + } + let mut pending: BTreeMap<_, usize> = fixed.stored.plan.pages.keys().map(|p| (*p, 0)).collect(); + let mut parents: BTreeMap<_, Vec<_>> = BTreeMap::new(); + for &(parent, child) in &fixed.stored.plan.edges { + *pending + .get_mut(&parent) + .ok_or_else(|| integrity("canonical parent is outside its fixed plan"))? += 1; + parents.entry(child).or_default().push(parent); + } + let mut ready: BTreeSet<_> = pending + .iter() + .filter_map(|(p, count)| (*count == 0).then_some(*p)) + .collect(); + let mut ordered = Vec::with_capacity(pending.len()); + while let Some(page) = ready.pop_first() { + ordered.push(json!({"page":hex::encode(page),"generation":fixed.bindings.0[&page].0})); + for parent in parents.get(&page).into_iter().flatten() { + let count = pending + .get_mut(parent) + .ok_or_else(|| integrity("canonical ancestor is outside its fixed plan"))?; + *count = count + .checked_sub(1) + .ok_or_else(|| integrity("canonical edge accounting underflow"))?; + if *count == 0 { + ready.insert(*parent); + } + } + } + if ordered.len() != pending.len() { + return Err(integrity("canonical certification order contains a cycle")); + } + let certified: i32 = txn + .query_one_raw(statement( + "SELECT mst2_metadata_certify_batch($1,$2::jsonb) AS certified", + [ + fixed.intent.prepare_id().into(), + serde_json::to_string(&ordered).map_err(internal)?.into(), + ], + )) + .await + .map_err(internal)? + .ok_or_else(|| integrity("canonical certification batch is missing"))? + .try_get("", "certified") + .map_err(internal)?; + if certified as usize != ordered.len() { + return Err(integrity( + "canonical certification did not cover its exact cold DAG", + )); + } + retain_existing_roots(txn, &fixed.intent).await?; + verify_graph(txn, fixed).await?; + let changed = txn + .execute_raw(statement( + "UPDATE mst2_metadata_prepare SET state='COMMITTED',committed_at=clock_timestamp() + WHERE prepare_id=$1 AND state='PREPARING' AND storage_seal=$2", + [ + fixed.intent.prepare_id().into(), + fixed.intent.storage_seal.to_vec().into(), + ], + )) + .await + .map_err(internal)?; + if changed.rows_affected() != 1 { + return Err(integrity("canonical finalize lost its prepare CAS")); + } + fixed.stored.receipt() +} + +pub(super) async fn finalize_graph( + txn: &DatabaseTransaction, + fixed: &FixedPlan, + dag: &ValidatedMetadataDag, +) -> Result { + validate_fixed_observation(fixed, dag)?; + if fixed.stored.record.state == "COMMITTED" { + verify_graph(txn, fixed).await?; + return fixed.stored.receipt(); + } + txn.execute_raw(statement( + "INSERT INTO mst2_metadata_graph_node(page_id,generation,state,metadata_codec,bytes) + SELECT m.page_id,m.generation,'LIVE',$2,m.expected_size FROM mst2_metadata_prepare_page m + WHERE m.prepare_id=$1 AND NOT EXISTS(SELECT 1 FROM mst2_metadata_graph_node n WHERE n.page_id=m.page_id AND n.generation=m.generation)", + [fixed.intent.prepare_id().into(),fixed.stored.record.metadata_codec.into()], + )).await.map_err(internal)?; + let edges:Vec<_>=fixed.stored.plan.edges.iter().map(|(p,c)|json!({"parent":hex::encode(p),"pg":fixed.bindings.0[p].0,"child":hex::encode(c),"cg":fixed.bindings.0[c].0})).collect(); + txn.execute_raw(statement( + "INSERT INTO mst2_metadata_graph_edge(parent_page,parent_generation,child_page,child_generation) + SELECT decode(e.parent,'hex'),e.pg,decode(e.child,'hex'),e.cg + FROM jsonb_to_recordset($1::jsonb) e(parent text,pg bigint,child text,cg bigint) + ON CONFLICT(parent_page,parent_generation,child_page,child_generation) DO NOTHING", + [serde_json::to_string(&edges).map_err(internal)?.into()], + )).await.map_err(internal)?; + retain_existing_roots(txn, &fixed.intent).await?; + verify_graph(txn, fixed).await?; + let changed = txn + .execute_raw(statement( + "UPDATE mst2_metadata_prepare SET state='COMMITTED',committed_at=clock_timestamp() + WHERE prepare_id=$1 AND state='PREPARING' AND storage_seal=$2", + [ + fixed.intent.prepare_id().into(), + fixed.intent.storage_seal.to_vec().into(), + ], + )) + .await + .map_err(internal)?; + if changed.rows_affected() != 1 { + return Err(integrity("qualified finalize lost its prepare CAS")); + } + fixed.stored.receipt() +} + +pub(super) async fn verify_graph( + connection: &C, + fixed: &FixedPlan, +) -> Result<(), SnapshotError> { + let rows=connection.query_all_raw(statement( + "SELECT n.page_id,n.generation,n.state,n.metadata_codec,n.bytes,n.incoming_refs, + (SELECT count(*) FROM mst2_metadata_graph_edge e WHERE e.child_page=n.page_id AND e.child_generation=n.generation) AS actual_refs + FROM mst2_metadata_prepare_page m JOIN mst2_metadata_graph_node n USING(page_id,generation) + WHERE m.prepare_id=$1 ORDER BY n.page_id LIMIT 4097", + [fixed.intent.prepare_id().into()], + )).await.map_err(internal)?; + if rows.len() != fixed.bindings.0.len() { + return Err(unavailable("qualified graph coverage is incomplete")); + } + for row in rows { + let page: Vec = row.try_get("", "page_id").map_err(internal)?; + let page: [u8; 32] = page.as_slice().try_into().map_err(internal)?; + let &(generation, size) = fixed + .bindings + .0 + .get(&page) + .ok_or_else(|| integrity("qualified graph has an extra node"))?; + if row.try_get::("", "generation").map_err(internal)? != generation + || row.try_get::("", "state").map_err(internal)? != "LIVE" + || row.try_get::("", "metadata_codec").map_err(internal)? + != fixed.stored.record.metadata_codec + || row.try_get::("", "bytes").map_err(internal)? != size as i64 + || row.try_get::("", "incoming_refs").map_err(internal)? + != row.try_get::("", "actual_refs").map_err(internal)? + { + return Err(integrity("qualified graph profile or counter drift")); + } + } + let rows=connection.query_all_raw(statement( + "SELECT e.parent_page,e.parent_generation,e.child_page,e.child_generation FROM mst2_metadata_prepare_page m + JOIN mst2_metadata_graph_edge e ON e.parent_page=m.page_id AND e.parent_generation=m.generation + WHERE m.prepare_id=$1 LIMIT 16385", + [fixed.intent.prepare_id().into()], + )).await.map_err(internal)?; + let mut actual = BTreeSet::new(); + for row in rows { + let parent: Vec = row.try_get("", "parent_page").map_err(internal)?; + let child: Vec = row.try_get("", "child_page").map_err(internal)?; + actual.insert(( + parent, + row.try_get::("", "parent_generation") + .map_err(internal)?, + child, + row.try_get::("", "child_generation") + .map_err(internal)?, + )); + } + let expected: BTreeSet<_> = fixed + .stored + .plan + .edges + .iter() + .map(|(p, c)| { + ( + p.to_vec(), + fixed.bindings.0[p].0, + c.to_vec(), + fixed.bindings.0[c].0, + ) + }) + .collect(); + if actual != expected { + return Err(integrity( + "qualified graph edges differ from immutable plan", + )); + } + verify_roots(connection, fixed, true).await +} + +pub(super) async fn verify_roots( + connection: &C, + fixed: &FixedPlan, + complete: bool, +) -> Result<(), SnapshotError> { + let rows=connection.query_all_raw(statement( + "SELECT page_id,generation,storage_seal FROM mst2_metadata_graph_root WHERE prepare_id=$1 LIMIT 4097", + [fixed.intent.prepare_id().into()], + )).await.map_err(internal)?; + let mut actual = BTreeSet::new(); + for row in rows { + let page: Vec = row.try_get("", "page_id").map_err(internal)?; + let page: [u8; 32] = page.as_slice().try_into().map_err(internal)?; + let generation: i64 = row.try_get("", "generation").map_err(internal)?; + let seal: Vec = row.try_get("", "storage_seal").map_err(internal)?; + if seal.as_slice() != fixed.intent.storage_seal + || fixed.bindings.0.get(&page).map(|b| b.0) != Some(generation) + { + return Err(integrity( + "qualified prepare root differs from its immutable seal", + )); + } + actual.insert((page, generation)); + } + if complete + && actual + != fixed + .bindings + .0 + .iter() + .map(|(p, (g, _))| (*p, *g)) + .collect() + { + return Err(unavailable( + "qualified prepare lost or transferred original roots", + )); + } + Ok(()) +} + +#[cfg(test)] +#[path = "native_metadata_qualified_tests.rs"] +mod tests; diff --git a/src/jupiter/storage/native_metadata_qualified_tests.rs b/src/jupiter/storage/native_metadata_qualified_tests.rs new file mode 100644 index 00000000..a7b73493 --- /dev/null +++ b/src/jupiter/storage/native_metadata_qualified_tests.rs @@ -0,0 +1,1261 @@ +use std::sync::Arc; + +use mst2_codec::metapage::{Entry, EntryKind}; +use sea_orm::Database; +use sea_orm_migration::MigratorTrait; + +use super::{ + super::history::{MetadataTerminalAction, MetadataTerminalError, MetadataTerminalObservation}, + *, +}; +use crate::{ + ceres::snapshot::retention_dag::MetadataDagBuilder, + jupiter::{ + migration::Migrator, + tests::{TestSchemaGuard, test_db_config}, + }, +}; + +fn prepared() -> PreparedNativeMetadataRetention { + let entries = [Entry::file(EntryKind::Regular, b"file", 3, [42; 32])]; + let child = Page::build(&entries).unwrap(); + let roots = [ + Entry::dir(b"one", page_id(&child)), + Entry::dir(b"two", page_id(&child)), + ]; + let root = Page::build(&roots).unwrap(); + let mut builder = MetadataDagBuilder::new(MetadataDagLimits::default()); + builder.add_directory(&child, &entries).unwrap(); + builder.add_directory(&root, &roots).unwrap(); + PreparedNativeMetadataRetention::test_installation( + Arc::new(builder.finish(page_id(&root)).unwrap()), + "/", + ) +} + +async fn fixture() -> (DatabaseConnection, DatabaseConnection, TestSchemaGuard) { + let temp = tempfile::tempdir().unwrap(); + let (config, schema) = test_db_config(temp.path()).await; + let first = Database::connect(config.db_url.clone()).await.unwrap(); + Migrator::up(&first, None).await.unwrap(); + let second = Database::connect(config.db_url).await.unwrap(); + (first, second, schema) +} + +async fn scalar(db: &C, sql: &str) -> i64 { + db.query_one_raw(statement(sql, [])) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap() +} + +async fn installed( + db: &DatabaseConnection, + operation: &str, +) -> ( + PostgresQualifiedMetadataRepository, + PreparedNativeMetadataRetention, + GenerationMetadataReceipt, +) { + let repo = PostgresQualifiedMetadataRepository::new(db.clone()) + .await + .unwrap(); + let pages = prepared(); + let intent = repo.begin_intent(operation, &pages).await.unwrap(); + repo.install_pages(&intent, pages.dag().payloads()) + .await + .unwrap(); + let receipt = repo.finalize(&intent).await.unwrap(); + (repo, pages, receipt) +} + +async fn claim_root( + repo: &PostgresQualifiedMetadataRepository, + receipt: &GenerationMetadataReceipt, +) -> MetadataGcClaim { + let lifetime = repo.lifetime(receipt.metadata_root()).await.unwrap(); + repo.claim(&uuid::Uuid::new_v4().to_string(), &lifetime) + .await + .unwrap() +} + +async fn wait_for_waiter(txn: &DatabaseTransaction) { + let deadline = tokio::time::Instant::now() + Duration::from_secs(4); + loop { + let waiting:bool=txn.query_one_raw(statement( + "SELECT EXISTS(SELECT 1 FROM pg_locks WHERE locktype='advisory' AND NOT granted + AND classid=$1::integer::oid AND objid=hashtext(current_schema())::oid + AND database=(SELECT oid FROM pg_database WHERE datname=current_database())) AS waiting", + [RETENTION_LOCK_KEY.into()], + )).await.unwrap().unwrap().try_get("","waiting").unwrap(); + if waiting { + break; + } + assert!( + tokio::time::Instant::now() < deadline, + "real independent connection must wait on retention barrier" + ); + tokio::task::yield_now().await; + } +} + +#[tokio::test] +async fn qualified_graph_exact_fks_unique_edges_and_permanent_domains() { + let (first, second, _schema) = fixture().await; + let (repo, pages, receipt) = installed(&first, "qualified-graph").await; + repo.install_pages(receipt.intent(), pages.dag().payloads()) + .await + .unwrap(); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_graph_node").await, + 2 + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_graph_edge").await, + 1 + ); + assert_eq!( + scalar( + &first, + "SELECT sum(incoming_refs)::bigint FROM mst2_metadata_graph_node" + ) + .await, + 1 + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_retention_node").await, + 0 + ); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_lifetime WHERE graph_domain='qualified-v1'" + ) + .await, + 2 + ); + repo.retire_prepare_coverage(&receipt).await.unwrap(); + let generic = PostgresMetadataGenerationRepository::new(second.clone()) + .await + .unwrap(); + assert!( + generic + .begin_intent("cannot-adopt-retired-qualified", &pages) + .await + .is_err() + ); + for sql in [ + "UPDATE mst2_metadata_lifetime SET graph_domain='generic-v1'", + "UPDATE mst2_metadata_graph_node SET incoming_refs=incoming_refs+1", + "UPDATE mst2_metadata_graph_edge SET child_generation=child_generation+1", + "DELETE FROM mst2_metadata_current", + "UPDATE mst2_metadata_payload SET payload=payload", + "DELETE FROM mst2_metadata_storage_scope", + ] { + assert!(second.execute_unprepared(sql).await.is_err(), "{sql}"); + } + let claim = claim_root(&repo, &receipt).await; + let applied = repo.apply(&claim).await.unwrap(); + assert_eq!(applied.claim(), &claim); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_gc_op WHERE state='APPLIED'" + ) + .await, + 1 + ); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_prepare_page WHERE generation=1" + ) + .await, + 2 + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_lifetime").await, + 2 + ); +} + +#[tokio::test] +async fn qualified_shared_child_and_other_prepare_are_real_collection_vetoes() { + let (first, second, _schema) = fixture().await; + let (repo, pages, receipt) = installed(&first, "shared-owner-one").await; + let other = PostgresQualifiedMetadataRepository::new(second) + .await + .unwrap(); + let intent = other + .begin_intent("shared-owner-two", &pages) + .await + .unwrap(); + let shared = other.finalize(&intent).await.unwrap(); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_graph_root").await, + 4 + ); + repo.retire_prepare_coverage(&receipt).await.unwrap(); + let root = repo.lifetime(receipt.metadata_root()).await.unwrap(); + assert!( + repo.claim(&uuid::Uuid::new_v4().to_string(), &root) + .await + .is_err() + ); + other.retire_prepare_coverage(&shared).await.unwrap(); + let child = pages + .dag() + .payloads() + .iter() + .find(|p| p.id != receipt.metadata_root()) + .unwrap(); + let child = repo.lifetime(child.id).await.unwrap(); + assert!( + repo.claim(&uuid::Uuid::new_v4().to_string(), &child) + .await + .is_err() + ); + let root_claim = repo + .claim(&uuid::Uuid::new_v4().to_string(), &root) + .await + .unwrap(); + repo.apply(&root_claim).await.unwrap(); + assert_eq!( + scalar(&first, "SELECT incoming_refs FROM mst2_metadata_graph_node").await, + 0 + ); + let child_claim = repo + .claim(&uuid::Uuid::new_v4().to_string(), &child) + .await + .unwrap(); + repo.apply(&child_claim).await.unwrap(); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_payload").await, + 0 + ); +} + +#[tokio::test] +async fn qualified_pending_preparation_and_lost_roots_fail_closed() { + let (first, second, _schema) = fixture().await; + let (repo, pages, receipt) = installed(&first, "lost-roots").await; + second + .execute_raw(statement( + "DELETE FROM mst2_metadata_graph_root WHERE prepare_id=$1", + [receipt.intent().prepare_id().into()], + )) + .await + .unwrap(); + assert!(repo.retire_prepare_coverage(&receipt).await.is_err()); + let lifetime = repo.lifetime(receipt.metadata_root()).await.unwrap(); + assert!( + repo.claim(&uuid::Uuid::new_v4().to_string(), &lifetime) + .await + .is_err() + ); + let preparing = repo.begin_intent("pending-owner", &pages).await.unwrap(); + assert!( + repo.claim(&uuid::Uuid::new_v4().to_string(), &lifetime) + .await + .is_err() + ); + repo.abort(&preparing).await.unwrap(); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_payload").await, + 2 + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_gc_op").await, + 0 + ); +} + +#[tokio::test] +async fn qualified_explicit_aborted_orphans_with_and_without_payload_collect() { + for partial in [false, true] { + let (first, second, _schema) = fixture().await; + let repo = PostgresQualifiedMetadataRepository::new(first.clone()) + .await + .unwrap(); + let pages = prepared(); + let intent = repo.begin_intent("orphan", &pages).await.unwrap(); + if partial { + repo.install_pages(&intent, pages.dag().payloads()) + .await + .unwrap(); + } + let lifetime = repo.lifetime(pages.dag().root()).await.unwrap(); + assert!( + repo.claim(&uuid::Uuid::new_v4().to_string(), &lifetime) + .await + .is_err() + ); + repo.abort(&intent).await.unwrap(); + let claim = repo + .claim(&uuid::Uuid::new_v4().to_string(), &lifetime) + .await + .unwrap(); + assert!(!claim.graph_present); + assert_eq!(claim.had_payload, partial); + repo.apply(&claim).await.unwrap(); + assert_eq!( + repo.inspect_gc( + &second, + claim.operation_id(), + &lifetime, + MetadataGcPhase::Apply + ) + .await + .unwrap(), + MetadataGcObservation::Applied(Box::new(repo.apply(&claim).await.unwrap())) + ); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_lifetime WHERE state='REMOVED'" + ) + .await, + 1 + ); + } +} + +#[tokio::test] +async fn qualified_abort_and_collection_fence_a_late_installer_at_actual_pg_lock() { + let (first, second, _schema) = fixture().await; + let repo = PostgresQualifiedMetadataRepository::new(first.clone()) + .await + .unwrap(); + let pages = prepared(); + let intent = repo.begin_intent("late-installer", &pages).await.unwrap(); + repo.install_pages(&intent, pages.dag().payloads()) + .await + .unwrap(); + let lifetime = repo.lifetime(pages.dag().root()).await.unwrap(); + let held = first.begin().await.unwrap(); + repo.inner.inner.barrier(&held).await.unwrap(); + let late_intent = intent.clone(); + let late_pages = pages.dag().payloads().to_vec(); + let late = tokio::spawn(async move { + let late_repo = PostgresQualifiedMetadataRepository::new(second) + .await + .unwrap(); + late_repo.install_pages(&late_intent, &late_pages).await + }); + wait_for_waiter(&held).await; + repo.terminate_in_txn(&held, &intent, MetadataTerminalAction::Abort, None) + .await + .unwrap(); + let claim = repo + .claim_in_txn(&held, &uuid::Uuid::new_v4().to_string(), &lifetime) + .await + .unwrap(); + repo.apply_in_txn(&held, &claim).await.unwrap(); + held.commit().await.unwrap(); + assert!(late.await.unwrap().is_err()); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_payload").await, + 1 + ); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_gc_op WHERE state='APPLIED'" + ) + .await, + 1 + ); +} + +#[tokio::test] +async fn qualified_statement_barrier_raw_writer_waits_before_row_lock_and_rechecks_domain() { + for generic in [true, false] { + let (first, second, _schema) = fixture().await; + let (repo, _pages, receipt) = installed(&first, "raw-writer").await; + repo.retire_prepare_coverage(&receipt).await.unwrap(); + let root = repo.lifetime(receipt.metadata_root()).await.unwrap(); + let held = first.begin().await.unwrap(); + repo.inner.inner.barrier(&held).await.unwrap(); + let sql = if generic { + format!( + "INSERT INTO mst2_retention_node(node_id,kind,bytes,state) VALUES('page:sha256:{}','page',{},'LIVE')", + hex::encode(root.page_id), + root.expected_size + ) + } else { + format!( + "UPDATE mst2_metadata_graph_node SET state='LIVE' WHERE page_id=decode('{}','hex') AND generation={}", + hex::encode(root.page_id), + root.generation + ) + }; + let writer = tokio::spawn(async move { second.execute_unprepared(&sql).await }); + wait_for_waiter(&held).await; + held.query_one_raw(statement("SELECT page_id FROM mst2_metadata_graph_node WHERE page_id=$1 AND generation=$2 FOR UPDATE NOWAIT", + [root.page_id.to_vec().into(),root.generation.into()])).await.unwrap().unwrap(); + let claim = repo + .claim_in_txn(&held, &uuid::Uuid::new_v4().to_string(), &root) + .await + .unwrap(); + held.commit().await.unwrap(); + assert!(writer.await.unwrap().is_err()); + repo.apply(&claim).await.unwrap(); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_retention_node").await, + 0 + ); + } +} + +#[tokio::test] +async fn qualified_claim_outer_rollback_and_lost_commit_response_use_fresh_barrier() { + let (first, second, _schema) = fixture().await; + let (repo, _pages, receipt) = installed(&first, "claim-recovery").await; + repo.retire_prepare_coverage(&receipt).await.unwrap(); + let lifetime = repo.lifetime(receipt.metadata_root()).await.unwrap(); + let id = uuid::Uuid::new_v4().to_string(); + let held = first.begin().await.unwrap(); + repo.claim_in_txn(&held, &id, &lifetime).await.unwrap(); + held.rollback().await.unwrap(); + assert_eq!( + repo.inspect_gc(&second, &id, &lifetime, MetadataGcPhase::Claim) + .await + .unwrap(), + MetadataGcObservation::Absent + ); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_graph_node WHERE state='LIVE'" + ) + .await, + 2 + ); + let held = first.begin().await.unwrap(); + let claim = repo.claim_in_txn(&held, &id, &lifetime).await.unwrap(); + held.commit().await.unwrap(); // The caller deliberately discards the commit response. + assert_eq!( + repo.inspect_gc(&second, &id, &lifetime, MetadataGcPhase::Claim) + .await + .unwrap(), + MetadataGcObservation::Pending(Box::new(claim.clone())) + ); + assert_eq!(repo.pending_gc(1).await.unwrap(), vec![claim.clone()]); + repo.apply(&claim).await.unwrap(); +} + +#[tokio::test] +async fn qualified_apply_faults_rollback_bytes_graph_counters_and_receipt_with_guards_enabled() { + for (table, event, predicate) in [ + ("mst2_metadata_payload", "AFTER DELETE", "true"), + ("mst2_metadata_graph_edge", "AFTER DELETE", "true"), + ( + "mst2_metadata_gc_op", + "BEFORE UPDATE", + "NEW.state='APPLIED'", + ), + ] { + let (first, second, _schema) = fixture().await; + let (repo, _pages, receipt) = installed(&first, "apply-fault").await; + repo.retire_prepare_coverage(&receipt).await.unwrap(); + let claim = claim_root(&repo, &receipt).await; + let granularity = if table == "mst2_metadata_graph_edge" { + "STATEMENT" + } else { + "ROW" + }; + let sql = format!( + "CREATE FUNCTION test_gc_fault() RETURNS trigger LANGUAGE plpgsql AS $$ BEGIN IF {predicate} THEN RAISE EXCEPTION 'injected atomic GC failure'; END IF; RETURN NEW; END $$; CREATE TRIGGER zzz_test_gc_fault {event} ON {table} FOR EACH {granularity} EXECUTE FUNCTION test_gc_fault()" + ); + second.execute_unprepared(&sql).await.unwrap(); + assert!(repo.apply(&claim).await.is_err()); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_payload").await, + 2 + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_graph_edge").await, + 1 + ); + assert_eq!( + scalar( + &first, + "SELECT sum(incoming_refs)::bigint FROM mst2_metadata_graph_node" + ) + .await, + 1 + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_graph_node").await, + 2 + ); + assert_eq!(scalar(&first,"SELECT count(*) FROM mst2_metadata_gc_op WHERE state='PENDING' AND payload_delete_xid IS NULL").await,1); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_lifetime WHERE state='DELETING'" + ) + .await, + 1 + ); + assert_eq!(scalar(&first,"SELECT count(*) FROM pg_trigger WHERE tgname IN ('mst2_metadata_payload_fenced','mst2_metadata_payload_removed','mst2_metadata_graph_node_guard','mst2_metadata_gc_op_guard') AND tgenabled='O' AND tgrelid IN (SELECT oid FROM pg_class WHERE relnamespace=(SELECT oid FROM pg_namespace WHERE nspname=current_schema()))").await,4); + second + .execute_unprepared(&format!("DROP TRIGGER zzz_test_gc_fault ON {table}")) + .await + .unwrap(); + repo.apply(&claim).await.unwrap(); + } +} + +#[tokio::test] +async fn qualified_direct_selected_delete_completes_all_bookkeeping_and_guc_is_not_authority() { + let (first, second, _schema) = fixture().await; + let (repo, _pages, receipt) = installed(&first, "direct-delete").await; + repo.retire_prepare_coverage(&receipt).await.unwrap(); + let claim = claim_root(&repo, &receipt).await; + for sql in [ + "UPDATE mst2_metadata_gc_op SET had_payload=false", + "UPDATE mst2_metadata_gc_op SET graph_present=false", + "UPDATE mst2_metadata_gc_op SET generation=generation+1", + ] { + assert!(second.execute_unprepared(sql).await.is_err(), "{sql}"); + } + assert!( + second + .query_one_raw(statement( + "SELECT mst2_metadata_gc_finish($1::uuid)", + [claim.operation_id.clone().into()] + )) + .await + .is_err() + ); + second + .execute_unprepared("UPDATE mst2_metadata_gc_op SET payload_delete_xid=1") + .await + .unwrap(); + let marked = second.begin().await.unwrap(); + marked + .execute_raw(statement( + "UPDATE mst2_metadata_gc_op SET payload_delete_xid=42 WHERE operation_id=$1::uuid", + [claim.operation_id.clone().into()], + )) + .await + .unwrap(); + let same_xid:bool=marked.query_one_raw(statement( + "SELECT payload_delete_xid=txid_current() AS same_xid FROM mst2_metadata_gc_op WHERE operation_id=$1::uuid", + [claim.operation_id.clone().into()], + )).await.unwrap().unwrap().try_get("","same_xid").unwrap(); + assert!( + same_xid, + "database must replace a caller marker with the actual transaction identity" + ); + let error = marked + .query_one_raw(statement( + "SELECT mst2_metadata_gc_finish($1::uuid)", + [claim.operation_id.clone().into()], + )) + .await + .unwrap_err(); + assert!( + error.to_string().contains("payload presence changed"), + "{error}" + ); + marked.rollback().await.unwrap(); + assert!( + second + .query_one_raw(statement( + "SELECT mst2_metadata_gc_finish($1::uuid)", + [claim.operation_id.clone().into()] + )) + .await + .is_err() + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_payload").await, + 2 + ); + assert!( + second + .execute_unprepared("DELETE FROM mst2_metadata_payload") + .await + .is_err() + ); + let txn = second.begin().await.unwrap(); + txn.execute_raw(statement( + "SELECT set_config('mega2.metadata_gc_operation',$1,true)", + [uuid::Uuid::new_v4().to_string().into()], + )) + .await + .unwrap(); + assert!( + txn.execute_raw(statement( + "DELETE FROM mst2_metadata_payload WHERE page_id=$1 AND generation=$2", + [ + claim.lifetime.page_id.to_vec().into(), + claim.lifetime.generation.into() + ] + )) + .await + .is_err() + ); + txn.rollback().await.unwrap(); + let txn = second.begin().await.unwrap(); + txn.execute_raw(statement( + "SELECT set_config('mega2.metadata_gc_operation',$1,true)", + [claim.operation_id.clone().into()], + )) + .await + .unwrap(); + txn.execute_raw(statement( + "DELETE FROM mst2_metadata_payload WHERE page_id=$1 AND generation=$2", + [ + claim.lifetime.page_id.to_vec().into(), + claim.lifetime.generation.into(), + ], + )) + .await + .unwrap(); + assert_eq!( + scalar( + &txn, + "SELECT count(*) FROM mst2_metadata_gc_op WHERE state='APPLIED'" + ) + .await, + 1 + ); + assert_eq!( + scalar(&txn, "SELECT count(*) FROM mst2_metadata_graph_edge").await, + 0 + ); + assert_eq!( + scalar(&txn, "SELECT incoming_refs FROM mst2_metadata_graph_node").await, + 0 + ); + txn.commit().await.unwrap(); + repo.apply(&claim).await.unwrap(); + assert!( + second + .execute_unprepared("DELETE FROM mst2_metadata_gc_op") + .await + .is_err() + ); +} + +#[tokio::test] +async fn qualified_fresh_aba_replays_old_receipts_without_deleting_g2_or_reviving_old_intents() { + let (first, second, _schema) = fixture().await; + let (repo, pages, receipt) = installed(&first, "aba-old").await; + let observation = repo.observe_installed_dag(receipt.intent()).await.unwrap(); + let terminal = repo.retire_prepare_coverage(&receipt).await.unwrap(); + let claim = claim_root(&repo, &receipt).await; + second.execute_unprepared("CREATE TABLE test_payload_deletes(n integer NOT NULL); CREATE FUNCTION test_record_delete() RETURNS trigger LANGUAGE plpgsql AS $$ BEGIN INSERT INTO test_payload_deletes VALUES(1); RETURN NULL; END $$; CREATE TRIGGER zzz_test_record_delete AFTER DELETE ON mst2_metadata_payload FOR EACH ROW EXECUTE FUNCTION test_record_delete()").await.unwrap(); + let held = first.begin().await.unwrap(); + let applied = repo.apply_in_txn(&held, &claim).await.unwrap(); + held.commit().await.unwrap(); // Deliberately lose delivery; inspect recovers the exact durable receipt. + assert_eq!( + repo.inspect_gc( + &second, + claim.operation_id(), + claim.lifetime(), + MetadataGcPhase::Apply + ) + .await + .unwrap(), + MetadataGcObservation::Applied(Box::new(applied.clone())) + ); + assert!( + repo.begin_intent("ordinary-cannot-reopen", &pages) + .await + .is_err() + ); + let fresh = repo + .begin_fresh_intent("aba-fresh", &pages, std::slice::from_ref(&applied)) + .await + .unwrap(); + assert_eq!( + repo.lifetime(receipt.metadata_root()) + .await + .unwrap() + .generation(), + 2 + ); + repo.install_pages(&fresh, pages.dag().payloads()) + .await + .unwrap(); + repo.finalize(&fresh).await.unwrap(); + assert!( + repo.install_pages(receipt.intent(), pages.dag().payloads()) + .await + .is_err() + ); + assert!( + repo.finalize_observation(receipt.intent(), &observation) + .await + .is_err() + ); + let restarted = PostgresQualifiedMetadataRepository::new(second.clone()) + .await + .unwrap(); + let old = restarted + .historical_lifetime(claim.lifetime.page_id, 1) + .await + .unwrap(); + let recovered = restarted + .inspect_gc(&second, claim.operation_id(), &old, MetadataGcPhase::Apply) + .await + .unwrap(); + let MetadataGcObservation::Applied(recovered) = recovered else { + panic!("restart must recover historical APPLIED receipt"); + }; + assert_eq!(restarted.apply(recovered.claim()).await.unwrap(), applied); + assert!( + restarted + .claim(&uuid::Uuid::new_v4().to_string(), &old) + .await + .is_err() + ); + let terminal_intent = restarted + .capture_terminal_intent( + receipt.intent().operation_id(), + receipt.intent().manifest_digest(), + ) + .await + .unwrap(); + assert_eq!( + restarted + .inspect_terminal( + &second, + &terminal_intent, + MetadataTerminalAction::RetireCoverage + ) + .await + .unwrap(), + MetadataTerminalObservation::Terminated(Box::new(terminal.clone())) + ); + assert_eq!( + repo.retire_prepare_coverage(&receipt).await.unwrap(), + terminal + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM test_payload_deletes").await, + 1 + ); + let stale = second.begin().await.unwrap(); + stale + .execute_raw(statement( + "SELECT set_config('mega2.metadata_gc_operation',$1,true)", + [claim.operation_id.clone().into()], + )) + .await + .unwrap(); + assert!( + stale + .execute_raw(statement( + "DELETE FROM mst2_metadata_payload WHERE page_id=$1 AND generation=2", + [claim.lifetime.page_id.to_vec().into()] + )) + .await + .is_err() + ); + stale.rollback().await.unwrap(); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_payload").await, + 2 + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_lifetime").await, + 3 + ); + assert_eq!( + repo.begin_fresh_intent("aba-fresh", &pages, std::slice::from_ref(&applied)) + .await + .unwrap(), + fresh + ); + let mut forged = applied.clone(); + forged.created_at += chrono::Duration::seconds(1); + assert!( + repo.begin_fresh_intent("aba-fresh", &pages, &[forged]) + .await + .is_err() + ); + assert!( + repo.begin_fresh_intent("stale-fresh-proof", &pages, &[applied]) + .await + .is_err() + ); +} + +#[tokio::test] +async fn qualified_multi_page_fresh_failure_rolls_back_all_watermarks_and_new_history() { + let (first, second, _schema) = fixture().await; + let (repo, pages, receipt) = installed(&first, "multi-old").await; + repo.retire_prepare_coverage(&receipt).await.unwrap(); + let root = claim_root(&repo, &receipt).await; + let root = repo.apply(&root).await.unwrap(); + let child_id = pages + .dag() + .payloads() + .iter() + .find(|p| p.id != receipt.metadata_root()) + .unwrap() + .id; + let child = repo.lifetime(child_id).await.unwrap(); + let child = repo + .claim(&uuid::Uuid::new_v4().to_string(), &child) + .await + .unwrap(); + let child = repo.apply(&child).await.unwrap(); + second.execute_unprepared("CREATE FUNCTION test_fresh_fault() RETURNS trigger LANGUAGE plpgsql AS $$ BEGIN IF NEW.generation=2 AND NEW.page_id=(SELECT page_id FROM mst2_metadata_current ORDER BY page_id DESC LIMIT 1) THEN RAISE EXCEPTION 'injected multi-page fresh CAS failure'; END IF; RETURN NEW; END $$; CREATE TRIGGER zzz_test_fresh_fault BEFORE UPDATE ON mst2_metadata_current FOR EACH ROW EXECUTE FUNCTION test_fresh_fault()").await.unwrap(); + assert!( + repo.begin_fresh_intent("multi-fresh", &pages, &[root.clone(), child.clone()]) + .await + .is_err() + ); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_current WHERE generation=1" + ) + .await, + 2 + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_lifetime").await, + 2 + ); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_prepare WHERE operation_id='multi-fresh'" + ) + .await, + 0 + ); + second + .execute_unprepared("DROP TRIGGER zzz_test_fresh_fault ON mst2_metadata_current") + .await + .unwrap(); + let intent = repo + .begin_fresh_intent("multi-fresh", &pages, &[root, child]) + .await + .unwrap(); + repo.install_pages(&intent, pages.dag().payloads()) + .await + .unwrap(); + repo.finalize(&intent).await.unwrap(); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_current WHERE generation=2" + ) + .await, + 2 + ); +} + +#[tokio::test] +async fn qualified_wrong_primary_and_repeatable_read_never_prove_absence_or_authorize_delete() { + let (first, second, _schema) = fixture().await; + let (other, _other_second, _other_schema) = fixture().await; + let (repo, _pages, receipt) = installed(&first, "scope").await; + repo.retire_prepare_coverage(&receipt).await.unwrap(); + let claim = claim_root(&repo, &receipt).await; + assert!(matches!( + repo.inspect_gc( + &other, + claim.operation_id(), + claim.lifetime(), + MetadataGcPhase::Apply + ) + .await, + Err(MetadataGcError::CommitUncertain { .. }) + )); + let txn = second + .begin_with_config(Some(IsolationLevel::RepeatableRead), None) + .await + .unwrap(); + assert!( + txn.execute_unprepared("DELETE FROM mst2_metadata_payload") + .await + .is_err() + ); + txn.rollback().await.unwrap(); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_payload").await, + 2 + ); + repo.apply(&claim).await.unwrap(); +} + +#[tokio::test] +async fn qualified_null_bytes_and_retired_generic_incarnations_cannot_be_adopted() { + let (first, second, _schema) = fixture().await; + let pages = prepared(); + let payload = &pages.dag().payloads()[0]; + first.execute_raw(statement("INSERT INTO mst2_metadata_payload(page_id,metadata_codec,byte_size,payload) VALUES($1,1,$2,$3)", + [payload.id.to_vec().into(),(payload.size as i32).into(),payload.bytes.clone().into()])).await.unwrap(); + let qualified = PostgresQualifiedMetadataRepository::new(second.clone()) + .await + .unwrap(); + assert!( + qualified + .begin_intent("cannot-adopt-null", &pages) + .await + .is_err() + ); + assert!( + second + .execute_unprepared("DELETE FROM mst2_metadata_payload") + .await + .is_err() + ); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_payload WHERE generation IS NULL" + ) + .await, + 1 + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_current").await, + 0 + ); + let (first, second, _second_schema) = fixture().await; + let generic = PostgresMetadataGenerationRepository::new(first.clone()) + .await + .unwrap(); + let intent = generic + .begin_intent("generic-history", &pages) + .await + .unwrap(); + generic + .install_pages(&intent, pages.dag().payloads()) + .await + .unwrap(); + let receipt = generic.finalize(&intent).await.unwrap(); + generic.retire_prepare_coverage(&receipt).await.unwrap(); + let qualified = PostgresQualifiedMetadataRepository::new(second) + .await + .unwrap(); + assert!( + qualified + .begin_intent("cannot-adopt-retired-generic", &pages) + .await + .is_err() + ); + assert!(qualified.lifetime(payload.id).await.is_err()); +} + +#[tokio::test] +async fn qualified_committed_partial_payload_is_rejected_without_repair() { + let (first, second, _schema) = fixture().await; + let repo = PostgresQualifiedMetadataRepository::new(first.clone()) + .await + .unwrap(); + let pages = prepared(); + let intent = repo + .begin_intent("corrupt-partial-committed", &pages) + .await + .unwrap(); + repo.install_page(&intent, &pages.dag().payloads()[0]) + .await + .unwrap(); + second.execute_raw(statement("UPDATE mst2_metadata_prepare SET state='COMMITTED',committed_at=clock_timestamp() WHERE prepare_id=$1", + [intent.prepare_id().into()])).await.unwrap(); + assert!( + repo.install_pages(&intent, pages.dag().payloads()) + .await + .is_err() + ); + let lifetime = repo.lifetime(pages.dag().root()).await.unwrap(); + assert!( + repo.claim(&uuid::Uuid::new_v4().to_string(), &lifetime) + .await + .is_err() + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_payload").await, + 1 + ); +} + +#[tokio::test] +async fn qualified_cross_action_terminal_inspection_conflicts_and_production_adoption_stays_closed() +{ + let (first, second, _schema) = fixture().await; + let (repo, pages, receipt) = installed(&first, "cross-action-committed").await; + assert!(matches!( + repo.inspect_terminal(&second, receipt.intent(), MetadataTerminalAction::Abort) + .await, + Err(MetadataTerminalError::Rejected(SnapshotError { + code: SnapshotErrorCode::Conflict, + .. + })) + )); + let pending = repo + .begin_intent("cross-action-aborted", &pages) + .await + .unwrap(); + repo.abort(&pending).await.unwrap(); + assert!(matches!( + repo.inspect_terminal(&second, &pending, MetadataTerminalAction::RetireCoverage) + .await, + Err(MetadataTerminalError::Rejected(SnapshotError { + code: SnapshotErrorCode::Conflict, + .. + })) + )); + assert!(repo.abort(receipt.intent()).await.is_err()); + let sql="INSERT INTO mst2_snapshot_context(snapshot_id,canonical_descriptor,instance_id,commit_oid,root_tree_oid, + metadata_root,prepare_id,publication_sequence,writer_epoch,state) VALUES('blocked',$1,'i','c','r',$2,$3,0,1,'READY')"; + assert!( + second + .execute_raw(statement( + sql, + [ + vec![1_u8].into(), + receipt.metadata_root().to_vec().into(), + receipt.intent().prepare_id().into() + ] + )) + .await + .is_err() + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_snapshot_context").await, + 0 + ); + assert_eq!( + repo.inspect_terminal( + &second, + receipt.intent(), + MetadataTerminalAction::RetireCoverage + ) + .await + .unwrap(), + MetadataTerminalObservation::Active + ); +} + +#[tokio::test] +async fn qualified_deferred_current_guard_denies_unprotected_raw_fresh_commit() { + let (first, second, _schema) = fixture().await; + let (repo, _pages, receipt) = installed(&first, "raw-fresh").await; + repo.retire_prepare_coverage(&receipt).await.unwrap(); + let claim = claim_root(&repo, &receipt).await; + repo.apply(&claim).await.unwrap(); + let txn = second.begin().await.unwrap(); + txn.execute_raw(statement("INSERT INTO mst2_metadata_lifetime(page_id,node_id,generation,state,metadata_codec,expected_size,graph_domain) VALUES($1,$2,2,'RESERVED',1,$3,'qualified-v1')", + [claim.lifetime.page_id.to_vec().into(),node_id(&claim.lifetime.page_id).into(),claim.lifetime.expected_size.into()])).await.unwrap(); + txn.execute_raw(statement( + "UPDATE mst2_metadata_current SET generation=2 WHERE page_id=$1", + [claim.lifetime.page_id.to_vec().into()], + )) + .await + .unwrap(); + assert!(txn.commit().await.is_err()); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_current WHERE generation=2" + ) + .await, + 0 + ); + assert_eq!( + scalar( + &first, + "SELECT count(*) FROM mst2_metadata_lifetime WHERE generation=2" + ) + .await, + 0 + ); +} + +#[tokio::test] +async fn qualified_statement_dag_audit_accepts_wide_and_chain_graphs_and_rejects_a_cycle() { + for wide in [true, false] { + let (first, second, _schema) = fixture().await; + let repo = PostgresQualifiedMetadataRepository::new(first.clone()) + .await + .unwrap(); + let mut builder = MetadataDagBuilder::new(MetadataDagLimits::default()); + let mut root_entries = Vec::new(); + let mut last = [0_u8; 32]; + for i in 0..64_u8 { + let entries = if wide || i == 0 { + vec![Entry::file(EntryKind::Regular, b"file", 3, [i; 32])] + } else { + vec![Entry::dir(b"child", last)] + }; + let page = Page::build(&entries).unwrap(); + builder.add_directory(&page, &entries).unwrap(); + last = page_id(&page); + root_entries.push(Entry::dir(format!("d{i:03}").as_bytes(), last)); + } + let root = if wide { + let root = Page::build(&root_entries).unwrap(); + builder.add_directory(&root, &root_entries).unwrap(); + page_id(&root) + } else { + last + }; + let pages = PreparedNativeMetadataRetention::test_installation( + Arc::new(builder.finish(root).unwrap()), + "/", + ); + let intent = repo.begin_intent("statement-dag", &pages).await.unwrap(); + repo.install_pages(&intent, &pages.dag().payloads()[..64]) + .await + .unwrap(); + if wide { + repo.install_page(&intent, &pages.dag().payloads()[64]) + .await + .unwrap(); + } + let receipt = repo.finalize(&intent).await.unwrap(); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_graph_node").await, + if wide { 65 } else { 64 } + ); + assert_eq!( + scalar( + &first, + "SELECT sum(incoming_refs)::bigint FROM mst2_metadata_graph_node" + ) + .await, + if wide { 64 } else { 63 } + ); + let pending = repo.begin_intent("cycle-attempt", &pages).await.unwrap(); + let leaf = pages + .dag() + .payloads() + .iter() + .find(|p| { + !pages + .dag() + .edges() + .iter() + .any(|e| e.parent == node_id(&p.id)) + }) + .unwrap() + .id; + let error=second.execute_raw(statement( + "INSERT INTO mst2_metadata_graph_edge(parent_page,parent_generation,child_page,child_generation) VALUES($1,1,$2,1)", + [leaf.to_vec().into(),root.to_vec().into()], + )).await.unwrap_err(); + assert!(error.to_string().contains("cycle"), "{error}"); + assert_eq!( + scalar( + &first, + "SELECT sum(incoming_refs)::bigint FROM mst2_metadata_graph_node" + ) + .await, + if wide { 64 } else { 63 } + ); + repo.abort(&pending).await.unwrap(); + repo.retire_prepare_coverage(&receipt).await.unwrap(); + } +} + +#[tokio::test] +async fn qualified_statement_dag_overflow_rolls_back_without_disabling_production_guards() { + let (first, second, _schema) = fixture().await; + let repo = PostgresQualifiedMetadataRepository::new(first.clone()) + .await + .unwrap(); + let pages = prepared(); + let intent = repo.begin_intent("overflow-proof", &pages).await.unwrap(); + assert_eq!(MetadataDagLimits::default().nodes, 4096); + assert_eq!(MetadataDagLimits::default().edges, 16384); + let txn = second.begin().await.unwrap(); + txn.execute_raw(statement( + "INSERT INTO mst2_metadata_lifetime(page_id,node_id,generation,state,metadata_codec,expected_size,graph_domain) + SELECT sha256(convert_to('bound-node-'||i,'UTF8')),'page:sha256:'||encode(sha256(convert_to('bound-node-'||i,'UTF8')),'hex'), + 1,'RESERVED',1,$1,'qualified-v1' FROM generate_series(1,4097) x(i)", + [(pages.dag().payloads()[0].size as i32).into()], + )).await.unwrap(); + txn.execute_unprepared("INSERT INTO mst2_metadata_current(page_id,generation) SELECT l.page_id,l.generation FROM mst2_metadata_lifetime l WHERE NOT EXISTS(SELECT 1 FROM mst2_metadata_current c WHERE c.page_id=l.page_id)").await.unwrap(); + txn.execute_raw(statement( + "INSERT INTO mst2_metadata_prepare_page(prepare_id,page_id,generation,expected_size) + SELECT $1,l.page_id,l.generation,l.expected_size FROM mst2_metadata_lifetime l + WHERE NOT EXISTS(SELECT 1 FROM mst2_metadata_prepare_page m WHERE m.prepare_id=$1 AND m.page_id=l.page_id)", + [intent.prepare_id().into()], + )).await.unwrap(); + txn.execute_raw(statement( + "INSERT INTO mst2_metadata_graph_node(page_id,generation,state,metadata_codec,bytes) + SELECT page_id,generation,'LIVE',1,expected_size FROM mst2_metadata_prepare_page WHERE prepare_id=$1", + [intent.prepare_id().into()], + )).await.unwrap(); + let error=txn.execute_unprepared( + "INSERT INTO mst2_metadata_graph_edge(parent_page,parent_generation,child_page,child_generation) + SELECT p.page_id,1,c.page_id,1 FROM (SELECT page_id FROM mst2_metadata_graph_node ORDER BY page_id LIMIT 1) p + CROSS JOIN mst2_metadata_graph_node c WHERE c.page_id<>p.page_id" + ).await.unwrap_err(); + assert!(error.to_string().contains("overflow"), "{error}"); + txn.rollback().await.unwrap(); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_graph_node").await, + 0 + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_current").await, + 2 + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_lifetime").await, + 2 + ); + assert_eq!( + scalar(&first, "SELECT count(*) FROM mst2_metadata_prepare_page").await, + 2 + ); + let captured = repo.scan_current_lifetimes(None, 1).await.unwrap(); + assert_eq!(captured.len(), 1); + let rest = repo + .scan_current_lifetimes(Some((captured[0].page_id(), captured[0].generation())), 1) + .await + .unwrap(); + assert_eq!(rest.len(), 1); + assert_ne!(rest[0].page_id(), captured[0].page_id()); +} + +#[tokio::test] +async fn qualified_aborted_intent_is_recovered_from_history_after_repository_restart() { + let (first, second, _schema) = fixture().await; + let repo = PostgresQualifiedMetadataRepository::new(first) + .await + .unwrap(); + let pages = prepared(); + let intent = repo.begin_intent("restart-aborted", &pages).await.unwrap(); + let terminal = repo.abort(&intent).await.unwrap(); + drop(repo); + let restarted = PostgresQualifiedMetadataRepository::new(second.clone()) + .await + .unwrap(); + let recovered = restarted + .capture_terminal_intent("restart-aborted", intent.manifest_digest()) + .await + .unwrap(); + assert_eq!( + restarted + .inspect_terminal(&second, &recovered, MetadataTerminalAction::Abort) + .await + .unwrap(), + MetadataTerminalObservation::Terminated(Box::new(terminal)) + ); + assert!( + restarted + .install_pages(&recovered, pages.dag().payloads()) + .await + .is_err() + ); + assert_eq!( + scalar(&second, "SELECT count(*) FROM mst2_metadata_payload").await, + 0 + ); +} diff --git a/src/jupiter/storage/native_publication_storage.rs b/src/jupiter/storage/native_publication_storage.rs new file mode 100644 index 00000000..979c7314 --- /dev/null +++ b/src/jupiter/storage/native_publication_storage.rs @@ -0,0 +1,664 @@ +//! Native `/` publication certificates; origin path receipts retain their v1 identity. + +use git_internal::hash::{ObjectHash, get_hash_kind}; +use sea_orm::{ + ActiveModelTrait, ActiveValue::Set, ColumnTrait, ConnectionTrait, DatabaseTransaction, + EntityTrait, QueryFilter, QueryResult, Statement, TransactionTrait, +}; + +use crate::{ + callisto::{mega_refs, mst2_native_publication, mst2_publication, push_queue}, + common::utils::MEGA_BRANCH_NAME, + jupiter::storage::{ + base_storage::StorageConnector, + mono_storage::MonoStorage, + mst2_publication_storage::{CommittedPublication, PublicationReceiptError}, + push_queue_storage::PushQueueStorage, + }, +}; + +const NATIVE_EPOCH: i64 = 1; + +// Used by both ordinary reads and the single UPDATE that captures a B2.5 token. +pub(crate) const NATIVE_OBSERVATION_CTES: &str = r#" + native_roots AS ( + SELECT count(*)::bigint AS root_count, + max(ref_commit_hash) AS root_commit, max(ref_tree_hash) AS root_tree + FROM mega_refs WHERE path = '/' AND ref_name = $1 AND is_cl = false + ), native_observation AS ( + SELECT roots.*, h.instance_id, h.sequence, h.writer_epoch, h.state, + h.root_commit AS head_commit, h.root_tree AS head_tree, + h.certificate_receipt_id, + c.receipt_id AS certificate_id, c.namespace AS certificate_namespace, + c.instance_id AS certificate_instance, c.sequence AS certificate_sequence, + c.writer_epoch AS certificate_epoch, + c.old_root_commit AS certificate_old_commit, + c.root_commit AS certificate_commit, c.root_tree AS certificate_tree, + c.origin_path, c.origin_ref, c.old_path_commit, c.path_commit, + r.id AS receipt_id, r.namespace AS receipt_namespace, + r.sequence AS receipt_sequence, r.operation_id, + r.old_oid, r.new_oid, r.writer_epoch AS receipt_epoch, r.writer_kind, + r.request_digest, r.request_digest_version, r.native_certificate_version, + o.id AS outbox_id, o.namespace AS outbox_namespace, o.sequence AS outbox_sequence + FROM native_roots roots + LEFT JOIN mst2_native_head h ON h.namespace = '/' + LEFT JOIN mst2_native_publication c ON c.receipt_id = h.certificate_receipt_id + LEFT JOIN mst2_publication r ON r.id = c.receipt_id + LEFT JOIN mst2_publication_outbox o ON o.operation_id = r.operation_id + ) +"#; + +#[derive(Clone, Debug, PartialEq, Eq)] +pub(crate) struct NativeRoot { + pub(crate) commit: String, + pub(crate) tree: String, +} + +#[derive(Clone, Debug, PartialEq, Eq)] +pub(crate) struct NativePublicationToken { + pub(crate) sequence: i64, + pub(crate) epoch: i64, + pub(crate) certificate: Option, +} + +#[derive(Clone, Debug)] +pub(crate) struct NativePublicationHead { + pub(crate) root: NativeRoot, + pub(crate) instance_id: String, + pub(crate) token: NativePublicationToken, + ready: bool, +} + +#[derive(Clone, Debug)] +pub(crate) struct NativeObservation { + pub(crate) root: Option, + pub(crate) head: Option, +} + +#[derive(Debug)] +pub(crate) struct PreparedNativePublication { + head: NativePublicationHead, + path: String, + old_path: Option, + transaction_id: i64, +} + +fn integrity(message: &str) -> PublicationReceiptError { + PublicationReceiptError::Integrity(message.to_owned()) +} + +fn canonical_instance(instance: &str) -> Result { + uuid::Uuid::parse_str(instance) + .map(|id| id.to_string()) + .map_err(|_| integrity("invalid native deployment instance")) +} + +fn validate_root(root: &NativeRoot) -> Result<(), PublicationReceiptError> { + for oid in [&root.commit, &root.tree] { + ObjectHash::from_hex_for_kind(get_hash_kind(), oid) + .map_err(|_| integrity("invalid native object identity"))?; + } + Ok(()) +} + +pub(crate) fn decode_native_observation( + row: &QueryResult, +) -> Result { + let root_count: i64 = row.try_get("", "root_count")?; + if !(0..=1).contains(&root_count) { + return Err(integrity("native root is ambiguous")); + } + let root = if root_count == 1 { + let root = NativeRoot { + commit: row.try_get("", "root_commit")?, + tree: row.try_get("", "root_tree")?, + }; + validate_root(&root)?; + Some(root) + } else { + None + }; + let instance: Option = row.try_get("", "instance_id")?; + let Some(instance_id) = instance else { + return Ok(NativeObservation { root, head: None }); + }; + if canonical_instance(&instance_id)? != instance_id { + return Err(integrity("native head instance is not canonical")); + } + let head_root = NativeRoot { + commit: row.try_get("", "head_commit")?, + tree: row.try_get("", "head_tree")?, + }; + validate_root(&head_root)?; + if root.as_ref() != Some(&head_root) { + return Err(integrity("native root bypassed its publication head")); + } + let token = NativePublicationToken { + sequence: row.try_get("", "sequence")?, + epoch: row.try_get("", "writer_epoch")?, + certificate: row.try_get("", "certificate_receipt_id")?, + }; + if token.sequence < 0 || token.epoch != NATIVE_EPOCH { + return Err(PublicationReceiptError::Conflict( + "native writer epoch is fenced".into(), + )); + } + let state: String = row.try_get("", "state")?; + let ready = match state.as_str() { + "INITIALIZING" if token.certificate.is_none() => false, + "READY" if token.certificate.is_some() => { + let certificate_id: Option = row.try_get("", "certificate_id")?; + let receipt_id: Option = row.try_get("", "receipt_id")?; + let certificate_sequence: Option = row.try_get("", "certificate_sequence")?; + let certificate_epoch: Option = row.try_get("", "certificate_epoch")?; + let certificate_namespace: Option = row.try_get("", "certificate_namespace")?; + let certificate_instance: Option = row.try_get("", "certificate_instance")?; + let certificate_commit: Option = row.try_get("", "certificate_commit")?; + let certificate_tree: Option = row.try_get("", "certificate_tree")?; + let marker: Option = row.try_get("", "native_certificate_version")?; + let version: Option = row.try_get("", "request_digest_version")?; + let digest: Option = row.try_get("", "request_digest")?; + let writer_kind: Option = row.try_get("", "writer_kind")?; + let receipt_epoch: Option = row.try_get("", "receipt_epoch")?; + let origin_path: Option = row.try_get("", "origin_path")?; + let origin_ref: Option = row.try_get("", "origin_ref")?; + let receipt_namespace: Option = row.try_get("", "receipt_namespace")?; + let old_oid: Option = row.try_get("", "old_oid")?; + let old_root: Option = row.try_get("", "certificate_old_commit")?; + let new_oid: Option = row.try_get("", "new_oid")?; + let path_commit: Option = row.try_get("", "path_commit")?; + let old_path: Option = row.try_get("", "old_path_commit")?; + let outbox: Option = row.try_get("", "outbox_id")?; + let outbox_namespace: Option = row.try_get("", "outbox_namespace")?; + let outbox_sequence: Option = row.try_get("", "outbox_sequence")?; + let receipt_sequence: Option = row.try_get("", "receipt_sequence")?; + let digest_valid = digest.as_deref().is_some_and(|value| { + value.len() == 71 + && value.starts_with("sha256:") + && value[7..] + .bytes() + .all(|byte| byte.is_ascii_digit() || (b'a'..=b'f').contains(&byte)) + }); + if certificate_id != token.certificate + || receipt_id != token.certificate + || certificate_sequence != Some(token.sequence) + || certificate_epoch != Some(token.epoch) + || certificate_namespace.as_deref() != Some("/") + || certificate_instance.as_deref() != Some(instance_id.as_str()) + || certificate_commit.as_deref() != Some(head_root.commit.as_str()) + || certificate_tree.as_deref() != Some(head_root.tree.as_str()) + || marker != Some(1) + || version != Some(1) + || !digest_valid + || writer_kind.as_deref() != Some("trunk_push") + || receipt_epoch != Some(token.epoch) + || origin_ref.as_deref() != Some(MEGA_BRANCH_NAME) + || origin_path.is_none() + || origin_path != receipt_namespace + || new_oid.is_none() + || new_oid != path_commit + || old_oid.is_none() + || old_oid != old_root + || old_path == path_commit + || outbox.is_none() + || outbox_namespace != receipt_namespace + || outbox_sequence != receipt_sequence + { + return Err(integrity( + "native publication association is incomplete or inconsistent", + )); + } + true + } + _ => return Err(integrity("invalid native head state")), + }; + Ok(NativeObservation { + root, + head: Some(NativePublicationHead { + root: head_root, + instance_id, + token, + ready, + }), + }) +} + +pub(crate) async fn observe( + connection: &C, +) -> Result { + let rows = connection + .query_all_raw(Statement::from_sql_and_values( + connection.get_database_backend(), + format!("WITH {NATIVE_OBSERVATION_CTES} SELECT * FROM native_observation"), + [MEGA_BRANCH_NAME.into()], + )) + .await?; + if rows.len() != 1 { + return Err(integrity("ambiguous native publication observation")); + } + decode_native_observation(&rows[0]) +} + +async fn selected_ref( + connection: &C, + path: &str, +) -> Result, PublicationReceiptError> { + let refs = mega_refs::Entity::find() + .filter(mega_refs::Column::Path.eq(path)) + .filter(mega_refs::Column::RefName.eq(MEGA_BRANCH_NAME)) + .filter(mega_refs::Column::IsCl.eq(false)) + .all(connection) + .await?; + match refs.as_slice() { + [] => Ok(None), + [row] => { + let root = NativeRoot { + commit: row.ref_commit_hash.clone(), + tree: row.ref_tree_hash.clone(), + }; + validate_root(&root)?; + Ok(Some(root)) + } + _ => Err(integrity("selected native ref is ambiguous")), + } +} + +impl MonoStorage { + pub(crate) async fn read_native_publication_head( + &self, + instance: &str, + ) -> Result { + Self::read_native_publication_head_from(self.get_connection(), instance).await + } + + pub(crate) async fn read_native_publication_head_from( + connection: &C, + instance: &str, + ) -> Result { + let observation = observe(connection).await?; + let head = observation + .head + .ok_or_else(|| integrity("native publication is not initialized"))?; + if !head.ready { + return Err(integrity("native publication is not ready")); + } + if head.instance_id != canonical_instance(instance)? { + return Err(integrity("native instance changed")); + } + Ok(head) + } + + /// The command confirms stopped writers; this transaction checks the + /// paused admission gate, drained queue and exact native root. + pub(crate) async fn initialize_native_publication_for_maintenance( + &self, + instance: &str, + expected_root: &NativeRoot, + ) -> Result<(), PublicationReceiptError> { + let instance = canonical_instance(instance)?; + if uuid::Uuid::parse_str(&instance).is_ok_and(|id| id.is_nil()) { + return Err(integrity("native deployment instance must not be nil")); + } + for oid in [&expected_root.commit, &expected_root.tree] { + if oid.len() != 40 + || !oid + .bytes() + .all(|byte| byte.is_ascii_digit() || (b'a'..=b'f').contains(&byte)) + { + return Err(integrity( + "maintenance requires canonical SHA-1 root identities", + )); + } + } + let txn = self.get_connection().begin().await?; + if !PushQueueStorage::try_mono_write_lock(&txn) + .await + .map_err(|error| integrity(&error.to_string()))? + { + return Err(PublicationReceiptError::Conflict( + "native initialization refused while a writer holds the mono write lock".into(), + )); + } + let control = txn + .query_one_raw(Statement::from_string( + txn.get_database_backend(), + "SELECT paused, hard_stopped FROM queue_control WHERE id=1 FOR UPDATE NOWAIT" + .to_owned(), + )) + .await? + .ok_or_else(|| { + integrity( + "queue_control row missing; maintenance cannot establish admission exclusion", + ) + })?; + if !control.try_get::("", "paused")? { + return Err(integrity( + "queue admission must be paused before native initialization", + )); + } + if control.try_get::("", "hard_stopped")? { + return Err(integrity( + "hard-stopped queue must be investigated before native initialization", + )); + } + let history = txn + .query_one_raw(Statement::from_string( + txn.get_database_backend(), + "SELECT EXISTS(SELECT 1 FROM mst2_native_publication) OR \ + EXISTS(SELECT 1 FROM mst2_publication WHERE native_certificate_version IS NOT NULL) AS present" + .to_owned(), + )) + .await? + .ok_or_else(|| integrity("native history observation missing"))?; + if history.try_get::("", "present")? { + txn.rollback().await?; + return Err(integrity( + "native publication history exists; initialization cannot repair or replace it", + )); + } + let objects = txn + .query_one_raw(Statement::from_sql_and_values( + txn.get_database_backend(), + "SELECT (SELECT count(*) FROM mega_commit WHERE commit_id=$1) AS commits, \ + (SELECT max(tree) FROM mega_commit WHERE commit_id=$1) AS commit_tree, \ + (SELECT count(*) FROM mega_tree WHERE tree_id=$2) AS trees", + [ + expected_root.commit.clone().into(), + expected_root.tree.clone().into(), + ], + )) + .await? + .ok_or_else(|| integrity("native object observation missing"))?; + if objects.try_get::("", "commits")? != 1 + || objects + .try_get::>("", "commit_tree")? + .as_deref() + != Some(expected_root.tree.as_str()) + || objects.try_get::("", "trees")? != 1 + { + return Err(integrity( + "native root requires a unique stored commit with the expected tree and a unique stored tree", + )); + } + self.initialize_native_publication_in_txn(&txn, &instance, Some(expected_root)) + .await?; + txn.commit().await?; + Ok(()) + } + + #[cfg(test)] + pub(crate) async fn initialize_native_publication( + &self, + instance: &str, + ) -> Result<(), PublicationReceiptError> { + let instance = canonical_instance(instance)?; + let txn = self.get_connection().begin().await?; + PushQueueStorage::acquire_mono_write_lock(&txn) + .await + .map_err(|error| integrity(&error.to_string()))?; + txn.execute_unprepared("SELECT id FROM queue_control WHERE id=1 FOR UPDATE") + .await?; + self.initialize_native_publication_in_txn(&txn, &instance, None) + .await?; + txn.commit().await?; + Ok(()) + } + + async fn initialize_native_publication_in_txn( + &self, + txn: &DatabaseTransaction, + instance: &str, + expected_root: Option<&NativeRoot>, + ) -> Result<(), PublicationReceiptError> { + let observation = observe(txn).await?; + if observation.head.is_some() { + return Err(integrity("native head already initialized")); + } + let root = observation + .root + .ok_or_else(|| integrity("native root missing"))?; + if expected_root.is_some_and(|expected| *expected != root) { + return Err(PublicationReceiptError::Conflict( + "native root differs from the maintenance command's expected commit/tree".into(), + )); + } + let row = txn.query_one_raw(Statement::from_string(txn.get_database_backend(), + "SELECT count(*)::bigint AS active FROM push_queue WHERE status IN ('Queued', 'Running')".to_owned())).await? + .ok_or_else(|| integrity("queue observation missing"))?; + if row.try_get::("", "active")? != 0 { + return Err(integrity( + "queue must be drained before native initialization", + )); + } + let row = txn + .query_one_raw(Statement::from_string( + txn.get_database_backend(), + r#" + SELECT min(sequence) AS minimum, max(sequence) AS maximum FROM ( + SELECT sequence FROM mst2_namespace_seq UNION ALL SELECT sequence FROM mst2_publication + UNION ALL SELECT observed_sequence AS sequence FROM mst2_queue_noop_receipt + ) counters + "# + .to_owned(), + )) + .await? + .ok_or_else(|| integrity("native sequence floor missing"))?; + let minimum: Option = row.try_get("", "minimum")?; + let floor = row.try_get::>("", "maximum")?.unwrap_or(0); + if minimum.is_some_and(|value| value < 0) || floor == i64::MAX { + return Err(integrity("native sequence floor is invalid or exhausted")); + } + txn.execute_raw(Statement::from_sql_and_values(txn.get_database_backend(), + "INSERT INTO mst2_native_head (namespace, instance_id, sequence, writer_epoch, root_commit, root_tree, state) \ + VALUES ('/', $1, $2, $3, $4, $5, 'INITIALIZING')", + [instance.into(), floor.into(), NATIVE_EPOCH.into(), root.commit.into(), root.tree.into()], + )).await?; + Ok(()) + } + + pub(crate) async fn reserve_native_publication_in_txn( + &self, + txn: &DatabaseTransaction, + queue: &push_queue::Model, + instance: &str, + ) -> Result { + let owner = txn.query_one_raw(Statement::from_string(txn.get_database_backend(), + "SELECT txid_current() AS owner FROM mst2_native_head WHERE namespace = '/' FOR UPDATE".to_owned())) + .await?.ok_or_else(|| integrity("native publication is not initialized"))?; + let observation = observe(txn).await?; + let head = observation + .head + .ok_or_else(|| integrity("native publication is not initialized"))?; + if head.instance_id != canonical_instance(instance)? { + return Err(integrity("native instance changed")); + } + if queue.expected_native_sequence != Some(head.token.sequence) + || queue.expected_native_epoch != Some(head.token.epoch) + || queue.expected_native_certificate != head.token.certificate + || queue.expected_commit_hash.as_deref() != Some(head.root.commit.as_str()) + || queue.expected_tree_hash.as_deref() != Some(head.root.tree.as_str()) + { + return Err(PublicationReceiptError::Conflict( + "claimed native publication token is stale or missing".into(), + )); + } + let old_path = selected_ref(txn, &queue.path).await?; + Ok(PreparedNativePublication { + head, + path: queue.path.clone(), + old_path, + transaction_id: owner.try_get("", "owner")?, + }) + } + + async fn check_native_reservation( + &self, + txn: &DatabaseTransaction, + prepared: &PreparedNativePublication, + ) -> Result<(), PublicationReceiptError> { + let row = txn.query_one_raw(Statement::from_sql_and_values(txn.get_database_backend(), + "SELECT sequence FROM mst2_native_head WHERE namespace = '/' AND sequence = $1 AND writer_epoch = $2 \ + AND certificate_receipt_id IS NOT DISTINCT FROM $3 AND root_commit = $4 AND root_tree = $5 \ + AND txid_current() = $6 FOR UPDATE", + [prepared.head.token.sequence.into(), prepared.head.token.epoch.into(), prepared.head.token.certificate.into(), + prepared.head.root.commit.clone().into(), prepared.head.root.tree.clone().into(), prepared.transaction_id.into()], + )).await?; + if row.is_none() { + return Err(PublicationReceiptError::Conflict( + "native reservation is stale or belongs to another transaction".into(), + )); + } + Ok(()) + } + + pub(crate) async fn finish_native_noop_in_txn( + &self, + txn: &DatabaseTransaction, + prepared: PreparedNativePublication, + ) -> Result<(), PublicationReceiptError> { + self.check_native_reservation(txn, &prepared).await?; + if selected_ref(txn, "/").await?.as_ref() != Some(&prepared.head.root) + || selected_ref(txn, &prepared.path).await? != prepared.old_path + { + return Err(integrity("native no-op changed selected refs")); + } + Ok(()) + } + + pub(crate) async fn record_native_publication_in_txn( + &self, + txn: &DatabaseTransaction, + prepared: PreparedNativePublication, + committed: &CommittedPublication, + ) -> Result<(), PublicationReceiptError> { + self.check_native_reservation(txn, &prepared).await?; + let root = selected_ref(txn, "/") + .await? + .ok_or_else(|| integrity("published native root missing"))?; + let path = selected_ref(txn, &prepared.path) + .await? + .ok_or_else(|| integrity("published native path missing"))?; + if prepared + .old_path + .as_ref() + .is_some_and(|old| old.commit == path.commit) + || committed.receipt.namespace != prepared.path + || committed.receipt.new_oid != path.commit + || committed.receipt.old_oid != prepared.head.root.commit + || committed.receipt.writer_kind != "trunk_push" + || committed.receipt.writer_epoch != NATIVE_EPOCH + || committed.receipt.native_certificate_version.is_some() + || committed.outbox.operation_id != committed.receipt.operation_id + || committed.outbox.namespace != committed.receipt.namespace + || committed.outbox.sequence != committed.receipt.sequence + { + return Err(integrity( + "operation did not change its selected native ref", + )); + } + let resolved_path_tree = self + .resolve_path_tree_hash_in_txn(&root.tree, &prepared.path, txn) + .await + .map_err(|_| integrity("native path tree lookup failed"))? + .ok_or_else(|| integrity("native path tree is not materialized"))?; + if resolved_path_tree != path.tree { + return Err(integrity("native path tree does not match selected ref")); + } + let next = prepared + .head + .token + .sequence + .checked_add(1) + .ok_or_else(|| integrity("native sequence exhausted"))?; + let marked = txn.execute_raw(Statement::from_sql_and_values(txn.get_database_backend(), + "UPDATE mst2_publication SET native_certificate_version = 1 WHERE id = $1 AND native_certificate_version IS NULL", + [committed.receipt.id.into()], + )).await?; + if marked.rows_affected() != 1 { + return Err(integrity("native receipt marker changed")); + } + mst2_native_publication::ActiveModel { + receipt_id: Set(committed.receipt.id), + namespace: Set("/".to_owned()), + instance_id: Set(prepared.head.instance_id), + sequence: Set(next), + writer_epoch: Set(NATIVE_EPOCH), + old_root_commit: Set(prepared.head.root.commit.clone()), + old_root_tree: Set(prepared.head.root.tree.clone()), + root_commit: Set(root.commit.clone()), + root_tree: Set(root.tree.clone()), + origin_path: Set(prepared.path), + origin_ref: Set(MEGA_BRANCH_NAME.to_owned()), + old_path_commit: Set(prepared.old_path.as_ref().map(|old| old.commit.clone())), + old_path_tree: Set(prepared.old_path.map(|old| old.tree)), + path_commit: Set(path.commit), + path_tree: Set(path.tree), + } + .insert(txn) + .await?; + #[cfg(all(test, unix))] + tests::crash_checkpoint("native-certificate-written"); + let updated = txn.execute_raw(Statement::from_sql_and_values(txn.get_database_backend(), + "UPDATE mst2_native_head SET sequence = $1, root_commit = $2, root_tree = $3, state = 'READY', certificate_receipt_id = $4 \ + WHERE namespace = '/' AND sequence = $5 AND writer_epoch = $6 \ + AND certificate_receipt_id IS NOT DISTINCT FROM $7 AND root_commit = $8 AND root_tree = $9 AND txid_current() = $10", + [next.into(), root.commit.into(), root.tree.into(), committed.receipt.id.into(), prepared.head.token.sequence.into(), + NATIVE_EPOCH.into(), prepared.head.token.certificate.into(), prepared.head.root.commit.into(), + prepared.head.root.tree.into(), prepared.transaction_id.into()], + )).await?; + if updated.rows_affected() != 1 { + return Err(PublicationReceiptError::Conflict( + "native head CAS failed".into(), + )); + } + #[cfg(all(test, unix))] + tests::crash_checkpoint("native-head-written"); + Ok(()) + } + + pub(crate) async fn validate_native_historical_receipt_in_txn( + &self, + txn: &DatabaseTransaction, + receipt: &mst2_publication::Model, + ) -> Result<(), PublicationReceiptError> { + match receipt.native_certificate_version { + None => return Ok(()), + Some(1) => {} + _ => return Err(integrity("unsupported native certificate version")), + } + let certificate = mst2_native_publication::Entity::find_by_id(receipt.id) + .one(txn) + .await? + .ok_or_else(|| integrity("historical native certificate missing"))?; + if certificate.namespace != "/" + || certificate.sequence <= 0 + || certificate.writer_epoch != receipt.writer_epoch + || certificate.origin_path != receipt.namespace + || certificate.origin_ref != MEGA_BRANCH_NAME + || receipt.writer_kind != "trunk_push" + || certificate.old_root_commit != receipt.old_oid + || certificate.path_commit != receipt.new_oid + || certificate.old_path_commit.as_ref() == Some(&certificate.path_commit) + || canonical_instance(&certificate.instance_id)? != certificate.instance_id + { + return Err(integrity("historical native certificate identity mismatch")); + } + validate_root(&NativeRoot { + commit: certificate.root_commit, + tree: certificate.root_tree, + })?; + validate_root(&NativeRoot { + commit: certificate.old_root_commit, + tree: certificate.old_root_tree, + })?; + validate_root(&NativeRoot { + commit: certificate.path_commit, + tree: certificate.path_tree, + })?; + Ok(()) + } +} + +#[cfg(test)] +#[path = "native_publication_tests.rs"] +mod tests; diff --git a/src/jupiter/storage/native_publication_tests.rs b/src/jupiter/storage/native_publication_tests.rs new file mode 100644 index 00000000..c3e83323 --- /dev/null +++ b/src/jupiter/storage/native_publication_tests.rs @@ -0,0 +1,854 @@ +use std::sync::Arc; + +use git_internal::{ + hash::{ObjectHash, get_hash_kind}, + internal::object::tree::{TreeItem, TreeItemMode}, +}; +use sea_orm::{DatabaseTransaction, PaginatorTrait}; + +use super::*; +use crate::{ + callisto::{ + mst2_native_head, mst2_publication_outbox, sea_orm_active_enums::PushQueueKindEnum, + }, + jupiter::{ + migration::apply_migrations, + storage::{ + base_storage::BaseStorage, + mst2_publication_storage::{PublicationPreparation, PublicationRequest}, + push_queue_storage::{ClaimOutcome, EnqueueOutcome, EnqueueParams}, + }, + tests::test_db_connection, + }, +}; + +const INSTANCE: &str = "6ab219b0-4275-45ba-9d7b-7b0b633018cd"; + +async fn fixture() -> (tempfile::TempDir, MonoStorage, PushQueueStorage) { + let temp = tempfile::tempdir().unwrap(); + let db = test_db_connection(temp.path()).await; + apply_migrations(&db, false).await.unwrap(); + let base = BaseStorage::new(Arc::new(db)); + let mono = MonoStorage { base: base.clone() }; + for (id, path, commit, tree) in [(1_i64, "/", 'a', 'b'), (2, "/project", 'c', 'd')] { + mono.get_connection().execute_raw(Statement::from_sql_and_values( + mono.get_connection().get_database_backend(), + "INSERT INTO mega_refs (id,path,ref_name,ref_commit_hash,ref_tree_hash,created_at,updated_at,is_cl) \ + VALUES ($1,$2,$3,$4,$5,now(),now(),false)", + [id.into(), path.into(), MEGA_BRANCH_NAME.into(), commit.to_string().repeat(40).into(), tree.to_string().repeat(40).into()], + )).await.unwrap(); + } + let child_tree = "d".repeat(40); + let root_tree = "b".repeat(40); + let project = TreeItem::new( + TreeItemMode::Tree, + ObjectHash::from_hex_for_kind(get_hash_kind(), &child_tree).unwrap(), + "project".to_owned(), + ); + for (id, tree, sub_trees) in [ + (1_i64, child_tree, Vec::new()), + (2_i64, root_tree, project.to_data()), + ] { + mono.get_connection().execute_raw(Statement::from_sql_and_values( + mono.get_connection().get_database_backend(), + "INSERT INTO mega_tree (id,tree_id,sub_trees,size,created_at,pack_id,pack_offset,commit_id) \ + VALUES ($1,$2,$3,0,now(),$4,0,$5)", + [id.into(), tree.into(), sub_trees.into(), "".into(), "".into()], + )).await.unwrap(); + } + mono.initialize_native_publication(INSTANCE).await.unwrap(); + ( + temp, + mono, + PushQueueStorage::new(base).with_native_publication(true), + ) +} + +async fn enqueue_claim( + queue: &PushQueueStorage, + old: &str, + new: &str, + n: u64, +) -> push_queue::Model { + let operation = format!("{old}->{new}"); + let inserted = queue + .enqueue_atomic(EnqueueParams { + kind: PushQueueKindEnum::Push, + operation_id: &operation, + path: "/project", + old_id: old, + new_id: new, + requester: Some("alice"), + payload: serde_json::json!({"n":n,"commits":[new]}), + }) + .await + .unwrap(); + let EnqueueOutcome::Inserted { id } = inserted else { + panic!("fresh operation"); + }; + assert_eq!( + queue.claim_for_execution(id).await.unwrap(), + ClaimOutcome::Claimed + ); + queue.get_by_id(id).await.unwrap().unwrap() +} + +async fn publish_same_root( + mono: &MonoStorage, + txn: &DatabaseTransaction, + row: &push_queue::Model, +) -> i64 { + let request = PublicationRequest::from_trunk_queue(row).unwrap(); + let PublicationPreparation::Prepared(origin) = + mono.begin_publication_in_txn(txn, request).await.unwrap() + else { + panic!("fresh publication"); + }; + let native = mono + .reserve_native_publication_in_txn(txn, row, INSTANCE) + .await + .unwrap(); + assert!( + mono.cas_update_root_main_ref_in_txn( + txn, + row.expected_commit_hash.as_deref(), + row.expected_tree_hash.as_deref(), + row.expected_commit_hash.as_deref().unwrap(), + row.expected_tree_hash.as_deref().unwrap() + ) + .await + .unwrap() + ); + txn.execute_raw(Statement::from_sql_and_values(txn.get_database_backend(), + "UPDATE mega_refs SET ref_commit_hash = $1 WHERE path='/project' AND ref_name=$2 AND is_cl=false", + [row.new_id.clone().into(), MEGA_BRANCH_NAME.into()], + )).await.unwrap(); + let committed = mono + .record_publication_in_txn( + txn, + origin, + row.expected_commit_hash.as_deref().unwrap(), + &row.new_id, + ) + .await + .unwrap(); + mono.record_native_publication_in_txn(txn, native, &committed) + .await + .unwrap(); + txn.execute_raw(Statement::from_sql_and_values( + txn.get_database_backend(), + "UPDATE push_queue SET status='Done', landed_commit_id=$1 WHERE id=$2", + [row.new_id.clone().into(), row.id.into()], + )) + .await + .unwrap(); + committed.receipt.id +} + +#[tokio::test] +async fn same_root_changes_advance_global_head_and_historical_replay_ignores_latest() { + let (_temp, mono, queue) = fixture().await; + let root_before = mono.get_main_ref("/").await.unwrap().unwrap(); + let first = enqueue_claim(&queue, &"c".repeat(40), &"e".repeat(40), 1).await; + assert_eq!(first.expected_native_sequence, Some(0)); + assert_eq!(first.expected_native_epoch, Some(1)); + assert_eq!(first.expected_native_certificate, None); + let txn = mono.get_connection().begin().await.unwrap(); + let first_receipt = publish_same_root(&mono, &txn, &first).await; + txn.commit().await.unwrap(); + let first_head = mono.read_native_publication_head(INSTANCE).await.unwrap(); + assert_eq!(first_head.token.sequence, 1); + assert_eq!(first_head.root.commit, root_before.ref_commit_hash); + assert_eq!(first_head.root.tree, root_before.ref_tree_hash); + assert_eq!(first_head.token.certificate, Some(first_receipt)); + let second = enqueue_claim(&queue, &first.new_id, &"f".repeat(40), 1).await; + let txn = mono.get_connection().begin().await.unwrap(); + publish_same_root(&mono, &txn, &second).await; + txn.commit().await.unwrap(); + let txn = mono.get_connection().begin().await.unwrap(); + let replay = mono + .begin_publication_in_txn(&txn, PublicationRequest::from_trunk_queue(&first).unwrap()) + .await + .unwrap(); + assert!( + matches!(replay, PublicationPreparation::AlreadyCommitted(ref committed) if committed.receipt.id == first_receipt) + ); + txn.rollback().await.unwrap(); + assert_eq!( + mono.read_native_publication_head(INSTANCE) + .await + .unwrap() + .token + .sequence, + 2 + ); + assert_eq!( + mst2_native_publication::Entity::find() + .count(mono.get_connection()) + .await + .unwrap(), + 2 + ); + assert_eq!( + mst2_publication_outbox::Entity::find() + .count(mono.get_connection()) + .await + .unwrap(), + 2 + ); +} + +#[tokio::test] +async fn stale_same_root_claim_token_rejects_before_mutation_and_reset_clears_the_group() { + let (_temp, mono, queue) = fixture().await; + let row = enqueue_claim(&queue, &"c".repeat(40), &"e".repeat(40), 1).await; + // Independent transaction advances the head without changing either root OID. + mono.get_connection() + .execute_unprepared("UPDATE mst2_native_head SET sequence=1 WHERE namespace='/'") + .await + .unwrap(); + let txn = mono.get_connection().begin().await.unwrap(); + assert!(matches!( + mono.reserve_native_publication_in_txn(&txn, &row, INSTANCE) + .await, + Err(PublicationReceiptError::Conflict(_)) + )); + txn.rollback().await.unwrap(); + assert_eq!( + mono.get_main_ref("/project") + .await + .unwrap() + .unwrap() + .ref_commit_hash, + "c".repeat(40) + ); + assert!(queue.reset_running_to_queued(row.id).await.unwrap()); + let reset = queue.get_by_id(row.id).await.unwrap().unwrap(); + assert_eq!( + ( + reset.expected_commit_hash, + reset.expected_tree_hash, + reset.expected_native_sequence, + reset.expected_native_epoch, + reset.expected_native_certificate + ), + (None, None, None, None, None) + ); +} + +#[tokio::test] +async fn an_uncommitted_certificate_never_exposes_a_partial_head_to_another_connection() { + let (_temp, mono, queue) = fixture().await; + let first = enqueue_claim(&queue, &"c".repeat(40), &"e".repeat(40), 1).await; + let txn = mono.get_connection().begin().await.unwrap(); + publish_same_root(&mono, &txn, &first).await; + txn.commit().await.unwrap(); + let row = enqueue_claim(&queue, &first.new_id, &"f".repeat(40), 1).await; + let txn = mono.get_connection().begin().await.unwrap(); + publish_same_root(&mono, &txn, &row).await; + // The pool has two actual connections: one is held by txn, this read is on the other. + let before = tokio::time::timeout( + std::time::Duration::from_secs(5), + mono.read_native_publication_head(INSTANCE), + ) + .await + .unwrap() + .unwrap(); + assert_eq!(before.token.sequence, 1); + txn.commit().await.unwrap(); + let after = mono.read_native_publication_head(INSTANCE).await.unwrap(); + assert_eq!(after.token.sequence, 2); + assert_ne!(before.token.certificate, after.token.certificate); +} + +#[tokio::test] +async fn marked_history_fails_without_certificate_but_literal_v1_history_replays() { + let (_temp, mono, queue) = fixture().await; + let row = enqueue_claim(&queue, &"c".repeat(40), &"e".repeat(40), 1).await; + let txn = mono.get_connection().begin().await.unwrap(); + let receipt_id = publish_same_root(&mono, &txn, &row).await; + txn.commit().await.unwrap(); + mono.get_connection() + .execute_unprepared("DELETE FROM mst2_native_head; DELETE FROM mst2_native_publication;") + .await + .unwrap(); + let txn = mono.get_connection().begin().await.unwrap(); + assert!(matches!( + mono.begin_publication_in_txn(&txn, PublicationRequest::from_trunk_queue(&row).unwrap()) + .await, + Err(PublicationReceiptError::Integrity(_)) + )); + txn.rollback().await.unwrap(); + mono.get_connection() + .execute_raw(Statement::from_sql_and_values( + mono.get_connection().get_database_backend(), + "UPDATE mst2_publication SET native_certificate_version=NULL WHERE id=$1", + [receipt_id.into()], + )) + .await + .unwrap(); + let txn = mono.get_connection().begin().await.unwrap(); + assert!(matches!( + mono.begin_publication_in_txn(&txn, PublicationRequest::from_trunk_queue(&row).unwrap()) + .await + .unwrap(), + PublicationPreparation::AlreadyCommitted(_) + )); + txn.rollback().await.unwrap(); +} + +#[tokio::test] +async fn initialization_uses_only_a_counter_floor_and_cannot_reinitialize_or_serve_it() { + let (_temp, mono, queue) = fixture().await; + assert!(mono.read_native_publication_head(INSTANCE).await.is_err()); + assert!(mono.initialize_native_publication(INSTANCE).await.is_err()); + mono.get_connection().execute_unprepared("DELETE FROM mst2_native_head; INSERT INTO mst2_namespace_seq(namespace,sequence,epoch) VALUES('/older',19,1)").await.unwrap(); + mono.initialize_native_publication(INSTANCE).await.unwrap(); + let head = mst2_native_head::Entity::find() + .one(mono.get_connection()) + .await + .unwrap() + .unwrap(); + assert_eq!(head.sequence, 19); + assert_eq!(head.state, "INITIALIZING"); + assert_eq!(head.certificate_receipt_id, None); + let row = enqueue_claim(&queue, &"c".repeat(40), &"e".repeat(40), 1).await; + let txn = mono.get_connection().begin().await.unwrap(); + publish_same_root(&mono, &txn, &row).await; + txn.commit().await.unwrap(); + assert_eq!( + mono.read_native_publication_head(INSTANCE) + .await + .unwrap() + .token + .sequence, + 20 + ); +} + +#[tokio::test] +async fn reservation_owner_and_unchanged_selected_ref_cannot_issue_a_certificate() { + let (_temp, mono, queue) = fixture().await; + let row = enqueue_claim(&queue, &"c".repeat(40), &"e".repeat(40), 1).await; + let txn = mono.get_connection().begin().await.unwrap(); + let native = mono + .reserve_native_publication_in_txn(&txn, &row, INSTANCE) + .await + .unwrap(); + txn.rollback().await.unwrap(); + let other = mono.get_connection().begin().await.unwrap(); + assert!(matches!( + mono.finish_native_noop_in_txn(&other, native).await, + Err(PublicationReceiptError::Conflict(_)) + )); + other.rollback().await.unwrap(); + let txn = mono.get_connection().begin().await.unwrap(); + let native = mono + .reserve_native_publication_in_txn(&txn, &row, INSTANCE) + .await + .unwrap(); + let PublicationPreparation::Prepared(origin) = mono + .begin_publication_in_txn(&txn, PublicationRequest::from_trunk_queue(&row).unwrap()) + .await + .unwrap() + else { + panic!("fresh publication"); + }; + let committed = mono + .record_publication_in_txn( + &txn, + origin, + row.expected_commit_hash.as_deref().unwrap(), + &row.old_id, + ) + .await + .unwrap(); + assert!( + mono.record_native_publication_in_txn(&txn, native, &committed) + .await + .is_err() + ); + txn.rollback().await.unwrap(); + assert_eq!( + mst2_native_publication::Entity::find() + .count(mono.get_connection()) + .await + .unwrap(), + 0 + ); + assert_eq!( + mst2_publication_outbox::Entity::find() + .count(mono.get_connection()) + .await + .unwrap(), + 0 + ); +} + +#[tokio::test] +async fn missing_native_path_tree_keeps_head_initializing_and_rolls_back_receipt() { + let (_temp, mono, queue) = fixture().await; + let row = enqueue_claim(&queue, &"c".repeat(40), &"e".repeat(40), 1).await; + mono.get_connection() + .execute_raw(Statement::from_sql_and_values( + mono.get_connection().get_database_backend(), + "DELETE FROM mega_tree WHERE tree_id = $1", + ["d".repeat(40).into()], + )) + .await + .unwrap(); + + let txn = mono.get_connection().begin().await.unwrap(); + let request = PublicationRequest::from_trunk_queue(&row).unwrap(); + let PublicationPreparation::Prepared(origin) = + mono.begin_publication_in_txn(&txn, request).await.unwrap() + else { + panic!("fresh publication"); + }; + let native = mono + .reserve_native_publication_in_txn(&txn, &row, INSTANCE) + .await + .unwrap(); + assert!( + mono.cas_update_root_main_ref_in_txn( + &txn, + row.expected_commit_hash.as_deref(), + row.expected_tree_hash.as_deref(), + row.expected_commit_hash.as_deref().unwrap(), + row.expected_tree_hash.as_deref().unwrap(), + ) + .await + .unwrap() + ); + txn.execute_raw(Statement::from_sql_and_values( + txn.get_database_backend(), + "UPDATE mega_refs SET ref_commit_hash = $1 WHERE path='/project' AND ref_name=$2 AND is_cl=false", + [row.new_id.clone().into(), MEGA_BRANCH_NAME.into()], + )) + .await + .unwrap(); + let committed = mono + .record_publication_in_txn( + &txn, + origin, + row.expected_commit_hash.as_deref().unwrap(), + &row.new_id, + ) + .await + .unwrap(); + let error = mono + .record_native_publication_in_txn(&txn, native, &committed) + .await + .unwrap_err(); + assert!( + matches!(error, PublicationReceiptError::Integrity(message) if message.contains("path tree")) + ); + txn.rollback().await.unwrap(); + + assert!(mono.read_native_publication_head(INSTANCE).await.is_err()); + assert_eq!( + mst2_native_publication::Entity::find() + .count(mono.get_connection()) + .await + .unwrap(), + 0 + ); + assert_eq!( + mst2_publication_outbox::Entity::find() + .count(mono.get_connection()) + .await + .unwrap(), + 0 + ); +} + +#[cfg(unix)] +pub(super) fn crash_checkpoint(phase: &str) { + if std::env::var("MEGA_MST2_NATIVE_CRASH_PHASE") + .ok() + .as_deref() + == Some(phase) + { + unsafe { + libc::raise(libc::SIGKILL); + } + panic!("SIGKILL did not terminate the native worker"); + } +} + +async fn maintenance_fixture() -> (tempfile::TempDir, MonoStorage, PushQueueStorage, NativeRoot) { + use git_internal::{ + hash::HashKind, + internal::object::{ + blob::Blob, + commit::Commit, + tree::{Tree, TreeItem, TreeItemMode}, + }, + }; + + let (temp, mono, queue) = fixture().await; + mono.get_connection() + .execute_unprepared("DELETE FROM mst2_native_head") + .await + .unwrap(); + queue + .set_control_flags(Some(true), None, None) + .await + .unwrap(); + let blob = + Blob::from_content_bytes_with_kind(HashKind::Sha1, b"maintenance root".to_vec()).unwrap(); + let tree = Tree::from_tree_items_with_kind( + HashKind::Sha1, + vec![TreeItem::new( + TreeItemMode::Blob, + blob.id, + ".gitkeep".to_owned(), + )], + ) + .unwrap(); + let commit = + Commit::from_tree_id_with_kind(HashKind::Sha1, tree.id, vec![], "maintenance root") + .unwrap(); + mono.save_mega_trees(vec![tree.clone()], commit.id, None) + .await + .unwrap(); + mono.save_mega_commits(vec![commit.clone()], None) + .await + .unwrap(); + let root = NativeRoot { + commit: commit.id.to_string(), + tree: tree.id.to_string(), + }; + mono.get_connection().execute_raw(Statement::from_sql_and_values( + mono.get_connection().get_database_backend(), + "UPDATE mega_refs SET ref_commit_hash=$1,ref_tree_hash=$2 WHERE path='/' AND ref_name=$3 AND is_cl=false", + [root.commit.clone().into(), root.tree.clone().into(), MEGA_BRANCH_NAME.into()], + )).await.unwrap(); + (temp, mono, queue, root) +} + +async fn assert_no_maintenance_head(mono: &MonoStorage) { + assert_eq!( + mst2_native_head::Entity::find() + .count(mono.get_connection()) + .await + .unwrap(), + 0 + ); + assert_eq!( + mst2_native_publication::Entity::find() + .count(mono.get_connection()) + .await + .unwrap(), + 0 + ); + assert_eq!( + mst2_publication_outbox::Entity::find() + .count(mono.get_connection()) + .await + .unwrap(), + 0 + ); +} + +#[tokio::test] +async fn maintenance_initializes_only_a_floor_and_preserves_the_paused_gate() { + let (_temp, mono, queue, root) = maintenance_fixture().await; + mono.get_connection() + .execute_unprepared( + "INSERT INTO mst2_namespace_seq(namespace,sequence,epoch) VALUES('/older',23,1)", + ) + .await + .unwrap(); + mono.initialize_native_publication_for_maintenance(INSTANCE, &root) + .await + .unwrap(); + let before = mst2_native_head::Entity::find() + .one(mono.get_connection()) + .await + .unwrap() + .unwrap(); + assert_eq!(before.instance_id, INSTANCE); + assert_eq!(before.sequence, 23); + assert_eq!(before.writer_epoch, 1); + assert_eq!( + (&before.root_commit, &before.root_tree), + (&root.commit, &root.tree) + ); + assert_eq!(before.state, "INITIALIZING"); + assert_eq!(before.certificate_receipt_id, None); + assert!(queue.get_control().await.unwrap().paused); + assert!(mono.read_native_publication_head(INSTANCE).await.is_err()); + assert!( + mono.initialize_native_publication_for_maintenance(INSTANCE, &root) + .await + .is_err() + ); + assert!( + mono.initialize_native_publication_for_maintenance( + "11111111-2222-4333-8444-555555555555", + &root + ) + .await + .is_err() + ); + assert_eq!( + mst2_native_head::Entity::find() + .one(mono.get_connection()) + .await + .unwrap() + .unwrap(), + before + ); + assert_eq!( + selected_ref(mono.get_connection(), "/").await.unwrap(), + Some(root) + ); + assert_eq!( + mst2_native_publication::Entity::find() + .count(mono.get_connection()) + .await + .unwrap(), + 0 + ); + assert_eq!( + mst2_publication_outbox::Entity::find() + .count(mono.get_connection()) + .await + .unwrap(), + 0 + ); +} + +#[tokio::test] +async fn maintenance_rejects_unpaused_missing_or_hard_stopped_control_without_writes() { + for state in ["unpaused", "missing", "hard-stopped"] { + let (_temp, mono, queue, root) = maintenance_fixture().await; + match state { + "unpaused" => queue + .set_control_flags(Some(false), None, None) + .await + .unwrap(), + "hard-stopped" => queue + .set_control_flags(None, Some(true), None) + .await + .unwrap(), + "missing" => { + mono.get_connection() + .execute_unprepared("DELETE FROM queue_control") + .await + .unwrap(); + } + _ => unreachable!(), + } + assert!( + mono.initialize_native_publication_for_maintenance(INSTANCE, &root) + .await + .is_err(), + "{state}" + ); + assert_no_maintenance_head(&mono).await; + let control = crate::callisto::queue_control::Entity::find() + .one(mono.get_connection()) + .await + .unwrap(); + match state { + "missing" => assert!(control.is_none()), + "hard-stopped" => assert!(control.unwrap().hard_stopped), + _ => assert!(!control.unwrap().paused), + } + } +} + +#[tokio::test] +async fn maintenance_refuses_queued_and_running_work_until_drained() { + for status in ["Queued", "Running"] { + let (_temp, mono, queue, root) = maintenance_fixture().await; + queue + .set_control_flags(Some(false), None, None) + .await + .unwrap(); + let EnqueueOutcome::Inserted { id } = queue + .enqueue_atomic(EnqueueParams { + kind: PushQueueKindEnum::Push, + operation_id: "maintenance-pending", + path: "/project", + old_id: &"c".repeat(40), + new_id: &"e".repeat(40), + requester: None, + payload: serde_json::json!({"n":1,"commits":["e".repeat(40)]}), + }) + .await + .unwrap() + else { + panic!("fresh pending operation"); + }; + mono.get_connection() + .execute_raw(Statement::from_sql_and_values( + mono.get_connection().get_database_backend(), + "UPDATE push_queue SET status=$1::push_queue_status_enum WHERE id=$2", + [status.into(), id.into()], + )) + .await + .unwrap(); + queue + .set_control_flags(Some(true), None, None) + .await + .unwrap(); + assert!( + mono.initialize_native_publication_for_maintenance(INSTANCE, &root) + .await + .is_err(), + "{status}" + ); + assert_no_maintenance_head(&mono).await; + assert!(queue.get_control().await.unwrap().paused); + } +} + +#[tokio::test] +async fn maintenance_refuses_busy_writer_or_admission_locks_and_can_retry() { + for lock in ["writer", "admission"] { + let (_temp, mono, _queue, root) = maintenance_fixture().await; + let other = mono.get_connection().begin().await.unwrap(); + if lock == "writer" { + PushQueueStorage::acquire_mono_write_lock(&other) + .await + .unwrap(); + } else { + other + .execute_unprepared("SELECT id FROM queue_control WHERE id=1 FOR UPDATE") + .await + .unwrap(); + } + let result = tokio::time::timeout( + std::time::Duration::from_secs(5), + mono.initialize_native_publication_for_maintenance(INSTANCE, &root), + ) + .await; + assert!( + result + .expect("maintenance lock refusal must be bounded") + .is_err(), + "{lock}" + ); + assert_no_maintenance_head(&mono).await; + other.rollback().await.unwrap(); + mono.initialize_native_publication_for_maintenance(INSTANCE, &root) + .await + .unwrap(); + } +} + +#[tokio::test] +async fn maintenance_rejects_wrong_or_ambiguous_root_and_invalid_counters() { + for damage in [ + "commit", + "tree", + "missing", + "ambiguous", + "negative", + "exhausted", + ] { + let (_temp, mono, _queue, mut root) = maintenance_fixture().await; + match damage { + "commit" => root.commit = "e".repeat(40), + "tree" => root.tree = "e".repeat(40), + "missing" => { + mono.get_connection() + .execute_unprepared("DELETE FROM mega_refs WHERE path='/'") + .await + .unwrap(); + } + "ambiguous" => { + // Simulate a corrupted/old schema, independently of the normal + // uniqueness protection on selected refs. + mono.get_connection() + .execute_unprepared("DROP INDEX uniq_mref_path") + .await + .unwrap(); + mono.get_connection().execute_unprepared("INSERT INTO mega_refs(id,path,ref_name,ref_commit_hash,ref_tree_hash,created_at,updated_at,is_cl) SELECT 3,path,ref_name,ref_commit_hash,ref_tree_hash,created_at,updated_at,is_cl FROM mega_refs WHERE path='/'").await.unwrap(); + } + "negative" | "exhausted" => { + let floor = if damage == "negative" { + -1_i64 + } else { + i64::MAX + }; + mono.get_connection().execute_raw(Statement::from_sql_and_values( + mono.get_connection().get_database_backend(), + "INSERT INTO mst2_namespace_seq(namespace,sequence,epoch) VALUES('/older',$1,1)", [floor.into()], + )).await.unwrap(); + } + _ => unreachable!(), + } + assert!( + mono.initialize_native_publication_for_maintenance(INSTANCE, &root) + .await + .is_err(), + "{damage}" + ); + assert_no_maintenance_head(&mono).await; + } +} + +#[tokio::test] +async fn maintenance_rejects_nil_instance_and_non_sha1_expected_roots() { + let (_temp, mono, _queue, root) = maintenance_fixture().await; + for instance in ["invalid", "00000000-0000-0000-0000-000000000000"] { + assert!( + mono.initialize_native_publication_for_maintenance(instance, &root) + .await + .is_err() + ); + } + for commit in ["a".repeat(64), "A".repeat(40)] { + let wrong = NativeRoot { + commit, + tree: root.tree.clone(), + }; + assert!( + mono.initialize_native_publication_for_maintenance(INSTANCE, &wrong) + .await + .is_err() + ); + } + assert_no_maintenance_head(&mono).await; +} + +#[tokio::test] +async fn maintenance_refuses_missing_commit_tree_or_a_mismatched_commit_tree() { + for damage in ["commit", "tree", "wrong-tree"] { + let (_temp, mono, _queue, root) = maintenance_fixture().await; + let sql = match damage { + "commit" => "DELETE FROM mega_commit WHERE commit_id=$1", + "tree" => "DELETE FROM mega_tree WHERE tree_id=$1", + "wrong-tree" => "UPDATE mega_commit SET tree=repeat('e',40) WHERE commit_id=$1", + _ => unreachable!(), + }; + let id = if damage == "tree" { + &root.tree + } else { + &root.commit + }; + mono.get_connection() + .execute_raw(Statement::from_sql_and_values( + mono.get_connection().get_database_backend(), + sql, + [id.clone().into()], + )) + .await + .unwrap(); + assert!( + mono.initialize_native_publication_for_maintenance(INSTANCE, &root) + .await + .is_err(), + "{damage}" + ); + assert_no_maintenance_head(&mono).await; + assert_eq!( + selected_ref(mono.get_connection(), "/").await.unwrap(), + Some(root) + ); + } +} diff --git a/src/jupiter/storage/native_snapshot_metadata_routes.rs b/src/jupiter/storage/native_snapshot_metadata_routes.rs new file mode 100644 index 00000000..8326fca6 --- /dev/null +++ b/src/jupiter/storage/native_snapshot_metadata_routes.rs @@ -0,0 +1,492 @@ +//! Fixed-root persisted META reads under the original generic lease authority. + +use std::collections::{BTreeSet, HashMap}; + +use mst2_codec::metapage::{HEADER_LEN, PAGE_MAX_BYTES, Page, page_id}; +use sea_orm::{ConnectionTrait, DatabaseTransaction, QueryResult}; + +use super::{ + PostgresNativeSessionRepository, SnapshotContext, SnapshotError, SnapshotErrorCode, + decode_context, expired, finish, integrity, internal, routes, statement, +}; +use crate::ceres::snapshot::{ + retention_dag::MetadataDagLimits, view::validate_scope_relative_path, +}; + +pub(crate) struct MetadataRouteRequest<'a> { + pub directory_path: &'a str, + pub route: &'a [u8], + pub expected_digest: Option<&'a str>, +} + +#[derive(Debug, Default, Clone, Copy)] +pub(crate) struct PersistedMetadataReadWork { + pub page_queries: u64, + pub pages_loaded: u64, + pub payload_bytes: u64, + pub walk_visits: u64, + pub edge_references_checked: u64, +} + +pub(crate) struct PersistedMetadataRouteBatch { + pub pages: Vec<([u8; 32], Vec)>, + pub work: PersistedMetadataReadWork, +} + +#[cfg(test)] +tokio::task_local! { + static METADATA_READ_BARRIERS: (std::sync::Arc, std::sync::Arc); +} + +#[cfg(test)] +pub(crate) async fn with_metadata_read_barriers( + captured: std::sync::Arc, + release: std::sync::Arc, + future: F, +) -> F::Output { + METADATA_READ_BARRIERS + .scope((captured, release), future) + .await +} + +impl PostgresNativeSessionRepository { + pub(crate) async fn metadata_routes( + &self, + context: &SnapshotContext, + requests: &[MetadataRouteRequest<'_>], + ) -> Result { + if requests.is_empty() || requests.len() > 64 { + return Err(limit("items must hold 1..64 entries")); + } + for request in requests { + validate_scope_relative_path(request.directory_path)?; + let scope = &context.built.descriptor.scope; + let absolute = if scope == "/" { + request.directory_path.to_owned() + } else if request.directory_path == "/" { + scope.clone() + } else { + format!("{scope}{}", request.directory_path) + }; + validate_scope_relative_path(&absolute)?; + } + let installer = self.installer().await?; + let txn = self.transaction().await?; + let result = async { + installer.verify_primary_connection(&txn).await?; + if routes::lease(&txn, installer, &context.lease_id) + .await? + .as_deref() + != Some(context.built.snapshot_id.as_str()) + { + return Err(expired()); + } + // This barrier captures the namespace and uses bounded lock acquisition. + // Do not acquire the publication/route writer lock after retention. + installer.metadata_read_barrier(&txn).await?; + if routes::lease(&txn, installer, &context.lease_id) + .await? + .as_deref() + != Some(context.built.snapshot_id.as_str()) + { + return Err(expired()); + } + let row = txn + .query_one_raw(statement( + self.session_sql(installer).await, + [ + context.built.snapshot_id.clone().into(), + context.lease_id.clone().into(), + ], + )) + .await + .map_err(internal)? + .ok_or_else(expired)?; + installer.verify_primary_scope_row(&row)?; + let current = decode_context( + &row, + &context.built.snapshot_id, + &context.lease_id, + &context.built.instance_id, + )?; + if current.built.descriptor != context.built.descriptor + || current.commit_oid != context.commit_oid + || current.root_tree_oid != context.root_tree_oid + || current.authorization_epoch != context.authorization_epoch + { + return Err(integrity("fixed session changed during metadata read")); + } + let prepare_id: String = row.try_get("", "prepare_id").map_err(internal)?; + let mut reader = Reader { + txn: &txn, + prepare_id, + codec: context.built.descriptor.metadata_codec() as i16, + cache: HashMap::new(), + work: PersistedMetadataReadWork::default(), + limits: MetadataDagLimits::default(), + }; + let mut seen = BTreeSet::new(); + let mut pages = Vec::new(); + for request in requests { + let root = reader + .directory( + context.built.descriptor.metadata_root, + request.directory_path, + ) + .await?; + let route = reader.route(root, request.route).await?; + let reached = route + .last() + .ok_or_else(|| integrity("metadata route is empty"))?; + if let Some(expected) = request.expected_digest + && expected != format!("sha256:{}", hex::encode(reached)) + { + return Err(SnapshotError::new( + SnapshotErrorCode::DigestMismatch, + format!( + "{}: route does not reach expected_digest", + request.directory_path + ), + )); + } + for id in route { + if seen.insert(id) { + let page = reader + .cache + .get(&id) + .ok_or_else(|| integrity("read metadata page is missing"))?; + pages.push((id, page.bytes.clone())); + } + } + } + Ok(PersistedMetadataRouteBatch { + pages, + work: reader.work, + }) + } + .await; + // All bytes are owned before releasing protection. Frame delivery keeps + // the existing per-frame authentication and lease revalidation. + finish(txn, result).await + } +} + +struct StoredPage { + bytes: Vec, + page: Page, + entries: u64, +} + +struct Reader<'a> { + txn: &'a DatabaseTransaction, + prepare_id: String, + codec: i16, + cache: HashMap<[u8; 32], StoredPage>, + work: PersistedMetadataReadWork, + limits: MetadataDagLimits, +} + +impl Reader<'_> { + async fn load(&mut self, id: [u8; 32]) -> Result<(), SnapshotError> { + self.work.walk_visits += 1; + if self.work.walk_visits > self.limits.prepare_entry_visits as u64 { + return Err(limit("metadata route work budget exceeded")); + } + if self.cache.contains_key(&id) { + return Ok(()); + } + if self.cache.len() >= self.limits.nodes { + return Err(limit("metadata route page budget exceeded")); + } + self.work.page_queries += 1; + let row = self + .txn + .query_one_raw(statement( + PAGE_SQL, + [self.prepare_id.clone().into(), id.to_vec().into()], + )) + .await + .map_err(internal)? + .ok_or_else(|| unavailable("metadata page is not a prepared member"))?; + let page = decode_page(&row, id, self.codec)?; + self.work.payload_bytes = self + .work + .payload_bytes + .checked_add(page.bytes.len() as u64) + .filter(|bytes| *bytes <= self.limits.payload_bytes) + .ok_or_else(|| limit("metadata route payload budget exceeded"))?; + let expected = page_references(&page.page); + self.work.edge_references_checked += expected.len() as u64; + if self.work.edge_references_checked > self.limits.edges as u64 { + return Err(limit("metadata route edge budget exceeded")); + } + let edges: String = row.try_get("", "outgoing_edges").map_err(internal)?; + let edges: Vec = serde_json::from_str(&edges).map_err(internal)?; + let actual: BTreeSet<_> = edges.iter().cloned().collect(); + if edges.len() != actual.len() || actual != expected { + return Err(integrity( + "metadata graph edges disagree with fixed page references", + )); + } + self.work.pages_loaded += 1; + self.cache.insert(id, page); + #[cfg(test)] + if self.cache.len() == 1 + && let Ok((captured, release)) = METADATA_READ_BARRIERS.try_with(Clone::clone) + { + captured.wait().await; + release.wait().await; + } + Ok(()) + } + + async fn descend(&mut self, parent: [u8; 32], label: u8) -> Result<[u8; 32], SnapshotError> { + self.load(parent).await?; + let (prefix, child) = match &self.cache[&parent].page { + Page::Branch { + prefix, children, .. + } => { + let child = children + .iter() + .find(|child| child.label == label) + .ok_or_else(|| absent("route label is absent in fixed metadata page"))?; + (prefix.clone(), child.clone()) + } + Page::Leaf { .. } => return Err(absent("route descends past a leaf page")), + }; + let id = child.child_page_id; + self.load(id).await?; + let received = &self.cache[&id]; + let mut partition = prefix; + partition.push(label); + let valid_prefix = match &received.page { + Page::Leaf { entries } => entries + .iter() + .all(|entry| entry.name.starts_with(&partition)), + Page::Branch { prefix, .. } => prefix.starts_with(&partition), + }; + if received.entries != child.subtree_entries || !valid_prefix { + return Err(integrity( + "metadata child differs from fixed radix partition", + )); + } + Ok(id) + } + + async fn directory( + &mut self, + mut root: [u8; 32], + path: &str, + ) -> Result<[u8; 32], SnapshotError> { + if path == "/" { + return Ok(root); + } + for name in path[1..].split('/') { + let mut id = root; + loop { + self.load(id).await?; + let entry = match &self.cache[&id].page { + Page::Leaf { entries } => entries + .iter() + .find(|entry| entry.name == name.as_bytes()) + .cloned(), + Page::Branch { + prefix, terminal, .. + } => { + if name.as_bytes() == prefix { + terminal.clone() + } else { + if !name.as_bytes().starts_with(prefix) { + return Err(absent("name is absent in fixed directory")); + } + let label = name + .as_bytes() + .get(prefix.len()) + .copied() + .ok_or_else(|| absent("name is absent in fixed directory"))?; + id = self.descend(id, label).await?; + continue; + } + } + } + .ok_or_else(|| absent("name is absent in fixed directory"))?; + if !entry.is_dir() { + return Err(SnapshotError::new( + SnapshotErrorCode::NotDirectory, + format!("{path} is not a directory"), + )); + } + root = entry.child_root; + break; + } + } + Ok(root) + } + + async fn route( + &mut self, + root: [u8; 32], + labels: &[u8], + ) -> Result, SnapshotError> { + self.load(root).await?; + let mut pages = vec![root]; + let mut current = root; + for label in labels { + current = self.descend(current, *label).await?; + pages.push(current); + } + Ok(pages) + } +} + +const PAGE_SQL: &str = "SELECT m.expected_size,m.generation AS member_generation,p.graph_domain AS prepare_graph_domain, + b.metadata_codec AS payload_codec,b.byte_size AS payload_size,octet_length(b.payload) AS actual_size, + CASE WHEN octet_length(b.payload) BETWEEN 20 AND 16384 THEN b.payload END AS payload, + b.generation AS payload_generation,c.generation AS current_generation, + l.generation AS lifetime_generation,l.graph_domain,l.state AS lifetime_state, + l.metadata_codec AS lifetime_codec,l.expected_size AS lifetime_size, + n.state AS graph_state,n.kind AS graph_kind,n.bytes AS graph_bytes, + EXISTS(SELECT 1 FROM mst2_retention_gc_op g WHERE g.node_id='page:sha256:'||encode(m.page_id,'hex') + AND g.operation='REMOVE' AND g.state IN ('PENDING','APPLIED')) AS tombstone, + COALESCE((SELECT jsonb_agg(e.child_id) FROM + (SELECT child_id FROM mst2_retention_edge WHERE parent_id='page:sha256:'||encode(m.page_id,'hex') + ORDER BY child_id LIMIT 258) e),'[]'::jsonb)::text AS outgoing_edges + FROM mst2_metadata_prepare_page m + JOIN mst2_metadata_prepare p ON p.prepare_id=m.prepare_id + LEFT JOIN mst2_metadata_payload b ON b.page_id=m.page_id + LEFT JOIN mst2_metadata_current c ON c.page_id=m.page_id + LEFT JOIN mst2_metadata_lifetime l ON l.page_id=c.page_id AND l.generation=c.generation + LEFT JOIN mst2_retention_node n ON n.node_id='page:sha256:'||encode(m.page_id,'hex') + WHERE m.prepare_id=$1 AND m.page_id=$2"; + +fn decode_page(row: &QueryResult, id: [u8; 32], codec: i16) -> Result { + if row + .try_get::>("", "prepare_graph_domain") + .map_err(internal)? + .as_deref() + .is_some_and(|domain| domain != "generic-v1") + { + return Err(unavailable( + "fixed metadata preparation is outside the generic namespace", + )); + } + let size: i32 = row.try_get("", "expected_size").map_err(internal)?; + if !(HEADER_LEN..=PAGE_MAX_BYTES).contains(&(size as usize)) + || row + .try_get::>("", "payload_codec") + .map_err(internal)? + != Some(codec) + || row + .try_get::>("", "payload_size") + .map_err(internal)? + != Some(size) + || row + .try_get::>("", "actual_size") + .map_err(internal)? + != Some(size) + { + return Err(unavailable( + "fixed metadata payload profile or size is unavailable", + )); + } + if row.try_get::("", "tombstone").map_err(internal)? + || row + .try_get::>("", "graph_state") + .map_err(internal)? + .as_deref() + != Some("LIVE") + || row + .try_get::>("", "graph_kind") + .map_err(internal)? + .as_deref() + != Some("page") + || row + .try_get::>("", "graph_bytes") + .map_err(internal)? + != Some(size as i64) + { + return Err(unavailable("fixed generic metadata graph is unavailable")); + } + let member: Option = row.try_get("", "member_generation").map_err(internal)?; + let payload: Option = row.try_get("", "payload_generation").map_err(internal)?; + let current: Option = row.try_get("", "current_generation").map_err(internal)?; + let lifetime: Option = row.try_get("", "lifetime_generation").map_err(internal)?; + if payload != current || payload != lifetime || member.is_some() && member != payload { + return Err(integrity( + "fixed generic page differs from its exact physical lifetime", + )); + } + if let Some(generation) = payload + && (generation <= 0 + || row + .try_get::>("", "graph_domain") + .map_err(internal)? + .as_deref() + != Some("generic-v1") + || row + .try_get::>("", "lifetime_state") + .map_err(internal)? + .as_deref() + != Some("LIVE") + || row + .try_get::>("", "lifetime_codec") + .map_err(internal)? + != Some(codec) + || row + .try_get::>("", "lifetime_size") + .map_err(internal)? + != Some(size)) + { + return Err(unavailable( + "fixed metadata lifetime cannot be served in the generic namespace", + )); + } + let bytes: Vec = row + .try_get::>>("", "payload") + .map_err(internal)? + .ok_or_else(|| unavailable("fixed metadata payload is missing"))?; + if page_id(&bytes) != id { + return Err(integrity("fixed metadata payload digest mismatch")); + } + let (page, entries) = Page::decode(&bytes) + .map_err(|_| integrity("fixed metadata payload is not a canonical page"))?; + Ok(StoredPage { + bytes, + page, + entries, + }) +} + +fn page_references(page: &Page) -> BTreeSet { + let mut edges = BTreeSet::new(); + let entries = match page { + Page::Leaf { entries } => entries.as_slice(), + Page::Branch { + terminal, children, .. + } => { + edges.extend( + children + .iter() + .map(|child| format!("page:sha256:{}", hex::encode(child.child_page_id))), + ); + terminal.as_slice() + } + }; + edges.extend( + entries + .iter() + .filter(|entry| entry.is_dir()) + .map(|entry| format!("page:sha256:{}", hex::encode(entry.child_root))), + ); + edges +} + +fn absent(message: &str) -> SnapshotError { + SnapshotError::new(SnapshotErrorCode::PathNotFound, message) +} +fn unavailable(message: &str) -> SnapshotError { + SnapshotError::new(SnapshotErrorCode::ObjectUnavailable, message) +} +fn limit(message: &str) -> SnapshotError { + SnapshotError::new(SnapshotErrorCode::LimitExceeded, message) +} diff --git a/src/jupiter/storage/native_snapshot_routes.rs b/src/jupiter/storage/native_snapshot_routes.rs new file mode 100644 index 00000000..ac818034 --- /dev/null +++ b/src/jupiter/storage/native_snapshot_routes.rs @@ -0,0 +1,122 @@ +//! Permanent namespace selection precedes the generic physical session proof. + +use super::{ + ConnectionTrait, DatabaseTransaction, PostgresMetadataInstallRepository, SESSION_SQL, + SnapshotError, integrity, internal, statement, +}; + +fn function(installer: &PostgresMetadataInstallRepository, name: &str) -> String { + format!( + "\"{}\".{name}", + installer.captured_schema().replace('"', "\"\"") + ) +} + +pub(super) async fn enter( + txn: &DatabaseTransaction, + installer: &PostgresMetadataInstallRepository, +) -> Result<(), SnapshotError> { + installer.verify_primary_connection(txn).await?; + txn.execute_raw(statement( + &format!( + "SELECT {}(current_schema())", + function(installer, "mst2_route_enter") + ), + [], + )) + .await + .map_err(internal)?; + generic_path(txn, installer).await?; + Ok(()) +} + +pub(super) async fn generic_path( + txn: &DatabaseTransaction, + installer: &PostgresMetadataInstallRepository, +) -> Result<(), SnapshotError> { + let schema = installer.captured_schema().replace('"', "\"\""); + txn.execute_unprepared(&format!( + "SET LOCAL search_path=\"{schema}\",pg_catalog,pg_temp" + )) + .await + .map_err(internal)?; + Ok(()) +} + +pub(super) fn session_sql(installer: &PostgresMetadataInstallRepository) -> String { + let schema = installer.captured_schema().replace('"', "\"\""); + let mut sql = SESSION_SQL.to_owned(); + for table in [ + "mst2_metadata_storage_scope", + "mst2_snapshot_context", + "mst2_snapshot_lease", + "mst2_retention_node", + "mst2_retention_root", + "mst2_metadata_prepare", + "mst2_native_publication", + "mst2_publication", + "mst2_publication_outbox", + ] { + for join in ["FROM", "JOIN"] { + sql = sql.replace( + &format!("{join} {table} "), + &format!("{join} \"{schema}\".{table} "), + ); + } + } + sql +} + +pub(super) async fn snapshot( + connection: &C, + installer: &PostgresMetadataInstallRepository, + sid: &str, +) -> Result<(), SnapshotError> { + let row = connection + .query_one_raw(statement( + &format!( + "SELECT * FROM {}($1,current_schema())", + function(installer, "mst2_route_select_snapshot") + ), + [sid.into()], + )) + .await + .map_err(internal)? + .ok_or_else(|| integrity("snapshot storage route selection is missing"))?; + let context: bool = row.try_get("", "context_present").map_err(internal)?; + let route: bool = row.try_get("", "route_present").map_err(internal)?; + let valid: bool = row.try_get("", "valid").map_err(internal)?; + if context != route || context && !valid { + return Err(integrity( + "snapshot storage route conflicts with its immutable generic session", + )); + } + Ok(()) +} + +pub(super) async fn lease( + connection: &C, + installer: &PostgresMetadataInstallRepository, + lease_id: &str, +) -> Result, SnapshotError> { + let row = connection + .query_one_raw(statement( + &format!( + "SELECT * FROM {}($1,current_schema())", + function(installer, "mst2_route_select_lease") + ), + [lease_id.into()], + )) + .await + .map_err(internal)? + .ok_or_else(|| integrity("lease storage route selection is missing"))?; + let sid: Option = row.try_get("", "actual_sid").map_err(internal)?; + let route: bool = row.try_get("", "route_present").map_err(internal)?; + let valid: bool = row.try_get("", "valid").map_err(internal)?; + if sid.is_some() != route || sid.is_some() && !valid { + return Err(integrity( + "lease storage route conflicts with its exact generic incarnation", + )); + } + Ok(sid) +} diff --git a/src/jupiter/storage/native_snapshot_session.rs b/src/jupiter/storage/native_snapshot_session.rs new file mode 100644 index 00000000..a13180bd --- /dev/null +++ b/src/jupiter/storage/native_snapshot_session.rs @@ -0,0 +1,831 @@ +//! Durable native HTTP sessions. Deployment bearer authentication is enforced +//! by the router; leases retain data and do not grant per-principal permission. + +use std::{collections::HashMap, sync::Arc}; + +use mst2_codec::descriptor::ServingDescriptor; +use sea_orm::{ + ConnectionTrait, DatabaseConnection, DatabaseTransaction, DbBackend, EntityTrait, + IsolationLevel, QueryResult, Statement, TransactionTrait, +}; +use tokio::sync::{Mutex, OnceCell}; + +use super::{ + mono_storage::MonoStorage, + mst2_retention::{PostgresRetentionRepository, RETENTION_LOCK_KEY}, + native_metadata_install::{ + MetadataInstallError, PostgresMetadataInstallRepository, PreparedMetadataReceipt, + }, + native_publication_storage::{NativePublicationHead, decode_native_observation}, +}; +use crate::{ + callisto::mst2_snapshot_context, + ceres::snapshot::{ + descriptor::BuiltDescriptor, + error::{SnapshotError, SnapshotErrorCode}, + pages::PreparedNativeMetadataRetention, + retention::RetentionRoot, + runtime::{LeaseRenewed, SnapshotContext}, + view::{SnapshotView, hex}, + }, +}; + +pub(crate) struct PostgresNativeSessionRepository { + connection: DatabaseConnection, + installer: OnceCell, + qualified_session_sql: OnceCell, + restored: Mutex>>>, +} + +impl PostgresNativeSessionRepository { + pub(crate) fn new(connection: DatabaseConnection) -> Self { + Self { + connection, + installer: OnceCell::new(), + qualified_session_sql: OnceCell::new(), + restored: Mutex::new(HashMap::new()), + } + } + + async fn installer(&self) -> Result<&PostgresMetadataInstallRepository, SnapshotError> { + self.installer + .get_or_try_init(|| PostgresMetadataInstallRepository::new(self.connection.clone())) + .await + } + + async fn session_sql(&self, installer: &PostgresMetadataInstallRepository) -> &str { + self.qualified_session_sql + .get_or_init(|| async { routes::session_sql(installer) }) + .await + } + + pub(crate) async fn install( + &self, + built: &BuiltDescriptor, + prepared: &PreparedNativeMetadataRetention, + ) -> Result { + let (receipt, work) = self.install_with_work(built, prepared).await?; + tracing::debug!( + snapshot_id = %built.snapshot_id, + requested_pages = work.requested_pages, + payload_pages_omitted = work.payload_pages_omitted, + payload_pages_encoded = work.payload_pages_encoded, + requested_payload_bytes_validated = work.requested_payload_bytes_validated, + payload_bytes_omitted = work.payload_bytes_omitted, + payload_bytes_encoded = work.payload_bytes_encoded, + metadata_parameter_bytes = work.metadata_parameter_bytes, + payload_parameter_bytes = work.payload_parameter_bytes, + payload_transactions = work.transactions, + registration_queries = work.registration_queries, + requested_member_queries = work.requested_member_queries, + classification_batches = work.classification_batches, + insert_statements = work.insert_statements, + byte_comparison_queries = work.byte_comparison_queries, + committed_replay_pages = work.committed_replay_pages, + payload_batch_elapsed_micros = work.elapsed_micros, + "native metadata payload batch work" + ); + Ok(receipt) + } + + pub(crate) async fn install_with_work( + &self, + built: &BuiltDescriptor, + prepared: &PreparedNativeMetadataRetention, + ) -> Result< + ( + PreparedMetadataReceipt, + super::native_metadata_install::LegacyPayloadInstallWork, + ), + SnapshotError, + > { + if prepared.dag().root() != built.descriptor.metadata_root + || prepared.scope() != built.descriptor.scope + { + return Err(integrity( + "prepared DAG differs from the serving descriptor", + )); + } + let installer = self.installer().await?; + let intent = installer + .begin_intent(&format!("http:{}", built.snapshot_id), prepared) + .await + .map_err(install_error)?; + let capability = installer + .mint_legacy_install_capability(&intent) + .await + .map_err(install_error)?; + let mut work = super::native_metadata_install::LegacyPayloadInstallWork::default(); + for pages in prepared.dag().payloads().chunks(64) { + let batch = installer + .install_missing_pages_validated(&capability, pages) + .await + .map_err(install_error)?; + work.record(batch); + } + let receipt = installer.finalize(&intent).await.map_err(install_error)?; + Ok((receipt, work)) + } + + /// Projection and payload installation happen before this boundary. + /// Cold handoff hashes a <=2 MiB plan, without a page/edge database scan. + /// The formal native head and LIVE root are selected with lease creation. + pub(crate) async fn open( + &self, + expected: &NativePublicationHead, + built: &BuiltDescriptor, + receipt: Option<&PreparedMetadataReceipt>, + lease_seconds: u64, + ) -> Result, SnapshotError> { + let installer = self.installer().await?; + let txn = self.transaction().await?; + let result = async { + routes::enter(&txn, installer).await?; + let current = MonoStorage::read_native_publication_head_from(&txn, &expected.instance_id) + .await.map_err(|_| not_ready("native publication is not ready"))?; + if current.root != expected.root || current.token != expected.token { + return Err(not_ready("native publication advanced during preparation; retry resolve")); + } + retention_lock(&txn).await?; + routes::snapshot(&txn, installer, &built.snapshot_id).await?; + let existing = mst2_snapshot_context::Entity::find_by_id(built.snapshot_id.clone()) + .one(&txn).await.map_err(internal)?; + let prepare_id = if let Some(existing) = existing { + if existing.canonical_descriptor != built.descriptor.encode().map_err(internal)? + || existing.commit_oid != expected.root.commit || existing.root_tree_oid != expected.root.tree + || existing.instance_id != expected.instance_id + { + return Err(integrity("immutable session source conflicts with selected publication")); + } + if existing.state != "READY" || existing.authorization_epoch != 1 { + return Err(forbidden()); + } + existing.prepare_id + } else { + let Some(receipt) = receipt else { return Ok(None); }; + if receipt.metadata_root() != built.descriptor.metadata_root { + return Err(integrity("metadata receipt root differs from descriptor")); + } + let prepare_id = receipt.intent().prepare_id().to_owned(); + let tagged_tree=git_internal::hash::ObjectHash::from_hex_for_kind( + git_internal::hash::get_hash_kind(),&expected.root.tree).map_err(internal)?.to_tagged_string(); + self.installer().await?.verify_receipt_in_txn(&txn, receipt,&tagged_tree,&built.descriptor.scope).await?; + txn.execute_raw(statement( + "INSERT INTO mst2_snapshot_context(snapshot_id,canonical_descriptor,instance_id,commit_oid, + root_tree_oid,metadata_root,prepare_id,publication_sequence,writer_epoch,certificate_receipt_id, + authorization_epoch,state) VALUES($1,$2,$3,$4,$5,$6,$7,$8,$9,$10,1,'READY')", + [built.snapshot_id.clone().into(),built.descriptor.encode().map_err(internal)?.into(), + expected.instance_id.clone().into(),expected.root.commit.clone().into(),expected.root.tree.clone().into(), + built.descriptor.metadata_root.to_vec().into(),prepare_id.clone().into(),expected.token.sequence.into(), + expected.token.epoch.into(),expected.token.certificate.into()], + )).await.map_err(internal)?; + prepare_id + }; + let node = root_node(built); + let lease_id = uuid::Uuid::new_v4().to_string(); + PostgresRetentionRepository::acquire_existing_roots_in_txn(&txn,&node,&[ + RetentionRoot::Pin(format!("session:{}",built.snapshot_id)),RetentionRoot::Lease(lease_id.clone()) + ]).await?; + PostgresRetentionRepository::release_root_in_txn(&txn,&RetentionRoot::Prepare(prepare_id)).await?; + let row = txn.query_one_raw(statement( + "INSERT INTO mst2_snapshot_lease(lease_id,snapshot_id,authorization_epoch,expires_at_unix,state, + publication_sequence,writer_epoch,certificate_receipt_id) + VALUES($1,$2,1,floor(extract(epoch FROM clock_timestamp()))::bigint+$3,'ACTIVE',$4,$5,$6) + RETURNING expires_at_unix", + [lease_id.clone().into(),built.snapshot_id.clone().into(),(lease_seconds.clamp(1,3600) as i64).into(), + expected.token.sequence.into(),expected.token.epoch.into(),expected.token.certificate.into()], + )).await.map_err(internal)?.ok_or_else(|| internal("lease insert returned no deadline"))?; + expire_locked(&txn).await?; + Ok(Some(SnapshotContext { built:built.clone(),commit_oid:expected.root.commit.clone(), + root_tree_oid:expected.root.tree.clone(),lease_id, + lease_expires_at_unix:row.try_get::("","expires_at_unix").map_err(internal)? as u64, + authorization_epoch:1 })) + }.await; + let context = finish(txn, result).await?; + if let Some(context) = &context + && receipt.is_some() + { + let _ = self + .verification_cell(&context.built.snapshot_id) + .await + .set(()); + } + Ok(context) + } + + pub(crate) async fn context( + &self, + snapshot_id: &str, + lease_id: &str, + instance_id: &str, + ) -> Result { + let installer = self.installer().await?; + if routes::lease(&self.connection, installer, lease_id) + .await? + .as_deref() + != Some(snapshot_id) + { + return Err(expired()); + } + let row = self + .connection + .query_one_raw(statement( + self.session_sql(installer).await, + [snapshot_id.into(), lease_id.into()], + )) + .await + .map_err(internal)? + .ok_or_else(expired)?; + installer.verify_primary_scope_row(&row)?; + let context = match decode_context(&row, snapshot_id, lease_id, instance_id) { + Ok(context) => context, + Err(error) => { + if error.code == SnapshotErrorCode::LeaseExpired { + let txn = self.transaction().await?; + let result = async { + installer.verify_primary_connection(&txn).await?; + routes::lease(&txn, installer, lease_id).await?; + routes::generic_path(&txn, installer).await?; + retention_lock(&txn).await?; + expire_specific_locked(&txn, lease_id).await + } + .await; + finish(txn, result).await?; + } + return Err(error); + } + }; + let prepare_id: String = row.try_get("", "prepare_id").map_err(internal)?; + let cell = self.verification_cell(snapshot_id).await; + cell.get_or_try_init(|| async { + self.installer() + .await? + .restore_session_dag(&prepare_id) + .await + }) + .await?; + Ok(context) + } + + pub(crate) async fn renew( + &self, + lease_id: &str, + seconds: u64, + instance: &str, + ) -> Result { + let installer = self.installer().await?; + let txn = self.transaction().await?; + let result = async { + installer.verify_primary_connection(&txn).await?; + let selected_sid = routes::lease(&txn, installer, lease_id).await? + .ok_or_else(|| SnapshotError::new(SnapshotErrorCode::LeaseUnknown,"unknown lease_id"))?; + routes::generic_path(&txn, installer).await?; + retention_lock(&txn).await?; + let lease = txn.query_one_raw(statement( + "SELECT snapshot_id FROM mst2_snapshot_lease WHERE lease_id=$1 FOR UPDATE", + [lease_id.into()], + )).await.map_err(internal)?.ok_or_else(|| SnapshotError::new(SnapshotErrorCode::LeaseUnknown,"unknown lease_id"))?; + let sid: String = lease.try_get("","snapshot_id").map_err(internal)?; + if sid != selected_sid { + return Err(integrity("lease changed after immutable route selection")); + } + let row = txn.query_one_raw(statement(self.session_sql(installer).await,[sid.clone().into(),lease_id.into()])) + .await.map_err(internal)?.ok_or_else(expired)?; + self.installer().await?.verify_primary_scope_row(&row)?; + if let Err(error)=decode_context(&row,&sid,lease_id,instance) { + if error.code==SnapshotErrorCode::LeaseExpired { + expire_specific_locked(&txn,lease_id).await?; + return Ok(Err(error)); + } + return Err(error); + } + let updated = txn.query_one_raw(statement( + "UPDATE mst2_snapshot_lease SET expires_at_unix=greatest(expires_at_unix, + floor(extract(epoch FROM clock_timestamp()))::bigint)+$2 WHERE lease_id=$1 + AND state='ACTIVE' AND expires_at_unix>=floor(extract(epoch FROM clock_timestamp()))::bigint + RETURNING expires_at_unix", + [lease_id.into(),(seconds.clamp(1,3600) as i64).into()], + )).await.map_err(internal)?; + let Some(updated)=updated else { + expire_specific_locked(&txn,lease_id).await?; + return Ok(Err(expired())); + }; + Ok(Ok(LeaseRenewed { lease_id:lease_id.into(),snapshot_id:sid, + expires_at_unix:updated.try_get::("","expires_at_unix").map_err(internal)? as u64 })) + }.await; + finish(txn, result).await? + } + + pub(crate) async fn release(&self, lease_id: &str) -> Result { + let installer = self.installer().await?; + let txn = self.transaction().await?; + let result = async { + installer.verify_primary_connection(&txn).await?; + if routes::lease(&txn, installer, lease_id).await?.is_none() { + return Ok(false); + } + routes::generic_path(&txn, installer).await?; + retention_lock(&txn).await?; + let changed=txn.query_one_raw(statement( + "UPDATE mst2_snapshot_lease SET state='RELEASED' WHERE lease_id=$1 AND state='ACTIVE' RETURNING snapshot_id", + [lease_id.into()], + )).await.map_err(internal)?; + PostgresRetentionRepository::release_root_in_txn(&txn,&RetentionRoot::Lease(lease_id.into())).await?; + expire_locked(&txn).await?; + if let Some(row)=&changed { + let sid:String=row.try_get("","snapshot_id").map_err(internal)?; + retire_session_pin_locked(&txn,&sid).await?; + } + Ok(changed.is_some()) + }.await; + finish(txn, result).await + } + + async fn verification_cell(&self, sid: &str) -> Arc> { + let mut restored = self.restored.lock().await; + if restored.len() >= 128 + && !restored.contains_key(sid) + && let Some(old) = restored.keys().next().cloned() + { + restored.remove(&old); + } + restored.entry(sid.to_owned()).or_default().clone() + } + + async fn transaction(&self) -> Result { + self.connection + .begin_with_config(Some(IsolationLevel::ReadCommitted), None) + .await + .map_err(internal) + } +} + +#[path = "native_snapshot_routes.rs"] +mod routes; + +#[path = "native_snapshot_metadata_routes.rs"] +mod metadata_routes; + +#[cfg(test)] +pub(crate) use metadata_routes::with_metadata_read_barriers; +pub(crate) use metadata_routes::{ + MetadataRouteRequest, PersistedMetadataReadWork, PersistedMetadataRouteBatch, +}; + +const SESSION_SQL: &str = "SELECT s.snapshot_id,s.canonical_descriptor,s.commit_oid,s.root_tree_oid, + (SELECT storage_uuid FROM mst2_metadata_storage_scope WHERE singleton=1) AS authority_storage_uuid, + current_database() AS authority_database, + (SELECT oid::bigint FROM pg_catalog.pg_database WHERE datname=current_database()) AS authority_database_oid, + current_schema() AS authority_schema, + (SELECT oid::bigint FROM pg_catalog.pg_namespace WHERE nspname=current_schema()) AS authority_schema_oid, + inet_server_addr()::text AS authority_server_address,inet_server_port() AS authority_server_port, + pg_is_in_recovery() AS authority_replica, + s.metadata_root,s.prepare_id,s.authorization_epoch,s.state AS session_state,l.lease_id, + l.authorization_epoch AS lease_epoch,l.expires_at_unix,l.state AS lease_state, + floor(extract(epoch FROM clock_timestamp()))::bigint AS db_now,n.state AS root_state, + EXISTS(SELECT 1 FROM mst2_retention_root rr WHERE rr.node_id=n.node_id AND rr.root_key='lease:'||l.lease_id + AND rr.root_kind='lease') AS lease_covered, + EXISTS(SELECT 1 FROM mst2_retention_root rr WHERE rr.node_id=n.node_id AND rr.root_key='pin:session:'||s.snapshot_id + AND rr.root_kind='pin') AS session_covered, + p.state AS prepare_state,p.metadata_root AS prepared_root,p.tagged_root_tree_oid,p.scope AS prepared_scope, + p.source_domain,p.schema_version,p.metadata_codec,p.materialization_policy,p.fs_semantics,p.access_projection, + p.verification_revision,p.projection_revision, + 1::bigint AS root_count,s.commit_oid AS root_commit,s.root_tree_oid AS root_tree,s.instance_id, + l.publication_sequence AS sequence,l.writer_epoch,'READY'::text AS state,s.commit_oid AS head_commit, + s.root_tree_oid AS head_tree,l.certificate_receipt_id,c.receipt_id AS certificate_id, + c.namespace AS certificate_namespace,c.instance_id AS certificate_instance,c.sequence AS certificate_sequence, + c.writer_epoch AS certificate_epoch,c.old_root_commit AS certificate_old_commit, + c.root_commit AS certificate_commit,c.root_tree AS certificate_tree,c.origin_path,c.origin_ref,c.old_path_commit,c.path_commit, + r.id AS receipt_id,r.namespace AS receipt_namespace,r.sequence AS receipt_sequence,r.operation_id,r.old_oid,r.new_oid, + r.writer_epoch AS receipt_epoch,r.writer_kind,r.request_digest,r.request_digest_version,r.native_certificate_version, + o.id AS outbox_id,o.namespace AS outbox_namespace,o.sequence AS outbox_sequence + FROM mst2_snapshot_context s JOIN mst2_snapshot_lease l ON l.snapshot_id=s.snapshot_id AND l.lease_id=$2 + LEFT JOIN mst2_retention_node n ON n.node_id='page:sha256:'||encode(s.metadata_root,'hex') + LEFT JOIN mst2_metadata_prepare p ON p.prepare_id=s.prepare_id + LEFT JOIN mst2_native_publication c ON c.receipt_id=l.certificate_receipt_id + LEFT JOIN mst2_publication r ON r.id=c.receipt_id + LEFT JOIN mst2_publication_outbox o ON o.operation_id=r.operation_id WHERE s.snapshot_id=$1"; + +fn decode_context( + row: &QueryResult, + sid: &str, + lease: &str, + instance: &str, +) -> Result { + let deadline: i64 = row.try_get("", "expires_at_unix").map_err(internal)?; + if row.try_get::("", "lease_state").map_err(internal)? != "ACTIVE" + || deadline < row.try_get::("", "db_now").map_err(internal)? + { + return Err(expired()); + } + let epoch: i64 = row.try_get("", "authorization_epoch").map_err(internal)?; + if row + .try_get::("", "session_state") + .map_err(internal)? + != "READY" + || epoch != 1 + || row.try_get::("", "lease_epoch").map_err(internal)? != epoch + { + return Err(forbidden()); + } + if row + .try_get::>("", "root_state") + .map_err(internal)? + .as_deref() + != Some("LIVE") + || !row.try_get::("", "lease_covered").map_err(internal)? + || !row + .try_get::("", "session_covered") + .map_err(internal)? + { + return Err(SnapshotError::new( + SnapshotErrorCode::ObjectUnavailable, + "snapshot root protection is unavailable", + )); + } + let head = decode_native_observation(row) + .map_err(|_| integrity("durable native source certificate is invalid"))? + .head + .ok_or_else(|| integrity("durable native source is missing"))?; + if head.instance_id != instance { + return Err(forbidden()); + } + let canonical: Vec = row.try_get("", "canonical_descriptor").map_err(internal)?; + let descriptor = ServingDescriptor::decode(&canonical) + .map_err(|_| integrity("invalid durable serving descriptor"))?; + let view = SnapshotView::from_commit(&head.root.commit, &head.root.tree); + let root: Vec = row.try_get("", "metadata_root").map_err(internal)?; + let prepared_root: Option> = row.try_get("", "prepared_root").map_err(internal)?; + let tagged_tree = git_internal::hash::ObjectHash::from_hex_for_kind( + git_internal::hash::get_hash_kind(), + &head.root.tree, + ) + .map_err(|_| integrity("invalid durable tree identity"))? + .to_tagged_string(); + if format!( + "sha256:{}", + hex(&descriptor.snapshot_id().map_err(internal)?) + ) != sid + || uuid::Uuid::from_bytes(descriptor.instance_uuid).to_string() != instance + || format!("sha256:{}", hex(&descriptor.namespace_view_id)) != view.view_id + || descriptor.metadata_root.as_slice() != root.as_slice() + || prepared_root.as_deref() != Some(root.as_slice()) + || row + .try_get::>("", "prepare_state") + .map_err(internal)? + .as_deref() + != Some("COMMITTED") + || row + .try_get::>("", "tagged_root_tree_oid") + .map_err(internal)? + .as_deref() + != Some(tagged_tree.as_str()) + || row + .try_get::>("", "prepared_scope") + .map_err(internal)? + .as_deref() + != Some(descriptor.scope.as_str()) + { + return Err(integrity( + "durable snapshot identity or source proof is inconsistent", + )); + } + let profile = [ + ("schema_version", mst2_codec::descriptor::SCHEMA_VERSION), + ("metadata_codec", mst2_codec::descriptor::METADATA_CODEC), + ( + "materialization_policy", + mst2_codec::descriptor::MATERIALIZATION_POLICY_GIT_RAW_V1, + ), + ( + "fs_semantics", + mst2_codec::descriptor::FS_SEMANTICS_LINUX_CODE_V1, + ), + ( + "access_projection", + mst2_codec::descriptor::ACCESS_PROJECTION_EXACT_FULL, + ), + ( + "projection_revision", + crate::ceres::snapshot::projection_observation::NATIVE_PROJECTION_REVISION, + ), + ]; + if profile.iter().any(|(column, value)| { + row.try_get::>("", column).ok().flatten() != Some(*value as i16) + }) || row + .try_get::>("", "source_domain") + .map_err(internal)? + .as_deref() + != Some("native-git") + || row + .try_get::>("", "verification_revision") + .map_err(internal)? + != Some(super::mono_storage::MST2_VERIFICATION_VERSION) + { + return Err(integrity( + "durable snapshot materialization profile changed", + )); + } + Ok(SnapshotContext { + built: BuiltDescriptor { + instance_id: instance.into(), + snapshot_id: sid.into(), + metadata_root: format!("sha256:{}", hex(&descriptor.metadata_root)), + descriptor, + }, + commit_oid: head.root.commit, + root_tree_oid: head.root.tree, + lease_id: lease.into(), + lease_expires_at_unix: deadline as u64, + authorization_epoch: epoch as u64, + }) +} + +async fn retention_lock(txn: &DatabaseTransaction) -> Result<(), SnapshotError> { + txn.execute_raw(statement( + "SELECT pg_advisory_xact_lock($1,hashtext(current_schema()))", + [RETENTION_LOCK_KEY.into()], + )) + .await + .map_err(internal) + .map(|_| ()) +} + +async fn retire_session_pin_locked( + txn: &DatabaseTransaction, + sid: &str, +) -> Result<(), SnapshotError> { + txn.execute_raw(statement( + "DELETE FROM mst2_retention_root r WHERE r.root_kind='pin' AND r.root_key='pin:session:'||$1 + AND NOT EXISTS(SELECT 1 FROM mst2_snapshot_lease l WHERE l.snapshot_id=$1 AND l.state='ACTIVE' + AND l.expires_at_unix>=floor(extract(epoch FROM clock_timestamp()))::bigint)",[sid.into()] + )).await.map_err(internal).map(|_| ()) +} + +async fn expire_specific_locked( + txn: &DatabaseTransaction, + lease: &str, +) -> Result<(), SnapshotError> { + let row=txn.query_one_raw(statement( + "UPDATE mst2_snapshot_lease SET state='EXPIRED' WHERE lease_id=$1 AND state='ACTIVE' + AND expires_at_unix("", "snapshot_id").map_err(internal)?, + ) + .await?; + } + Ok(()) +} + +async fn expire_locked(txn: &DatabaseTransaction) -> Result<(), SnapshotError> { + let rows = txn + .query_all_raw(statement( + "UPDATE mst2_snapshot_lease SET state='EXPIRED' WHERE lease_id IN + (SELECT lease_id FROM mst2_snapshot_lease WHERE state='ACTIVE' + AND expires_at_unix("","lease_id").map_err(internal)?, + "sid":row.try_get::("","snapshot_id").map_err(internal)? + })) + }) + .collect::, SnapshotError>>()?; + let encoded = serde_json::to_string(&rows).map_err(internal)?; + txn.execute_raw(statement( + "DELETE FROM mst2_retention_root r USING jsonb_to_recordset($1::jsonb) AS p(lease text,sid text) + WHERE r.root_kind='lease' AND r.root_key='lease:'||p.lease",[encoded.clone().into()] + )).await.map_err(internal)?; + txn.execute_raw(statement( + "DELETE FROM mst2_retention_root r USING jsonb_to_recordset($1::jsonb) AS p(lease text,sid text) + WHERE r.root_kind='pin' AND r.root_key='pin:session:'||p.sid + AND NOT EXISTS(SELECT 1 FROM mst2_snapshot_lease l WHERE l.snapshot_id=p.sid AND l.state='ACTIVE' + AND l.expires_at_unix>=floor(extract(epoch FROM clock_timestamp()))::bigint)",[encoded.into()] + )).await.map_err(internal).map(|_| ()) +} + +async fn finish( + txn: DatabaseTransaction, + result: Result, +) -> Result { + match result { + Ok(value) => { + txn.commit().await.map_err(|error| { + tracing::error!(error=%error,"native session commit outcome unknown"); + SnapshotError::new( + SnapshotErrorCode::TemporaryUnavailable, + "session commit outcome unknown; retry resolve", + ) + })?; + Ok(value) + } + Err(error) => { + txn.rollback().await.map_err(internal)?; + Err(error) + } + } +} +fn statement(sql: &str, values: [sea_orm::Value; N]) -> Statement { + Statement::from_sql_and_values(DbBackend::Postgres, sql, values) +} +fn root_node(built: &BuiltDescriptor) -> String { + format!("page:{}", built.metadata_root) +} +fn internal(error: impl std::fmt::Display) -> SnapshotError { + SnapshotError::new(SnapshotErrorCode::Internal, error.to_string()) +} +fn integrity(message: &str) -> SnapshotError { + SnapshotError::new(SnapshotErrorCode::IntegrityError, message) +} +fn not_ready(message: &str) -> SnapshotError { + SnapshotError::new(SnapshotErrorCode::SnapshotNotReady, message) +} +fn expired() -> SnapshotError { + SnapshotError::new( + SnapshotErrorCode::LeaseExpired, + "lease is not active for this snapshot; re-resolve", + ) +} +fn forbidden() -> SnapshotError { + SnapshotError::new( + SnapshotErrorCode::ScopeForbidden, + "snapshot serving state or authorization epoch changed", + ) +} +pub(crate) fn install_error(error: MetadataInstallError) -> SnapshotError { + match error { + MetadataInstallError::Rejected(error) if error.code == SnapshotErrorCode::Internal => { + tracing::error!(%error,"native metadata installation unavailable"); + #[cfg(test)] + eprintln!("MST2 test native metadata installation rejected: {error}"); + SnapshotError::new( + SnapshotErrorCode::TemporaryUnavailable, + "native metadata installation unavailable", + ) + } + MetadataInstallError::Rejected(error) => error, + MetadataInstallError::CommitUncertain { + operation_id, + phase, + .. + } => { + tracing::error!(%operation_id,?phase,"native metadata commit outcome unknown"); + SnapshotError::new( + SnapshotErrorCode::TemporaryUnavailable, + "metadata commit outcome unknown; retry resolve", + ) + } + } +} + +impl super::Storage { + pub(crate) async fn snapshot_metadata_family( + &self, + identity: &str, + lease: bool, + ) -> Result, SnapshotError> + { + use super::base_storage::StorageConnector; + super::qualified_metadata_family::select_snapshot_family( + self.mono_storage().get_connection(), + identity, + lease, + ) + .await + } + + pub(crate) async fn snapshot_metadata_routes( + &self, + context: &SnapshotContext, + requests: &[MetadataRouteRequest<'_>], + ) -> Result { + if self + .snapshot_metadata_family(&context.lease_id, true) + .await? + == Some(super::qualified_metadata_family::SnapshotMetadataFamily::Rooted) + { + return self + .rooted_qualified_metadata_writer() + .await + .map_err(internal)? + .metadata_routes(context, requests) + .await; + } + self.snapshot_sessions() + .await + .metadata_routes(context, requests) + .await + } + + pub(crate) async fn snapshot_sessions(&self) -> &PostgresNativeSessionRepository { + use super::base_storage::StorageConnector; + self.native_snapshot_sessions + .get_or_init(|| async { + PostgresNativeSessionRepository::new(self.mono_storage().get_connection().clone()) + }) + .await + } + + pub(crate) async fn snapshot_context( + &self, + sid: &str, + lease: &str, + ) -> Result { + let config = self.config(); + if config.mst2.publication_enabled { + let instance = config + .mst2 + .instance_uuid + .as_deref() + .ok_or_else(|| not_ready("native instance missing"))?; + let instance = uuid::Uuid::parse_str(instance) + .map_err(|_| not_ready("invalid native instance"))? + .to_string(); + if self.snapshot_metadata_family(lease, true).await? + == Some(super::qualified_metadata_family::SnapshotMetadataFamily::Rooted) + { + return self + .rooted_qualified_metadata_writer() + .await + .map_err(internal)? + .context(sid, lease, &instance) + .await; + } + self.snapshot_sessions() + .await + .context(sid, lease, &instance) + .await + } else { + let runtime = crate::ceres::snapshot::runtime::runtime(); + runtime.validate_lease(sid, lease)?; + runtime.context(sid) + } + } + + pub(crate) async fn snapshot_renew( + &self, + lease: &str, + seconds: u64, + ) -> Result { + let config = self.config(); + if config.mst2.publication_enabled { + let instance = config + .mst2 + .instance_uuid + .as_deref() + .ok_or_else(|| not_ready("native instance missing"))?; + let instance = uuid::Uuid::parse_str(instance) + .map_err(|_| not_ready("invalid native instance"))? + .to_string(); + if self.snapshot_metadata_family(lease, true).await? + == Some(super::qualified_metadata_family::SnapshotMetadataFamily::Rooted) + { + return self + .rooted_qualified_metadata_writer() + .await + .map_err(internal)? + .renew(lease, seconds, &instance) + .await; + } + self.snapshot_sessions() + .await + .renew(lease, seconds, &instance) + .await + } else { + crate::ceres::snapshot::runtime::runtime().renew_lease(lease, seconds) + } + } + + pub(crate) async fn snapshot_release(&self, lease: &str) -> Result { + if self.config().mst2.publication_enabled { + if self.snapshot_metadata_family(lease, true).await? + == Some(super::qualified_metadata_family::SnapshotMetadataFamily::Rooted) + { + return self + .rooted_qualified_metadata_writer() + .await + .map_err(internal)? + .release(lease) + .await; + } + self.snapshot_sessions().await.release(lease).await + } else { + Ok(crate::ceres::snapshot::runtime::runtime().release_lease(lease)) + } + } +} diff --git a/src/jupiter/storage/object_storage.rs b/src/jupiter/storage/object_storage.rs index f77fbc86..891232cf 100644 --- a/src/jupiter/storage/object_storage.rs +++ b/src/jupiter/storage/object_storage.rs @@ -35,6 +35,155 @@ pub async fn build_object_storage( }) } +#[cfg(test)] +mod exact_range_tests { + use super::*; + use crate::orbit_api::object_storage::ObjectNamespace; + + #[tokio::test] + async fn memory_exact_range_has_no_clipped_or_empty_success() { + let storage = mock_object_storage(); + let key = ObjectKey { + namespace: ObjectNamespace::Git, + key: "abcdef1234567890".to_string(), + }; + storage + .inner + .put_stream( + &key, + Box::pin(futures::stream::iter([Ok(Bytes::from_static(b"abcdef"))])), + ObjectMeta { + size: 6, + ..Default::default() + }, + ) + .await + .unwrap(); + let (mut stream, meta) = storage + .inner + .get_range_stream_exact(&key, 2, 5) + .await + .unwrap() + .unwrap(); + assert_eq!(meta.size, 6); + assert_eq!(stream.next().await.unwrap().unwrap(), b"cde".as_slice()); + assert!(stream.next().await.is_none()); + for (start, end) in [(5, 7), (6, 7), (4, 4), (u64::MAX, u64::MAX)] { + assert!( + storage + .inner + .get_range_stream_exact(&key, start, end) + .await + .is_err() + ); + } + } + + #[tokio::test] + async fn memory_chunk_receipts_reject_all_mutation_routes() { + let storage = mock_object_storage(); + super::assert_immutable_chunk_receipt_contract(storage.inner.as_ref()).await; + } +} + +#[cfg(test)] +pub(crate) async fn assert_immutable_chunk_receipt_contract( + storage: &dyn crate::orbit_api::factory::MegaObjectStorageWithLog, +) { + let key = ObjectKey { + namespace: crate::orbit_api::object_storage::ObjectNamespace::ChunkMapReceipt, + key: "a".repeat(64), + }; + let first = Bytes::from_static(b"trusted immutable receipt"); + storage + .put_metadata_atomic_create(&key, first.clone(), ObjectMeta::default()) + .await + .unwrap(); + storage + .put_metadata_atomic_create( + &key, + Bytes::from_static(b"conflicting replay"), + ObjectMeta::default(), + ) + .await + .unwrap(); + let input = || { + Box::pin(futures::stream::iter([Ok(Bytes::from_static( + b"overwrite", + ))])) as ObjectByteStream + }; + assert!( + storage + .put_stream(&key, input(), ObjectMeta::default()) + .await + .is_err() + ); + assert!( + storage + .put_stream_bounded(&key, input(), ObjectMeta::default()) + .await + .is_err() + ); + assert!( + storage + .put_metadata_atomic( + &key, + Bytes::from_static(b"overwrite"), + ObjectMeta::default() + ) + .await + .is_err() + ); + assert!( + storage + .put_metadata_atomic_create( + &key, + Bytes::from(vec![ + 0; + crate::orbit_api::object_storage::MAX_METADATA_ATOMIC_BYTES + + 1 + ]), + ObjectMeta::default() + ) + .await + .is_err() + ); + assert!( + storage + .append(&key, input(), ObjectMeta::default()) + .await + .is_err() + ); + assert!( + storage + .append_concurrently(&key, input(), ObjectMeta::default()) + .await + .is_err() + ); + assert!(storage.delete(&key).await.is_err()); + for method in [Method::PUT, Method::POST, Method::DELETE, Method::PATCH] { + assert!( + storage + .signed_url(&key, method, std::time::Duration::from_secs(60)) + .await + .is_err() + ); + } + assert!( + storage + .signed_url(&key, Method::GET, std::time::Duration::from_secs(60)) + .await + .is_ok() + ); + let (mut stream, meta) = storage.get_stream(&key).await.unwrap(); + assert_eq!(meta.size, first.len() as i64); + let mut observed = Vec::new(); + while let Some(part) = stream.next().await { + observed.extend_from_slice(&part.unwrap()); + } + assert_eq!(observed, first.as_ref()); +} + #[derive(Default)] struct InMemoryObjectStorage { objects: Mutex>, @@ -50,7 +199,9 @@ impl InMemoryObjectStorage { } fn stream_bytes(bytes: Bytes) -> ObjectByteStream { - Box::pin(futures::stream::once(async move { Ok(bytes) })) + crate::orbit_api::object_storage::fragment_object_stream(Box::pin(futures::stream::once( + async move { Ok(bytes) }, + ))) } } @@ -60,12 +211,17 @@ pub fn mock_object_storage() -> MegaObjectStorageWrapper { #[async_trait::async_trait] impl MegaObjectStorage for InMemoryObjectStorage { + fn supports_chunk_map_retention(&self) -> bool { + true + } + async fn put_stream( &self, key: &ObjectKey, data: ObjectByteStream, mut meta: ObjectMeta, ) -> OrbitResult<()> { + reject_receipt_mutation(key)?; let bytes = Self::read_stream(data).await?; meta.size = bytes.len() as i64; self.objects @@ -81,6 +237,7 @@ impl MegaObjectStorage for InMemoryObjectStorage { bytes: Bytes, mut meta: ObjectMeta, ) -> OrbitResult<()> { + reject_receipt_mutation(key)?; use crate::orbit_api::object_storage::MAX_METADATA_ATOMIC_BYTES; if bytes.len() > MAX_METADATA_ATOMIC_BYTES { return Err(IoOrbitError::Other(format!( @@ -95,6 +252,27 @@ impl MegaObjectStorage for InMemoryObjectStorage { Ok(()) } + async fn put_metadata_atomic_create( + &self, + key: &ObjectKey, + bytes: Bytes, + mut meta: ObjectMeta, + ) -> OrbitResult<()> { + if bytes.len() > crate::orbit_api::object_storage::MAX_METADATA_ATOMIC_BYTES { + return Err(IoOrbitError::Other( + "immutable atomic metadata exceeds its byte limit".into(), + )); + } + key.validate()?; + meta.size = bytes.len() as i64; + self.objects + .lock() + .map_err(|_| IoOrbitError::Other("object storage lock poisoned".into()))? + .entry(key.clone()) + .or_insert((bytes, meta)); + Ok(()) + } + async fn get_stream(&self, key: &ObjectKey) -> OrbitResult<(ObjectByteStream, ObjectMeta)> { let (bytes, meta) = self .objects @@ -106,6 +284,56 @@ impl MegaObjectStorage for InMemoryObjectStorage { Ok((Self::stream_bytes(bytes), meta)) } + async fn chunk_map_receipt_inventory( + &self, + ) -> OrbitResult { + use crate::orbit_api::object_storage::{ + ChunkMapReceiptInventory, MAX_CHUNK_MAP_RECEIPT_BYTES, MAX_CHUNK_MAP_RECEIPTS, + ObjectNamespace, + }; + let objects = self + .objects + .lock() + .map_err(|_| IoOrbitError::Other("object storage lock poisoned".into()))?; + let mut inventory = ChunkMapReceiptInventory { + objects: Vec::new(), + bytes: 0, + }; + for (key, (bytes, _)) in objects + .iter() + .filter(|(key, _)| key.namespace == ObjectNamespace::ChunkMapReceipt) + { + if inventory.objects.len() == MAX_CHUNK_MAP_RECEIPTS { + return Err(IoOrbitError::ChunkMapRetentionCapacityExceeded); + } + inventory.bytes = inventory + .bytes + .checked_add(bytes.len() as u64) + .filter(|bytes| *bytes <= MAX_CHUNK_MAP_RECEIPT_BYTES) + .ok_or(IoOrbitError::ChunkMapRetentionCapacityExceeded)?; + inventory.objects.push((key.clone(), bytes.len() as u64)); + } + Ok(inventory) + } + + async fn delete_chunk_map_receipt( + &self, + authority: &crate::orbit_api::object_storage::ChunkMapReceiptDeletion, + ) -> OrbitResult { + let mut objects = self + .objects + .lock() + .map_err(|_| IoOrbitError::Other("object storage lock poisoned".into()))?; + let Some((bytes, _)) = objects.get(authority.key()) else { + return Ok(false); + }; + if bytes.as_ref() != authority.expected_bytes() { + return Err(IoOrbitError::Other("retired receipt body changed".into())); + } + objects.remove(authority.key()); + Ok(true) + } + async fn get_range_stream( &self, key: &ObjectKey, @@ -127,6 +355,32 @@ impl MegaObjectStorage for InMemoryObjectStorage { Ok((Self::stream_bytes(bytes.slice(start..end)), meta)) } + async fn get_range_stream_exact( + &self, + key: &ObjectKey, + start: u64, + end: u64, + ) -> OrbitResult> { + let (bytes, meta) = self + .objects + .lock() + .map_err(|_| IoOrbitError::Other("object storage lock poisoned".to_string()))? + .get(key) + .cloned() + .ok_or_else(|| IoOrbitError::object_store_not_found(key.default_sharding()))?; + let start = + usize::try_from(start).map_err(|_| IoOrbitError::object_store("range overflow"))?; + let end = usize::try_from(end).map_err(|_| IoOrbitError::object_store("range overflow"))?; + if start >= end || end > bytes.len() { + return Err(std::io::Error::new( + std::io::ErrorKind::InvalidData, + "exact object range is outside stored bytes", + ) + .into()); + } + Ok(Some((Self::stream_bytes(bytes.slice(start..end)), meta))) + } + async fn exists(&self, key: &ObjectKey) -> OrbitResult { Ok(self .objects @@ -137,14 +391,18 @@ impl MegaObjectStorage for InMemoryObjectStorage { async fn signed_url( &self, - _key: &ObjectKey, - _method: Method, + key: &ObjectKey, + method: Method, _expires_in: std::time::Duration, ) -> OrbitResult> { + if method != Method::GET { + reject_receipt_mutation(key)?; + } Ok(None) } async fn delete(&self, key: &ObjectKey) -> OrbitResult<()> { + reject_receipt_mutation(key)?; let deleted = self .objects .lock() @@ -159,6 +417,16 @@ impl MegaObjectStorage for InMemoryObjectStorage { } } +fn reject_receipt_mutation(key: &ObjectKey) -> OrbitResult<()> { + if key.namespace == crate::orbit_api::object_storage::ObjectNamespace::ChunkMapReceipt { + Err(IoOrbitError::Other( + "chunk map receipts require immutable atomic creation".into(), + )) + } else { + Ok(()) + } +} + #[async_trait::async_trait] impl LogStorage for InMemoryObjectStorage { async fn append( @@ -167,6 +435,7 @@ impl LogStorage for InMemoryObjectStorage { data: ObjectByteStream, mut meta: ObjectMeta, ) -> OrbitResult<()> { + reject_receipt_mutation(key)?; let bytes = Self::read_stream(data).await?; let mut objects = self .objects diff --git a/src/jupiter/storage/push_queue_storage.rs b/src/jupiter/storage/push_queue_storage.rs index 06878580..5be79779 100644 --- a/src/jupiter/storage/push_queue_storage.rs +++ b/src/jupiter/storage/push_queue_storage.rs @@ -8,7 +8,7 @@ use serde_json::Value as JsonValue; use crate::{ callisto::{ - authz_notify_outbox, mega_refs, push_queue, queue_control, + authz_notify_outbox, push_queue, queue_control, sea_orm_active_enums::{PushQueueKindEnum, PushQueueStatusEnum}, }, common::{errors::MegaError, utils::MEGA_BRANCH_NAME}, @@ -101,6 +101,7 @@ pub struct MonoWriteLockHolder { #[derive(Clone)] pub struct PushQueueStorage { base: BaseStorage, + native_publication_enabled: bool, } impl Deref for PushQueueStorage { @@ -123,7 +124,15 @@ pub struct EnqueueParams<'a> { impl PushQueueStorage { pub fn new(base: BaseStorage) -> Self { - Self { base } + Self { + base, + native_publication_enabled: false, + } + } + + pub(crate) fn with_native_publication(mut self, enabled: bool) -> Self { + self.native_publication_enabled = enabled; + self } /// B1: lock `queue_control`, conditional INSERT, classify on zero rows. @@ -241,8 +250,9 @@ impl PushQueueStorage { }); } // operation_states race: winner already committed — classify by read. + let replay_txn = self.get_connection().begin().await?; if let Some(existing) = push_queue::Entity::find() - .filter(push_queue::Column::Kind.eq(kind)) + .filter(push_queue::Column::Kind.eq(kind.clone())) .filter(push_queue::Column::Path.eq(path)) .filter(push_queue::Column::OperationId.eq(operation_id)) .filter(push_queue::Column::Status.is_in([ @@ -250,9 +260,23 @@ impl PushQueueStorage { PushQueueStatusEnum::Running, PushQueueStatusEnum::Done, ])) - .one(self.get_connection()) + .one(&replay_txn) .await? { + if matches!(&kind, PushQueueKindEnum::Push | PushQueueKindEnum::Merge) { + let mut candidate = existing.clone(); + candidate.old_id = old_id.to_owned(); + candidate.new_id = new_id.to_owned(); + candidate.requester = requester.map(str::to_owned); + candidate.payload = payload.clone(); + crate::jupiter::storage::mono_storage::MonoStorage { + base: self.base.clone(), + } + .validate_queue_publication_replay_in_txn(&replay_txn, &candidate) + .await + .map_err(|error| MegaError::Other(error.to_string()))?; + } + replay_txn.commit().await?; return match existing.status { PushQueueStatusEnum::Done => Ok(EnqueueOutcome::Replay { id: existing.id, @@ -264,6 +288,7 @@ impl PushQueueStorage { _ => unreachable!("filter restricts status"), }; } + replay_txn.rollback().await?; return Err(MegaError::Other(format!( "unique conflict without classifiable row: {msg}" ))); @@ -283,6 +308,29 @@ impl PushQueueStorage { Some(outcome) => { match &outcome { EnqueueOutcome::Adopted { .. } | EnqueueOutcome::Replay { .. } => { + if matches!(&kind, PushQueueKindEnum::Push | PushQueueKindEnum::Merge) { + let id = match &outcome { + EnqueueOutcome::Adopted { id } + | EnqueueOutcome::Replay { id, .. } => *id, + _ => unreachable!("adopt or replay"), + }; + let mut candidate = push_queue::Entity::find_by_id(id) + .one(&txn) + .await? + .ok_or_else(|| { + MegaError::Other("queue replay row missing".into()) + })?; + candidate.old_id = old_id.to_owned(); + candidate.new_id = new_id.to_owned(); + candidate.requester = requester.map(str::to_owned); + candidate.payload = payload.clone(); + crate::jupiter::storage::mono_storage::MonoStorage { + base: self.base.clone(), + } + .validate_queue_publication_replay_in_txn(&txn, &candidate) + .await + .map_err(|error| MegaError::Other(error.to_string()))?; + } txn.commit().await?; } EnqueueOutcome::Rejected { .. } => { @@ -722,66 +770,63 @@ impl PushQueueStorage { return Ok(ClaimOutcome::HardStopped); } - let claimed = txn - .execute_raw(Statement::from_sql_and_values( + // One statement snapshot supplies both the successful claim and its + // root/publication token. Updating the same queue row twice in a + // data-modifying CTE is deliberately avoided. + let sql = format!( + r#" + WITH {native_ctes}, claimed AS ( + UPDATE push_queue q + SET status = 'Running'::push_queue_status_enum, + started_at = now(), heartbeat_at = now(), updated_at = now(), + expected_commit_hash = observation.root_commit, + expected_tree_hash = observation.root_tree, + expected_native_sequence = observation.sequence, + expected_native_epoch = observation.writer_epoch, + expected_native_certificate = observation.certificate_receipt_id + FROM native_observation observation + WHERE q.id = $2 AND q.status = 'Queued'::push_queue_status_enum + AND (SELECT NOT hard_stopped FROM queue_control WHERE id = 1) + AND NOT EXISTS (SELECT 1 FROM push_queue WHERE status = 'Running'::push_queue_status_enum) + AND q.id = (SELECT min(id) FROM push_queue WHERE status = 'Queued'::push_queue_status_enum) + RETURNING q.id + ) + SELECT observation.* FROM native_observation observation + CROSS JOIN claimed + "#, + native_ctes = super::native_publication_storage::NATIVE_OBSERVATION_CTES + .replace("h.namespace = '/'", "h.namespace = '/' AND $3") + ); + let observations = txn + .query_all_raw(Statement::from_sql_and_values( DbBackend::Postgres, - r#" - UPDATE push_queue - SET status = 'Running'::push_queue_status_enum, - started_at = now(), - heartbeat_at = now(), - updated_at = now() - WHERE id = $1 - AND status = 'Queued'::push_queue_status_enum - AND (SELECT NOT hard_stopped FROM queue_control WHERE id = 1) - AND NOT EXISTS ( - SELECT 1 FROM push_queue - WHERE status = 'Running'::push_queue_status_enum - ) - AND id = ( - SELECT min(id) FROM push_queue - WHERE status = 'Queued'::push_queue_status_enum - ) - "#, - [Value::from(id)], + sql, + [ + MEGA_BRANCH_NAME.into(), + id.into(), + self.native_publication_enabled.into(), + ], )) .await?; - - if claimed.rows_affected() == 0 { + if observations.is_empty() { txn.rollback().await?; return Ok(ClaimOutcome::Missed); } - - // Snapshot root AFTER the claim UPDATE succeeds (still in this txn). - // Ordering matters: - // - Before UPDATE: a legitimate predecessor B3 can commit Done + new - // root in the gap → stale expected_* → false QueueBypassDetected. - // - FOR SHARE before UPDATE: blocks admissions while a writer holds - // the root row exclusively (stalls past wait_timeout). - // - After UPDATE: NOT EXISTS(Running) already held, so no legitimate - // B3 can still be writing; plain SELECT needs no row lock. - let root = mega_refs::Entity::find() - .filter(mega_refs::Column::Path.eq("/")) - .filter(mega_refs::Column::RefName.eq(MEGA_BRANCH_NAME.to_owned())) - .one(&txn) - .await?; - let (commit, tree) = match root { - Some(r) => (Some(r.ref_commit_hash), Some(r.ref_tree_hash)), - None => (None, None), - }; - - txn.execute_raw(Statement::from_sql_and_values( - DbBackend::Postgres, - r#" - UPDATE push_queue - SET expected_commit_hash = $2, - expected_tree_hash = $3, - updated_at = now() - WHERE id = $1 - "#, - [Value::from(id), Value::from(commit), Value::from(tree)], - )) - .await?; + if observations.len() != 1 { + return Err(MegaError::Other( + "ambiguous native claim observation".into(), + )); + } + if self.native_publication_enabled { + let observation = + super::native_publication_storage::decode_native_observation(&observations[0]) + .map_err(|error| MegaError::Other(error.to_string()))?; + if observation.head.is_none() { + return Err(MegaError::Other( + "native publication is not initialized".into(), + )); + } + } txn.commit().await?; Ok(ClaimOutcome::Claimed) @@ -899,6 +944,9 @@ impl PushQueueStorage { started_at = NULL, expected_commit_hash = NULL, expected_tree_hash = NULL, + expected_native_sequence = NULL, + expected_native_epoch = NULL, + expected_native_certificate = NULL, heartbeat_at = now(), updated_at = now() WHERE id = $1 @@ -985,6 +1033,9 @@ impl PushQueueStorage { started_at = NULL, expected_commit_hash = NULL, expected_tree_hash = NULL, + expected_native_sequence = NULL, + expected_native_epoch = NULL, + expected_native_certificate = NULL, pending_action = NULL, heartbeat_at = now(), updated_at = now() @@ -1363,9 +1414,12 @@ mod tests { use serde_json::json; use super::*; - use crate::jupiter::{ - migration::apply_migrations, storage::base_storage::StorageConnector, - tests::test_db_connection, + use crate::{ + callisto::mega_refs, + jupiter::{ + migration::apply_migrations, storage::base_storage::StorageConnector, + tests::test_db_connection, + }, }; async fn storage() -> (tempfile::TempDir, PushQueueStorage) { diff --git a/src/jupiter/storage/qualified_family_catalog.sql b/src/jupiter/storage/qualified_family_catalog.sql new file mode 100644 index 00000000..acb8cce9 --- /dev/null +++ b/src/jupiter/storage/qualified_family_catalog.sql @@ -0,0 +1,57 @@ +WITH selected_namespaces AS ( + SELECT oid,nspname,nspowner,nspacl FROM pg_catalog.pg_namespace + WHERE oid IN ($CORE_OID$,$Q_OID$) +), selected_relations AS ( + SELECT c.oid,c.relnamespace,c.relname,c.relkind,c.relpersistence,c.relowner,c.relacl, + c.relrowsecurity,c.relforcerowsecurity,c.reloptions,c.relam,c.relhasrules,c.relispartition + FROM pg_catalog.pg_class c + WHERE c.relnamespace=$Q_OID$ + OR (c.relnamespace=$CORE_OID$ AND (c.relname IN ('mst2_metadata_namespace','mst2_qualified_family_policy', + 'mega_tree','mst2_rooted_source_tree_revision','mst2_verified_object','mst2_retention_node','mst2_retention_gc_op', + 'mega_commit','mega_refs','mst2_native_head','mst2_native_publication','mst2_publication','mst2_publication_outbox', + 'mst2_snapshot_storage_route','mst2_lease_storage_route','mst2_generic_session_storage_binding', + 'mst2_snapshot_context','mst2_snapshot_lease') + OR c.oid IN (SELECT i.indexrelid FROM pg_catalog.pg_index i JOIN pg_catalog.pg_class parent ON parent.oid=i.indrelid + WHERE parent.relnamespace=$CORE_OID$ AND parent.relname IN ('mst2_metadata_namespace','mst2_qualified_family_policy', + 'mega_tree','mst2_rooted_source_tree_revision','mst2_verified_object','mst2_retention_node','mst2_retention_gc_op', + 'mega_commit','mega_refs','mst2_native_head','mst2_native_publication','mst2_publication','mst2_publication_outbox', + 'mst2_snapshot_storage_route','mst2_lease_storage_route','mst2_generic_session_storage_binding', + 'mst2_snapshot_context','mst2_snapshot_lease')))) +), selected_triggers AS ( + SELECT t.* FROM pg_catalog.pg_trigger t + WHERE (t.tgrelid IN (SELECT oid FROM selected_relations) + OR EXISTS(SELECT 1 FROM pg_catalog.pg_constraint x + WHERE x.oid=t.tgconstraint AND x.conrelid IN (SELECT oid FROM pg_catalog.pg_class WHERE relnamespace=$Q_OID$))) + AND NOT ($Q_OID$=0 AND $EXEMPT_Q_OID$<>0 AND t.tgisinternal + AND EXISTS(SELECT 1 FROM pg_catalog.pg_constraint fk + JOIN pg_catalog.pg_class source ON source.oid=fk.conrelid + JOIN pg_catalog.pg_class target ON target.oid=fk.confrelid + WHERE fk.oid=t.tgconstraint AND fk.contype='f' AND fk.connamespace=$EXEMPT_Q_OID$ + AND source.relnamespace=$EXEMPT_Q_OID$ AND target.relnamespace=$CORE_OID$ + AND t.tgrelid=fk.confrelid AND t.tgconstrrelid=fk.conrelid)) +), objects AS ( + SELECT 'namespace' AS kind,oid::text AS key,pg_catalog.to_jsonb(n) AS value FROM selected_namespaces n + UNION ALL SELECT 'relation',oid::text,pg_catalog.to_jsonb(c) FROM selected_relations c + UNION ALL SELECT 'column',a.attrelid||':'||a.attnum,pg_catalog.to_jsonb(a) + FROM pg_catalog.pg_attribute a JOIN selected_relations c ON c.oid=a.attrelid WHERE a.attnum>0 + UNION ALL SELECT 'default',d.oid::text,pg_catalog.to_jsonb(d) + FROM pg_catalog.pg_attrdef d JOIN selected_relations c ON c.oid=d.adrelid + UNION ALL SELECT 'constraint',x.oid::text,pg_catalog.to_jsonb(x) + FROM pg_catalog.pg_constraint x JOIN selected_relations c ON c.oid=x.conrelid + UNION ALL SELECT 'index',i.indexrelid::text,pg_catalog.to_jsonb(i) + FROM pg_catalog.pg_index i JOIN selected_relations c ON c.oid=i.indrelid + UNION ALL SELECT 'trigger',t.oid::text,pg_catalog.to_jsonb(t) + FROM selected_triggers t + UNION ALL SELECT 'function',p.oid::text,pg_catalog.to_jsonb(p) + FROM pg_catalog.pg_proc p WHERE p.pronamespace=$Q_OID$ + OR (p.pronamespace=$CORE_OID$ AND (pg_catalog.left(p.proname,11)='mst2_route_' + OR p.proname='mst2_metadata_has_generic_overlap')) + OR EXISTS(SELECT 1 FROM selected_triggers t WHERE t.tgfoid=p.oid) + UNION ALL SELECT 'policy',p.oid::text,pg_catalog.to_jsonb(p) + FROM pg_catalog.pg_policy p JOIN selected_relations c ON c.oid=p.polrelid + UNION ALL SELECT 'rule',r.oid::text,pg_catalog.to_jsonb(r) + FROM pg_catalog.pg_rewrite r JOIN selected_relations c ON c.oid=r.ev_class +) +SELECT pg_catalog.sha256(pg_catalog.convert_to( + pg_catalog.jsonb_agg(pg_catalog.jsonb_build_array(kind,key,value) ORDER BY kind,key)::text,'UTF8')) AS fingerprint +FROM objects diff --git a/src/jupiter/storage/qualified_family_shape.sql b/src/jupiter/storage/qualified_family_shape.sql new file mode 100644 index 00000000..4269fe74 --- /dev/null +++ b/src/jupiter/storage/qualified_family_shape.sql @@ -0,0 +1,65 @@ +WITH q AS ( + SELECT n.oid,n.nspname FROM pg_catalog.pg_namespace n WHERE n.oid=q_oid +), relations AS ( + SELECT c.* FROM pg_catalog.pg_class c WHERE c.relnamespace=q_oid +), objects AS ( + SELECT 'relation' AS kind,c.relname AS key,jsonb_build_array(c.relname,c.relkind,c.relpersistence,c.relowner,c.relacl, + c.relrowsecurity,c.relforcerowsecurity,c.reloptions,c.relispartition,a.amname) AS value + FROM relations c LEFT JOIN pg_catalog.pg_am a ON a.oid=c.relam + UNION ALL SELECT 'column',c.relname||':'||a.attnum,jsonb_build_array(c.relname,a.attnum,a.attname, + CASE WHEN tn.oid=q_oid THEN '$Q_SCHEMA$' ELSE tn.nspname END,t.typname,a.attnotnull,a.attisdropped,a.attidentity,a.attgenerated,a.attislocal,a.attinhcount, + a.atttypmod,a.attndims,a.attacl,a.attlen,a.attbyval,a.attalign,a.attstorage,a.attcompression, + CASE WHEN cn.oid=q_oid THEN '$Q_SCHEMA$' ELSE cn.nspname END,co.collname, + pg_catalog.replace(pg_catalog.pg_get_expr(d.adbin,d.adrelid),(SELECT nspname FROM q),'$Q_SCHEMA$')) + FROM pg_catalog.pg_attribute a JOIN relations c ON c.oid=a.attrelid + JOIN pg_catalog.pg_type t ON t.oid=a.atttypid JOIN pg_catalog.pg_namespace tn ON tn.oid=t.typnamespace + LEFT JOIN pg_catalog.pg_collation co ON co.oid=a.attcollation LEFT JOIN pg_catalog.pg_namespace cn ON cn.oid=co.collnamespace + LEFT JOIN pg_catalog.pg_attrdef d ON d.adrelid=a.attrelid AND d.adnum=a.attnum WHERE a.attnum>0 + UNION ALL SELECT 'constraint',c.relname||':'||x.conname,jsonb_build_array(c.relname,x.conname,x.contype, + pg_catalog.replace(pg_catalog.pg_get_constraintdef(x.oid,true),(SELECT nspname FROM q),'$Q_SCHEMA$'), + x.condeferrable,x.condeferred,x.convalidated,x.conislocal,x.coninhcount,x.connoinherit) + FROM pg_catalog.pg_constraint x JOIN relations c ON c.oid=x.conrelid + UNION ALL SELECT 'index',c.relname||':'||ic.relname,jsonb_build_array(c.relname,ic.relname, + pg_catalog.replace(pg_catalog.pg_get_indexdef(i.indexrelid),(SELECT nspname FROM q),'$Q_SCHEMA$'), + i.indisunique,i.indisprimary,i.indisexclusion,i.indimmediate,i.indisvalid,i.indisready,i.indislive, + i.indnullsnotdistinct,i.indisclustered,i.indisreplident) + FROM pg_catalog.pg_index i JOIN relations c ON c.oid=i.indrelid JOIN relations ic ON ic.oid=i.indexrelid + UNION ALL SELECT 'function',p.proname||':'||pg_catalog.replace(pg_catalog.pg_get_function_identity_arguments(p.oid),(SELECT nspname FROM q),'$Q_SCHEMA$'),jsonb_build_array( + p.proname,pg_catalog.replace(pg_catalog.pg_get_function_identity_arguments(p.oid),(SELECT nspname FROM q),'$Q_SCHEMA$'), + pg_catalog.replace(pg_catalog.pg_get_function_result(p.oid),(SELECT nspname FROM q),'$Q_SCHEMA$'),l.lanname, + p.proowner,p.proacl,p.prokind,p.provolatile,p.proisstrict,p.prosecdef,p.proleakproof,p.proparallel, + p.procost,p.prorows,p.probin,p.pronargdefaults, + pg_catalog.replace(pg_catalog.pg_get_expr(p.proargdefaults,0),(SELECT nspname FROM q),'$Q_SCHEMA$'), + pg_catalog.replace(pg_catalog.replace(pg_catalog.replace(pg_catalog.replace(p.prosrc, + pg_catalog.quote_literal((SELECT nspname FROM q)),pg_catalog.quote_literal('$Q_SCHEMA$')), + pg_catalog.quote_literal(q_oid::text),pg_catalog.quote_literal('$Q_OID$')), + pg_catalog.quote_literal(n_uuid::text),pg_catalog.quote_literal('$NAMESPACE_UUID$')), + pg_catalog.quote_literal(s_uuid),pg_catalog.quote_literal('$STORAGE_UUID$')), + pg_catalog.replace(pg_catalog.array_to_string(p.proconfig,E'\n'),(SELECT nspname FROM q),'$Q_SCHEMA$')) + FROM pg_catalog.pg_proc p JOIN pg_catalog.pg_language l ON l.oid=p.prolang WHERE p.pronamespace=q_oid + UNION ALL SELECT 'trigger',c.relname||':'||p.proname||':'||t.tgtype||':'||coalesce(x.conname,t.tgname),jsonb_build_array( + CASE WHEN c.relnamespace=q_oid THEN '$Q_SCHEMA$' ELSE cns.nspname END,c.relname, + CASE WHEN t.tgisinternal THEN NULL ELSE t.tgname END,t.tgtype,t.tgenabled,t.tgisinternal, + CASE WHEN pn.oid=q_oid THEN '$Q_SCHEMA$' ELSE pn.nspname END,p.proname, + pg_catalog.replace(pg_catalog.pg_get_function_identity_arguments(p.oid),(SELECT nspname FROM q),'$Q_SCHEMA$'), + t.tgdeferrable,t.tginitdeferred,t.tgnargs, + pg_catalog.replace(encode(t.tgargs,'hex'),encode(pg_catalog.convert_to((SELECT nspname FROM q),'UTF8'),'hex'), + encode(pg_catalog.convert_to('$Q_SCHEMA$','UTF8'),'hex')),t.tgattr::text, + pg_catalog.replace(pg_catalog.pg_get_expr(t.tgqual,t.tgrelid),(SELECT nspname FROM q),'$Q_SCHEMA$'),t.tgoldtable,t.tgnewtable,x.conname, + CASE WHEN rn.oid=q_oid THEN '$Q_SCHEMA$' ELSE rn.nspname END,rc.relname, + CASE WHEN t.tgparentid=0 THEN NULL ELSE 'unexpected parent trigger' END) + FROM pg_catalog.pg_trigger t JOIN pg_catalog.pg_class c ON c.oid=t.tgrelid + JOIN pg_catalog.pg_namespace cns ON cns.oid=c.relnamespace + JOIN pg_catalog.pg_proc p ON p.oid=t.tgfoid JOIN pg_catalog.pg_namespace pn ON pn.oid=p.pronamespace + LEFT JOIN pg_catalog.pg_constraint x ON x.oid=t.tgconstraint + LEFT JOIN pg_catalog.pg_class rc ON rc.oid=t.tgconstrrelid LEFT JOIN pg_catalog.pg_namespace rn ON rn.oid=rc.relnamespace + WHERE c.relnamespace=q_oid OR EXISTS(SELECT 1 FROM pg_catalog.pg_constraint fk + WHERE fk.oid=t.tgconstraint AND fk.conrelid IN (SELECT oid FROM relations)) + UNION ALL SELECT 'rule',c.relname||':'||r.rulename,pg_catalog.to_jsonb(r) + FROM pg_catalog.pg_rewrite r JOIN relations c ON c.oid=r.ev_class + UNION ALL SELECT 'policy',c.relname||':'||p.polname,pg_catalog.to_jsonb(p) + FROM pg_catalog.pg_policy p JOIN relations c ON c.oid=p.polrelid +) +SELECT pg_catalog.sha256(pg_catalog.convert_to( + pg_catalog.jsonb_agg(pg_catalog.jsonb_build_array(kind,key,value) ORDER BY kind,key,value::text)::text,'UTF8')) AS fingerprint +FROM objects diff --git a/src/jupiter/storage/qualified_metadata_anchors.sql b/src/jupiter/storage/qualified_metadata_anchors.sql new file mode 100644 index 00000000..e7db38e1 --- /dev/null +++ b/src/jupiter/storage/qualified_metadata_anchors.sql @@ -0,0 +1,438 @@ +CREATE TABLE mst2_metadata_source_root_attestation ( + attestation_id uuid PRIMARY KEY,namespace_uuid uuid NOT NULL REFERENCES mst2_metadata_family_identity(namespace_uuid), + origin_prepare_id text NOT NULL REFERENCES mst2_metadata_prepare(prepare_id), + tagged_tree_oid text NOT NULL CHECK(tagged_tree_oid ~ '^(sha1:[0-9a-f]{40}|sha256:[0-9a-f]{64}|blake3:[0-9a-f]{64})$'), + source_profile jsonb NOT NULL,profile_digest bytea NOT NULL CHECK(octet_length(profile_digest)=32), + source_body_digest bytea NOT NULL CHECK(octet_length(source_body_digest)=32), + source_revision uuid NOT NULL, + root_page bytea NOT NULL,root_generation bigint NOT NULL, + root_certificate_digest bytea NOT NULL CHECK(octet_length(root_certificate_digest)=32), + source_proof jsonb NOT NULL,attestation_digest bytea NOT NULL CHECK(octet_length(attestation_digest)=32), + FOREIGN KEY(root_page,root_generation,root_certificate_digest) + REFERENCES mst2_metadata_page_certificate(page_id,generation,certificate_digest), + UNIQUE(attestation_id,root_page,root_generation,attestation_digest) +); +CREATE INDEX mst2_metadata_source_root_lookup ON mst2_metadata_source_root_attestation(profile_digest,tagged_tree_oid,root_page,root_generation); +CREATE TABLE mst2_metadata_prepare_reuse_root ( + prepare_id text NOT NULL REFERENCES mst2_metadata_prepare(prepare_id),root_page bytea NOT NULL, + root_generation bigint NOT NULL,attestation_id uuid NOT NULL,attestation_digest bytea NOT NULL, + PRIMARY KEY(prepare_id,root_page), + FOREIGN KEY(attestation_id,root_page,root_generation,attestation_digest) + REFERENCES mst2_metadata_source_root_attestation(attestation_id,root_page,root_generation,attestation_digest) +); +CREATE INDEX mst2_metadata_prepare_reuse_root_page ON mst2_metadata_prepare_reuse_root(root_page,root_generation,prepare_id); +CREATE TABLE mst2_metadata_reuse_index ( + profile_digest bytea NOT NULL,tagged_tree_oid text NOT NULL,attestation_id uuid NOT NULL, + root_page bytea NOT NULL,root_generation bigint NOT NULL,attestation_digest bytea NOT NULL, + PRIMARY KEY(profile_digest,tagged_tree_oid), + FOREIGN KEY(attestation_id,root_page,root_generation,attestation_digest) + REFERENCES mst2_metadata_source_root_attestation(attestation_id,root_page,root_generation,attestation_digest) +); +CREATE INDEX mst2_metadata_reuse_index_page ON mst2_metadata_reuse_index(root_page,root_generation); + +ALTER TABLE mst2_qualified_session_incarnation ADD COLUMN canonical_descriptor bytea NOT NULL, + ADD COLUMN attestation_id uuid NOT NULL,ADD COLUMN attestation_digest bytea NOT NULL, + ADD COLUMN state text NOT NULL CHECK(state IN ('READY','RETIRED')), + ADD FOREIGN KEY(attestation_id,metadata_root,root_generation,attestation_digest) + REFERENCES mst2_metadata_source_root_attestation(attestation_id,root_page,root_generation,attestation_digest); +ALTER TABLE mst2_qualified_lease_binding ADD COLUMN namespace_uuid uuid NOT NULL, + ADD COLUMN authorization_epoch bigint NOT NULL,ADD COLUMN publication_sequence bigint NOT NULL, + ADD COLUMN writer_epoch bigint NOT NULL,ADD COLUMN certificate_receipt_id bigint NOT NULL, + ADD COLUMN expires_at_unix bigint NOT NULL,ADD COLUMN state text NOT NULL CHECK(state IN ('ACTIVE','RELEASED','EXPIRED')), + ADD COLUMN lease_epoch bigint NOT NULL CHECK(lease_epoch>0); +CREATE INDEX mst2_qualified_lease_active_incarnation ON mst2_qualified_lease_binding(snapshot_id,session_incarnation,state,lease_id); +CREATE TABLE mst2_metadata_reader_operation ( + operation_id uuid PRIMARY KEY,lease_id text NOT NULL REFERENCES mst2_qualified_lease_binding(lease_id), + snapshot_id text NOT NULL,session_incarnation uuid NOT NULL,root_page bytea NOT NULL,root_generation bigint NOT NULL, + lease_epoch bigint NOT NULL CHECK(lease_epoch>0),hard_deadline_unix bigint NOT NULL, + state text NOT NULL CHECK(state IN ('ACTIVE','FINISHED','EXPIRED')), + reader_issuance bigint NOT NULL CHECK(reader_issuance>=0),terminal_xid bigint, + CONSTRAINT mst2_metadata_reader_identity UNIQUE(operation_id,reader_issuance), + CONSTRAINT mst2_metadata_reader_terminal_state CHECK((state='ACTIVE')=(terminal_xid IS NULL)), + FOREIGN KEY(snapshot_id,session_incarnation) REFERENCES mst2_qualified_session_incarnation(snapshot_id,session_incarnation) +); +CREATE TABLE mst2_metadata_reader_issuance ( + singleton smallint PRIMARY KEY CHECK(singleton=1),high_water bigint NOT NULL CHECK(high_water>=0) +); +INSERT INTO mst2_metadata_reader_issuance VALUES(1,0); +CREATE INDEX mst2_metadata_reader_terminal ON mst2_metadata_reader_operation(terminal_xid,operation_id) WHERE state IN ('FINISHED','EXPIRED'); +CREATE INDEX mst2_metadata_reader_active_lease ON mst2_metadata_reader_operation(lease_id,state,operation_id); +CREATE INDEX mst2_metadata_reader_active_deadline ON mst2_metadata_reader_operation(hard_deadline_unix,operation_id) WHERE state='ACTIVE'; +CREATE TABLE mst2_metadata_root_anchor ( + anchor_id uuid PRIMARY KEY,anchor_kind text NOT NULL CHECK(anchor_kind IN ('PREPARE','REUSE','SESSION','LEASE','REQUEST','READER')), + owner_key text NOT NULL CHECK(octet_length(owner_key) BETWEEN 1 AND 512), + root_page bytea NOT NULL,root_generation bigint NOT NULL,root_certificate_digest bytea NOT NULL, + prepare_id text REFERENCES mst2_metadata_prepare(prepare_id), + snapshot_id text,session_incarnation uuid,lease_id text REFERENCES mst2_qualified_lease_binding(lease_id), + reader_operation_id uuid,reader_issuance bigint, + CONSTRAINT mst2_metadata_root_anchor_reader_identity FOREIGN KEY(reader_operation_id,reader_issuance) REFERENCES mst2_metadata_reader_operation(operation_id,reader_issuance) MATCH FULL, + CONSTRAINT mst2_metadata_root_anchor_reader_kind CHECK((anchor_kind IN ('REQUEST','READER'))=(reader_operation_id IS NOT NULL)), + CONSTRAINT mst2_metadata_root_anchor_reader_fields CHECK((reader_operation_id IS NULL)=(reader_issuance IS NULL)), + UNIQUE(anchor_kind,owner_key,root_page,root_generation), + FOREIGN KEY(root_page,root_generation) REFERENCES mst2_metadata_graph_node(page_id,generation), + FOREIGN KEY(root_page,root_generation,root_certificate_digest) + REFERENCES mst2_metadata_page_certificate(page_id,generation,certificate_digest), + FOREIGN KEY(snapshot_id,session_incarnation) REFERENCES mst2_qualified_session_incarnation(snapshot_id,session_incarnation) +); +CREATE INDEX mst2_metadata_root_anchor_reader ON mst2_metadata_root_anchor(reader_operation_id,reader_issuance) WHERE reader_operation_id IS NOT NULL; +CREATE INDEX mst2_metadata_root_anchor_page ON mst2_metadata_root_anchor(root_page,root_generation,anchor_kind,owner_key); +CREATE INDEX mst2_metadata_root_anchor_prepare ON mst2_metadata_root_anchor(prepare_id,anchor_kind,anchor_id); +CREATE INDEX mst2_metadata_root_anchor_lease ON mst2_metadata_root_anchor(lease_id,anchor_kind,anchor_id); + +CREATE FUNCTION mst2_metadata_session_covers_prepare(pid text) RETURNS boolean LANGUAGE sql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ + SELECT EXISTS(SELECT 1 FROM mst2_metadata_prepare q + JOIN mst2_metadata_current cur ON cur.page_id=q.metadata_root + JOIN mst2_metadata_page_certificate certificate USING(page_id,generation) + JOIN mst2_qualified_session_incarnation session ON session.metadata_root=q.metadata_root + AND session.root_generation=cur.generation + JOIN mst2_metadata_source_root_attestation source ON source.attestation_id=session.attestation_id + AND source.root_page=session.metadata_root AND source.root_generation=session.root_generation + AND source.attestation_digest=session.attestation_digest + AND source.root_certificate_digest=certificate.certificate_digest + JOIN mst2_metadata_root_anchor anchor ON anchor.snapshot_id=session.snapshot_id + AND anchor.session_incarnation=session.session_incarnation AND anchor.anchor_kind='SESSION' + AND anchor.owner_key=session.snapshot_id||':'||session.session_incarnation::text + AND anchor.root_page=session.metadata_root AND anchor.root_generation=session.root_generation + AND anchor.root_certificate_digest=certificate.certificate_digest + WHERE q.prepare_id=pid AND q.plan_kind='ROOTED' AND q.state='COMMITTED' AND session.state='READY' + AND session.namespace_uuid=(SELECT namespace_uuid FROM mst2_metadata_family_identity WHERE singleton=1) + AND convert_from(session.source_profile,'UTF8')::jsonb=$CORE_SCHEMA$.mst2_route_profile( + q.source_domain,q.tagged_root_tree_oid,q.scope,q.schema_version,q.metadata_codec, + q.materialization_policy,q.fs_semantics,q.access_projection,q.verification_revision,q.projection_revision)) +$$; + +CREATE FUNCTION mst2_metadata_native_profile(pid text) RETURNS jsonb LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE q record; +BEGIN + SELECT source_domain,tagged_root_tree_oid,schema_version,metadata_codec,materialization_policy, + fs_semantics,access_projection,verification_revision,projection_revision + INTO q FROM mst2_metadata_prepare WHERE prepare_id=pid AND state IN ('PREPARING','COMMITTED') + AND graph_domain='qualified-v1' AND mst2_metadata_scope_matches(primary_scope); + IF NOT FOUND OR q.source_domain<>'native-git' OR q.schema_version<>2 OR q.metadata_codec<>1 + OR q.materialization_policy<>1 OR q.fs_semantics<>1 OR q.access_projection<>0 + OR q.verification_revision<>2 OR q.projection_revision<>1 + OR q.tagged_root_tree_oid !~ '^(sha1:[0-9a-f]{40}|sha256:[0-9a-f]{64}|blake3:[0-9a-f]{64})$' THEN + RAISE EXCEPTION 'source attestation requires the exact current native metadata profile'; + END IF; + RETURN jsonb_build_object('source_domain',q.source_domain,'hash_kind',split_part(q.tagged_root_tree_oid,':',1), + 'schema_version',q.schema_version,'metadata_codec',q.metadata_codec,'materialization_policy',q.materialization_policy, + 'fs_semantics',q.fs_semantics,'access_projection',q.access_projection, + 'verification_revision',q.verification_revision,'projection_revision',q.projection_revision); +END $$; + +CREATE FUNCTION mst2_metadata_decode_git_tree(b bytea,hash_kind text) RETURNS jsonb +LANGUAGE plpgsql IMMUTABLE STRICT SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE width integer; p integer:=0; start integer; mode text; name bytea; oid bytea; kind integer; + result jsonb[]:=ARRAY[]::jsonb[]; sorted jsonb; count integer:=0; +BEGIN + width:=CASE hash_kind WHEN 'sha1' THEN 20 WHEN 'sha256' THEN 32 WHEN 'blake3' THEN 32 ELSE 0 END; + IF width=0 OR octet_length(b)>67108864 THEN RAISE EXCEPTION 'source Git tree kind or byte budget is invalid'; END IF; + WHILE p32 AND p-start<6 LOOP p:=p+1; END LOOP; + IF p>=octet_length(b) OR get_byte(b,p)<>32 THEN RAISE EXCEPTION 'source Git tree mode is malformed'; END IF; + mode:=convert_from(substring(b FROM start+1 FOR p-start),'UTF8'); p:=p+1; start:=p; + WHILE p0 AND p-start<=255 LOOP p:=p+1; END LOOP; + IF p>=octet_length(b) OR get_byte(b,p)<>0 THEN RAISE EXCEPTION 'source Git tree name is malformed'; END IF; + name:=substring(b FROM start+1 FOR p-start); PERFORM mst2_metadata_valid_name(name); p:=p+1; + IF p>octet_length(b)-width THEN RAISE EXCEPTION 'source Git tree object identity is truncated'; END IF; + oid:=substring(b FROM p+1 FOR width); p:=p+width; + kind:=CASE mode WHEN '40000' THEN 4 WHEN '100644' THEN 1 WHEN '100664' THEN 1 WHEN '100640' THEN 1 + WHEN '100755' THEN 2 WHEN '120000' THEN 3 ELSE 0 END; + IF kind=0 THEN RAISE EXCEPTION 'source Git tree contains an unsupported entry'; END IF; + result:=array_append(result,jsonb_build_object('kind',kind,'name',encode(name,'hex'), + 'oid',hash_kind||':'||encode(oid,'hex'))); count:=count+1; + IF count>131072 THEN RAISE EXCEPTION 'source Git tree entry budget exceeded'; END IF; + END LOOP; + IF EXISTS(SELECT 1 FROM unnest(result) AS rows(value) GROUP BY value->>'name' HAVING count(*)>1) THEN + RAISE EXCEPTION 'source Git tree has duplicate names'; + END IF; + SELECT coalesce(jsonb_agg(value ORDER BY decode(value->>'name','hex')),'[]'::jsonb) INTO sorted + FROM unnest(result) AS rows(value); + RETURN sorted; +END $$; + +CREATE FUNCTION mst2_metadata_compute_source_proof(pid text,tree_oid text,p bytea,g bigint) RETURNS jsonb +LANGUAGE plpgsql VOLATILE SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE profile jsonb:=mst2_metadata_native_profile(pid); profile_hash bytea; body bytea; decoded jsonb; + item jsonb; mapped jsonb; entry_rows jsonb[]:=ARRAY[]::jsonb[]; entries jsonb; child_root bytea; + child_binding record; source_entries jsonb; reference_rows jsonb[]:=ARRAY[]::jsonb[]; + reference jsonb; fact record; built jsonb; proof jsonb; source_revision uuid; body_digest bytea; +BEGIN + IF split_part(tree_oid,':',1) IS DISTINCT FROM profile->>'hash_kind' + OR tree_oid !~ '^(sha1:[0-9a-f]{40}|sha256:[0-9a-f]{64}|blake3:[0-9a-f]{64})$' THEN + RAISE EXCEPTION 'source tree attestation crossed its tagged hash kind'; + END IF; + profile_hash:=sha256(convert_to('mega.mst2.native-profile.v1','UTF8')||decode('00','hex')||convert_to(profile::text,'UTF8')); + SELECT sub_trees INTO body FROM $CORE_SCHEMA$.mega_tree WHERE tree_id=split_part(tree_oid,':',2); + IF NOT FOUND THEN RAISE EXCEPTION 'source Git tree is missing from its captured core'; END IF; + body_digest:=sha256(body); + source_revision:=$CORE_SCHEMA$.mst2_route_capture_source_tree(split_part(tree_oid,':',2),body_digest); + decoded:=mst2_metadata_decode_git_tree(body,profile->>'hash_kind'); + FOR item IN SELECT value FROM jsonb_array_elements(decoded) LOOP + mapped:=jsonb_build_object('kind',(item->>'kind')::integer,'name',item->>'name'); + reference:=jsonb_build_object('kind',(item->>'kind')::integer,'git_oid',item->>'oid'); + IF (item->>'kind')::integer=4 THEN + SELECT a.root_page,a.root_generation,a.root_certificate_digest INTO child_binding FROM mst2_metadata_source_root_attestation a + JOIN mst2_metadata_current cur ON cur.page_id=a.root_page AND cur.generation=a.root_generation + JOIN mst2_metadata_lifetime life USING(page_id,generation) + JOIN mst2_metadata_graph_node node USING(page_id,generation) + JOIN mst2_metadata_prepare origin ON origin.prepare_id=a.origin_prepare_id + JOIN $CORE_SCHEMA$.mega_tree child_tree ON child_tree.tree_id=split_part(a.tagged_tree_oid,':',2) + WHERE a.tagged_tree_oid=item->>'oid' AND a.profile_digest=profile_hash AND a.source_profile=profile + AND a.namespace_uuid=(SELECT namespace_uuid FROM mst2_metadata_family_identity WHERE singleton=1) + AND node.state='LIVE' AND node.certificate_digest=a.root_certificate_digest + AND $CORE_SCHEMA$.mst2_route_source_tree_matches(split_part(a.tagged_tree_oid,':',2),a.source_revision,a.source_body_digest) + AND (life.state='LIVE' AND origin.state='COMMITTED' OR life.state='RESERVED' AND origin.prepare_id=pid AND origin.state='PREPARING') + AND NOT EXISTS(SELECT 1 FROM mst2_metadata_gc_op gc WHERE gc.page_id=a.root_page AND gc.generation=a.root_generation) + ORDER BY a.attestation_id LIMIT 1; + IF NOT FOUND THEN RAISE EXCEPTION 'source directory lacks its exact-profile certified child root'; END IF; + child_root:=child_binding.root_page; + mapped:=mapped||jsonb_build_object('child',encode(child_root,'hex')); + reference:=reference||jsonb_build_object('child_root',encode(child_root,'hex'), + 'child_generation',child_binding.root_generation,'child_certificate',encode(child_binding.root_certificate_digest,'hex')); + ELSE + SELECT size,raw_sha256 INTO fact FROM $CORE_SCHEMA$.mst2_verified_object + WHERE storage_domain='git' AND git_oid=split_part(item->>'oid',':',2) AND object_kind='blob' + AND state='VERIFIED' AND verification_version=2 FOR SHARE NOWAIT; + IF NOT FOUND OR fact.size<0 OR fact.size>8796093022208 OR octet_length(fact.raw_sha256)<>32 + OR (item->>'kind')::integer=3 AND fact.size NOT BETWEEN 1 AND 4095 THEN + RAISE EXCEPTION 'source file lacks its valid current verified-object fact'; + END IF; + mapped:=mapped||jsonb_build_object('size',fact.size,'content_id',encode(fact.raw_sha256,'hex')); + reference:=reference||jsonb_build_object('size',fact.size,'content_digest',encode(fact.raw_sha256,'hex')); + END IF; + reference_rows:=array_append(reference_rows,reference||jsonb_build_object('name',item->>'name')); + entry_rows:=array_append(entry_rows,mapped); + END LOOP; + entries:=to_jsonb(entry_rows); + SELECT coalesce(jsonb_object_agg(value->>'name',value-'name'),'{}'::jsonb) INTO source_entries + FROM unnest(reference_rows) input(value); + built:=mst2_metadata_build_map(entries); + IF decode(built->>'page_id','hex')<>p OR NOT EXISTS(SELECT 1 FROM mst2_metadata_page_certificate c + JOIN mst2_metadata_current cur USING(page_id,generation) JOIN mst2_metadata_graph_node n USING(page_id,generation) + WHERE c.page_id=p AND c.generation=g AND n.state='LIVE' AND n.certificate_digest=c.certificate_digest) THEN + RAISE EXCEPTION 'source projection differs from its independently canonical certified root'; + END IF; + proof:=jsonb_build_object('namespace',(SELECT namespace_uuid::text FROM mst2_metadata_family_identity WHERE singleton=1), + 'source_profile',profile,'profile_digest',encode(profile_hash,'hex'),'tagged_tree_oid',tree_oid, + 'source_body_digest',encode(body_digest,'hex'),'source_revision',source_revision::text,'root_page',encode(p,'hex'),'root_generation',g, + 'root_certificate',(SELECT encode(certificate_digest,'hex') FROM mst2_metadata_page_certificate WHERE page_id=p AND generation=g), + 'source_entry_count',jsonb_array_length(entries),'source_entries',source_entries,'source_work_units',built->'source_work_units', + 'encoded_map_digest',encode(sha256(convert_to(entries::text,'UTF8')),'hex')); + RETURN proof||jsonb_build_object('attestation',encode(sha256(convert_to('mega.mst2.source-root.v1','UTF8') + ||decode('00','hex')||convert_to(proof::text,'UTF8')),'hex')); +END $$; + +CREATE FUNCTION mst2_metadata_source_attestation_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE proof jsonb; +BEGIN + IF TG_OP<>'INSERT' THEN RAISE EXCEPTION 'source root attestation history is immutable'; END IF; + proof:=mst2_metadata_compute_source_proof(NEW.origin_prepare_id,NEW.tagged_tree_oid,NEW.root_page,NEW.root_generation); + IF NEW.source_revision IS NULL THEN NEW.source_revision:=(proof->>'source_revision')::uuid; END IF; + IF NEW.namespace_uuid::text IS DISTINCT FROM proof->>'namespace' OR NEW.source_profile IS DISTINCT FROM proof->'source_profile' + OR NEW.profile_digest IS DISTINCT FROM decode(proof->>'profile_digest','hex') + OR NEW.source_body_digest IS DISTINCT FROM decode(proof->>'source_body_digest','hex') + OR NEW.source_revision IS DISTINCT FROM (proof->>'source_revision')::uuid + OR NEW.root_certificate_digest IS DISTINCT FROM decode(proof->>'root_certificate','hex') + OR NEW.attestation_digest IS DISTINCT FROM decode(proof->>'attestation','hex') OR NEW.source_proof IS DISTINCT FROM proof THEN + RAISE EXCEPTION 'source attestation was not independently derived from the captured core and canonical root'; + END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_metadata_source_attestation_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_metadata_source_root_attestation + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_source_attestation_guard(); + +CREATE FUNCTION mst2_metadata_source_attestation_committed() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + IF NOT EXISTS(SELECT 1 FROM mst2_metadata_prepare WHERE prepare_id=NEW.origin_prepare_id AND state='COMMITTED') THEN + RAISE EXCEPTION 'source attestation cannot commit without its definitive finalized preparation'; + END IF; + IF NOT $CORE_SCHEMA$.mst2_route_source_tree_matches(split_part(NEW.tagged_tree_oid,':',2),NEW.source_revision,NEW.source_body_digest) + OR NEW.source_profile IS DISTINCT FROM mst2_metadata_native_profile(NEW.origin_prepare_id) THEN + RAISE EXCEPTION 'source attestation cannot commit after its exact captured source body or profile changed'; + END IF; + RETURN NULL; +END $$; +CREATE CONSTRAINT TRIGGER mst2_metadata_source_attestation_committed AFTER INSERT ON mst2_metadata_source_root_attestation + DEFERRABLE INITIALLY DEFERRED FOR EACH ROW EXECUTE FUNCTION mst2_metadata_source_attestation_committed(); + +CREATE FUNCTION mst2_metadata_reuse_root_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE profile jsonb; profile_hash bytea; +BEGIN + IF TG_OP<>'INSERT' THEN RAISE EXCEPTION 'rooted reuse membership history is immutable'; END IF; + profile:=mst2_metadata_native_profile(NEW.prepare_id); + profile_hash:=sha256(convert_to('mega.mst2.native-profile.v1','UTF8')||decode('00','hex')||convert_to(profile::text,'UTF8')); + IF NOT EXISTS(SELECT 1 FROM mst2_metadata_prepare q,mst2_metadata_source_root_attestation a + JOIN mst2_metadata_current cur ON cur.page_id=a.root_page AND cur.generation=a.root_generation + JOIN mst2_metadata_lifetime life USING(page_id,generation) + JOIN mst2_metadata_graph_node node USING(page_id,generation) + JOIN mst2_metadata_prepare origin ON origin.prepare_id=a.origin_prepare_id + JOIN $CORE_SCHEMA$.mega_tree tree ON tree.tree_id=split_part(a.tagged_tree_oid,':',2) + WHERE q.prepare_id=NEW.prepare_id AND q.state='PREPARING' AND a.attestation_id=NEW.attestation_id + AND a.root_page=NEW.root_page AND a.root_generation=NEW.root_generation AND a.attestation_digest=NEW.attestation_digest + AND a.profile_digest=profile_hash AND a.source_profile=profile + AND $CORE_SCHEMA$.mst2_route_source_tree_matches(split_part(a.tagged_tree_oid,':',2),a.source_revision,a.source_body_digest) + AND origin.state='COMMITTED' AND life.state='LIVE' AND life.graph_domain='qualified-v1' + AND node.state='LIVE' AND node.certificate_digest=a.root_certificate_digest + AND NOT EXISTS(SELECT 1 FROM mst2_metadata_gc_op gc WHERE gc.page_id=a.root_page AND gc.generation=a.root_generation)) THEN + RAISE EXCEPTION 'rooted reuse membership lacks its exact finalized source and current lifetime'; + END IF; + IF (SELECT count(*) FROM mst2_metadata_prepare_reuse_root WHERE prepare_id=NEW.prepare_id)>=4096 THEN + RAISE EXCEPTION 'rooted reuse boundary budget exceeded'; + END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_metadata_reuse_root_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_metadata_prepare_reuse_root + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_reuse_root_guard(); + +CREATE FUNCTION mst2_metadata_reuse_index_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + IF TG_OP='UPDATE' THEN RAISE EXCEPTION 'rooted reuse index cannot retarget an exact lifetime'; END IF; + IF TG_OP='DELETE' THEN + IF NOT EXISTS(SELECT 1 FROM mst2_metadata_gc_op WHERE page_id=OLD.root_page AND generation=OLD.root_generation AND state='PENDING') THEN + RAISE EXCEPTION 'rooted reuse index retirement requires its exact GC claim'; + END IF; + RETURN OLD; + END IF; + IF NOT EXISTS(SELECT 1 FROM mst2_metadata_source_root_attestation a + JOIN mst2_metadata_prepare q ON q.prepare_id=a.origin_prepare_id + JOIN mst2_metadata_current cur ON cur.page_id=a.root_page AND cur.generation=a.root_generation + JOIN mst2_metadata_lifetime life USING(page_id,generation) + JOIN mst2_metadata_graph_node node USING(page_id,generation) + WHERE a.attestation_id=NEW.attestation_id AND a.profile_digest=NEW.profile_digest AND a.tagged_tree_oid=NEW.tagged_tree_oid + AND a.root_page=NEW.root_page AND a.root_generation=NEW.root_generation AND a.attestation_digest=NEW.attestation_digest + AND q.state='COMMITTED' AND life.state='LIVE' AND node.state='LIVE' AND node.certificate_digest=a.root_certificate_digest + AND NOT EXISTS(SELECT 1 FROM mst2_metadata_gc_op gc WHERE gc.page_id=a.root_page AND gc.generation=a.root_generation)) THEN + RAISE EXCEPTION 'rooted reuse index is not bound to its definitive exact current source proof'; + END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_metadata_reuse_index_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_metadata_reuse_index + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_reuse_index_guard(); + +CREATE FUNCTION mst2_metadata_root_anchor_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + IF TG_OP='UPDATE' THEN RAISE EXCEPTION 'rooted anchors cannot retarget immutable owner or root identities'; END IF; + IF TG_OP='DELETE' THEN + IF OLD.anchor_kind IN ('PREPARE','REUSE') THEN + IF NOT EXISTS(SELECT 1 FROM mst2_metadata_prepare WHERE prepare_id=OLD.prepare_id AND state IN ('PREPARING','COMMITTED','ABORTED')) THEN + RAISE EXCEPTION 'temporary anchor owner history is missing'; + END IF; + ELSIF OLD.anchor_kind='SESSION' THEN + IF NOT EXISTS(SELECT 1 FROM mst2_qualified_session_incarnation s + WHERE s.snapshot_id=OLD.snapshot_id AND s.session_incarnation=OLD.session_incarnation AND s.state='RETIRED') + OR EXISTS(SELECT 1 FROM mst2_qualified_lease_binding l WHERE l.snapshot_id=OLD.snapshot_id + AND l.session_incarnation=OLD.session_incarnation AND l.state='ACTIVE') THEN + RAISE EXCEPTION 'session root still has its active incarnation or leases'; + END IF; + ELSIF OLD.anchor_kind='LEASE' THEN + IF NOT EXISTS(SELECT 1 FROM mst2_qualified_lease_binding WHERE lease_id=OLD.lease_id AND state IN ('RELEASED','EXPIRED')) + OR EXISTS(SELECT 1 FROM mst2_metadata_reader_operation WHERE lease_id=OLD.lease_id AND state='ACTIVE') THEN + RAISE EXCEPTION 'lease root still has its active lease or readers'; + END IF; + ELSE + IF NOT EXISTS(SELECT 1 FROM mst2_metadata_reader_operation r WHERE r.operation_id=OLD.reader_operation_id AND r.reader_issuance=OLD.reader_issuance + AND (r.state='FINISHED' OR r.state='EXPIRED' AND r.hard_deadline_unix<=floor(extract(epoch FROM clock_timestamp()))::bigint)) THEN + RAISE EXCEPTION 'reader root still has an active operation'; + END IF; + END IF; + RETURN OLD; + END IF; + IF NOT EXISTS(SELECT 1 FROM mst2_metadata_graph_node node + JOIN mst2_metadata_current cur USING(page_id,generation) + JOIN mst2_metadata_lifetime life USING(page_id,generation) + JOIN mst2_metadata_page_certificate proof USING(page_id,generation) + WHERE node.page_id=NEW.root_page AND node.generation=NEW.root_generation AND node.state='LIVE' + AND node.certificate_digest=NEW.root_certificate_digest AND proof.certificate_digest=NEW.root_certificate_digest + AND life.state IN ('RESERVED','LIVE') AND life.graph_domain='qualified-v1' + AND NOT EXISTS(SELECT 1 FROM mst2_metadata_gc_op gc WHERE gc.page_id=node.page_id AND gc.generation=node.generation)) THEN + RAISE EXCEPTION 'rooted anchor does not protect its exact canonical current graph'; + END IF; + IF NEW.anchor_kind IN ('PREPARE','REUSE') THEN + IF NEW.owner_key IS DISTINCT FROM NEW.prepare_id OR NEW.snapshot_id IS NOT NULL OR NEW.session_incarnation IS NOT NULL + OR NEW.lease_id IS NOT NULL OR NEW.reader_operation_id IS NOT NULL + OR NOT EXISTS(SELECT 1 FROM mst2_metadata_prepare WHERE prepare_id=NEW.prepare_id + AND state IN ('PREPARING','COMMITTED') AND coverage_retired_at IS NULL) + OR NEW.anchor_kind='PREPARE' AND NOT (EXISTS(SELECT 1 FROM mst2_metadata_prepare_page + WHERE prepare_id=NEW.prepare_id AND page_id=NEW.root_page AND generation=NEW.root_generation) + OR EXISTS(SELECT 1 FROM mst2_metadata_prepare q JOIN mst2_metadata_prepare_reuse_root r USING(prepare_id) + WHERE q.prepare_id=NEW.prepare_id AND q.plan_kind='ROOTED' AND q.metadata_root=NEW.root_page + AND r.root_page=NEW.root_page AND r.root_generation=NEW.root_generation)) + OR NEW.anchor_kind='REUSE' AND NOT EXISTS(SELECT 1 FROM mst2_metadata_prepare_reuse_root + WHERE prepare_id=NEW.prepare_id AND root_page=NEW.root_page AND root_generation=NEW.root_generation) THEN + RAISE EXCEPTION 'temporary anchor differs from its immutable delta or reused-root owner'; + END IF; + ELSIF NEW.anchor_kind='SESSION' THEN + IF NEW.owner_key IS DISTINCT FROM NEW.snapshot_id||':'||NEW.session_incarnation::text OR NEW.prepare_id IS NOT NULL + OR NEW.lease_id IS NOT NULL OR NEW.reader_operation_id IS NOT NULL + OR NOT EXISTS(SELECT 1 FROM mst2_qualified_session_incarnation s WHERE s.snapshot_id=NEW.snapshot_id + AND s.session_incarnation=NEW.session_incarnation AND s.metadata_root=NEW.root_page AND s.root_generation=NEW.root_generation AND s.state='READY') THEN + RAISE EXCEPTION 'session anchor differs from its exact ready incarnation'; + END IF; + ELSIF NEW.anchor_kind='LEASE' THEN + IF NEW.owner_key IS DISTINCT FROM NEW.lease_id OR NEW.prepare_id IS NOT NULL OR NEW.reader_operation_id IS NOT NULL + OR NOT EXISTS(SELECT 1 FROM mst2_qualified_lease_binding l WHERE l.lease_id=NEW.lease_id + AND l.snapshot_id=NEW.snapshot_id AND l.session_incarnation=NEW.session_incarnation + AND l.metadata_root=NEW.root_page AND l.root_generation=NEW.root_generation AND l.state='ACTIVE' + AND l.expires_at_unix>floor(extract(epoch FROM clock_timestamp()))::bigint) THEN + RAISE EXCEPTION 'lease anchor differs from its exact active lease'; + END IF; + ELSE + IF NEW.owner_key IS DISTINCT FROM NEW.reader_operation_id::text OR NEW.prepare_id IS NOT NULL + OR NOT EXISTS(SELECT 1 FROM mst2_metadata_reader_operation r JOIN mst2_qualified_lease_binding l USING(lease_id) + WHERE r.operation_id=NEW.reader_operation_id AND r.reader_issuance=NEW.reader_issuance AND r.lease_id=NEW.lease_id AND r.snapshot_id=NEW.snapshot_id + AND r.session_incarnation=NEW.session_incarnation AND r.root_page=NEW.root_page AND r.root_generation=NEW.root_generation + AND r.state='ACTIVE' AND l.state='ACTIVE' AND l.lease_epoch=r.lease_epoch + AND r.hard_deadline_unix<=l.expires_at_unix AND r.hard_deadline_unix>floor(extract(epoch FROM clock_timestamp()))::bigint) THEN + RAISE EXCEPTION 'reader anchor differs from its exact active lease operation'; + END IF; + END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_metadata_root_anchor_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_metadata_root_anchor + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_root_anchor_guard(); + +CREATE FUNCTION mst2_metadata_reuse_root_protected() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + IF EXISTS(SELECT 1 FROM mst2_metadata_prepare WHERE prepare_id=NEW.prepare_id + AND state IN ('PREPARING','COMMITTED') AND coverage_retired_at IS NULL) + AND NOT EXISTS(SELECT 1 FROM mst2_metadata_root_anchor WHERE anchor_kind='REUSE' AND prepare_id=NEW.prepare_id + AND root_page=NEW.root_page AND root_generation=NEW.root_generation) THEN + RAISE EXCEPTION 'reused boundary must commit with continuously owned root protection'; + END IF; + RETURN NULL; +END $$; +CREATE CONSTRAINT TRIGGER mst2_metadata_reuse_root_protected AFTER INSERT ON mst2_metadata_prepare_reuse_root + DEFERRABLE INITIALLY DEFERRED FOR EACH ROW EXECUTE FUNCTION mst2_metadata_reuse_root_protected(); + +CREATE FUNCTION mst2_metadata_temporary_anchor_continuity() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE q mst2_metadata_prepare%ROWTYPE; +BEGIN + IF OLD.anchor_kind NOT IN ('PREPARE','REUSE') THEN RETURN NULL; END IF; + SELECT * INTO STRICT q FROM mst2_metadata_prepare WHERE prepare_id=OLD.prepare_id; + IF q.state='ABORTED' THEN RETURN NULL; END IF; + IF q.coverage_retired_at IS NULL THEN + IF NOT EXISTS(SELECT 1 FROM mst2_metadata_root_anchor a WHERE a.anchor_kind=OLD.anchor_kind + AND a.owner_key=OLD.owner_key AND a.prepare_id=OLD.prepare_id + AND a.root_page=OLD.root_page AND a.root_generation=OLD.root_generation + AND a.root_certificate_digest=OLD.root_certificate_digest) THEN + RAISE EXCEPTION 'active preparation cannot lose its continuously owned canonical root'; + END IF; + ELSIF q.state<>'COMMITTED' OR NOT (mst2_metadata_session_covers_prepare(q.prepare_id) + OR mst2_metadata_orphan_prepare_retired(q.prepare_id)) THEN + RAISE EXCEPTION 'retired preparation coverage requires a definitive independent session root'; + END IF; + RETURN NULL; +END $$; +CREATE CONSTRAINT TRIGGER mst2_metadata_temporary_anchor_continuity AFTER DELETE ON mst2_metadata_root_anchor + DEFERRABLE INITIALLY DEFERRED FOR EACH ROW EXECUTE FUNCTION mst2_metadata_temporary_anchor_continuity(); diff --git a/src/jupiter/storage/qualified_metadata_canonical.sql b/src/jupiter/storage/qualified_metadata_canonical.sql new file mode 100644 index 00000000..bc04dbf3 --- /dev/null +++ b/src/jupiter/storage/qualified_metadata_canonical.sql @@ -0,0 +1,227 @@ +CREATE FUNCTION mst2_metadata_read_le(b bytea,p integer,w integer) RETURNS numeric +LANGUAGE plpgsql IMMUTABLE STRICT SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE v numeric:=0; power numeric:=1; i integer; +BEGIN + IF w NOT IN (2,4,8) OR p<0 OR p>octet_length(b)-w THEN + RAISE EXCEPTION 'MTP2 integer is out of bounds'; + END IF; + FOR i IN 0..w-1 LOOP + v:=v+get_byte(b,p+i)*power; power:=power*256; + END LOOP; + RETURN v; +END $$; + +CREATE FUNCTION mst2_metadata_write_le(v numeric,w integer) RETURNS bytea +LANGUAGE plpgsql IMMUTABLE STRICT SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE result bytea; i integer; +BEGIN + IF w NOT IN (2,4,8) OR v<0 OR v<>trunc(v) OR v>=power(256::numeric,w) THEN + RAISE EXCEPTION 'MTP2 output integer is out of range'; + END IF; + result:=decode(repeat('00',w),'hex'); + FOR i IN 0..w-1 LOOP result:=set_byte(result,i,mod(v,256)::integer); v:=div(v,256); END LOOP; + RETURN result; +END $$; + +CREATE FUNCTION mst2_metadata_encode_entry(e jsonb) RETURNS bytea +LANGUAGE plpgsql IMMUTABLE STRICT SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE kind integer:=(e->>'kind')::integer; name bytea:=decode(e->>'name','hex'); target bytea; result bytea; +BEGIN + IF kind IS NULL OR kind NOT BETWEEN 1 AND 4 OR name IS NULL THEN RAISE EXCEPTION 'MTP2 output entry is incomplete'; END IF; + PERFORM mst2_metadata_valid_name(name); + result:=set_byte(decode('00','hex'),0,kind)||mst2_metadata_write_le(octet_length(name),2)||name; + IF kind=4 THEN + target:=decode(e->>'child','hex'); + IF target IS NULL OR octet_length(target)<>32 OR target=decode(repeat('00',32),'hex') THEN + RAISE EXCEPTION 'MTP2 output directory reference is invalid'; + END IF; + ELSE + target:=decode(e->>'content_id','hex'); + IF target IS NULL OR octet_length(target)<>32 OR e->>'size' IS NULL THEN + RAISE EXCEPTION 'MTP2 output file reference is invalid'; + END IF; + result:=result||mst2_metadata_write_le((e->>'size')::numeric,8); + END IF; + RETURN result||target; +END $$; + +CREATE FUNCTION mst2_metadata_build_map(items jsonb,budget bigint DEFAULT 67108864) RETURNS jsonb +LANGUAGE plpgsql IMMUTABLE STRICT SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE n integer; item jsonb; previous bytea; name bytea; minimum bytea; maximum bytea; + raw bytea:=''::bytea; result bytea; prefix bytea; terminal jsonb; + partition record; grouped jsonb:='[]'::jsonb; child jsonb; + visits bigint:=0; encoded_bytes bigint:=0; pages bigint:=1; total_bytes bigint; payload bytea; children integer:=0; +BEGIN + IF jsonb_typeof(items)<>'array' OR budget NOT BETWEEN 1 AND 67108864 THEN + RAISE EXCEPTION 'MTP2 canonical map input or work budget is invalid'; + END IF; + n:=jsonb_array_length(items); + IF n>131072 THEN RAISE EXCEPTION 'MTP2 canonical map exceeds its entry budget'; END IF; + FOR item IN SELECT value FROM jsonb_array_elements(items) LOOP + name:=decode(item->>'name','hex'); + IF previous IS NOT NULL AND previous>=name THEN RAISE EXCEPTION 'MTP2 build names are not strictly byte ordered'; END IF; + previous:=name; maximum:=name; IF minimum IS NULL THEN minimum:=name; END IF; + payload:=mst2_metadata_encode_entry(item); visits:=visits+1+octet_length(payload); + IF visits>budget THEN RAISE EXCEPTION 'MTP2 canonical source work budget exceeded'; END IF; + encoded_bytes:=encoded_bytes+octet_length(payload); + IF n<=128 AND 20+encoded_bytes<=16384 THEN raw:=raw||payload; END IF; + END LOOP; + IF n<=128 AND 20+encoded_bytes<=16384 THEN + result:=decode('4d5450320000','hex')||mst2_metadata_write_le(n,2)||mst2_metadata_write_le(n,8) + ||mst2_metadata_write_le(octet_length(raw),4)||raw; + RETURN jsonb_build_object('bytes',encode(result,'hex'),'page_id',encode(sha256( + convert_to('mega.mst2.metapage','UTF8')||decode('00','hex')||result),'hex'), + 'source_work_units',visits,'pages',pages,'metadata_bytes',octet_length(result)); + END IF; + prefix:=mst2_metadata_lcp(minimum,maximum); + SELECT value INTO terminal FROM jsonb_array_elements(items) WHERE decode(value->>'name','hex')=prefix; + visits:=visits+2*n; + IF visits>budget THEN RAISE EXCEPTION 'MTP2 canonical source grouping work budget exceeded'; END IF; + payload:=mst2_metadata_write_le(octet_length(prefix),2)||prefix; + IF terminal IS NULL THEN payload:=payload||decode('00','hex'); + ELSE payload:=payload||decode('01','hex')||mst2_metadata_encode_entry(terminal); END IF; + total_bytes:=0; + FOR partition IN SELECT get_byte(decode(value->>'name','hex'),octet_length(prefix)) AS label, + jsonb_agg(value ORDER BY decode(value->>'name','hex')) AS members + FROM jsonb_array_elements(items) WHERE octet_length(decode(value->>'name','hex'))>octet_length(prefix) + GROUP BY get_byte(decode(value->>'name','hex'),octet_length(prefix)) ORDER BY label LOOP + grouped:=partition.members; + IF visits>=budget THEN RAISE EXCEPTION 'MTP2 canonical source work budget exceeded'; END IF; + child:=mst2_metadata_build_map(grouped,budget-visits); visits:=visits+(child->>'source_work_units')::bigint; + pages:=pages+(child->>'pages')::bigint; total_bytes:=total_bytes+(child->>'metadata_bytes')::bigint; + IF pages>4096 OR total_bytes>67108864 THEN RAISE EXCEPTION 'MTP2 canonical source encoding exceeds metadata budget'; END IF; + payload:=payload||set_byte(decode('00','hex'),0,partition.label)||mst2_metadata_write_le(jsonb_array_length(grouped),8) + ||decode(child->>'page_id','hex'); children:=children+1; + END LOOP; + IF children+(terminal IS NOT NULL)::integer<2 OR 20+octet_length(payload)>16384 THEN + RAISE EXCEPTION 'MTP2 canonical branch shape or size is invalid'; + END IF; + result:=decode('4d5450320100','hex')||mst2_metadata_write_le(children,2)||mst2_metadata_write_le(n,8) + ||mst2_metadata_write_le(octet_length(payload),4)||payload; + total_bytes:=total_bytes+octet_length(result); + IF total_bytes>67108864 THEN RAISE EXCEPTION 'MTP2 canonical source encoding exceeds metadata byte budget'; END IF; + RETURN jsonb_build_object('bytes',encode(result,'hex'),'page_id',encode(sha256( + convert_to('mega.mst2.metapage','UTF8')||decode('00','hex')||result),'hex'), + 'source_work_units',visits,'pages',pages,'metadata_bytes',total_bytes); +END $$; + +CREATE FUNCTION mst2_metadata_valid_name(n bytea) RETURNS boolean +LANGUAGE plpgsql IMMUTABLE STRICT SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE i integer; +BEGIN + IF octet_length(n) NOT BETWEEN 1 AND 255 OR n IN (decode('2e','hex'),decode('2e2e','hex')) THEN + RAISE EXCEPTION 'MTP2 name has an invalid length or dot component'; + END IF; + FOR i IN 0..octet_length(n)-1 LOOP + IF get_byte(n,i) IN (0,47) THEN RAISE EXCEPTION 'MTP2 name contains NUL or slash'; END IF; + END LOOP; + PERFORM convert_from(n,'UTF8'); + RETURN true; +END $$; + +CREATE FUNCTION mst2_metadata_decode_entry(b bytea,p integer) RETURNS jsonb +LANGUAGE plpgsql IMMUTABLE STRICT SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE start integer:=p; kind integer; n integer; name bytea; target bytea; size numeric; +BEGIN + IF p<0 OR p>=octet_length(b) THEN RAISE EXCEPTION 'MTP2 entry kind is truncated'; END IF; + kind:=get_byte(b,p); p:=p+1; + IF kind NOT BETWEEN 1 AND 4 THEN RAISE EXCEPTION 'MTP2 entry kind is invalid'; END IF; + n:=mst2_metadata_read_le(b,p,2)::integer; p:=p+2; + IF n>octet_length(b)-p THEN RAISE EXCEPTION 'MTP2 entry name is truncated'; END IF; + name:=substring(b FROM p+1 FOR n); p:=p+n; + PERFORM mst2_metadata_valid_name(name); + IF kind=4 THEN + IF p>octet_length(b)-32 THEN RAISE EXCEPTION 'MTP2 directory reference is truncated'; END IF; + target:=substring(b FROM p+1 FOR 32); p:=p+32; + IF target=decode(repeat('00',32),'hex') THEN RAISE EXCEPTION 'MTP2 empty directory has a zero reference'; END IF; + RETURN jsonb_build_object('end',p,'kind',kind,'name',encode(name,'hex'), + 'encoded_bytes',p-start,'child',encode(target,'hex')); + END IF; + size:=mst2_metadata_read_le(b,p,8); p:=p+8; + IF p>octet_length(b)-32 THEN RAISE EXCEPTION 'MTP2 file content reference is truncated'; END IF; + target:=substring(b FROM p+1 FOR 32); p:=p+32; + RETURN jsonb_build_object('end',p,'kind',kind,'name',encode(name,'hex'), + 'encoded_bytes',p-start,'size',size,'content_id',encode(target,'hex')); +END $$; + +CREATE FUNCTION mst2_metadata_decode_local(b bytea) RETURNS jsonb +LANGUAGE plpgsql IMMUTABLE STRICT SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE kind integer; n integer; count numeric; declared numeric; p integer:=20; plen integer; + i integer; label integer; previous_label integer:=-1; previous_name bytea; name bytea; prefix bytea; + terminal integer:=0; item jsonb; entries jsonb:='[]'::jsonb; children jsonb:='[]'::jsonb; + refs jsonb:='[]'::jsonb; child bytea; child_count numeric; entry_bytes bigint:=0; +BEGIN + IF octet_length(b) NOT BETWEEN 20 AND 16384 THEN RAISE EXCEPTION 'MTP2 page length is invalid'; END IF; + IF substring(b FROM 1 FOR 4)<>decode('4d545032','hex') OR get_byte(b,5)<>0 THEN + RAISE EXCEPTION 'MTP2 magic or flags are invalid'; + END IF; + kind:=get_byte(b,4); n:=mst2_metadata_read_le(b,6,2)::integer; + count:=mst2_metadata_read_le(b,8,8); + IF count>9223372036854775807 OR mst2_metadata_read_le(b,16,4)<>octet_length(b)-20 THEN + RAISE EXCEPTION 'MTP2 header count or payload length is invalid'; + END IF; + IF kind=0 THEN + IF n>128 OR count<>n THEN RAISE EXCEPTION 'MTP2 leaf count is invalid'; END IF; + IF n>0 THEN + FOR i IN 1..n LOOP + item:=mst2_metadata_decode_entry(b,p); p:=(item->>'end')::integer; + name:=decode(item->>'name','hex'); + IF previous_name IS NOT NULL AND previous_name>=name THEN + RAISE EXCEPTION 'MTP2 leaf names are not strictly byte ordered'; + END IF; + previous_name:=name; entries:=entries||jsonb_build_array(item-'end'); + entry_bytes:=entry_bytes+(item->>'encoded_bytes')::bigint; + IF (item->>'kind')::integer=4 THEN + refs:=refs||jsonb_build_array(jsonb_build_object('kind','DIRECTORY','name',item->>'name','child',item->>'child')); + END IF; + END LOOP; + END IF; + ELSIF kind=1 THEN + plen:=mst2_metadata_read_le(b,p,2)::integer; p:=p+2; + IF plen>octet_length(b)-p THEN RAISE EXCEPTION 'MTP2 branch prefix is truncated'; END IF; + prefix:=substring(b FROM p+1 FOR plen); p:=p+plen; + IF p>=octet_length(b) THEN RAISE EXCEPTION 'MTP2 terminal flag is truncated'; END IF; + terminal:=get_byte(b,p); p:=p+1; + IF terminal NOT IN (0,1) THEN RAISE EXCEPTION 'MTP2 terminal flag is invalid'; END IF; + IF terminal=1 THEN + item:=mst2_metadata_decode_entry(b,p); p:=(item->>'end')::integer; + IF decode(item->>'name','hex')<>prefix THEN RAISE EXCEPTION 'MTP2 terminal name differs from prefix'; END IF; + entries:=entries||jsonb_build_array(item-'end'); entry_bytes:=(item->>'encoded_bytes')::bigint; + IF (item->>'kind')::integer=4 THEN + refs:=refs||jsonb_build_array(jsonb_build_object('kind','DIRECTORY','name',item->>'name','child',item->>'child')); + END IF; + END IF; + IF n>256 OR n+terminal<2 THEN RAISE EXCEPTION 'MTP2 branch group count is invalid'; END IF; + declared:=terminal; + IF n>0 THEN + FOR i IN 1..n LOOP + IF p>=octet_length(b) THEN RAISE EXCEPTION 'MTP2 branch child label is truncated'; END IF; + label:=get_byte(b,p); p:=p+1; + IF label<=previous_label THEN RAISE EXCEPTION 'MTP2 branch labels are not strictly ordered'; END IF; + previous_label:=label; child_count:=mst2_metadata_read_le(b,p,8); p:=p+8; + IF child_count NOT BETWEEN 1 AND 9223372036854775807 THEN RAISE EXCEPTION 'MTP2 child count is invalid'; END IF; + IF p>octet_length(b)-32 THEN RAISE EXCEPTION 'MTP2 branch child digest is truncated'; END IF; + child:=substring(b FROM p+1 FOR 32); p:=p+32; + declared:=declared+child_count; + IF declared>9223372036854775807 THEN RAISE EXCEPTION 'MTP2 branch count overflows'; END IF; + item:=jsonb_build_object('kind','RADIX','label',label,'count',child_count,'child',encode(child,'hex')); + children:=children||jsonb_build_array(item); refs:=refs||jsonb_build_array(item); + END LOOP; + END IF; + IF declared<>count THEN RAISE EXCEPTION 'MTP2 branch header count differs from children'; END IF; + ELSE + RAISE EXCEPTION 'MTP2 page kind is invalid'; + END IF; + IF p<>octet_length(b) THEN RAISE EXCEPTION 'MTP2 page has trailing bytes'; END IF; + RETURN jsonb_build_object('kind',kind,'count',count,'prefix',encode(prefix,'hex'), + 'entries',entries,'children',children,'refs',refs,'direct_entry_bytes',entry_bytes, + 'page_id',encode(sha256(convert_to('mega.mst2.metapage','UTF8')||decode('00','hex')||b),'hex')); +END $$; + +CREATE FUNCTION mst2_metadata_lcp(a bytea,b bytea) RETURNS bytea +LANGUAGE plpgsql IMMUTABLE STRICT SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE n integer:=least(octet_length(a),octet_length(b)); i integer:=0; +BEGIN + WHILE i Value { + Value::Array( + entries + .iter() + .map(|entry| { + json!({"kind":1,"name":hex::encode(&entry.name),"size":u64::MAX, + "content_id":hex::encode([17;32])}) + }) + .collect(), + ) +} + +fn files(names: impl IntoIterator) -> Vec { + let mut entries: Vec<_> = names + .into_iter() + .map(|name| Entry::file(EntryKind::Regular, name.as_bytes(), u64::MAX, [17; 32])) + .collect(); + entries.sort_by(|left, right| left.name.cmp(&right.name)); + entries +} + +#[tokio::test] +async fn database_canonical_builder_matches_pinned_codec_at_split_and_name_boundaries() { + let (_config, _core, _namespace, q, _guard) = fixture().await; + for (width, value) in [ + (2i32, 0u64), + (2, 255), + (2, 256), + (2, u16::MAX as u64), + (4, u32::MAX as u64), + (8, (1u64 << 53) + 1), + (8, (1u64 << 56) - 1), + (8, i64::MAX as u64), + (8, (1u64 << 63) + 1), + (8, u64::MAX - 1), + (8, u64::MAX), + ] { + let row = q.query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT mst2_metadata_write_le($1::numeric,$2) AS encoded,mst2_metadata_read_le(mst2_metadata_write_le($1::numeric,$2),0,$2)::text AS decoded", + [value.to_string().into(), width.into()], + )).await.unwrap().unwrap(); + assert_eq!( + row.try_get::>("", "encoded").unwrap(), + value.to_le_bytes()[..width as usize] + ); + assert_eq!( + row.try_get::("", "decoded").unwrap(), + value.to_string() + ); + } + for (value, width) in [ + ("-1", 8i32), + ("0.5", 8), + ("65536", 2), + ("4294967296", 4), + ("18446744073709551616", 8), + ("1", 1), + ] { + assert!( + q.query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT mst2_metadata_write_le($1::numeric,$2)", + [value.into(), width.into()], + )) + .await + .is_err() + ); + } + let terminal_names = + std::iter::once("a".to_owned()).chain((0..129).map(|index| format!("a{index:03}"))); + let cases = [ + Vec::new(), + files((0..128).map(|index| format!("f{index:03}"))), + files((0..129).map(|index| format!("f{index:03}"))), + files((0..128).map(|index| format!("{}{index:03}", "x".repeat(197)))), + files(terminal_names), + files((0..160).map(|index| format!("目录{index:03}"))), + ]; + for entries in cases { + let bytes = Page::build(&entries).unwrap(); + let proof: Value = q + .query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT mst2_metadata_build_map($1::jsonb) AS proof", + [encoded_entries(&entries).to_string().into()], + )) + .await + .unwrap() + .unwrap() + .try_get("", "proof") + .unwrap(); + assert_eq!(proof["bytes"].as_str().unwrap(), hex::encode(&bytes)); + assert_eq!( + proof["page_id"].as_str().unwrap(), + hex::encode(page_id(&bytes)) + ); + let decoded: Value = q + .query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT mst2_metadata_decode_local($1) AS decoded", + [bytes.into()], + )) + .await + .unwrap() + .unwrap() + .try_get("", "decoded") + .unwrap(); + assert_eq!(decoded["count"].as_u64().unwrap(), entries.len() as u64); + } +} + +#[tokio::test] +async fn database_parser_rejects_nonexact_bytes_and_invalid_names() { + let (_config, _core, _namespace, q, _guard) = fixture().await; + let bytes = Page::build(&files(["file".to_owned()])).unwrap(); + let mut trailing = bytes.clone(); + trailing.push(0); + let mut wrong_length = bytes.clone(); + wrong_length[16] = wrong_length[16].wrapping_add(1); + let mut flags = bytes.clone(); + flags[5] = 1; + let mut wrong_count = bytes.clone(); + wrong_count[8] = 2; + let mut invalid_utf8 = bytes.clone(); + invalid_utf8[23] = 255; + let mut slash = bytes.clone(); + slash[23] = b'/'; + for malformed in [ + trailing, + wrong_length, + flags, + wrong_count, + invalid_utf8, + slash, + ] { + assert!( + q.query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT mst2_metadata_decode_local($1)", + [malformed.into()], + )) + .await + .is_err() + ); + } + let mut entries = encoded_entries(&files(["one".to_owned(), "two".to_owned()])); + entries.as_array_mut().unwrap().reverse(); + assert!( + q.query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT mst2_metadata_build_map($1::jsonb)", + [entries.to_string().into()], + )) + .await + .is_err() + ); +} + +#[tokio::test] +async fn exact_typed_certificates_keep_shared_directory_occurrences_and_dedup_physical_edges() { + let (config, core, _namespace, q, _guard) = fixture().await; + let writer = ShadowQualifiedMetadataWriter::open(&core, &config) + .await + .unwrap(); + let pages = prepared(); + let receipt = write(&writer, "certified-shared-directory", &pages).await; + assert_eq!( + count(&q, "SELECT count(*) FROM mst2_metadata_page_certificate").await, + 2 + ); + assert_eq!( + count( + &q, + "SELECT count(*) FROM mst2_metadata_verified_ref WHERE reference_kind='DIRECTORY'" + ) + .await, + 2 + ); + assert_eq!( + count(&q, "SELECT count(*) FROM mst2_metadata_graph_edge").await, + 1 + ); + assert_eq!( + count( + &q, + "SELECT max(rank)::bigint FROM mst2_metadata_page_certificate" + ) + .await, + 1 + ); + assert_eq!(writer.finalize(receipt.intent()).await.unwrap(), receipt); + for table in [ + "mst2_metadata_page_certificate", + "mst2_metadata_verified_ref", + "mst2_metadata_source_root_attestation", + "mst2_metadata_prepare_reuse_root", + "mst2_metadata_reuse_index", + "mst2_metadata_root_anchor", + "mst2_metadata_reader_operation", + ] { + assert!( + q.execute_unprepared(&format!("TRUNCATE {table} CASCADE")) + .await + .is_err() + ); + } + for sql in [ + "UPDATE mst2_metadata_page_certificate SET rank=rank+1", + "DELETE FROM mst2_metadata_verified_ref", + "UPDATE mst2_metadata_verified_ref SET reference_ordinal=reference_ordinal+1", + ] { + assert!(q.execute_unprepared(sql).await.is_err()); + } +} + +#[tokio::test] +async fn raw_sql_cannot_certify_a_leafable_branch_from_valid_certified_children() { + let (config, core, namespace, q, _guard) = fixture().await; + let writer = ShadowQualifiedMetadataWriter::open(&core, &config) + .await + .unwrap(); + let receipt = write(&writer, "raw-canonical-oracle", &prepared()).await; + let child = Page::build(&[Entry::file(EntryKind::Regular, b"file", 3, [42; 32])]).unwrap(); + let mut payload = vec![1, 0, b'f', 1, 1, 1, 0, b'f']; + payload.extend_from_slice(&3_u64.to_le_bytes()); + payload.extend_from_slice(&[17; 32]); + payload.push(b'i'); + payload.extend_from_slice(&1_u64.to_le_bytes()); + payload.extend_from_slice(&page_id(&child)); + let mut branch = b"MTP2\x01\x00".to_vec(); + branch.extend_from_slice(&1_u16.to_le_bytes()); + branch.extend_from_slice(&2_u64.to_le_bytes()); + branch.extend_from_slice(&(payload.len() as u32).to_le_bytes()); + branch.extend_from_slice(&payload); + let root = page_id(&branch); + let pid = uuid::Uuid::new_v4().to_string(); + let txn = q + .begin_with_config(Some(IsolationLevel::ReadCommitted), None) + .await + .unwrap(); + namespace.enter(&txn).await.unwrap(); + txn.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "INSERT INTO mst2_metadata_prepare SELECT (jsonb_populate_record(NULL::mst2_metadata_prepare, + to_jsonb(q)||jsonb_build_object('prepare_id',$1::text,'operation_id','raw-leafable-branch','metadata_root',$2::bytea, + 'state','PREPARING','committed_at',NULL,'node_count',2,'edge_count',1,'total_bytes',$3::bigint))).* + FROM mst2_metadata_prepare q WHERE q.prepare_id=$4", + [pid.clone().into(),root.to_vec().into(),((branch.len()+child.len()) as i64).into(),receipt.intent().prepare_id().into()], + )).await.unwrap(); + txn.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "INSERT INTO mst2_metadata_lifetime(page_id,node_id,generation,state,metadata_codec,expected_size,graph_domain) + VALUES($1,'page:sha256:'||encode($1,'hex'),1,'RESERVED',1,$2,'qualified-v1')", + [root.to_vec().into(),(branch.len() as i32).into()], + )).await.unwrap(); + txn.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "INSERT INTO mst2_metadata_current VALUES($1,1)", + [root.to_vec().into()], + )) + .await + .unwrap(); + txn.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "INSERT INTO mst2_metadata_prepare_page SELECT $1,$2,1,$3 + UNION ALL SELECT $1,page_id,generation,expected_size + FROM mst2_metadata_prepare_page WHERE prepare_id=$4 AND page_id=$5", + [ + pid.clone().into(), + root.to_vec().into(), + (branch.len() as i32).into(), + receipt.intent().prepare_id().into(), + page_id(&child).to_vec().into(), + ], + )) + .await + .unwrap(); + txn.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "INSERT INTO mst2_metadata_payload(page_id,generation,metadata_codec,byte_size,payload) VALUES($1,1,1,$2,$3)", + [root.to_vec().into(),(branch.len() as i32).into(),branch.into()], + )).await.unwrap(); + let error = txn + .query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT mst2_metadata_certify_page($1,$2,1)", + [pid.into(), root.to_vec().into()], + )) + .await + .unwrap_err(); + assert!(error.to_string().contains("leafable"), "{error}"); + txn.rollback().await.unwrap(); +} + +#[tokio::test] +async fn temporary_root_cannot_disappear_in_a_later_transaction() { + let (config, core, namespace, q, _guard) = fixture().await; + let writer = ShadowQualifiedMetadataWriter::open(&core, &config) + .await + .unwrap(); + let receipt = write(&writer, "continuous-owned-root", &prepared()).await; + let txn = q + .begin_with_config(Some(IsolationLevel::ReadCommitted), None) + .await + .unwrap(); + namespace.enter(&txn).await.unwrap(); + txn.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "INSERT INTO mst2_metadata_root_anchor(anchor_id,anchor_kind,owner_key,root_page,root_generation,root_certificate_digest,prepare_id) + SELECT $1::uuid,'PREPARE',$2,c.page_id,c.generation,c.certificate_digest,$2 + FROM mst2_metadata_page_certificate c JOIN mst2_metadata_prepare q ON q.metadata_root=c.page_id + WHERE q.prepare_id=$2", + [uuid::Uuid::new_v4().to_string().into(),receipt.intent().prepare_id().into()], + )).await.unwrap(); + txn.commit().await.unwrap(); + let error = q + .execute_unprepared("DELETE FROM mst2_metadata_root_anchor WHERE anchor_kind='PREPARE'") + .await + .unwrap_err(); + assert!( + error + .to_string() + .contains("continuously owned canonical root"), + "{error}" + ); + assert_eq!( + count(&q, "SELECT count(*) FROM mst2_metadata_root_anchor").await, + 1 + ); +} + +#[tokio::test] +async fn source_root_is_derived_from_current_core_body_and_verified_blob_facts() { + let (config, core, namespace, q, _guard) = fixture().await; + let entries = [Entry::file(EntryKind::Regular, b"file", 3, [17; 32])]; + let page = Page::build(&entries).unwrap(); + let mut builder = MetadataDagBuilder::new(MetadataDagLimits::default()); + builder.add_directory(&page, &entries).unwrap(); + let prepared = PreparedNativeMetadataRetention::test_installation( + Arc::new(builder.finish(page_id(&page)).unwrap()), + "/", + ); + let mut body = b"100644 file\0".to_vec(); + body.extend_from_slice(&[187; 20]); + core.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "INSERT INTO mega_tree(id,tree_id,sub_trees,size,created_at,pack_id,pack_offset,commit_id) + VALUES(1,$1,$2,0,now(),'fixture',0,'fixture')", + ["a".repeat(40).into(), body.clone().into()], + )) + .await + .unwrap(); + core.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "INSERT INTO mst2_verified_object(storage_domain,git_oid,object_kind,raw_sha256,size,verification_version,state,created_at) + VALUES('git',$1,'blob',$2,3,2,'VERIFIED',now())", + ["b".repeat(40).into(),vec![17_u8;32].into()], + )).await.unwrap(); + let writer = ShadowQualifiedMetadataWriter::open(&core, &config) + .await + .unwrap(); + let receipt = write(&writer, "bound-real-core-source", &prepared).await; + let aid = uuid::Uuid::new_v4().to_string(); + let txn = q + .begin_with_config(Some(IsolationLevel::ReadCommitted), None) + .await + .unwrap(); + namespace.enter(&txn).await.unwrap(); + txn.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "INSERT INTO mst2_metadata_source_root_attestation(attestation_id,namespace_uuid,origin_prepare_id, + tagged_tree_oid,source_profile,profile_digest,source_body_digest,root_page,root_generation, + root_certificate_digest,source_proof,attestation_digest) + SELECT $1::uuid,(proof->>'namespace')::uuid,$2,$3,proof->'source_profile',decode(proof->>'profile_digest','hex'), + decode(proof->>'source_body_digest','hex'),$4,1,decode(proof->>'root_certificate','hex'),proof, + decode(proof->>'attestation','hex') FROM (SELECT mst2_metadata_compute_source_proof($2,$3,$4,1) AS proof) input", + [aid.into(),receipt.intent().prepare_id().into(),prepared.fixed_root_tree_oid().into(),page_id(&page).to_vec().into()], + )).await.unwrap(); + txn.commit().await.unwrap(); + assert_eq!( + count( + &q, + "SELECT count(*) FROM mst2_metadata_source_root_attestation" + ) + .await, + 1 + ); + body[7] = b'F'; + core.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "UPDATE mega_tree SET sub_trees=$1 WHERE tree_id=$2", + [body.into(), "a".repeat(40).into()], + )) + .await + .unwrap(); + let error = q + .query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT mst2_metadata_compute_source_proof($1,$2,$3,1)", + [ + receipt.intent().prepare_id().into(), + prepared.fixed_root_tree_oid().into(), + page_id(&page).to_vec().into(), + ], + )) + .await + .unwrap_err(); + assert!( + error + .to_string() + .contains("independently canonical certified root"), + "{error}" + ); +} + +pub(super) async fn seeded_rooted_plan( + core: &DatabaseConnection, + tree_char: char, + name: &str, + id: i64, +) -> (RootedMetadataInstallPlan, MetadataPagePayload) { + let entries = [Entry::file( + EntryKind::Regular, + name.as_bytes(), + 3, + [17; 32], + )]; + let bytes = Page::build(&entries).unwrap(); + let page = page_id(&bytes); + let mut builder = MetadataDagBuilder::new(MetadataDagLimits::default()); + builder.add_directory(&bytes, &entries).unwrap(); + let prepared = PreparedNativeMetadataRetention::test_installation( + Arc::new(builder.finish(page).unwrap()), + "/", + ); + let mut identity = prepared.install_plan().unwrap().identity; + identity.tagged_root_tree_oid = format!("sha1:{}", tree_char.to_string().repeat(40)); + let mut body = format!("100644 {name}\0").into_bytes(); + body.extend_from_slice(&[187; 20]); + core.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "INSERT INTO mega_tree(id,tree_id,sub_trees,size,created_at,pack_id,pack_offset,commit_id) + VALUES($1,$2,$3,0,now(),'fixture',0,'fixture')", + [ + id.into(), + tree_char.to_string().repeat(40).into(), + body.into(), + ], + )) + .await + .unwrap(); + core.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "INSERT INTO mst2_verified_object(storage_domain,git_oid,object_kind,raw_sha256,size,verification_version,state,created_at) + VALUES('git',$1,'blob',$2,3,2,'VERIFIED',now()) ON CONFLICT(storage_domain,git_oid,object_kind) DO NOTHING", + ["b".repeat(40).into(),vec![17_u8;32].into()], + )).await.unwrap(); + let source_roots = BTreeMap::from([(identity.tagged_root_tree_oid.clone(), page)]); + let plan = RootedMetadataInstallPlan::new( + identity, + page, + BTreeMap::from([(page, bytes.len() as u64)]), + BTreeSet::new(), + BTreeMap::new(), + source_roots, + ) + .unwrap(); + ( + plan, + MetadataPagePayload { + id: page, + size: bytes.len() as u64, + bytes, + }, + ) +} + +#[tokio::test] +async fn actual_rooted_cold_and_zero_delta_reuse_keep_one_canonical_graph() { + let (config, core, _namespace, q, _guard) = fixture().await; + let (cold, payload) = seeded_rooted_plan(&core, 'a', "file", 1).await; + let expected_payload = payload.bytes.clone(); + let writer = RootedQualifiedMetadataRepository::open(&core, &config) + .await + .unwrap(); + let intent = writer + .begin_intent("actual-rooted-cold", &cold) + .await + .unwrap(); + writer.install_pages(&intent, &[payload]).await.unwrap(); + let receipt = writer.finalize(&intent).await.unwrap(); + assert_eq!(receipt.metadata_root(), cold.root); + assert_eq!( + writer.recover("actual-rooted-cold", &cold).await.unwrap(), + Some(receipt.clone()) + ); + let proof=q.query_one_raw(Statement::from_string(DbBackend::Postgres, + "SELECT a.attestation_id::text,a.attestation_digest,a.root_certificate_digest FROM mst2_metadata_source_root_attestation a" + )).await.unwrap().unwrap(); + let reused = RootedReuseRoot { + generation: intent.root_generation(), + attestation_id: uuid::Uuid::parse_str( + &proof.try_get::("", "attestation_id").unwrap(), + ) + .unwrap(), + attestation_digest: proof + .try_get::>("", "attestation_digest") + .unwrap() + .try_into() + .unwrap(), + certificate_digest: proof + .try_get::>("", "root_certificate_digest") + .unwrap() + .try_into() + .unwrap(), + }; + let warm = RootedMetadataInstallPlan::new( + cold.identity.clone(), + cold.root, + BTreeMap::new(), + BTreeSet::new(), + BTreeMap::from([(cold.root, reused)]), + cold.source_roots.clone(), + ) + .unwrap(); + let warm_intent = writer + .begin_intent("actual-zero-delta-root", &warm) + .await + .unwrap(); + assert_eq!( + writer.finalize(&warm_intent).await.unwrap().metadata_root(), + cold.root + ); + assert_eq!( + count(&q, "SELECT count(*) FROM mst2_metadata_payload").await, + 1 + ); + assert_eq!( + count(&q, "SELECT count(*) FROM mst2_metadata_page_certificate").await, + 1 + ); + assert_eq!( + count( + &q, + "SELECT count(*) FROM mst2_metadata_source_root_attestation" + ) + .await, + 1 + ); + assert_eq!(count(&q,"SELECT count(*) FROM mst2_metadata_prepare WHERE plan_kind='ROOTED' AND state='COMMITTED' AND node_count=0").await,1); + for (operation, plan, delta_count, reused_count) in [ + ("actual-rooted-cold", &cold, 1usize, 0usize), + ("actual-zero-delta-root", &warm, 0usize, 1usize), + ] { + let row = q.query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT canonical_plan,mst2_metadata_decode_rooted_plan(canonical_plan) AS decoded FROM mst2_metadata_prepare WHERE operation_id=$1 AND state='COMMITTED'", + [operation.into()], + )).await.unwrap().unwrap(); + assert_eq!( + row.try_get::>("", "canonical_plan").unwrap(), + plan.encode().unwrap() + ); + let decoded: Value = row.try_get("", "decoded").unwrap(); + assert_eq!(decoded["root"], hex::encode(cold.root)); + assert_eq!(decoded["delta"].as_array().unwrap().len(), delta_count); + assert_eq!(decoded["reused"].as_array().unwrap().len(), reused_count); + assert_eq!(decoded["source_roots"].as_array().unwrap().len(), 1); + let member = if delta_count == 1 { + &decoded["delta"][0] + } else { + &decoded["reused"][0] + }; + assert_eq!(member["page"], hex::encode(cold.root)); + } + let resident = q + .query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT generation,payload FROM mst2_metadata_payload WHERE page_id=$1", + [cold.root.to_vec().into()], + )) + .await + .unwrap() + .unwrap(); + assert_eq!( + resident.try_get::>("", "payload").unwrap(), + expected_payload + ); + assert_eq!( + resident.try_get::("", "generation").unwrap(), + intent.root_generation() + ); + let error = q + .execute_unprepared("DELETE FROM mst2_metadata_root_anchor WHERE anchor_kind='REUSE'") + .await + .unwrap_err(); + assert!( + error + .to_string() + .contains("continuously owned canonical root"), + "{error}" + ); +} + +#[tokio::test] +async fn changed_ancestor_installs_only_delta_and_accepts_independently_attested_source_aliases() { + let (config, core, _namespace, q, _guard) = fixture().await; + let writer = RootedQualifiedMetadataRepository::open(&core, &config) + .await + .unwrap(); + let (first, payload) = seeded_rooted_plan(&core, 'a', "file", 1).await; + let first_intent = writer.begin_intent("alias-first", &first).await.unwrap(); + writer + .install_pages(&first_intent, &[payload]) + .await + .unwrap(); + writer.finalize(&first_intent).await.unwrap(); + let (alias, payload) = seeded_rooted_plan(&core, 'c', "file", 2).await; + assert_eq!(alias.root, first.root); + let alias_intent = writer.begin_intent("alias-second", &alias).await.unwrap(); + writer + .install_pages(&alias_intent, &[payload]) + .await + .unwrap(); + writer.finalize(&alias_intent).await.unwrap(); + let first_hint = writer + .lookup_reuse(&first.identity.tagged_root_tree_oid, &first.identity) + .await + .unwrap() + .unwrap(); + let alias_hint = writer + .lookup_reuse(&alias.identity.tagged_root_tree_oid, &alias.identity) + .await + .unwrap() + .unwrap(); + assert_eq!(first_hint.page_id, alias_hint.page_id); + assert_eq!( + first_hint.proof.certificate_digest, + alias_hint.proof.certificate_digest + ); + assert_ne!( + first_hint.proof.attestation_id, + alias_hint.proof.attestation_id + ); + + let entries = [ + Entry::dir(b"one", first.root), + Entry::dir(b"two", first.root), + ]; + let bytes = Page::build(&entries).unwrap(); + let root = page_id(&bytes); + let mut body = b"40000 one\0".to_vec(); + body.extend_from_slice(&[0xaa; 20]); + body.extend_from_slice(b"40000 two\0"); + body.extend_from_slice(&[0xcc; 20]); + core.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "INSERT INTO mega_tree(id,tree_id,sub_trees,size,created_at,pack_id,pack_offset,commit_id) + VALUES(3,$1,$2,0,now(),'fixture',0,'fixture')", + ["d".repeat(40).into(), body.into()], + )) + .await + .unwrap(); + let mut identity = first.identity.clone(); + identity.tagged_root_tree_oid = format!("sha1:{}", "d".repeat(40)); + let plan = RootedMetadataInstallPlan::new( + identity.clone(), + root, + BTreeMap::from([(root, bytes.len() as u64)]), + BTreeSet::from([(root, first.root)]), + BTreeMap::from([(first.root, first_hint.proof)]), + BTreeMap::from([ + (identity.tagged_root_tree_oid, root), + (first.identity.tagged_root_tree_oid.clone(), first.root), + (alias.identity.tagged_root_tree_oid.clone(), first.root), + ]), + ) + .unwrap(); + let intent = writer + .begin_intent("changed-ancestor-aliases", &plan) + .await + .unwrap(); + writer + .install_pages( + &intent, + &[MetadataPagePayload { + id: root, + size: bytes.len() as u64, + bytes, + }], + ) + .await + .unwrap(); + let receipt = writer.finalize(&intent).await.unwrap(); + assert_eq!(receipt.metadata_root(), root); + assert_eq!( + count(&q, "SELECT count(*) FROM mst2_metadata_payload").await, + 2 + ); + assert_eq!( + count(&q, "SELECT count(*) FROM mst2_metadata_graph_edge").await, + 1 + ); + assert_eq!( + count( + &q, + "SELECT count(*) FROM mst2_metadata_verified_ref WHERE reference_kind='DIRECTORY'" + ) + .await, + 2 + ); + assert_eq!( + count( + &q, + "SELECT count(*) FROM mst2_metadata_source_root_attestation" + ) + .await, + 3 + ); + let row = q.query_one_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "SELECT node_count,edge_count,(SELECT count(*) FROM mst2_metadata_prepare_reuse_root r + WHERE r.prepare_id=q.prepare_id)::bigint AS reuse_count FROM mst2_metadata_prepare q WHERE prepare_id=$1", + [intent.prepare_id().into()])).await.unwrap().unwrap(); + assert_eq!(row.try_get::("", "node_count").unwrap(), 1); + assert_eq!(row.try_get::("", "edge_count").unwrap(), 1); + assert_eq!(row.try_get::("", "reuse_count").unwrap(), 1); + assert_eq!( + writer + .recover("changed-ancestor-aliases", &plan) + .await + .unwrap(), + Some(receipt) + ); +} + +#[tokio::test] +async fn late_raw_delta_membership_is_rejected_after_intent_commit() { + let (config, core, _namespace, q, _guard) = fixture().await; + let (plan, payload) = seeded_rooted_plan(&core, 'a', "file", 1).await; + let (donor, donor_payload) = seeded_rooted_plan(&core, 'c', "other", 2).await; + let writer = RootedQualifiedMetadataRepository::open(&core, &config) + .await + .unwrap(); + let donor_intent = writer + .begin_intent("rooted-extra-donor", &donor) + .await + .unwrap(); + writer + .install_pages(&donor_intent, std::slice::from_ref(&donor_payload)) + .await + .unwrap(); + writer.finalize(&donor_intent).await.unwrap(); + let intent = writer + .begin_intent("rooted-bounded-intent", &plan) + .await + .unwrap(); + writer.install_pages(&intent, &[payload]).await.unwrap(); + let error=q.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "INSERT INTO mst2_metadata_prepare_page(prepare_id,page_id,generation,expected_size) VALUES($1,$2,1,$3)", + [intent.prepare_id().into(),donor.root.to_vec().into(),(donor_payload.size as i32).into()], + )).await.unwrap_err(); + assert!( + error + .to_string() + .contains("missing or extra exact delta/reuse"), + "{error}" + ); + assert_eq!( + writer.finalize(&intent).await.unwrap().metadata_root(), + plan.root + ); +} diff --git a/src/jupiter/storage/qualified_metadata_certificates.sql b/src/jupiter/storage/qualified_metadata_certificates.sql new file mode 100644 index 00000000..6ddd7e66 --- /dev/null +++ b/src/jupiter/storage/qualified_metadata_certificates.sql @@ -0,0 +1,293 @@ +CREATE TABLE mst2_metadata_page_certificate ( + page_id bytea NOT NULL,generation bigint NOT NULL,certificate_digest bytea NOT NULL CHECK(octet_length(certificate_digest)=32), + origin_prepare_id text NOT NULL REFERENCES mst2_metadata_prepare(prepare_id), + namespace_uuid uuid NOT NULL REFERENCES mst2_metadata_family_identity(namespace_uuid), + proof_revision integer NOT NULL CHECK(proof_revision=1),metadata_codec smallint NOT NULL CHECK(metadata_codec=1), + byte_size integer NOT NULL CHECK(byte_size BETWEEN 20 AND 16384), + rank integer NOT NULL CHECK(rank BETWEEN 0 AND 4095), + map_entry_count bigint NOT NULL CHECK(map_entry_count BETWEEN 0 AND 131072), + map_encoded_entry_bytes bigint NOT NULL CHECK(map_encoded_entry_bytes BETWEEN 0 AND 67108864), + min_name bytea,max_name bytea, + relative_path_bytes integer NOT NULL CHECK(relative_path_bytes BETWEEN 0 AND 4096), + relative_components integer NOT NULL CHECK(relative_components BETWEEN 0 AND 256), + closure_nodes_upper integer NOT NULL CHECK(closure_nodes_upper BETWEEN 1 AND 4096), + closure_edges_upper integer NOT NULL CHECK(closure_edges_upper BETWEEN 0 AND 16384), + closure_bytes_upper bigint NOT NULL CHECK(closure_bytes_upper BETWEEN 20 AND 67108864), + closure_entries_upper bigint NOT NULL CHECK(closure_entries_upper BETWEEN 0 AND 131072), + canonical_proof jsonb NOT NULL, + PRIMARY KEY(page_id,generation),UNIQUE(page_id,generation,certificate_digest), + FOREIGN KEY(page_id,generation) REFERENCES mst2_metadata_lifetime(page_id,generation), + CHECK((map_entry_count=0)=(min_name IS NULL AND max_name IS NULL)), + CHECK(map_entry_count=0 OR (min_name IS NOT NULL AND max_name IS NOT NULL AND min_name<=max_name)) +); +CREATE TABLE mst2_metadata_verified_ref ( + parent_page bytea NOT NULL,parent_generation bigint NOT NULL, + reference_ordinal integer NOT NULL CHECK(reference_ordinal BETWEEN 0 AND 256), + reference_kind text NOT NULL CHECK(reference_kind IN ('DIRECTORY','RADIX')), + name bytea,label integer,advertised_count bigint, + child_page bytea NOT NULL,child_generation bigint NOT NULL, + child_certificate_digest bytea NOT NULL CHECK(octet_length(child_certificate_digest)=32), + PRIMARY KEY(parent_page,parent_generation,reference_ordinal), + FOREIGN KEY(parent_page,parent_generation) REFERENCES mst2_metadata_page_certificate(page_id,generation), + FOREIGN KEY(child_page,child_generation,child_certificate_digest) + REFERENCES mst2_metadata_page_certificate(page_id,generation,certificate_digest), + CHECK((reference_kind='DIRECTORY' AND name IS NOT NULL AND label IS NULL AND advertised_count IS NULL) + OR (reference_kind='RADIX' AND name IS NULL AND label IS NOT NULL AND advertised_count IS NOT NULL + AND label BETWEEN 0 AND 255 AND advertised_count>0)) +); +CREATE INDEX mst2_metadata_verified_ref_child ON mst2_metadata_verified_ref(child_page,child_generation,parent_page,parent_generation); +ALTER TABLE mst2_metadata_graph_node ADD COLUMN certificate_digest bytea NOT NULL, + ADD FOREIGN KEY(page_id,generation,certificate_digest) + REFERENCES mst2_metadata_page_certificate(page_id,generation,certificate_digest); + +CREATE FUNCTION mst2_metadata_child_certificate(p bytea,g bigint,pid text) RETURNS jsonb +LANGUAGE plpgsql VOLATILE SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE c mst2_metadata_page_certificate%ROWTYPE; +BEGIN + SELECT proof.* INTO c FROM mst2_metadata_page_certificate proof + JOIN mst2_metadata_current cur USING(page_id,generation) + JOIN mst2_metadata_lifetime life USING(page_id,generation) + JOIN mst2_metadata_graph_node node USING(page_id,generation) + JOIN mst2_metadata_payload body USING(page_id,generation) + WHERE proof.page_id=p AND proof.generation=g AND life.graph_domain='qualified-v1' + AND node.state='LIVE' AND node.certificate_digest=proof.certificate_digest + AND node.metadata_codec=proof.metadata_codec AND body.metadata_codec=proof.metadata_codec + AND body.byte_size=proof.byte_size AND node.bytes=proof.byte_size AND life.expected_size=proof.byte_size + AND (life.state='LIVE' OR (life.state='RESERVED' AND proof.origin_prepare_id=pid + AND EXISTS(SELECT 1 FROM mst2_metadata_prepare q WHERE q.prepare_id=pid AND q.state='PREPARING'))) + AND NOT EXISTS(SELECT 1 FROM mst2_metadata_gc_op op WHERE op.page_id=p AND op.generation=g); + IF NOT FOUND THEN RAISE EXCEPTION 'MTP2 child lacks its exact certified current graph'; END IF; + RETURN jsonb_build_object('page',encode(c.page_id,'hex'),'generation',c.generation, + 'certificate',encode(c.certificate_digest,'hex'),'rank',c.rank,'map_entry_count',c.map_entry_count, + 'map_encoded_entry_bytes',c.map_encoded_entry_bytes,'min_name',encode(c.min_name,'hex'), + 'max_name',encode(c.max_name,'hex'),'relative_path_bytes',c.relative_path_bytes, + 'relative_components',c.relative_components,'closure_nodes_upper',c.closure_nodes_upper, + 'closure_edges_upper',c.closure_edges_upper,'closure_bytes_upper',c.closure_bytes_upper, + 'closure_entries_upper',c.closure_entries_upper); +END $$; + +CREATE FUNCTION mst2_metadata_compute_certificate(p bytea,g bigint,pid text) RETURNS jsonb +LANGUAGE plpgsql VOLATILE SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE body mst2_metadata_payload%ROWTYPE; q mst2_metadata_prepare%ROWTYPE; decoded jsonb; + ref jsonb; entry jsonb; child jsonb; bound jsonb; bound_refs jsonb:='[]'::jsonb; + seen jsonb:='[]'::jsonb; children jsonb:='{}'::jsonb; child_page bytea; child_generation bigint; + key text; name bytea; child_min bytea; child_max bytea; prefix bytea; minimum bytea; maximum bytea; + map_count bigint:=0; map_bytes bigint:=0; path_bytes integer:=0; components integer:=0; rank integer:=0; + nodes bigint:=1; edges bigint:=0; bytes bigint; entries bigint; proof jsonb; +BEGIN + SELECT * INTO q FROM mst2_metadata_prepare WHERE prepare_id=pid AND state='PREPARING' + AND graph_domain='qualified-v1' AND mst2_metadata_scope_matches(primary_scope); + IF NOT FOUND THEN RAISE EXCEPTION 'MTP2 certification needs an active exact preparation'; END IF; + SELECT b.* INTO body FROM mst2_metadata_payload b JOIN mst2_metadata_current cur USING(page_id,generation) + JOIN mst2_metadata_lifetime life USING(page_id,generation) + JOIN mst2_metadata_prepare_page member USING(page_id,generation) + WHERE b.page_id=p AND b.generation=g AND member.prepare_id=pid AND life.state='RESERVED' + AND life.graph_domain='qualified-v1' AND life.metadata_codec=q.metadata_codec + AND b.metadata_codec=q.metadata_codec AND b.byte_size=member.expected_size AND b.byte_size=life.expected_size + AND NOT EXISTS(SELECT 1 FROM mst2_metadata_gc_op op WHERE op.page_id=p AND op.generation=g); + IF NOT FOUND THEN RAISE EXCEPTION 'MTP2 certification crossed its durable payload lifetime'; END IF; + decoded:=mst2_metadata_decode_local(body.payload); + IF decode(decoded->>'page_id','hex')<>p THEN RAISE EXCEPTION 'MTP2 durable payload digest differs from its page'; END IF; + map_count:=jsonb_array_length(decoded->'entries'); map_bytes:=(decoded->>'direct_entry_bytes')::bigint; + bytes:=body.byte_size; entries:=map_count; + FOR ref IN SELECT value FROM jsonb_array_elements(decoded->'refs') LOOP + child_page:=decode(ref->>'child','hex'); + SELECT member.generation INTO child_generation FROM mst2_metadata_prepare_page member + WHERE member.prepare_id=pid AND member.page_id=child_page; + IF NOT FOUND THEN + SELECT reused.root_generation INTO child_generation FROM mst2_metadata_prepare_reuse_root reused + JOIN mst2_metadata_root_anchor anchor ON anchor.prepare_id=reused.prepare_id + AND anchor.anchor_kind='REUSE' AND anchor.owner_key=pid + AND anchor.root_page=reused.root_page AND anchor.root_generation=reused.root_generation + WHERE reused.prepare_id=pid AND reused.root_page=child_page; + IF NOT FOUND THEN RAISE EXCEPTION 'MTP2 reference is outside exact delta or reused-root membership'; END IF; + END IF; + child:=mst2_metadata_child_certificate(child_page,child_generation,pid); + bound:=ref||jsonb_build_object('generation',child_generation,'certificate',child->>'certificate'); + bound_refs:=bound_refs||jsonb_build_array(bound); + key:=encode(child_page,'hex')||':'||child_generation; + children:=children||jsonb_build_object(encode(child_page,'hex'),child); + IF NOT seen ? key THEN + seen:=seen||jsonb_build_array(key); rank:=greatest(rank,(child->>'rank')::integer+1); + nodes:=nodes+(child->>'closure_nodes_upper')::bigint; + edges:=edges+1+(child->>'closure_edges_upper')::bigint; + bytes:=bytes+(child->>'closure_bytes_upper')::bigint; + entries:=entries+(child->>'closure_entries_upper')::bigint; + END IF; + IF ref->>'kind'='RADIX' THEN + child_min:=decode(child->>'min_name','hex'); child_max:=decode(child->>'max_name','hex'); + prefix:=decode(decoded->>'prefix','hex'); + IF child_min IS NULL OR child_max IS NULL OR (child->>'map_entry_count')::bigint<>(ref->>'count')::bigint + OR octet_length(child_min)<=octet_length(prefix) OR octet_length(child_max)<=octet_length(prefix) + OR substring(child_min FROM 1 FOR octet_length(prefix))<>prefix + OR substring(child_max FROM 1 FOR octet_length(prefix))<>prefix + OR get_byte(child_min,octet_length(prefix))<>(ref->>'label')::integer + OR get_byte(child_max,octet_length(prefix))<>(ref->>'label')::integer THEN + RAISE EXCEPTION 'MTP2 radix child count or name partition differs from its canonical certificate'; + END IF; + IF minimum IS NULL OR child_minmaximum THEN maximum:=child_max; END IF; + map_count:=map_count+(child->>'map_entry_count')::bigint; + map_bytes:=map_bytes+(child->>'map_encoded_entry_bytes')::bigint; + path_bytes:=greatest(path_bytes,(child->>'relative_path_bytes')::integer); + components:=greatest(components,(child->>'relative_components')::integer); + END IF; + IF rank>4095 OR nodes>4096 OR edges>16384 OR bytes>67108864 OR entries>131072 THEN + RAISE EXCEPTION 'MTP2 certified closure upper bound exceeds its fixed budget'; + END IF; + END LOOP; + FOR entry IN SELECT value FROM jsonb_array_elements(decoded->'entries') LOOP + name:=decode(entry->>'name','hex'); + IF minimum IS NULL OR namemaximum THEN maximum:=name; END IF; + IF (entry->>'kind')::integer=4 THEN + child:=children->(entry->>'child'); + path_bytes:=greatest(path_bytes,1+octet_length(name)+(child->>'relative_path_bytes')::integer); + components:=greatest(components,1+(child->>'relative_components')::integer); + ELSE + path_bytes:=greatest(path_bytes,1+octet_length(name)); components:=greatest(components,1); + END IF; + END LOOP; + IF map_count<>(decoded->>'count')::bigint OR map_count>131072 OR map_bytes>67108864 + OR path_bytes>4096 OR components>256 THEN RAISE EXCEPTION 'MTP2 canonical map or path budget is invalid'; END IF; + IF (decoded->>'kind')::integer=1 AND ( + (map_count<=128 AND 20+map_bytes<=16384) + OR decode(decoded->>'prefix','hex') IS DISTINCT FROM mst2_metadata_lcp(minimum,maximum)) THEN + RAISE EXCEPTION 'MTP2 branch is leafable or its prefix is not the true canonical LCP'; + END IF; + proof:=jsonb_build_object('namespace',(SELECT namespace_uuid::text FROM mst2_metadata_family_identity WHERE singleton=1), + 'page',encode(p,'hex'),'generation',g,'codec',body.metadata_codec,'byte_size',body.byte_size, + 'proof_revision',1,'page_kind',(decoded->>'kind')::integer,'references',bound_refs,'rank',rank, + 'map_entry_count',map_count,'map_encoded_entry_bytes',map_bytes,'min_name',encode(minimum,'hex'), + 'max_name',encode(maximum,'hex'),'relative_path_bytes',path_bytes,'relative_components',components, + 'closure_nodes_upper',nodes,'closure_edges_upper',edges,'closure_bytes_upper',bytes,'closure_entries_upper',entries); + RETURN proof||jsonb_build_object('certificate',encode(sha256(convert_to('mega.mst2.canonical-proof.v1','UTF8') + ||decode('00','hex')||convert_to(proof::text,'UTF8')),'hex')); +END $$; + +CREATE FUNCTION mst2_metadata_certificate_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE proof jsonb; +BEGIN + IF TG_OP<>'INSERT' THEN RAISE EXCEPTION 'MTP2 canonical certificate history is immutable'; END IF; + proof:=mst2_metadata_compute_certificate(NEW.page_id,NEW.generation,NEW.origin_prepare_id); + IF NEW.certificate_digest<>decode(proof->>'certificate','hex') OR NEW.canonical_proof<>proof + OR NEW.namespace_uuid::text<>proof->>'namespace' OR NEW.proof_revision<>1 OR NEW.metadata_codec<>(proof->>'codec')::smallint + OR NEW.byte_size<>(proof->>'byte_size')::integer OR NEW.rank<>(proof->>'rank')::integer + OR NEW.map_entry_count<>(proof->>'map_entry_count')::bigint + OR NEW.map_encoded_entry_bytes<>(proof->>'map_encoded_entry_bytes')::bigint + OR NEW.min_name IS DISTINCT FROM decode(proof->>'min_name','hex') + OR NEW.max_name IS DISTINCT FROM decode(proof->>'max_name','hex') + OR NEW.relative_path_bytes<>(proof->>'relative_path_bytes')::integer + OR NEW.relative_components<>(proof->>'relative_components')::integer + OR NEW.closure_nodes_upper<>(proof->>'closure_nodes_upper')::integer + OR NEW.closure_edges_upper<>(proof->>'closure_edges_upper')::integer + OR NEW.closure_bytes_upper<>(proof->>'closure_bytes_upper')::bigint + OR NEW.closure_entries_upper<>(proof->>'closure_entries_upper')::bigint THEN + RAISE EXCEPTION 'MTP2 canonical certificate was not derived from its durable bytes and exact child proofs'; + END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_metadata_certificate_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_metadata_page_certificate + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_certificate_guard(); + +CREATE FUNCTION mst2_metadata_verified_ref_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE ref jsonb; +BEGIN + IF TG_OP<>'INSERT' THEN RAISE EXCEPTION 'MTP2 canonical reference history is immutable'; END IF; + SELECT canonical_proof->'references'->NEW.reference_ordinal INTO ref FROM mst2_metadata_page_certificate + WHERE page_id=NEW.parent_page AND generation=NEW.parent_generation; + IF ref IS NULL OR NEW.reference_kind<>ref->>'kind' OR NEW.child_page<>decode(ref->>'child','hex') + OR NEW.child_generation<>(ref->>'generation')::bigint + OR NEW.child_certificate_digest<>decode(ref->>'certificate','hex') + OR NEW.name IS DISTINCT FROM decode(ref->>'name','hex') + OR NEW.label IS DISTINCT FROM (ref->>'label')::integer + OR NEW.advertised_count IS DISTINCT FROM (ref->>'count')::bigint THEN + RAISE EXCEPTION 'MTP2 canonical reference differs from its independently derived occurrence'; + END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_metadata_verified_ref_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_metadata_verified_ref + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_verified_ref_guard(); + +CREATE FUNCTION mst2_metadata_certificate_complete() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE p bytea; g bigint; proof jsonb; +BEGIN + IF TG_TABLE_NAME IN ('mst2_metadata_verified_ref','mst2_metadata_graph_edge') THEN p:=NEW.parent_page; g:=NEW.parent_generation; + ELSE p:=NEW.page_id; g:=NEW.generation; END IF; + SELECT canonical_proof INTO proof FROM mst2_metadata_page_certificate WHERE page_id=p AND generation=g; + IF proof IS NULL OR (SELECT count(*) FROM mst2_metadata_verified_ref WHERE parent_page=p AND parent_generation=g) + <>jsonb_array_length(proof->'references') + OR NOT EXISTS(SELECT 1 FROM mst2_metadata_graph_node n WHERE n.page_id=p AND n.generation=g + AND n.state='LIVE' AND n.certificate_digest=decode(proof->>'certificate','hex')) + OR EXISTS((SELECT child_page,child_generation FROM mst2_metadata_verified_ref WHERE parent_page=p AND parent_generation=g) + EXCEPT (SELECT child_page,child_generation FROM mst2_metadata_graph_edge WHERE parent_page=p AND parent_generation=g)) + OR EXISTS((SELECT child_page,child_generation FROM mst2_metadata_graph_edge WHERE parent_page=p AND parent_generation=g) + EXCEPT (SELECT child_page,child_generation FROM mst2_metadata_verified_ref WHERE parent_page=p AND parent_generation=g)) THEN + RAISE EXCEPTION 'MTP2 canonical certificate must commit with its complete exact graph and typed references'; + END IF; + RETURN NULL; +END $$; +CREATE CONSTRAINT TRIGGER mst2_metadata_certificate_complete AFTER INSERT ON mst2_metadata_page_certificate + DEFERRABLE INITIALLY DEFERRED FOR EACH ROW EXECUTE FUNCTION mst2_metadata_certificate_complete(); +CREATE CONSTRAINT TRIGGER mst2_metadata_verified_refs_complete AFTER INSERT ON mst2_metadata_verified_ref + DEFERRABLE INITIALLY DEFERRED FOR EACH ROW EXECUTE FUNCTION mst2_metadata_certificate_complete(); +CREATE CONSTRAINT TRIGGER mst2_metadata_graph_refs_complete AFTER INSERT ON mst2_metadata_graph_edge + DEFERRABLE INITIALLY DEFERRED FOR EACH ROW EXECUTE FUNCTION mst2_metadata_certificate_complete(); + +CREATE FUNCTION mst2_metadata_certify_page(pid text,p bytea,g bigint) RETURNS bytea LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE proof jsonb; existing bytea; +BEGIN + SELECT c.certificate_digest INTO existing FROM mst2_metadata_page_certificate c + JOIN mst2_metadata_current cur USING(page_id,generation) JOIN mst2_metadata_lifetime life USING(page_id,generation) + JOIN mst2_metadata_graph_node n USING(page_id,generation) JOIN mst2_metadata_prepare_page member USING(page_id,generation) + JOIN mst2_metadata_prepare q USING(prepare_id) + WHERE c.page_id=p AND c.generation=g AND member.prepare_id=pid AND q.state='PREPARING' + AND mst2_metadata_scope_matches(q.primary_scope) AND life.state='LIVE' AND life.graph_domain='qualified-v1' + AND n.state='LIVE' AND n.certificate_digest=c.certificate_digest AND n.bytes=member.expected_size + AND n.metadata_codec=q.metadata_codec + AND NOT EXISTS(SELECT 1 FROM mst2_metadata_gc_op op WHERE op.page_id=p AND op.generation=g); + IF FOUND THEN RETURN existing; END IF; + proof:=mst2_metadata_compute_certificate(p,g,pid); + INSERT INTO mst2_metadata_page_certificate(page_id,generation,certificate_digest,origin_prepare_id,namespace_uuid, + proof_revision,metadata_codec,byte_size,rank,map_entry_count,map_encoded_entry_bytes,min_name,max_name, + relative_path_bytes,relative_components,closure_nodes_upper,closure_edges_upper,closure_bytes_upper,closure_entries_upper,canonical_proof) + VALUES(p,g,decode(proof->>'certificate','hex'),pid,(proof->>'namespace')::uuid,1,(proof->>'codec')::smallint, + (proof->>'byte_size')::integer,(proof->>'rank')::integer,(proof->>'map_entry_count')::bigint, + (proof->>'map_encoded_entry_bytes')::bigint,decode(proof->>'min_name','hex'),decode(proof->>'max_name','hex'), + (proof->>'relative_path_bytes')::integer,(proof->>'relative_components')::integer, + (proof->>'closure_nodes_upper')::integer,(proof->>'closure_edges_upper')::integer, + (proof->>'closure_bytes_upper')::bigint,(proof->>'closure_entries_upper')::bigint,proof); + INSERT INTO mst2_metadata_verified_ref(parent_page,parent_generation,reference_ordinal,reference_kind, + name,label,advertised_count,child_page,child_generation,child_certificate_digest) + SELECT p,g,ordinality::integer-1,value->>'kind',decode(value->>'name','hex'),(value->>'label')::integer, + (value->>'count')::bigint,decode(value->>'child','hex'),(value->>'generation')::bigint,decode(value->>'certificate','hex') + FROM jsonb_array_elements(proof->'references') WITH ORDINALITY refs(value,ordinality); + INSERT INTO mst2_metadata_graph_node(page_id,generation,state,metadata_codec,bytes,incoming_refs,certificate_digest) + VALUES(p,g,'LIVE',(proof->>'codec')::smallint,(proof->>'byte_size')::integer,0,decode(proof->>'certificate','hex')); + INSERT INTO mst2_metadata_graph_edge(parent_page,parent_generation,child_page,child_generation) + SELECT DISTINCT p,g,child_page,child_generation FROM mst2_metadata_verified_ref WHERE parent_page=p AND parent_generation=g; + RETURN decode(proof->>'certificate','hex'); +END $$; + +CREATE FUNCTION mst2_metadata_certify_batch(pid text,members jsonb) RETURNS integer LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE member jsonb; certified integer:=0; +BEGIN + IF members IS NULL OR jsonb_typeof(members)<>'array' OR jsonb_array_length(members) NOT BETWEEN 1 AND 4096 THEN + RAISE EXCEPTION 'canonical certification batch exceeds its fixed member budget'; + END IF; + -- Array order is supplied by the independently validated cold DAG. Each + -- invocation still proves its durable bytes and already certified children. + FOR member IN SELECT value FROM jsonb_array_elements(members) WITH ORDINALITY AS input(value,ordinal) + ORDER BY ordinal LOOP + IF jsonb_typeof(member)<>'object' OR member->>'page' !~ '^[0-9a-f]{64}$' + OR member->>'generation' IS NULL THEN RAISE EXCEPTION 'canonical certification batch member is malformed'; END IF; + PERFORM mst2_metadata_certify_page(pid,decode(member->>'page','hex'),(member->>'generation')::bigint); + certified:=certified+1; + END LOOP; + RETURN certified; +END $$; diff --git a/src/jupiter/storage/qualified_metadata_family.rs b/src/jupiter/storage/qualified_metadata_family.rs new file mode 100644 index 00000000..13ffb440 --- /dev/null +++ b/src/jupiter/storage/qualified_metadata_family.rs @@ -0,0 +1,639 @@ +//! Captured physical Q provisioning and the sealed rooted production repository. + +use std::sync::OnceLock; + +use mst2_codec::metapage::{HEADER_LEN, PAGE_MAX_BYTES}; +use sea_orm::{ + ConnectionTrait, DatabaseConnection, DatabaseTransaction, DbBackend, IsolationLevel, Statement, + TransactionTrait, +}; +use sha2::{Digest, Sha256}; +use url::Url; + +#[cfg(test)] +use super::native_metadata_install::generations::{ + GenerationMetadataReceipt, GenerationPrepareIntent, + qualified::PostgresQualifiedMetadataRepository, +}; +use super::{init::postgres_connection, native_metadata_install::MetadataInstallError}; +#[cfg(test)] +use crate::ceres::snapshot::pages::PreparedNativeMetadataRetention; +use crate::{ + ceres::snapshot::{error::SnapshotError, retention_dag::MetadataPagePayload}, + common::errors::MegaError, + config::DbConfig, +}; + +const FAMILY: &str = "v3-rooted-qualified-1"; +const FAMILY_SQL: &str = include_str!("qualified_metadata_family.sql"); +const CANONICAL_SQL: &str = include_str!("qualified_metadata_canonical.sql"); +const CERTIFICATES_SQL: &str = include_str!("qualified_metadata_certificates.sql"); +const ANCHORS_SQL: &str = include_str!("qualified_metadata_anchors.sql"); +const SOURCE_READ_SQL: &str = include_str!("qualified_metadata_source_read.sql"); +const SOURCE_REVISION_SQL: &str = include_str!("qualified_source_revision.sql"); +const ROOTED_SQL: &str = include_str!("qualified_metadata_rooted.sql"); +const SERVING_SQL: &str = include_str!("qualified_metadata_serving.sql"); +const GC_SQL: &str = include_str!("qualified_metadata_gc.sql"); +const READER_LIFECYCLE_SQL: &str = include_str!("qualified_metadata_reader_lifecycle.sql"); + +#[path = "qualified_metadata_rooted.rs"] +mod rooted; +pub(crate) use rooted::{RootedLookupStatus, RootedQualifiedMetadataRepository}; +#[cfg(test)] +#[path = "qualified_metadata_reader_previous_fixture.rs"] +pub(crate) mod reader_previous_fixture; +#[cfg(test)] +pub(crate) use rooted::{ + RootedPrepareIntent, with_rooted_reader_barriers, with_rooted_source_fact_barriers, + with_rooted_source_temporary_shadow, +}; + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(crate) enum SnapshotMetadataFamily { + Generic, + Rooted, +} + +pub(crate) async fn select_snapshot_family( + connection: &DatabaseConnection, + identity: &str, + lease: bool, +) -> Result, SnapshotError> { + let result = async { + let txn = connection + .begin_with_config(Some(IsolationLevel::ReadCommitted), None) + .await?; + let (schema, _) = captured_core(&txn).await?; + let function = if lease { + "mst2_route_family_for_lease" + } else { + "mst2_route_family_for_snapshot" + }; + let rows = txn + .query_all_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + format!( + "SELECT namespace_uuid::text,graph_domain FROM {}.{function}($1,$2)", + identifier(&schema) + ), + [identity.into(), schema.into()], + )) + .await?; + let family = match rows.as_slice() { + [] => None, + [row] => Some(match row.try_get::("", "graph_domain")?.as_str() { + "generic-v1" => SnapshotMetadataFamily::Generic, + "qualified-v1" => SnapshotMetadataFamily::Rooted, + _ => { + return Err(rejected( + "snapshot route selected an unsupported physical family", + )); + } + }), + _ => { + return Err(rejected( + "snapshot route selected multiple physical families", + )); + } + }; + txn.commit().await?; + Ok::<_, MegaError>(family) + } + .await; + result.map_err(|error| { + if let MegaError::Db(db_error) = &error + && rooted::is_lock_unavailable(db_error) + { + return SnapshotError::new( + crate::ceres::snapshot::error::SnapshotErrorCode::TemporaryUnavailable, + "fixed source is being updated; retry the operation", + ); + } + SnapshotError::new( + crate::ceres::snapshot::error::SnapshotErrorCode::IntegrityError, + error.to_string(), + ) + }) +} + +#[derive(Debug, Clone, PartialEq, Eq)] +pub(crate) struct VerifiedQualifiedNamespace { + core_schema: String, + core_oid: i64, + schema: String, + schema_oid: i64, + namespace_uuid: String, + storage_uuid: String, + catalog_fingerprint: Vec, +} + +fn identifier(value: &str) -> String { + format!("\"{}\"", value.replace('"', "\"\"")) +} +fn literal(value: &str) -> String { + format!("'{}'", value.replace('\'', "''")) +} +pub(crate) fn implementation_fingerprint() -> Vec { + static FINGERPRINT: OnceLock<[u8; 32]> = OnceLock::new(); + FINGERPRINT + .get_or_init(|| { + let mut hash = Sha256::new(); + hash.update(b"mega.mst2.rooted-qualified-implementation.v1\0"); + // Length-prefix every source component so source selection and proof + // revisions cannot change while retaining an accepted physical stamp. + for (name, bytes) in [ + ("family", FAMILY.as_bytes()), + ("family-ddl", FAMILY_SQL.as_bytes()), + ("canonical-proof-revision-1", CANONICAL_SQL.as_bytes()), + ("indexed-source-read-revision-1", SOURCE_READ_SQL.as_bytes()), + ("captured-source-revision-1", SOURCE_REVISION_SQL.as_bytes()), + ("rooted-collector-revision-1", GC_SQL.as_bytes()), + ( + "bounded-reader-lifecycle-revision-1", + READER_LIFECYCLE_SQL.as_bytes(), + ), + ("typed-certificate-revision-1", CERTIFICATES_SQL.as_bytes()), + ( + "source-attestation-and-anchor-revision-1", + ANCHORS_SQL.as_bytes(), + ), + ("rooted-plan-and-bindings-revision-1", ROOTED_SQL.as_bytes()), + ( + "rooted-session-and-reader-revision-1", + SERVING_SQL.as_bytes(), + ), + ( + "rooted-physical-route-revision-1", + include_bytes!("../migration/m20261008_000200_rooted_routes.sql").as_slice(), + ), + ( + "authority-catalog-selector", + include_bytes!("qualified_family_catalog.sql").as_slice(), + ), + ( + "normalized-family-shape", + include_bytes!("qualified_family_shape.sql").as_slice(), + ), + ( + "initial-core-registration", + include_bytes!("../migration/m20261008_000200_rooted_qualified_family.sql") + .as_slice(), + ), + ] { + hash.update((name.len() as u64).to_le_bytes()); + hash.update(name.as_bytes()); + hash.update((bytes.len() as u64).to_le_bytes()); + hash.update(bytes); + } + hash.finalize().into() + }) + .to_vec() +} +pub(crate) fn render_family( + core_schema: &str, + core_oid: i64, + q_schema: &str, + q_oid: i64, + n_uuid: &str, + s_uuid: &str, +) -> String { + FAMILY_SQL + .replace("$CANONICAL_SQL$", CANONICAL_SQL) + .replace("$CERTIFICATES_SQL$", CERTIFICATES_SQL) + .replace("$ANCHORS_SQL$", ANCHORS_SQL) + .replace("$SOURCE_READ_SQL$", SOURCE_READ_SQL) + .replace("$ROOTED_SQL$", ROOTED_SQL) + .replace("$SERVING_SQL$", SERVING_SQL) + .replace("$GC_SQL$", GC_SQL) + .replace("$READER_LIFECYCLE_SQL$", READER_LIFECYCLE_SQL) + .replace("$CORE_SCHEMA$", &identifier(core_schema)) + .replace("$CORE_LITERAL$", &literal(core_schema)) + .replace("$Q_SCHEMA$", &identifier(q_schema)) + .replace("$Q_LITERAL$", &literal(q_schema)) + .replace("$CORE_OID$", &core_oid.to_string()) + .replace("$Q_OID$", &q_oid.to_string()) + .replace("$NAMESPACE_UUID$", n_uuid) + .replace("$STORAGE_UUID$", s_uuid) + .replace( + "$IMPLEMENTATION_SHA$", + &hex::encode(implementation_fingerprint()), + ) + .replace("$HEADER_LEN$", &HEADER_LEN.to_string()) + .replace("$PAGE_MAX_BYTES$", &PAGE_MAX_BYTES.to_string()) +} +fn rejected(message: &str) -> MegaError { + MegaError::Other(message.into()) +} + +async fn captured_core(connection: &C) -> Result<(String, i64), MegaError> { + let row = connection.query_one_raw(Statement::from_string(DbBackend::Postgres, + "SELECT current_schema() AS schema,n.oid::bigint AS oid FROM pg_catalog.pg_namespace n WHERE n.nspname=current_schema()")) + .await?.ok_or_else(|| rejected("qualified family core schema is missing"))?; + let schema: String = row.try_get("", "schema")?; + let oid: i64 = row.try_get("", "oid")?; + let valid = connection + .query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + format!( + "SELECT {}.mst2_route_scope_valid($1) AS valid", + identifier(&schema) + ), + [schema.clone().into()], + )) + .await? + .ok_or_else(|| rejected("qualified family core identity is missing"))? + .try_get::("", "valid")?; + if !valid { + return Err(rejected( + "qualified family requires its registered primary core schema", + )); + } + Ok((schema, oid)) +} + +async fn catalog( + connection: &C, + core_oid: i64, + q_oid: i64, +) -> Result, MegaError> { + catalog_with_exemption(connection, core_oid, q_oid, 0).await +} + +async fn catalog_with_exemption( + connection: &C, + core_oid: i64, + q_oid: i64, + exempt_q_oid: i64, +) -> Result, MegaError> { + // Inspect the actual catalogs directly. A replaced helper function cannot + // turn a bad physical structure into a fresh accepted fingerprint. + let sql = include_str!("qualified_family_catalog.sql") + .replace("$CORE_OID$", "$1::bigint::oid") + .replace("$Q_OID$", "$2::bigint::oid") + .replace("$EXEMPT_Q_OID$", "$3::bigint::oid"); + let row = connection + .query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + sql, + [core_oid.into(), q_oid.into(), exempt_q_oid.into()], + )) + .await? + .ok_or_else(|| rejected("qualified family catalog is missing"))?; + Ok(row.try_get("", "fingerprint")?) +} + +async fn registered( + connection: &C, + core: &(String, i64), +) -> Result, MegaError> { + let policy=connection.query_one_raw(Statement::from_string(DbBackend::Postgres,format!( + "SELECT implementation_fingerprint,expected_shape,authority_catalog FROM {}.mst2_qualified_family_policy WHERE singleton=1",identifier(&core.0)))) + .await?.ok_or_else(||rejected("qualified trusted family policy is missing"))?; + if policy.try_get::>("", "implementation_fingerprint")? != implementation_fingerprint() + { + return Err(rejected("qualified trusted core authority catalog changed")); + } + let rows=connection.query_all_raw(Statement::from_string(DbBackend::Postgres,format!( + "SELECT n.namespace_uuid::text,n.metadata_schema,n.metadata_schema_oid::bigint,n.metadata_storage_uuid, + n.implementation_fingerprint,n.catalog_fingerprint, + n.family_identity,n.admission_state,n.collector_state,n.storage_uuid AS core_storage_uuid, + n.core_schema,n.core_schema_oid::bigint, + (SELECT count(*) FROM {c}.mst2_metadata_namespace)::bigint AS namespace_count, + EXISTS(SELECT 1 FROM pg_catalog.pg_namespace p WHERE p.oid=n.metadata_schema_oid AND p.nspname=n.metadata_schema) AS schema_present, + {c}.mst2_route_scope_valid({core_literal}) AS core_valid + FROM {c}.mst2_metadata_namespace n WHERE n.graph_domain='qualified-v1'",c=identifier(&core.0),core_literal=literal(&core.0)))) + .await?; + if rows.is_empty() { + if policy.try_get::>("", "authority_catalog")? + != catalog(connection, core.1, 0).await? + { + return Err(rejected("qualified trusted core authority catalog changed")); + } + return Ok(None); + } + if rows.len() != 1 { + return Err(rejected( + "qualified family registry exceeds its one-Q hard limit", + )); + } + let row = &rows[0]; + let schema: String = row.try_get("", "metadata_schema")?; + let namespace_uuid: String = row.try_get("", "namespace_uuid")?; + let storage_uuid: String = row.try_get("", "metadata_storage_uuid")?; + let schema_oid: i64 = row.try_get("", "metadata_schema_oid")?; + let expected_schema = format!("mst2q_{}", namespace_uuid.replace('-', "")); + let namespace_identity = uuid::Uuid::parse_str(&namespace_uuid) + .map_err(|_| rejected("qualified namespace UUID is invalid"))?; + let storage_identity = uuid::Uuid::parse_str(&storage_uuid) + .map_err(|_| rejected("qualified storage UUID is invalid"))?; + if schema != expected_schema + || namespace_identity.get_version_num() != 4 + || namespace_identity.get_variant() != uuid::Variant::RFC4122 + || namespace_identity.to_string() != namespace_uuid + || storage_identity.get_version_num() != 4 + || storage_identity.get_variant() != uuid::Variant::RFC4122 + || storage_identity.to_string() != storage_uuid + || storage_uuid == row.try_get::("", "core_storage_uuid")? + || row.try_get::("", "core_schema")? != core.0 + || row.try_get::("", "core_schema_oid")? != core.1 + || row.try_get::("", "family_identity")? != FAMILY + || row.try_get::("", "admission_state")? != "ROOTED_Q_ADMITTED" + || row.try_get::("", "collector_state")? != "ENABLED" + || row.try_get::("", "namespace_count")? != 2 + || !row.try_get::("", "schema_present")? + || !row.try_get::("", "core_valid")? + || row.try_get::>("", "implementation_fingerprint")? != implementation_fingerprint() + { + return Err(rejected( + "qualified family registry and physical identity disagree", + )); + } + // A single registered, physically present Q identity is the only permitted + // source of core-side RI triggers omitted from the core authority stamp. + // The complete catalog and trusted shape below still bind all of them. + if policy.try_get::>("", "authority_catalog")? + != catalog_with_exemption(connection, core.1, 0, schema_oid).await? + { + return Err(rejected("qualified trusted core authority catalog changed")); + } + let fingerprint: Vec = row.try_get("", "catalog_fingerprint")?; + if fingerprint.len() != 32 || catalog(connection, core.1, schema_oid).await? != fingerprint { + return Err(rejected( + "qualified family actual catalog fingerprint changed", + )); + } + let shape: Vec = connection + .query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + format!( + "SELECT {}.mst2_route_family_shape($1::bigint::oid,$2::uuid,$3) AS fingerprint", + identifier(&core.0) + ), + [ + schema_oid.into(), + namespace_uuid.clone().into(), + storage_uuid.clone().into(), + ], + )) + .await? + .ok_or_else(|| rejected("qualified family actual shape is missing"))? + .try_get("", "fingerprint")?; + if shape != policy.try_get::>("", "expected_shape")? { + return Err(rejected( + "qualified namespace does not have the trusted complete physical family shape", + )); + } + let stamp=connection.query_one_raw(Statement::from_sql_and_values(DbBackend::Postgres,format!( + "SELECT EXISTS(SELECT 1 FROM {q}.mst2_metadata_family_identity i JOIN {q}.mst2_metadata_storage_scope s USING(singleton) + WHERE i.singleton=1 AND i.namespace_uuid=$1::uuid AND i.storage_uuid=$2 AND s.storage_uuid=$2 + AND i.core_schema_oid=$3::bigint::oid AND i.metadata_schema_oid=$4::bigint::oid + AND i.family_identity=$5 AND i.implementation_fingerprint=$6) AS valid",q=identifier(&schema)), + [namespace_uuid.clone().into(),storage_uuid.clone().into(),core.1.into(),schema_oid.into(),FAMILY.into(),implementation_fingerprint().into()])) + .await?.ok_or_else(||rejected("qualified family physical stamp is missing"))?; + if !stamp.try_get::("", "valid")? { + return Err(rejected("qualified family physical stamp changed")); + } + Ok(Some(VerifiedQualifiedNamespace { + core_schema: core.0.clone(), + core_oid: core.1, + schema, + schema_oid, + namespace_uuid, + storage_uuid, + catalog_fingerprint: fingerprint, + })) +} + +/// Production bootstrap calls this after core migrations, before returning a +/// writable app connection. Existing registrations are verified, never reset. +pub(crate) async fn provision_or_verify_rooted_qualified_family( + connection: &DatabaseConnection, +) -> Result { + let core = captured_core(connection).await?; + for attempt in 0..2 { + let txn = connection + .begin_with_config(Some(IsolationLevel::ReadCommitted), None) + .await?; + if let Some(namespace) = registered(&txn, &core).await? { + txn.execute_unprepared(&format!( + "SELECT {}.mst2_route_enter({})", + identifier(&core.0), + literal(&core.0) + )) + .await?; + let locked = registered(&txn, &core) + .await? + .ok_or_else(|| rejected("qualified registration disappeared"))?; + if locked != namespace { + return Err(rejected("qualified registration changed during bootstrap")); + } + txn.commit().await?; + return Ok(locked); + } + let candidate = uuid::Uuid::new_v4().to_string(); + let schema = format!("mst2q_{}", candidate.replace('-', "")); + let enter = txn + .execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + format!( + "SELECT {}.mst2_route_family_candidate_enter($1,$2::uuid,$3)", + identifier(&core.0) + ), + [ + core.0.clone().into(), + candidate.clone().into(), + schema.clone().into(), + ], + )) + .await; + if let Err(error) = enter { + txn.rollback().await?; + if attempt == 0 && error.to_string().contains("lost bootstrap serialization") { + continue; + } + return Err(error.into()); + } + txn.execute_unprepared(&format!("CREATE SCHEMA {}", identifier(&schema))) + .await?; + let row = txn + .query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT oid::bigint AS oid FROM pg_catalog.pg_namespace WHERE nspname=$1", + [schema.clone().into()], + )) + .await? + .ok_or_else(|| rejected("qualified provisioning schema was not created"))?; + let schema_oid: i64 = row.try_get("", "oid")?; + let storage_uuid = uuid::Uuid::new_v4().to_string(); + let sql = render_family( + &core.0, + core.1, + &schema, + schema_oid, + &candidate, + &storage_uuid, + ); + txn.execute_unprepared(&sql).await?; + let fingerprint = catalog(&txn, core.1, schema_oid).await?; + txn.execute_unprepared(&format!( + "SET LOCAL search_path={},pg_catalog,pg_temp", + identifier(&core.0) + )) + .await?; + txn.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres,format!( + "INSERT INTO {c}.mst2_metadata_namespace(singleton,namespace_uuid,core_schema,core_schema_oid, + database_name,database_oid,storage_uuid,server_address,server_port,mono_lock_key2,metadata_schema, + metadata_schema_oid,family_identity,graph_domain,admission_state,collector_state, + metadata_storage_uuid,implementation_fingerprint,catalog_fingerprint) + SELECT NULL,$1::uuid,g.core_schema,g.core_schema_oid,g.database_name,g.database_oid,g.storage_uuid, + g.server_address,g.server_port,g.mono_lock_key2,$2,$3::bigint::oid,$4,'qualified-v1','ROOTED_Q_ADMITTED','ENABLED',$5,$6,$7 + FROM {c}.mst2_metadata_namespace g WHERE singleton=1",c=identifier(&core.0)), + [candidate.into(),schema.into(),schema_oid.into(),FAMILY.into(),storage_uuid.into(),implementation_fingerprint().into(),fingerprint.into()])).await?; + let namespace = registered(&txn, &core) + .await? + .ok_or_else(|| rejected("qualified provisioning did not register its family"))?; + txn.commit().await?; + return Ok(namespace); + } + Err(rejected( + "qualified provisioning could not serialize bootstrap", + )) +} + +impl VerifiedQualifiedNamespace { + pub(crate) async fn enter(&self, txn: &DatabaseTransaction) -> Result<(), SnapshotError> { + let result: Result<(), MegaError> = async { + let actual = txn + .query_one_raw(Statement::from_string( + DbBackend::Postgres, + "SELECT current_schema() AS schema", + )) + .await? + .ok_or_else(|| rejected("qualified writer schema is missing"))? + .try_get::("", "schema")?; + if actual != self.schema { + return Err(rejected( + "qualified writer escaped its physical pool schema", + )); + } + txn.execute_unprepared(&format!( + "SELECT {}.mst2_route_enter({})", + identifier(&self.core_schema), + literal(&self.core_schema) + )) + .await?; + if registered(txn, &(self.core_schema.clone(), self.core_oid)) + .await? + .as_ref() + != Some(self) + { + return Err(rejected("qualified writer physical namespace changed")); + } + Ok(()) + } + .await; + result.map_err(|e| { + if let MegaError::Db(error) = &e + && rooted::is_lock_unavailable(error) + { + return SnapshotError::new( + crate::ceres::snapshot::error::SnapshotErrorCode::TemporaryUnavailable, + "qualified source is being updated; retry the operation", + ); + } + SnapshotError::new( + crate::ceres::snapshot::error::SnapshotErrorCode::SnapshotNotReady, + e.to_string(), + ) + }) + } +} + +fn pool_url(db_url: &str, namespace: &VerifiedQualifiedNamespace) -> Result { + let mut url = Url::parse(db_url) + .map_err(|_| rejected("qualified writer requires a valid PostgreSQL URL"))?; + let mut options = Vec::new(); + let mut others = Vec::new(); + for (key, value) in url.query_pairs() { + if key == "options" { + options.push(value.into_owned()); + } else { + others.push((key.into_owned(), value.into_owned())); + } + } + // Only server-generated lowercase schema names can reach this option. + options.push(format!( + "-csearch_path={},pg_catalog,pg_temp", + namespace.schema + )); + url.set_query(None); + { + let mut pairs = url.query_pairs_mut(); + for (key, value) in others { + pairs.append_pair(&key, &value); + } + pairs.append_pair("options", &options.join(" ")); + } + Ok(url.to_string()) +} + +/// The adapter has no Deref or unsealed connection accessor. It permits only +/// test-only cold preparation writes, without a production factory. +#[cfg(test)] +pub(crate) struct ShadowQualifiedMetadataWriter { + repository: PostgresQualifiedMetadataRepository, +} +#[cfg(test)] +impl ShadowQualifiedMetadataWriter { + pub(crate) async fn open( + core: &DatabaseConnection, + config: &DbConfig, + ) -> Result { + let captured = captured_core(core).await?; + let namespace = registered(core, &captured) + .await? + .ok_or_else(|| rejected("qualified shadow family is not provisioned at bootstrap"))?; + let mut q_config = config.clone(); + q_config.db_url = pool_url(&config.db_url, &namespace)?; + q_config.max_connection = q_config.max_connection.clamp(1, 2); + q_config.min_connection = 1; + let connection = postgres_connection(&q_config).await?; + let txn = connection + .begin_with_config(Some(IsolationLevel::ReadCommitted), None) + .await?; + namespace + .enter(&txn) + .await + .map_err(|e| rejected(&e.to_string()))?; + txn.commit().await?; + let repository = + PostgresQualifiedMetadataRepository::registered_shadow(connection, namespace) + .await + .map_err(|e| rejected(&e.to_string()))?; + Ok(Self { repository }) + } + pub(crate) async fn begin_intent( + &self, + operation_id: &str, + prepared: &PreparedNativeMetadataRetention, + ) -> Result { + self.repository.begin_intent(operation_id, prepared).await + } + pub(crate) async fn install_pages( + &self, + intent: &GenerationPrepareIntent, + payloads: &[MetadataPagePayload], + ) -> Result<(), MetadataInstallError> { + self.repository.install_pages(intent, payloads).await + } + pub(crate) async fn finalize( + &self, + intent: &GenerationPrepareIntent, + ) -> Result { + self.repository.finalize(intent).await + } +} + +#[cfg(test)] +#[path = "qualified_metadata_family_tests.rs"] +mod tests; diff --git a/src/jupiter/storage/qualified_metadata_family.sql b/src/jupiter/storage/qualified_metadata_family.sql new file mode 100644 index 00000000..fae04446 --- /dev/null +++ b/src/jupiter/storage/qualified_metadata_family.sql @@ -0,0 +1,491 @@ +-- Dedicated physical v3 Q family with exact generations and owned roots. +SET LOCAL search_path=$Q_SCHEMA$,pg_catalog,pg_temp; +CREATE TABLE mst2_metadata_storage_scope ( + singleton smallint PRIMARY KEY CHECK(singleton=1),storage_uuid text NOT NULL UNIQUE +); +INSERT INTO mst2_metadata_storage_scope VALUES(1,'$STORAGE_UUID$'); +CREATE TABLE mst2_metadata_family_identity ( + singleton smallint PRIMARY KEY CHECK(singleton=1),namespace_uuid uuid NOT NULL UNIQUE, + storage_uuid text NOT NULL UNIQUE,core_schema_oid oid NOT NULL,metadata_schema_oid oid NOT NULL, + family_identity text NOT NULL CHECK(family_identity='v3-rooted-qualified-1'), + implementation_fingerprint bytea NOT NULL CHECK(octet_length(implementation_fingerprint)=32) +); +INSERT INTO mst2_metadata_family_identity VALUES(1,'$NAMESPACE_UUID$'::uuid,'$STORAGE_UUID$', + $CORE_OID$,$Q_OID$,'v3-rooted-qualified-1',decode('$IMPLEMENTATION_SHA$','hex')); +CREATE TABLE mst2_metadata_lifetime ( + page_id bytea NOT NULL CHECK(octet_length(page_id)=32), + node_id text NOT NULL CHECK(node_id='page:sha256:'||encode(page_id,'hex')), + generation bigint NOT NULL CHECK(generation>0), + state text NOT NULL CHECK(state IN ('RESERVED','LIVE','DELETING','REMOVED')), + metadata_codec smallint NOT NULL CHECK(metadata_codec=1), + expected_size integer NOT NULL CHECK(expected_size BETWEEN $HEADER_LEN$ AND $PAGE_MAX_BYTES$), + graph_domain text NOT NULL CHECK(graph_domain='qualified-v1'),PRIMARY KEY(page_id,generation) +); +CREATE INDEX idx_mst2_metadata_lifetime_state_page ON mst2_metadata_lifetime(state,page_id); +CREATE INDEX idx_mst2_metadata_lifetime_node_generation ON mst2_metadata_lifetime(node_id,generation); +CREATE TABLE mst2_metadata_current ( + page_id bytea PRIMARY KEY CHECK(octet_length(page_id)=32),generation bigint NOT NULL CHECK(generation>0), + FOREIGN KEY(page_id,generation) REFERENCES mst2_metadata_lifetime(page_id,generation) +); +CREATE TABLE mst2_metadata_payload ( + page_id bytea PRIMARY KEY CHECK(octet_length(page_id)=32),generation bigint NOT NULL CHECK(generation>0), + metadata_codec smallint NOT NULL CHECK(metadata_codec=1), + byte_size integer NOT NULL CHECK(byte_size BETWEEN $HEADER_LEN$ AND $PAGE_MAX_BYTES$), + payload bytea NOT NULL CHECK(octet_length(payload)=byte_size),created_at timestamptz NOT NULL DEFAULT now(), + FOREIGN KEY(page_id,generation) REFERENCES mst2_metadata_lifetime(page_id,generation) +); +CREATE TABLE mst2_metadata_prepare ( + prepare_id text PRIMARY KEY CHECK(prepare_id ~ '^[0-9a-f]{8}-[0-9a-f]{4}-4[0-9a-f]{3}-[89ab][0-9a-f]{3}-[0-9a-f]{12}$'), + operation_id text NOT NULL UNIQUE CHECK(octet_length(operation_id) BETWEEN 1 AND 255), + manifest_digest bytea NOT NULL CHECK(octet_length(manifest_digest)=32), + canonical_plan bytea NOT NULL CHECK(octet_length(canonical_plan)<=2097152), + source_domain text NOT NULL CHECK(source_domain='native-git'),tagged_root_tree_oid text NOT NULL, + plan_kind text NOT NULL DEFAULT 'COLD' CHECK(plan_kind IN ('COLD','ROOTED')), + bindings_revision bigint NOT NULL DEFAULT 0 CHECK(bindings_revision>=0), + scope text NOT NULL CHECK(octet_length(scope)<=4096),schema_version smallint NOT NULL, + metadata_codec smallint NOT NULL CHECK(metadata_codec=1),materialization_policy smallint NOT NULL, + fs_semantics smallint NOT NULL,access_projection smallint NOT NULL,verification_revision integer NOT NULL, + projection_revision smallint NOT NULL,metadata_root bytea NOT NULL CHECK(octet_length(metadata_root)=32), + node_count integer NOT NULL CHECK(node_count BETWEEN 0 AND 4096), + edge_count integer NOT NULL CHECK(edge_count BETWEEN 0 AND 16384), + total_bytes bigint NOT NULL CHECK(total_bytes BETWEEN 0 AND 67108864), + state text NOT NULL CHECK(state IN ('PREPARING','COMMITTED','ABORTED')), + created_at timestamptz NOT NULL DEFAULT now(),committed_at timestamptz,aborted_at timestamptz, + coverage_retired_at timestamptz, + canonical_bindings bytea NOT NULL CHECK(octet_length(canonical_bindings) BETWEEN 12 AND 196620), + bindings_digest bytea NOT NULL CHECK(octet_length(bindings_digest)=32), + primary_scope bytea NOT NULL CHECK(octet_length(primary_scope) BETWEEN 1 AND 16384), + storage_seal bytea NOT NULL CHECK(octet_length(storage_seal)=32), + graph_domain text NOT NULL CHECK(graph_domain='qualified-v1'),UNIQUE(prepare_id,storage_seal), + CHECK((state='COMMITTED')=(committed_at IS NOT NULL) AND (state='ABORTED')=(aborted_at IS NOT NULL) + AND (coverage_retired_at IS NULL OR state='COMMITTED')) +); +CREATE TABLE mst2_metadata_prepare_page ( + prepare_id text NOT NULL REFERENCES mst2_metadata_prepare(prepare_id), + page_id bytea NOT NULL CHECK(octet_length(page_id)=32),generation bigint NOT NULL CHECK(generation>0), + expected_size integer NOT NULL CHECK(expected_size BETWEEN $HEADER_LEN$ AND $PAGE_MAX_BYTES$), + PRIMARY KEY(prepare_id,page_id),UNIQUE(prepare_id,page_id,generation), + FOREIGN KEY(page_id,generation) REFERENCES mst2_metadata_lifetime(page_id,generation) +); +CREATE INDEX idx_mst2_metadata_prepare_page_lifetime ON mst2_metadata_prepare_page(page_id,generation,prepare_id); +CREATE TABLE mst2_metadata_graph_node ( + page_id bytea NOT NULL CHECK (octet_length(page_id)=32), + generation bigint NOT NULL CHECK (generation>0), + state text NOT NULL CHECK (state IN ('LIVE','DELETING')), + metadata_codec smallint NOT NULL CHECK (metadata_codec=1), + bytes bigint NOT NULL CHECK (bytes BETWEEN $HEADER_LEN$ AND $PAGE_MAX_BYTES$), + incoming_refs bigint NOT NULL DEFAULT 0 CHECK (incoming_refs>=0), + PRIMARY KEY(page_id,generation), + FOREIGN KEY(page_id,generation) REFERENCES mst2_metadata_lifetime(page_id,generation) +); +CREATE INDEX idx_mst2_metadata_graph_node_gc ON mst2_metadata_graph_node(state,incoming_refs,page_id,generation); +CREATE TABLE mst2_metadata_graph_edge ( + parent_page bytea NOT NULL, + parent_generation bigint NOT NULL, + child_page bytea NOT NULL, + child_generation bigint NOT NULL, + PRIMARY KEY(parent_page,parent_generation,child_page,child_generation), + FOREIGN KEY(parent_page,parent_generation) REFERENCES mst2_metadata_graph_node(page_id,generation), + FOREIGN KEY(child_page,child_generation) REFERENCES mst2_metadata_graph_node(page_id,generation), + CHECK (parent_page<>child_page OR parent_generation<>child_generation) +); +CREATE INDEX idx_mst2_metadata_graph_edge_child ON mst2_metadata_graph_edge(child_page,child_generation,parent_page,parent_generation); +CREATE TABLE mst2_metadata_graph_root ( + prepare_id text NOT NULL, + storage_seal bytea NOT NULL CHECK (octet_length(storage_seal)=32), + page_id bytea NOT NULL, + generation bigint NOT NULL, + PRIMARY KEY(prepare_id,page_id,generation), + FOREIGN KEY(prepare_id,storage_seal) REFERENCES mst2_metadata_prepare(prepare_id,storage_seal), + FOREIGN KEY(prepare_id,page_id,generation) REFERENCES mst2_metadata_prepare_page(prepare_id,page_id,generation), + FOREIGN KEY(page_id,generation) REFERENCES mst2_metadata_graph_node(page_id,generation) +); +CREATE INDEX idx_mst2_metadata_graph_root_page ON mst2_metadata_graph_root(page_id,generation,prepare_id); +CREATE TABLE mst2_metadata_gc_op ( + operation_id uuid PRIMARY KEY, + page_id bytea NOT NULL CHECK (octet_length(page_id)=32), + generation bigint NOT NULL CHECK (generation>0), + primary_scope bytea NOT NULL CHECK (octet_length(primary_scope) BETWEEN 1 AND 16384), + graph_domain text NOT NULL CHECK (graph_domain='qualified-v1'), + metadata_codec smallint NOT NULL CHECK (metadata_codec=1), + expected_size integer NOT NULL CHECK (expected_size BETWEEN $HEADER_LEN$ AND $PAGE_MAX_BYTES$), + graph_present boolean NOT NULL, + had_payload boolean NOT NULL, + payload_delete_xid bigint, + state text NOT NULL CHECK (state IN ('PENDING','APPLIED')), + created_at timestamptz NOT NULL DEFAULT clock_timestamp(), + completed_at timestamptz, + UNIQUE(page_id,generation), + FOREIGN KEY(page_id,generation) REFERENCES mst2_metadata_lifetime(page_id,generation), + CHECK ((state='APPLIED')=(completed_at IS NOT NULL)) +); +CREATE INDEX idx_mst2_metadata_gc_op_pending ON mst2_metadata_gc_op(created_at,operation_id) WHERE state='PENDING'; + +-- Incarnations are historical; SID alone is never a unique physical binding. +CREATE TABLE mst2_qualified_session_incarnation ( + snapshot_id text NOT NULL,session_incarnation uuid NOT NULL,namespace_uuid uuid NOT NULL, + prepare_id text NOT NULL,storage_seal bytea NOT NULL CHECK(octet_length(storage_seal)=32), + metadata_root bytea NOT NULL CHECK(octet_length(metadata_root)=32),root_generation bigint NOT NULL CHECK(root_generation>0), + source_profile bytea NOT NULL,instance_id text NOT NULL,commit_oid text NOT NULL,root_tree_oid text NOT NULL, + authorization_epoch bigint NOT NULL,publication_sequence bigint NOT NULL,writer_epoch bigint NOT NULL, + certificate_receipt_id bigint NOT NULL,PRIMARY KEY(snapshot_id,session_incarnation), + UNIQUE(snapshot_id,session_incarnation,prepare_id,storage_seal,metadata_root,root_generation), + FOREIGN KEY(snapshot_id,namespace_uuid) REFERENCES $CORE_SCHEMA$.mst2_snapshot_storage_route(snapshot_id,namespace_uuid), + FOREIGN KEY(prepare_id,storage_seal) REFERENCES mst2_metadata_prepare(prepare_id,storage_seal) +); +CREATE TABLE mst2_qualified_lease_binding ( + lease_id text PRIMARY KEY,snapshot_id text NOT NULL,session_incarnation uuid NOT NULL, + prepare_id text NOT NULL,storage_seal bytea NOT NULL,metadata_root bytea NOT NULL,root_generation bigint NOT NULL, + FOREIGN KEY(snapshot_id,session_incarnation,prepare_id,storage_seal,metadata_root,root_generation) + REFERENCES mst2_qualified_session_incarnation(snapshot_id,session_incarnation,prepare_id,storage_seal,metadata_root,root_generation) +); + +$CANONICAL_SQL$ +$CERTIFICATES_SQL$ +$ANCHORS_SQL$ +$SOURCE_READ_SQL$ +$ROOTED_SQL$ + +CREATE FUNCTION mst2_metadata_closed() RETURNS trigger LANGUAGE plpgsql +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN RAISE EXCEPTION 'qualified session serving and collector are closed'; END $$; +CREATE FUNCTION mst2_metadata_gc_apply(id uuid) RETURNS void LANGUAGE plpgsql +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN RAISE EXCEPTION 'qualified collector is closed'; END $$; +CREATE FUNCTION mst2_metadata_gc_finish(id uuid) RETURNS void LANGUAGE plpgsql +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN RAISE EXCEPTION 'qualified collector is closed'; END $$; +CREATE FUNCTION mst2_metadata_immutable() RETURNS trigger LANGUAGE plpgsql +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN RAISE EXCEPTION 'qualified family identity and history are immutable'; END $$; + +-- Observe the caller before entering fixed trusted functions. Core fallback +-- never enters the pool search path; all cross-family authority is explicit. +CREATE FUNCTION mst2_metadata_dml_barrier() RETURNS trigger LANGUAGE plpgsql VOLATILE AS $$ +BEGIN + IF pg_catalog.current_schema() IS DISTINCT FROM $Q_LITERAL$ OR TG_TABLE_SCHEMA IS DISTINCT FROM $Q_LITERAL$ + OR NOT EXISTS(SELECT 1 FROM pg_catalog.pg_class WHERE oid=TG_RELID AND relnamespace='$Q_OID$'::oid) + OR NOT EXISTS(SELECT 1 FROM pg_catalog.pg_namespace WHERE oid='$Q_OID$'::oid AND nspname=$Q_LITERAL$) + OR pg_catalog.pg_is_in_recovery() OR pg_catalog.current_setting('transaction_isolation')<>'read committed' THEN + RAISE EXCEPTION 'qualified mutation requires its captured primary family and READ COMMITTED'; + END IF; + PERFORM $CORE_SCHEMA$.mst2_route_enter($CORE_LITERAL$); + IF NOT EXISTS(SELECT 1 FROM $CORE_SCHEMA$.mst2_metadata_namespace n + WHERE n.namespace_uuid='$NAMESPACE_UUID$'::uuid AND n.metadata_schema=$Q_LITERAL$ + AND n.metadata_schema_oid='$Q_OID$'::oid AND n.core_schema_oid=$CORE_OID$ + AND n.metadata_storage_uuid='$STORAGE_UUID$' AND n.admission_state='ROOTED_Q_ADMITTED' + AND n.collector_state='ENABLED' AND n.implementation_fingerprint=pg_catalog.decode('$IMPLEMENTATION_SHA$','hex') + AND n.catalog_fingerprint=$CORE_SCHEMA$.mst2_route_family_catalog($CORE_OID$,'$Q_OID$'::oid)) THEN + RAISE EXCEPTION 'qualified physical family catalog fingerprint is unavailable'; + END IF; + RETURN NULL; +END $$; + + +CREATE FUNCTION mst2_metadata_scope_matches(s bytea) RETURNS boolean LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE actual jsonb; +BEGIN + IF pg_is_in_recovery() OR current_setting('transaction_isolation')<>'read committed' THEN RETURN false; END IF; + SELECT jsonb_build_array(x.storage_uuid,current_database(),d.oid::bigint,$Q_LITERAL$,'$Q_OID$'::bigint, + inet_server_addr()::text,inet_server_port()) INTO actual FROM mst2_metadata_storage_scope x + JOIN pg_database d ON d.datname=current_database() WHERE x.singleton=1; + RETURN actual IS NOT NULL AND convert_from(s,'UTF8')::jsonb=actual; +EXCEPTION WHEN OTHERS THEN RETURN false; +END $$; +CREATE FUNCTION mst2_metadata_has_generic_overlap(p bytea) RETURNS boolean LANGUAGE sql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ SELECT false $$; + +CREATE FUNCTION mst2_metadata_lifetime_guard() RETURNS trigger LANGUAGE plpgsql +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + IF TG_OP='DELETE' THEN RAISE EXCEPTION 'metadata lifetime watermark cannot be deleted'; END IF; + IF TG_OP='INSERT' THEN + IF NEW.generation<>1 OR NEW.state<>'RESERVED' OR EXISTS(SELECT 1 FROM mst2_metadata_current WHERE page_id=NEW.page_id) THEN + RAISE EXCEPTION 'qualified initial lifetime cannot adopt old history; collector is closed'; + END IF; + RETURN NEW; + END IF; + IF (to_jsonb(NEW)-'state') IS DISTINCT FROM (to_jsonb(OLD)-'state') THEN + RAISE EXCEPTION 'metadata lifetime identity is immutable'; + END IF; + IF NEW.state=OLD.state THEN RETURN NEW; END IF; + IF OLD.state='RESERVED' AND NEW.state='LIVE' THEN + IF NOT EXISTS(SELECT 1 FROM mst2_metadata_graph_node n JOIN mst2_metadata_payload b USING(page_id,generation) + WHERE n.page_id=OLD.page_id AND n.generation=OLD.generation AND n.state='LIVE' + AND n.metadata_codec=OLD.metadata_codec AND n.bytes=OLD.expected_size + AND b.metadata_codec=OLD.metadata_codec AND b.byte_size=OLD.expected_size) + OR NOT (EXISTS(SELECT 1 FROM mst2_metadata_graph_root r WHERE r.page_id=OLD.page_id AND r.generation=OLD.generation) + OR EXISTS(SELECT 1 FROM mst2_metadata_prepare_page member JOIN mst2_metadata_prepare q USING(prepare_id) + JOIN mst2_metadata_root_anchor anchor ON anchor.prepare_id=q.prepare_id AND anchor.anchor_kind='PREPARE' + AND anchor.owner_key=q.prepare_id AND anchor.root_page=q.metadata_root + WHERE member.page_id=OLD.page_id AND member.generation=OLD.generation AND q.plan_kind='ROOTED' + AND q.state='COMMITTED' AND q.coverage_retired_at IS NULL)) THEN + RAISE EXCEPTION 'qualified LIVE transition needs its graph payload and prepare root'; + END IF; + RETURN NEW; + END IF; + RAISE EXCEPTION 'qualified lifetime collection is closed'; +END $$; +CREATE TRIGGER mst2_metadata_lifetime_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_metadata_lifetime + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_lifetime_guard(); +CREATE FUNCTION mst2_metadata_current_guard() RETURNS trigger LANGUAGE plpgsql +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + IF TG_OP<>'INSERT' THEN RAISE EXCEPTION 'qualified current collection is closed'; END IF; + IF NEW.generation<>1 OR EXISTS(SELECT 1 FROM mst2_metadata_lifetime WHERE page_id=NEW.page_id AND generation<>1) + OR EXISTS(SELECT 1 FROM mst2_metadata_payload WHERE page_id=NEW.page_id) + OR EXISTS(SELECT 1 FROM mst2_metadata_graph_node WHERE page_id=NEW.page_id) THEN + RAISE EXCEPTION 'initial qualified current cannot reset or adopt old history'; + END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_metadata_current_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_metadata_current + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_current_guard(); +CREATE FUNCTION mst2_metadata_current_protected() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + IF EXISTS(SELECT 1 FROM mst2_metadata_lifetime WHERE page_id=NEW.page_id AND generation=NEW.generation AND graph_domain='qualified-v1') + AND NOT EXISTS(SELECT 1 FROM mst2_metadata_prepare_page m JOIN mst2_metadata_prepare q USING(prepare_id) + JOIN mst2_metadata_current c ON c.page_id=m.page_id AND c.generation=m.generation + WHERE m.page_id=NEW.page_id AND m.generation=NEW.generation AND q.graph_domain='qualified-v1' + AND q.storage_seal IS NOT NULL AND (q.state='PREPARING' OR (q.state='COMMITTED' AND + (q.coverage_retired_at IS NULL OR mst2_metadata_session_covers_prepare(q.prepare_id))))) THEN + RAISE EXCEPTION 'qualified current change must commit with fresh prepare protection'; + END IF; + RETURN NULL; +END $$; +CREATE CONSTRAINT TRIGGER mst2_metadata_current_protected AFTER INSERT OR UPDATE ON mst2_metadata_current + DEFERRABLE INITIALLY DEFERRED FOR EACH ROW EXECUTE FUNCTION mst2_metadata_current_protected(); + +CREATE FUNCTION mst2_metadata_prepare_guard() RETURNS trigger LANGUAGE plpgsql +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + IF TG_OP='DELETE' THEN RAISE EXCEPTION 'qualified preparation history is immutable'; END IF; + IF TG_OP='INSERT' THEN + IF NEW.state<>'PREPARING' OR NEW.committed_at IS NOT NULL OR NEW.aborted_at IS NOT NULL + OR NEW.coverage_retired_at IS NOT NULL OR NEW.bindings_revision<>0 OR NOT mst2_metadata_scope_matches(NEW.primary_scope) + OR NEW.manifest_digest IS DISTINCT FROM sha256(NEW.canonical_plan) + OR NEW.bindings_digest IS DISTINCT FROM sha256(NEW.canonical_bindings) THEN + RAISE EXCEPTION 'qualified prepare needs its actual fixed primary plan and bindings'; + END IF; + IF NEW.plan_kind='ROOTED' THEN PERFORM mst2_metadata_rooted_manifest(NEW); + ELSIF NEW.node_count=0 OR octet_length(NEW.canonical_bindings)<60 THEN + RAISE EXCEPTION 'cold preparation requires its nonempty full closure'; + END IF; + RETURN NEW; + END IF; + IF (to_jsonb(NEW)-ARRAY['state','committed_at','aborted_at','coverage_retired_at','bindings_revision']) IS DISTINCT FROM + (to_jsonb(OLD)-ARRAY['state','committed_at','aborted_at','coverage_retired_at','bindings_revision']) THEN + RAISE EXCEPTION 'qualified preparation complete identity is immutable'; + END IF; + IF NEW.bindings_revision<>OLD.bindings_revision AND (OLD.plan_kind<>'ROOTED' OR OLD.state<>'PREPARING' + OR NEW.state<>'PREPARING' OR NEW.bindings_revision<>OLD.bindings_revision+1) THEN + RAISE EXCEPTION 'rooted membership revision must advance its exact active preparation'; + END IF; + IF (OLD.state IN ('COMMITTED','ABORTED') AND NEW.state IS DISTINCT FROM OLD.state) + OR (OLD.committed_at IS NOT NULL AND NEW.committed_at IS DISTINCT FROM OLD.committed_at) + OR (OLD.aborted_at IS NOT NULL AND NEW.aborted_at IS DISTINCT FROM OLD.aborted_at) + OR (OLD.coverage_retired_at IS NOT NULL AND NEW.coverage_retired_at IS DISTINCT FROM OLD.coverage_retired_at) THEN + RAISE EXCEPTION 'qualified terminal receipt cannot be revived or rewritten'; + END IF; + IF OLD.state='PREPARING' AND NEW.state='COMMITTED' THEN + IF NEW.plan_kind='ROOTED' THEN + PERFORM mst2_metadata_rooted_finalize_proof(NEW.prepare_id); + ELSE + IF (SELECT count(*) FROM mst2_metadata_prepare_page WHERE prepare_id=NEW.prepare_id)<>NEW.node_count + OR (SELECT coalesce(sum(expected_size),0) FROM mst2_metadata_prepare_page WHERE prepare_id=NEW.prepare_id)<>NEW.total_bytes + OR NOT EXISTS(SELECT 1 FROM mst2_metadata_prepare_page WHERE prepare_id=NEW.prepare_id AND page_id=NEW.metadata_root) + OR EXISTS(SELECT 1 FROM mst2_metadata_prepare_page m + LEFT JOIN mst2_metadata_current c USING(page_id,generation) + LEFT JOIN mst2_metadata_lifetime l USING(page_id,generation) + LEFT JOIN mst2_metadata_payload b USING(page_id,generation) + LEFT JOIN mst2_metadata_graph_node n USING(page_id,generation) + LEFT JOIN mst2_metadata_graph_root r ON r.prepare_id=m.prepare_id AND r.page_id=m.page_id AND r.generation=m.generation + WHERE m.prepare_id=NEW.prepare_id AND (c.page_id IS NULL OR l.page_id IS NULL OR b.page_id IS NULL OR n.page_id IS NULL + OR l.state NOT IN ('RESERVED','LIVE') OR n.state<>'LIVE' OR l.metadata_codec<>NEW.metadata_codec + OR b.metadata_codec<>NEW.metadata_codec OR n.metadata_codec<>NEW.metadata_codec + OR l.expected_size<>m.expected_size OR b.byte_size<>m.expected_size OR n.bytes<>m.expected_size + OR r.storage_seal IS DISTINCT FROM NEW.storage_seal + OR n.incoming_refs<>(SELECT count(*) FROM mst2_metadata_graph_edge WHERE child_page=m.page_id AND child_generation=m.generation))) + OR (SELECT count(*) FROM mst2_metadata_prepare_page m JOIN mst2_metadata_graph_edge e + ON e.parent_page=m.page_id AND e.parent_generation=m.generation WHERE m.prepare_id=NEW.prepare_id)<>NEW.edge_count THEN + RAISE EXCEPTION 'qualified COMMITTED transition lacks its complete exact graph payload and root coverage'; + END IF; + END IF; + END IF; + IF NEW.state='ABORTED' THEN + IF EXISTS(SELECT 1 FROM mst2_metadata_graph_root WHERE prepare_id=NEW.prepare_id) + OR EXISTS(SELECT 1 FROM mst2_qualified_session_incarnation WHERE prepare_id=NEW.prepare_id) THEN + RAISE EXCEPTION 'qualified terminal transition still has exact coverage'; + END IF; + ELSIF NEW.coverage_retired_at IS NOT NULL AND OLD.coverage_retired_at IS NULL THEN + IF NEW.plan_kind<>'ROOTED' THEN + IF EXISTS(SELECT 1 FROM mst2_metadata_graph_root WHERE prepare_id=NEW.prepare_id) + OR EXISTS(SELECT 1 FROM mst2_qualified_session_incarnation WHERE prepare_id=NEW.prepare_id) THEN + RAISE EXCEPTION 'cold terminal transition still has exact coverage'; + END IF; + ELSIF NEW.state<>'COMMITTED' OR EXISTS(SELECT 1 FROM mst2_metadata_graph_root WHERE prepare_id=NEW.prepare_id) + OR NOT (mst2_metadata_session_covers_prepare(NEW.prepare_id) + OR mst2_metadata_orphan_prepare_eligible(NEW.prepare_id)) THEN + RAISE EXCEPTION 'rooted coverage retirement needs its definitive independent ready session root'; + END IF; + END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_metadata_prepare_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_metadata_prepare + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_prepare_guard(); +CREATE FUNCTION mst2_metadata_generation_mapping_guard() RETURNS trigger LANGUAGE plpgsql +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + IF TG_OP<>'INSERT' THEN RAISE EXCEPTION 'metadata generation mappings are immutable'; END IF; + IF NOT EXISTS(SELECT 1 FROM mst2_metadata_prepare q JOIN mst2_metadata_current c ON c.page_id=NEW.page_id AND c.generation=NEW.generation + JOIN mst2_metadata_lifetime l USING(page_id,generation) WHERE q.prepare_id=NEW.prepare_id AND q.state='PREPARING' + AND q.graph_domain='qualified-v1' AND q.storage_seal IS NOT NULL AND mst2_metadata_scope_matches(q.primary_scope) + AND l.state IN ('RESERVED','LIVE') AND l.expected_size=NEW.expected_size AND l.metadata_codec=q.metadata_codec) THEN + RAISE EXCEPTION 'qualified mapping cannot cross domain/current/state fence'; + END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_metadata_generation_mapping_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_metadata_prepare_page + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_generation_mapping_guard(); +CREATE FUNCTION mst2_metadata_payload_fenced() RETURNS trigger LANGUAGE plpgsql +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + IF TG_OP<>'INSERT' THEN RAISE EXCEPTION 'qualified payload mutation and collector are closed'; END IF; + IF NOT EXISTS(SELECT 1 FROM mst2_metadata_current c JOIN mst2_metadata_lifetime l USING(page_id,generation) + JOIN mst2_metadata_prepare_page m USING(page_id,generation) JOIN mst2_metadata_prepare q USING(prepare_id) + WHERE c.page_id=NEW.page_id AND c.generation=NEW.generation AND l.state IN ('RESERVED','LIVE') + AND l.metadata_codec=NEW.metadata_codec AND l.expected_size=NEW.byte_size AND q.state='PREPARING' + AND q.graph_domain='qualified-v1' AND mst2_metadata_scope_matches(q.primary_scope)) THEN + RAISE EXCEPTION 'qualified payload INSERT crossed its exact active generation'; + END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_metadata_payload_fenced BEFORE INSERT OR UPDATE OR DELETE ON mst2_metadata_payload + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_payload_fenced(); +CREATE FUNCTION mst2_metadata_graph_node_guard() RETURNS trigger LANGUAGE plpgsql +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE l mst2_metadata_lifetime%ROWTYPE; +BEGIN + IF TG_OP='DELETE' THEN RAISE EXCEPTION 'qualified graph collection is closed'; END IF; + SELECT l0.* INTO l FROM mst2_metadata_current c JOIN mst2_metadata_lifetime l0 USING(page_id,generation) + WHERE c.page_id=NEW.page_id AND c.generation=NEW.generation; + IF NOT FOUND OR l.graph_domain<>'qualified-v1' OR l.metadata_codec<>NEW.metadata_codec OR l.expected_size<>NEW.bytes + OR NEW.state<>'LIVE' OR l.state NOT IN ('RESERVED','LIVE') + OR NEW.incoming_refs<>(SELECT count(*) FROM mst2_metadata_graph_edge WHERE child_page=NEW.page_id AND child_generation=NEW.generation) THEN + RAISE EXCEPTION 'qualified node does not match exact lifetime and actual counters'; + END IF; + IF TG_OP='INSERT' AND NOT EXISTS(SELECT 1 FROM mst2_metadata_prepare_page m JOIN mst2_metadata_prepare q USING(prepare_id) + WHERE m.page_id=NEW.page_id AND m.generation=NEW.generation AND q.state='PREPARING') THEN + RAISE EXCEPTION 'qualified node creation needs active fixed preparation'; + END IF; + IF NOT EXISTS(SELECT 1 FROM mst2_metadata_page_certificate c WHERE c.page_id=NEW.page_id AND c.generation=NEW.generation + AND c.certificate_digest=NEW.certificate_digest AND c.metadata_codec=NEW.metadata_codec AND c.byte_size=NEW.bytes) THEN + RAISE EXCEPTION 'qualified graph node lacks its exact independently canonical certificate'; + END IF; + IF TG_OP='UPDATE' AND (to_jsonb(NEW)-'incoming_refs') IS DISTINCT FROM (to_jsonb(OLD)-'incoming_refs') THEN + RAISE EXCEPTION 'qualified node identity is immutable'; + END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_metadata_graph_node_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_metadata_graph_node + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_graph_node_guard(); +CREATE FUNCTION mst2_metadata_graph_root_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + IF TG_OP='UPDATE' THEN RAISE EXCEPTION 'qualified roots cannot be retargeted'; END IF; + IF TG_OP='DELETE' THEN + IF EXISTS(SELECT 1 FROM mst2_qualified_session_incarnation WHERE prepare_id=OLD.prepare_id) THEN + RAISE EXCEPTION 'qualified root still has historical incarnation coverage'; + END IF; + RETURN OLD; + END IF; + IF NOT EXISTS(SELECT 1 FROM mst2_metadata_prepare q JOIN mst2_metadata_current c ON c.page_id=NEW.page_id AND c.generation=NEW.generation + JOIN mst2_metadata_lifetime l USING(page_id,generation) JOIN mst2_metadata_graph_node n USING(page_id,generation) + WHERE q.prepare_id=NEW.prepare_id AND q.storage_seal=NEW.storage_seal AND q.state='PREPARING' + AND q.graph_domain='qualified-v1' AND l.graph_domain='qualified-v1' AND l.state IN ('RESERVED','LIVE') + AND n.state='LIVE' AND mst2_metadata_scope_matches(q.primary_scope)) THEN + RAISE EXCEPTION 'qualified root requires its active exact sealed preparation'; + END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_metadata_graph_root_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_metadata_graph_root + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_graph_root_guard(); + +CREATE FUNCTION mst2_metadata_graph_edge_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE op mst2_metadata_gc_op%ROWTYPE; +BEGIN + IF TG_OP='UPDATE' THEN RAISE EXCEPTION 'qualified edge identity is immutable'; END IF; + IF TG_OP='DELETE' THEN RAISE EXCEPTION 'qualified edge collection is closed'; END IF; + IF NOT EXISTS(SELECT 1 FROM mst2_metadata_verified_ref ref + JOIN mst2_metadata_page_certificate parent ON parent.page_id=ref.parent_page AND parent.generation=ref.parent_generation + JOIN mst2_metadata_page_certificate child ON child.page_id=ref.child_page AND child.generation=ref.child_generation + WHERE ref.parent_page=NEW.parent_page AND ref.parent_generation=NEW.parent_generation + AND ref.child_page=NEW.child_page AND ref.child_generation=NEW.child_generation + AND ref.child_certificate_digest=child.certificate_digest AND parent.rank>child.rank) THEN + RAISE EXCEPTION 'qualified edge differs from its exact canonical reference or strict certified rank'; + END IF; + IF (SELECT count(*) FROM mst2_metadata_graph_node n JOIN mst2_metadata_current c USING(page_id,generation) + JOIN mst2_metadata_lifetime l USING(page_id,generation) + WHERE ((n.page_id=NEW.parent_page AND n.generation=NEW.parent_generation) + OR (n.page_id=NEW.child_page AND n.generation=NEW.child_generation)) + AND n.state='LIVE' AND l.state IN ('RESERVED','LIVE') AND l.graph_domain='qualified-v1')<>2 + OR NOT EXISTS(SELECT 1 FROM mst2_metadata_prepare_page a JOIN mst2_metadata_prepare q USING(prepare_id) + WHERE a.page_id=NEW.parent_page AND a.generation=NEW.parent_generation AND q.state='PREPARING' + AND q.graph_domain='qualified-v1' AND mst2_metadata_scope_matches(q.primary_scope) + AND (EXISTS(SELECT 1 FROM mst2_metadata_prepare_page b WHERE b.prepare_id=a.prepare_id + AND b.page_id=NEW.child_page AND b.generation=NEW.child_generation) + OR EXISTS(SELECT 1 FROM mst2_metadata_prepare_reuse_root r JOIN mst2_metadata_root_anchor anchor + ON anchor.prepare_id=r.prepare_id AND anchor.anchor_kind='REUSE' AND anchor.owner_key=r.prepare_id + AND anchor.root_page=r.root_page AND anchor.root_generation=r.root_generation + WHERE r.prepare_id=a.prepare_id AND r.root_page=NEW.child_page AND r.root_generation=NEW.child_generation))) THEN + RAISE EXCEPTION 'edge creation requires active exact qualified endpoints'; + END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_metadata_graph_edge_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_metadata_graph_edge + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_graph_edge_guard(); + +CREATE FUNCTION mst2_metadata_edges_added() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + IF (SELECT count(*) FROM added_edges)>16384 THEN RAISE EXCEPTION 'qualified edge batch exceeds limit'; END IF; + IF NOT EXISTS(SELECT 1 FROM added_edges) THEN RETURN NULL; END IF; + IF EXISTS(SELECT 1 FROM mst2_metadata_graph_node n JOIN ( + SELECT child_page,child_generation,count(*) AS delta FROM added_edges GROUP BY child_page,child_generation + ) d ON d.child_page=n.page_id AND d.child_generation=n.generation + WHERE n.incoming_refs+d.delta<>(SELECT count(*) FROM mst2_metadata_graph_edge x WHERE x.child_page=n.page_id AND x.child_generation=n.generation)) THEN + RAISE EXCEPTION 'qualified insertion found preexisting counter drift'; + END IF; + UPDATE mst2_metadata_graph_node n SET incoming_refs=n.incoming_refs+d.delta FROM ( + SELECT child_page,child_generation,count(*) AS delta FROM added_edges GROUP BY child_page,child_generation + ) d WHERE n.page_id=d.child_page AND n.generation=d.child_generation; + RETURN NULL; +END $$; +CREATE TRIGGER mst2_metadata_edges_added AFTER INSERT ON mst2_metadata_graph_edge + REFERENCING NEW TABLE AS added_edges FOR EACH STATEMENT EXECUTE FUNCTION mst2_metadata_edges_added(); + +DO $$ DECLARE t text; BEGIN + FOREACH t IN ARRAY ARRAY['mst2_metadata_storage_scope','mst2_metadata_family_identity', + 'mst2_metadata_lifetime','mst2_metadata_current','mst2_metadata_payload','mst2_metadata_prepare', + 'mst2_metadata_prepare_page','mst2_metadata_graph_node','mst2_metadata_graph_edge','mst2_metadata_graph_root', + 'mst2_metadata_gc_op','mst2_qualified_session_incarnation','mst2_qualified_lease_binding', + 'mst2_metadata_page_certificate','mst2_metadata_verified_ref','mst2_metadata_source_root_attestation', + 'mst2_metadata_prepare_reuse_root','mst2_metadata_reuse_index','mst2_metadata_reader_operation','mst2_metadata_reader_issuance','mst2_metadata_root_anchor', + 'mst2_metadata_source_entry_reference','mst2_metadata_scope_source_reference'] LOOP + EXECUTE format('CREATE TRIGGER mst2_00_family_barrier BEFORE INSERT OR UPDATE OR DELETE ON %I FOR EACH STATEMENT EXECUTE FUNCTION mst2_metadata_dml_barrier()',t); + EXECUTE format('CREATE TRIGGER mst2_metadata_truncate_guard BEFORE TRUNCATE ON %I FOR EACH STATEMENT EXECUTE FUNCTION mst2_metadata_immutable()',t); + END LOOP; + FOREACH t IN ARRAY ARRAY['mst2_metadata_storage_scope','mst2_metadata_family_identity'] LOOP + EXECUTE format('CREATE TRIGGER mst2_metadata_identity_immutable BEFORE INSERT OR UPDATE OR DELETE ON %I FOR EACH STATEMENT EXECUTE FUNCTION mst2_metadata_immutable()',t); + END LOOP; + FOREACH t IN ARRAY ARRAY['mst2_metadata_gc_op','mst2_qualified_session_incarnation','mst2_qualified_lease_binding', + 'mst2_metadata_reader_operation'] LOOP + EXECUTE format('CREATE TRIGGER mst2_01_family_closed BEFORE INSERT OR UPDATE OR DELETE ON %I FOR EACH STATEMENT EXECUTE FUNCTION mst2_metadata_closed()',t); + END LOOP; +END $$; + +$SERVING_SQL$ +$GC_SQL$ diff --git a/src/jupiter/storage/qualified_metadata_family_tests.rs b/src/jupiter/storage/qualified_metadata_family_tests.rs new file mode 100644 index 00000000..176c465d --- /dev/null +++ b/src/jupiter/storage/qualified_metadata_family_tests.rs @@ -0,0 +1,855 @@ +use std::{sync::Arc, time::Duration}; + +use mst2_codec::metapage::{Entry, EntryKind, Page, page_id}; +use sea_orm::Database; +use sea_orm_migration::MigratorTrait; + +use super::*; +use crate::{ + ceres::snapshot::retention_dag::{MetadataDagBuilder, MetadataDagLimits}, + jupiter::{ + migration::Migrator, + tests::{TestSchemaGuard, test_db_config}, + }, +}; + +#[path = "qualified_metadata_canonical_tests.rs"] +mod canonical_tests; +#[path = "qualified_metadata_gc_tests.rs"] +mod gc_tests; +#[path = "qualified_metadata_source_revision_tests.rs"] +mod source_revision_tests; + +fn prepared() -> PreparedNativeMetadataRetention { + let entries = [Entry::file(EntryKind::Regular, b"file", 3, [42; 32])]; + let child = Page::build(&entries).unwrap(); + let root_entries = [ + Entry::dir(b"one", page_id(&child)), + Entry::dir(b"two", page_id(&child)), + ]; + let root = Page::build(&root_entries).unwrap(); + let mut builder = MetadataDagBuilder::new(MetadataDagLimits::default()); + builder.add_directory(&child, &entries).unwrap(); + builder.add_directory(&root, &root_entries).unwrap(); + PreparedNativeMetadataRetention::test_installation( + Arc::new(builder.finish(page_id(&root)).unwrap()), + "/", + ) +} + +async fn fixture() -> ( + DbConfig, + DatabaseConnection, + VerifiedQualifiedNamespace, + DatabaseConnection, + TestSchemaGuard, +) { + let temp = tempfile::tempdir().unwrap(); + let (mut config, guard) = test_db_config(temp.path()).await; + config.max_connection = 2; + config.min_connection = 1; + // Exercise the actual production bootstrap, not a test-only DDL shortcut. + let core = super::super::init::database_connection(&config) + .await + .unwrap(); + let namespace = provision_or_verify_rooted_qualified_family(&core) + .await + .unwrap(); + let mut q_config = config.clone(); + q_config.db_url = pool_url(&config.db_url, &namespace).unwrap(); + q_config.max_connection = 1; + let q = postgres_connection(&q_config).await.unwrap(); + (config, core, namespace, q, guard) +} + +async fn count(db: &C, sql: &str) -> i64 { + db.query_one_raw(Statement::from_string(DbBackend::Postgres, sql)) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap() +} +async fn write( + writer: &ShadowQualifiedMetadataWriter, + operation: &str, + pages: &PreparedNativeMetadataRetention, +) -> GenerationMetadataReceipt { + let intent = writer.begin_intent(operation, pages).await.unwrap(); + for batch in pages.dag().payloads().chunks(64) { + writer.install_pages(&intent, batch).await.unwrap(); + } + writer.finalize(&intent).await.unwrap() +} + +#[tokio::test] +async fn physical_q_shadow_writer_is_distinct_and_restart_reuses_its_exact_catalog() { + let (config, core, namespace, q, _guard) = fixture().await; + let pages = prepared(); + let generic=super::super::native_metadata_install::generations::PostgresMetadataGenerationRepository::new(core.clone()).await.unwrap(); + let g = generic.begin_intent("same-pages-g", &pages).await.unwrap(); + for batch in pages.dag().payloads().chunks(64) { + generic.install_pages(&g, batch).await.unwrap(); + } + let g_receipt = generic.finalize(&g).await.unwrap(); + let assembly_dir = tempfile::tempdir().unwrap(); + let mut app_config = crate::config::testing::isolated_config(assembly_dir.path()); + app_config.database = config.clone(); + let storage = super::super::Storage::new_with_connection( + Arc::new(app_config), + Arc::new(core.clone()), + super::super::object_storage::mock_object_storage(), + ) + .await + .unwrap(); + let writer = storage.shadow_qualified_metadata_writer().await.unwrap(); + let clone = storage.clone(); + assert!(std::ptr::eq( + writer, + clone.shadow_qualified_metadata_writer().await.unwrap() + )); + let q_receipt = write(writer, "same-pages-q", &pages).await; + assert_eq!(g_receipt.metadata_root(), q_receipt.metadata_root()); + assert_ne!(g.prepare_id(), q_receipt.intent().prepare_id()); + assert_eq!( + count(&core, "SELECT count(*) FROM mst2_metadata_payload").await, + 2 + ); + assert_eq!( + count( + &q, + "SELECT count(*) FROM mst2_metadata_payload WHERE generation=1" + ) + .await, + 2 + ); + assert_eq!(count(&q,"SELECT count(*) FROM mst2_metadata_prepare WHERE graph_domain='qualified-v1' AND state='COMMITTED'").await,1); + assert_eq!( + count( + &q, + "SELECT count(*) FROM mst2_qualified_session_incarnation" + ) + .await, + 0 + ); + assert_eq!( + count(&q, "SELECT count(*) FROM mst2_qualified_lease_binding").await, + 0 + ); + assert_eq!(count(&q,"SELECT count(*) FROM pg_catalog.pg_class WHERE relnamespace=(SELECT oid FROM pg_catalog.pg_namespace WHERE nspname=current_schema()) AND relkind='r'").await,23); + assert_eq!(count(&q,"SELECT count(*) FROM pg_catalog.pg_class WHERE relnamespace=(SELECT oid FROM pg_catalog.pg_namespace WHERE nspname=current_schema()) AND relname IN ('mst2_retention_node','mst2_retention_edge','mst2_snapshot_context','mst2_snapshot_lease','mega_refs','git_repo','seaql_migrations')").await,0); + let q_scope = q + .query_one_raw(Statement::from_string( + DbBackend::Postgres, + "SELECT primary_scope,storage_seal FROM mst2_metadata_prepare", + )) + .await + .unwrap() + .unwrap(); + let g_scope = core + .query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT primary_scope,storage_seal FROM mst2_metadata_prepare WHERE prepare_id=$1", + [g.prepare_id().into()], + )) + .await + .unwrap() + .unwrap(); + assert_ne!( + q_scope.try_get::>("", "primary_scope").unwrap(), + g_scope.try_get::>("", "primary_scope").unwrap() + ); + assert_ne!( + q_scope.try_get::>("", "storage_seal").unwrap(), + g_scope.try_get::>("", "storage_seal").unwrap() + ); + let restarted = super::super::init::database_connection(&config) + .await + .unwrap(); + assert_eq!( + provision_or_verify_rooted_qualified_family(&restarted) + .await + .unwrap(), + namespace + ); + let fresh = ShadowQualifiedMetadataWriter::open(&restarted, &config) + .await + .unwrap(); + assert_eq!(fresh.finalize(q_receipt.intent()).await.unwrap(), q_receipt); + assert_eq!( + catalog(&q, namespace.core_oid, namespace.schema_oid) + .await + .unwrap(), + namespace.catalog_fingerprint + ); +} + +#[tokio::test] +async fn q_catalog_trigger_function_and_index_tamper_are_never_refreshed() { + for tamper in [ + "ALTER TABLE {q}.mst2_metadata_payload DISABLE TRIGGER mst2_metadata_payload_fenced", + "CREATE OR REPLACE FUNCTION {q}.mst2_metadata_has_generic_overlap(p bytea) RETURNS boolean LANGUAGE sql AS 'SELECT true'", + "DROP INDEX {q}.idx_mst2_metadata_graph_edge_child", + "CREATE RULE forged_skip AS ON INSERT TO {q}.mst2_metadata_prepare DO INSTEAD NOTHING", + ] { + let (config, core, namespace, q, _guard) = fixture().await; + let writer = ShadowQualifiedMetadataWriter::open(&core, &config) + .await + .unwrap(); + core.execute_unprepared(&tamper.replace("{q}", &identifier(&namespace.schema))) + .await + .unwrap(); + assert!(writer.begin_intent("tampered", &prepared()).await.is_err()); + let error = provision_or_verify_rooted_qualified_family(&core) + .await + .unwrap_err(); + assert!( + error.to_string().contains("catalog fingerprint changed"), + "{error}" + ); + assert_eq!( + count(&q, "SELECT count(*) FROM mst2_metadata_prepare").await, + 0 + ); + assert_eq!( + count(&core, "SELECT count(*) FROM mst2_metadata_namespace").await, + 2 + ); + let stored:Vec=core.query_one_raw(Statement::from_string(DbBackend::Postgres, + "SELECT catalog_fingerprint FROM mst2_metadata_namespace WHERE graph_domain='qualified-v1'")) + .await.unwrap().unwrap().try_get_by_index(0).unwrap(); + assert_eq!(stored, namespace.catalog_fingerprint); + } +} + +#[tokio::test] +async fn q_schema_rename_fails_closed_without_reprovisioning_or_g_fallback() { + let (config, core, namespace, _q, _guard) = fixture().await; + let writer = ShadowQualifiedMetadataWriter::open(&core, &config) + .await + .unwrap(); + let renamed = format!("{}_renamed", namespace.schema); + core.execute_unprepared(&format!( + "ALTER SCHEMA {} RENAME TO {}", + identifier(&namespace.schema), + identifier(&renamed) + )) + .await + .unwrap(); + assert!(writer.begin_intent("renamed", &prepared()).await.is_err()); + assert!( + provision_or_verify_rooted_qualified_family(&core) + .await + .is_err() + ); + assert_eq!( + count(&core, "SELECT count(*) FROM mst2_metadata_namespace").await, + 2 + ); + assert_eq!( + count(&core, "SELECT count(*) FROM mst2_metadata_prepare").await, + 0 + ); + assert_eq!( + count( + &core, + &format!( + "SELECT count(*) FROM {}.mst2_metadata_prepare", + identifier(&renamed) + ) + ) + .await, + 0 + ); +} + +#[tokio::test] +async fn q_actual_schema_rr_temp_shadow_and_admitted_ledgers_keep_their_guards() { + let (_config, core, namespace, q, _guard) = fixture().await; + let bad = core + .execute_unprepared(&format!( + "INSERT INTO {}.mst2_metadata_gc_op SELECT * FROM {}.mst2_metadata_gc_op WHERE false", + identifier(&namespace.schema), + identifier(&namespace.schema) + )) + .await + .unwrap_err(); + assert!(bad.to_string().contains("captured primary family"), "{bad}"); + let rr = q + .begin_with_config(Some(IsolationLevel::RepeatableRead), None) + .await + .unwrap(); + let error = rr + .execute_unprepared( + "INSERT INTO mst2_metadata_gc_op SELECT * FROM mst2_metadata_gc_op WHERE false", + ) + .await + .unwrap_err(); + assert!(error.to_string().contains("READ COMMITTED"), "{error}"); + rr.rollback().await.unwrap(); + for statement in [ + "SELECT mst2_metadata_gc_apply(gen_random_uuid())", + "SELECT mst2_metadata_gc_finish(gen_random_uuid())", + ] { + let error = q.execute_unprepared(statement).await.unwrap_err(); + assert!( + error.to_string().contains("query returned no rows"), + "{error}" + ); + } + assert_eq!( + count(&q, "SELECT count(*) FROM mst2_metadata_gc_op").await, + 0 + ); + for table in [ + "mst2_qualified_session_incarnation", + "mst2_qualified_lease_binding", + ] { + assert!( + q.execute_unprepared(&format!("INSERT INTO {table} DEFAULT VALUES")) + .await + .is_err() + ); + assert_eq!(count(&q, &format!("SELECT count(*) FROM {table}")).await, 0); + } + q.execute_unprepared("CREATE TEMP TABLE mst2_metadata_prepare (prepare_id text)") + .await + .unwrap(); + let repository = + PostgresQualifiedMetadataRepository::registered_shadow(q.clone(), namespace.clone()) + .await + .unwrap(); + repository + .begin_intent("temp-shadow", &prepared()) + .await + .unwrap(); + assert_eq!( + count( + &q, + &format!( + "SELECT count(*) FROM {}.mst2_metadata_prepare", + identifier(&namespace.schema) + ) + ) + .await, + 1 + ); + assert_eq!( + count(&q, "SELECT count(*) FROM pg_temp.mst2_metadata_prepare").await, + 0 + ); + let duplicate=core.execute_unprepared("INSERT INTO mst2_metadata_namespace SELECT * FROM mst2_metadata_namespace WHERE graph_domain='qualified-v1'").await.unwrap_err(); + assert!( + duplicate + .to_string() + .contains("exact admitted rooted family scope"), + "{duplicate}" + ); + assert_eq!( + count(&core, "SELECT count(*) FROM mst2_metadata_namespace").await, + 2 + ); + let pk:String=q.query_one_raw(Statement::from_string(DbBackend::Postgres, + "SELECT conkey::text FROM pg_catalog.pg_constraint WHERE conrelid='mst2_qualified_session_incarnation'::regclass AND contype='p'")) + .await.unwrap().unwrap().try_get_by_index(0).unwrap(); + assert_eq!( + pk, "{1,2}", + "same SID can have future distinct incarnations" + ); +} + +#[tokio::test] +async fn provisioning_locks_candidate_and_registered_union_in_uuid_order() { + let temp = tempfile::tempdir().unwrap(); + let (config, _guard) = test_db_config(temp.path()).await; + let core = Database::connect(config.db_url.clone()).await.unwrap(); + Migrator::up(&core, None).await.unwrap(); + let other = Database::connect(config.db_url).await.unwrap(); + let captured = captured_core(&core).await.unwrap(); + let candidate = "00000000-0000-4000-8000-000000000001"; + let schema = format!("mst2q_{}", candidate.replace('-', "")); + let held = core.begin().await.unwrap(); + held.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT pg_catalog.pg_advisory_xact_lock(1296717362,pg_catalog.hashtext($1))", + [schema.clone().into()], + )) + .await + .unwrap(); + let (core_name, q_name) = (captured.0.clone(), schema.clone()); + let waiter = tokio::spawn(async move { + let txn = other.begin().await.unwrap(); + txn.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + format!( + "SELECT {}.mst2_route_family_candidate_enter($1,$2::uuid,$3)", + identifier(&core_name) + ), + [core_name.into(), candidate.into(), q_name.into()], + )) + .await + .unwrap(); + txn.rollback().await.unwrap(); + }); + let deadline = tokio::time::Instant::now() + Duration::from_secs(4); + loop { + let row=held.query_one_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "SELECT EXISTS(SELECT 1 FROM pg_catalog.pg_locks w WHERE w.locktype='advisory' AND NOT w.granted + AND w.classid=1296717362::oid AND w.objid=pg_catalog.hashtext($1)::oid AND w.objsubid=2 + AND w.database=(SELECT oid FROM pg_catalog.pg_database WHERE datname=current_database()) + AND NOT EXISTS(SELECT 1 FROM pg_catalog.pg_locks g WHERE g.pid=w.pid AND g.locktype='advisory' + AND g.classid=1296717362::oid AND g.objid=pg_catalog.hashtext($2)::oid AND g.objsubid=2 AND g.granted)) AS sorted_wait", + [schema.clone().into(),captured.0.clone().into()])).await.unwrap().unwrap(); + if row.try_get::("", "sorted_wait").unwrap() { + break; + } + assert!( + tokio::time::Instant::now() < deadline, + "candidate must be locked before G when its UUID sorts first" + ); + tokio::task::yield_now().await; + } + held.rollback().await.unwrap(); + waiter.await.unwrap(); + assert_eq!( + count(&core, "SELECT count(*) FROM mst2_metadata_namespace").await, + 1 + ); +} + +#[tokio::test] +async fn q_incomplete_commit_null_binding_and_unsealed_plan_are_rejected() { + let (config, core, namespace, q, _guard) = fixture().await; + let writer = ShadowQualifiedMetadataWriter::open(&core, &config) + .await + .unwrap(); + let intent = writer + .begin_intent("fixed-incomplete", &prepared()) + .await + .unwrap(); + let error=q.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "UPDATE mst2_metadata_prepare SET state='COMMITTED',committed_at=clock_timestamp() WHERE prepare_id=$1",[intent.prepare_id().into()])).await.unwrap_err(); + assert!( + error.to_string().contains("complete exact graph payload"), + "{error}" + ); + for mutation in [ + "graph_domain='generic-v1'", + "canonical_plan=canonical_plan||decode('ff','hex')", + "storage_seal=NULL", + "primary_scope=NULL", + ] { + let error = q + .execute_unprepared(&format!("UPDATE mst2_metadata_prepare SET {mutation}")) + .await + .unwrap_err(); + assert!( + error.to_string().contains("complete identity is immutable"), + "{error}" + ); + } + for batch in prepared().dag().payloads().chunks(64) { + writer.install_pages(&intent, batch).await.unwrap(); + } + writer.finalize(&intent).await.unwrap(); + assert_eq!( + count( + &q, + "SELECT count(*) FROM mst2_metadata_prepare WHERE state='COMMITTED'" + ) + .await, + 1 + ); + assert_eq!( + catalog(&q, namespace.core_oid, namespace.schema_oid) + .await + .unwrap(), + namespace.catalog_fingerprint + ); +} + +#[tokio::test] +async fn first_registration_cannot_self_sign_a_minimal_or_half_installed_family() { + for complete in [false, true] { + let temp = tempfile::tempdir().unwrap(); + let (config, _guard) = test_db_config(temp.path()).await; + let core = Database::connect(config.db_url).await.unwrap(); + Migrator::up(&core, None).await.unwrap(); + let captured = captured_core(&core).await.unwrap(); + let candidate = uuid::Uuid::new_v4().to_string(); + let schema = format!("mst2q_{}", candidate.replace('-', "")); + let storage = uuid::Uuid::new_v4().to_string(); + let txn = core.begin().await.unwrap(); + txn.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + format!( + "SELECT {}.mst2_route_family_candidate_enter($1,$2::uuid,$3)", + identifier(&captured.0) + ), + [ + captured.0.clone().into(), + candidate.clone().into(), + schema.clone().into(), + ], + )) + .await + .unwrap(); + txn.execute_unprepared(&format!("CREATE SCHEMA {}", identifier(&schema))) + .await + .unwrap(); + let oid: i64 = txn + .query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT oid::bigint FROM pg_catalog.pg_namespace WHERE nspname=$1", + [schema.clone().into()], + )) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap(); + if complete { + txn.execute_unprepared(&render_family( + &captured.0, + captured.1, + &schema, + oid, + &candidate, + &storage, + )) + .await + .unwrap(); + } else { + txn.execute_unprepared(&format!( + "CREATE TABLE {q}.mst2_metadata_storage_scope(singleton integer,storage_uuid text); + INSERT INTO {q}.mst2_metadata_storage_scope VALUES(1,{storage}); + CREATE TABLE {q}.mst2_metadata_family_identity(singleton integer,namespace_uuid uuid,storage_uuid text, + core_schema_oid oid,metadata_schema_oid oid,family_identity text,implementation_fingerprint bytea); + INSERT INTO {q}.mst2_metadata_family_identity VALUES(1,{candidate}::uuid,{storage},{core_oid},{oid},'{family}',decode('{implementation}','hex'))", + q=identifier(&schema),storage=literal(&storage),candidate=literal(&candidate),core_oid=captured.1, + family=FAMILY,implementation=hex::encode(implementation_fingerprint()))).await.unwrap(); + } + let fingerprint = if complete { + vec![0xff; 32] + } else { + catalog(&txn, captured.1, oid).await.unwrap() + }; + txn.execute_unprepared(&format!( + "SET LOCAL search_path={},pg_catalog,pg_temp", + identifier(&captured.0) + )) + .await + .unwrap(); + let error=txn.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres,format!( + "INSERT INTO {c}.mst2_metadata_namespace(singleton,namespace_uuid,core_schema,core_schema_oid,database_name, + database_oid,storage_uuid,server_address,server_port,mono_lock_key2,metadata_schema,metadata_schema_oid, + family_identity,graph_domain,admission_state,collector_state,metadata_storage_uuid,implementation_fingerprint,catalog_fingerprint) + SELECT NULL,$1::uuid,g.core_schema,g.core_schema_oid,g.database_name,g.database_oid,g.storage_uuid,g.server_address, + g.server_port,g.mono_lock_key2,$2,$3::bigint::oid,$4,'qualified-v1','ROOTED_Q_ADMITTED','ENABLED',$5,$6,$7 + FROM {c}.mst2_metadata_namespace g WHERE singleton=1",c=identifier(&captured.0)), + [candidate.into(),schema.clone().into(),oid.into(),FAMILY.into(),storage.into(),implementation_fingerprint().into(),fingerprint.into()])).await.unwrap_err(); + let expected = if complete { + "fingerprint disagrees" + } else { + "trusted complete physical family shape" + }; + assert!(error.to_string().contains(expected), "{error}"); + txn.rollback().await.unwrap(); + assert_eq!( + count(&core, "SELECT count(*) FROM mst2_metadata_namespace").await, + 1 + ); + assert_eq!( + count( + &core, + &format!( + "SELECT count(*) FROM pg_catalog.pg_namespace WHERE nspname={}", + literal(&schema) + ) + ) + .await, + 0 + ); + assert_eq!( + count(&core, "SELECT count(*) FROM mst2_qualified_family_policy").await, + 1 + ); + } +} + +#[tokio::test] +async fn simultaneous_production_bootstraps_reuse_one_physical_q_namespace() { + let temp = tempfile::tempdir().unwrap(); + let (config, _guard) = test_db_config(temp.path()).await; + let core = postgres_connection(&config).await.unwrap(); + Migrator::up(&core, None).await.unwrap(); + let (left, right) = tokio::join!( + super::super::init::database_connection(&config), + super::super::init::database_connection(&config) + ); + let left = left.unwrap(); + let right = right.unwrap(); + let a = provision_or_verify_rooted_qualified_family(&left) + .await + .unwrap(); + let b = provision_or_verify_rooted_qualified_family(&right) + .await + .unwrap(); + assert_eq!(a, b); + assert_eq!( + count(&core, "SELECT count(*) FROM mst2_metadata_namespace").await, + 2 + ); + assert_eq!( + count( + &core, + &format!( + "SELECT count(*) FROM pg_catalog.pg_namespace WHERE nspname={}", + literal(&a.schema) + ) + ) + .await, + 1 + ); +} + +#[tokio::test] +async fn trusted_shape_policy_and_core_validation_functions_are_immutable_or_fail_closed() { + let (config, core, namespace, _q, _guard) = fixture().await; + for sql in [ + "UPDATE mst2_qualified_family_policy SET expected_shape=decode(repeat('ff',32),'hex')", + "DELETE FROM mst2_qualified_family_policy", + "TRUNCATE mst2_qualified_family_policy", + ] { + let error = core.execute_unprepared(sql).await.unwrap_err(); + assert!(error.to_string().contains("immutable"), "{error}"); + } + let writer = ShadowQualifiedMetadataWriter::open(&core, &config) + .await + .unwrap(); + core.execute_unprepared(&format!( + "CREATE OR REPLACE FUNCTION {}.mst2_route_family_shape(q_oid oid,n_uuid uuid,s_uuid text) + RETURNS bytea LANGUAGE sql AS 'SELECT decode(repeat(''ff'',32),''hex'')'", + identifier(&namespace.core_schema) + )) + .await + .unwrap(); + let error = provision_or_verify_rooted_qualified_family(&core) + .await + .unwrap_err(); + assert!( + error + .to_string() + .contains("trusted core authority catalog changed"), + "{error}" + ); + assert!( + writer + .begin_intent("bad-core-validator", &prepared()) + .await + .is_err() + ); + assert_eq!( + count(&core, "SELECT count(*) FROM mst2_metadata_namespace").await, + 2 + ); +} + +#[tokio::test] +async fn random_q_templates_have_one_trusted_shape_and_changed_composite_signature_is_rejected() { + let temp = tempfile::tempdir().unwrap(); + let (config, _guard) = test_db_config(temp.path()).await; + let core = postgres_connection(&config).await.unwrap(); + Migrator::up(&core, None).await.unwrap(); + let captured = captured_core(&core).await.unwrap(); + let expected: Vec = core + .query_one_raw(Statement::from_string( + DbBackend::Postgres, + "SELECT expected_shape FROM mst2_qualified_family_policy WHERE singleton=1", + )) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap(); + let authority = catalog(&core, captured.1, 0).await.unwrap(); + let mut schemas = Vec::new(); + for _ in 0..2 { + let namespace = uuid::Uuid::new_v4().to_string(); + let schema = format!("mst2q_{}", namespace.replace('-', "")); + let storage = uuid::Uuid::new_v4().to_string(); + let txn = core.begin().await.unwrap(); + txn.execute_unprepared(&format!("CREATE SCHEMA {}", identifier(&schema))) + .await + .unwrap(); + let oid = count( + &txn, + &format!( + "SELECT oid::bigint FROM pg_catalog.pg_namespace WHERE nspname={}", + literal(&schema), + ), + ) + .await; + txn.execute_unprepared(&render_family( + &captured.0, + captured.1, + &schema, + oid, + &namespace, + &storage, + )) + .await + .unwrap(); + let shape = || { + Statement::from_sql_and_values( + DbBackend::Postgres, + format!( + "SELECT {}.mst2_route_family_shape($1::bigint::oid,$2::uuid,$3)", + identifier(&captured.0), + ), + [oid.into(), namespace.clone().into(), storage.clone().into()], + ) + }; + let actual: Vec = txn + .query_one_raw(shape()) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap(); + assert_eq!( + actual, expected, + "random template must retain its trusted structural signature" + ); + assert_eq!( + catalog_with_exemption(&txn, captured.1, 0, oid) + .await + .unwrap(), + authority + ); + assert_ne!( + catalog(&txn, captured.1, 0).await.unwrap(), + authority, + "an unregistered candidate is never implicitly exempted" + ); + txn.execute_unprepared(&format!( + "DROP FUNCTION {q}.mst2_metadata_rooted_manifest({q}.mst2_metadata_prepare); + CREATE FUNCTION {q}.mst2_metadata_rooted_manifest(q text) RETURNS jsonb + LANGUAGE sql IMMUTABLE STRICT AS 'SELECT to_jsonb(q)'", + q = identifier(&schema), + )) + .await + .unwrap(); + let altered: Vec = txn + .query_one_raw(shape()) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap(); + assert_ne!( + altered, expected, + "normalization must preserve the composite argument type" + ); + txn.rollback().await.unwrap(); + schemas.push(schema); + } + assert_ne!(schemas[0], schemas[1]); + assert_eq!(catalog(&core, captured.1, 0).await.unwrap(), authority); + assert!(registered(&core, &captured).await.unwrap().is_none()); +} + +#[tokio::test] +async fn fresh_q_core_authority_exempts_only_exact_q_ri_and_restart_keeps_full_catalog_binding() { + for tamper in ["fk", "ri", "external"] { + let (config, core, namespace, _q, _guard) = fixture().await; + let captured = (namespace.core_schema.clone(), namespace.core_oid); + let authority: Vec = core + .query_one_raw(Statement::from_string( + DbBackend::Postgres, + "SELECT authority_catalog FROM mst2_qualified_family_policy WHERE singleton=1", + )) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap(); + assert_eq!( + catalog_with_exemption(&core, captured.1, 0, namespace.schema_oid) + .await + .unwrap(), + authority + ); + assert_ne!(catalog(&core, captured.1, 0).await.unwrap(), authority); + for _ in 0..2 { + assert_eq!( + registered(&core, &captured).await.unwrap(), + Some(namespace.clone()) + ); + } + let restarted = postgres_connection(&config).await.unwrap(); + assert_eq!( + provision_or_verify_rooted_qualified_family(&restarted) + .await + .unwrap(), + namespace + ); + let txn = core.begin().await.unwrap(); + if tamper == "external" { + let external = format!("hostile_{}", uuid::Uuid::new_v4().simple()); + txn.execute_unprepared(&format!( + "CREATE SCHEMA {external}; CREATE TABLE {external}.forged_route(snapshot_id text,namespace_uuid uuid, + FOREIGN KEY(snapshot_id,namespace_uuid) REFERENCES {c}.mst2_snapshot_storage_route(snapshot_id,namespace_uuid))", + external=identifier(&external),c=identifier(&captured.0), + )).await.unwrap(); + assert_ne!( + catalog_with_exemption(&txn, captured.1, 0, namespace.schema_oid) + .await + .unwrap(), + authority, + "another namespace's internal RI triggers remain part of core authority" + ); + } else { + let row=txn.query_one_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "SELECT fk.conname,t.tgname FROM pg_catalog.pg_constraint fk + JOIN pg_catalog.pg_trigger t ON t.tgconstraint=fk.oid AND t.tgrelid=fk.confrelid AND t.tgisinternal + WHERE fk.connamespace=$1::bigint::oid AND fk.confrelid=$2::regclass AND fk.contype='f' LIMIT 1", + [namespace.schema_oid.into(),format!("{}.mst2_snapshot_storage_route",identifier(&captured.0)).into()])) + .await.unwrap().unwrap(); + let conname: String = row.try_get("", "conname").unwrap(); + let trigger: String = row.try_get("", "tgname").unwrap(); + let sql = if tamper == "fk" { + format!( + "ALTER TABLE {}.mst2_qualified_session_incarnation DROP CONSTRAINT {}", + identifier(&namespace.schema), + identifier(&conname) + ) + } else { + format!( + "ALTER TABLE {}.mst2_snapshot_storage_route DISABLE TRIGGER {}", + identifier(&captured.0), + identifier(&trigger) + ) + }; + txn.execute_unprepared(&sql).await.unwrap(); + assert_eq!( + catalog_with_exemption(&txn, captured.1, 0, namespace.schema_oid) + .await + .unwrap(), + authority + ); + assert_ne!( + catalog(&txn, captured.1, namespace.schema_oid) + .await + .unwrap(), + namespace.catalog_fingerprint + ); + } + assert!( + registered(&txn, &captured).await.is_err(), + "{tamper} tampering cannot refresh admission" + ); + txn.rollback().await.unwrap(); + assert_eq!(registered(&core, &captured).await.unwrap(), Some(namespace)); + } +} diff --git a/src/jupiter/storage/qualified_metadata_gc.rs b/src/jupiter/storage/qualified_metadata_gc.rs new file mode 100644 index 00000000..394d9335 --- /dev/null +++ b/src/jupiter/storage/qualified_metadata_gc.rs @@ -0,0 +1,540 @@ +//! Bounded rooted maintenance. Local retry state is a hint, never GC authority. + +use super::*; + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +enum GcPhase { + Claim, + Apply, +} + +#[derive(Debug, Clone, PartialEq, Eq)] +struct GcSeal { + operation_id: String, + page_id: [u8; 32], + generation: i64, + primary_scope: Vec, + metadata_codec: i16, + expected_size: i32, + graph_present: bool, + had_payload: bool, + certificate_digest: Option<[u8; 32]>, +} + +#[derive(Debug, Clone)] +struct RetryOperation { + seal: GcSeal, + phase: GcPhase, +} + +#[derive(Debug, Default)] +pub(super) struct RootedMaintenanceState { + retry: Option, + cursor: Option<([u8; 32], i64)>, +} + +#[derive(Debug, Default, Clone, serde::Serialize)] +pub(crate) struct RootedMaintenanceWork { + pub collector_enabled: bool, + pub examined: u16, + pub lifetimes_examined: u16, + pub owners_examined: u16, + pub pending_replayed: u16, + pub uncertain_receipts_observed: u16, + pub absent_claims_observed: u16, + pub gc_claimed: u16, + pub gc_applied: u16, + pub payload_pages_removed: u16, + pub payload_bytes_removed: u64, + pub readers_expired: u16, + pub leases_expired: u16, + pub prepares_aborted: u16, + pub handovers_retired: u16, + pub orphans_retired: u16, +} + +struct GcRecord { + seal: GcSeal, + applied: bool, +} + +fn uncertain(operation: &RetryOperation) -> SnapshotError { + SnapshotError::new( + crate::ceres::snapshot::error::SnapshotErrorCode::TemporaryUnavailable, + format!( + "qualified GC outcome unknown for operation {} generation {} phase {:?}; retry the sealed operation on the same primary", + operation.seal.operation_id, operation.seal.generation, operation.phase + ), + ) +} + +fn gc_record(row: &QueryResult) -> Result { + let operation_id: String = row.try_get("", "operation_id").map_err(internal)?; + let id = uuid::Uuid::parse_str(&operation_id).map_err(internal)?; + let certificate: Option> = row.try_get("", "certificate_digest").map_err(internal)?; + let seal = GcSeal { + operation_id, + page_id: digest_column(row, "page_id")?, + generation: row.try_get("", "generation").map_err(internal)?, + primary_scope: row.try_get("", "primary_scope").map_err(internal)?, + metadata_codec: row.try_get("", "metadata_codec").map_err(internal)?, + expected_size: row.try_get("", "expected_size").map_err(internal)?, + graph_present: row.try_get("", "graph_present").map_err(internal)?, + had_payload: row.try_get("", "had_payload").map_err(internal)?, + certificate_digest: certificate + .map(|bytes| bytes.as_slice().try_into().map_err(internal)) + .transpose()?, + }; + if id.get_version_num() != 4 + || id.to_string() != seal.operation_id + || seal.generation <= 0 + || seal.metadata_codec != 1 + || seal.expected_size <= 0 + || seal.graph_present != seal.certificate_digest.is_some() + || row + .try_get::("", "graph_domain") + .map_err(internal)? + != "qualified-v1" + { + return Err(integrity( + "qualified GC immutable operation profile is invalid", + )); + } + let state: String = row.try_get("", "state").map_err(internal)?; + let applied = match state.as_str() { + "PENDING" => false, + "APPLIED" => true, + _ => return Err(integrity("qualified GC operation has an unknown state")), + }; + Ok(GcRecord { seal, applied }) +} + +async fn load_gc( + db: &C, + operation: &str, +) -> Result, SnapshotError> { + db.query_one_raw(sql( + "SELECT operation_id::text,page_id,generation,primary_scope,graph_domain,metadata_codec, + expected_size,graph_present,had_payload,certificate_digest,state + FROM mst2_metadata_gc_op WHERE operation_id=$1::uuid", + [operation.into()], + )) + .await + .map_err(internal)? + .as_ref() + .map(gc_record) + .transpose() +} + +impl RootedQualifiedMetadataRepository { + pub(crate) fn start_maintenance(repository: &std::sync::Arc) { + let repository = std::sync::Arc::downgrade(repository); + tokio::spawn(async move { + let mut interval = tokio::time::interval(std::time::Duration::from_secs(30)); + interval.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Skip); + interval.tick().await; + loop { + interval.tick().await; + let Some(repository) = repository.upgrade() else { + break; + }; + match tokio::time::timeout( + std::time::Duration::from_secs(15), + repository.maintenance_tick(64), + ) + .await + { + Ok(Ok(_)) => {} + Ok(Err(error)) => { + tracing::warn!(code=?error.code, "rooted metadata maintenance failed; durable roots and sealed operations remain recoverable") + } + Err(_) => tracing::warn!( + "rooted metadata maintenance reached its deadline; retrying durable state on the next tick" + ), + } + } + }); + } + + async fn collector_enabled(&self) -> Result { + let txn = self.transaction().await?; + let result = async { + let row = txn + .query_one_raw(sql("SELECT mst2_metadata_gc_enabled() AS enabled", [])) + .await + .map_err(internal)? + .ok_or_else(|| integrity("qualified collector policy is missing"))?; + row.try_get("", "enabled").map_err(internal) + } + .await; + let _ = txn.rollback().await; + result + } + + fn require_gc_seal(&self, seal: &GcSeal) -> Result<(), SnapshotError> { + if seal.primary_scope != self.primary_scope { + return Err(integrity( + "qualified GC seal belongs to another captured primary", + )); + } + Ok(()) + } + + async fn prove_gc_record( + &self, + txn: &DatabaseTransaction, + record: &GcRecord, + ) -> Result<(), SnapshotError> { + self.require_gc_seal(&record.seal)?; + let seal = &record.seal; + txn.query_one_raw(sql( + "SELECT mst2_metadata_gc_proof($1,$2,$3,$4,$5,$6)", + [ + seal.page_id.to_vec().into(), + seal.generation.into(), + if record.applied { "APPLIED" } else { "PENDING" }.into(), + (!record.applied && seal.graph_present).into(), + (!record.applied && seal.had_payload).into(), + seal.certificate_digest.map(|value| value.to_vec()).into(), + ], + )) + .await + .map_err(internal)?; + Ok(()) + } + + async fn observe_gc_retry( + &self, + operation: &RetryOperation, + ) -> Result, SnapshotError> { + self.require_gc_seal(&operation.seal)?; + // transaction() acquires the same-primary route completion barrier. + // A failed barrier or read never proves that an uncertain claim is absent. + let txn = self.transaction().await.map_err(|_| uncertain(operation))?; + let result = async { + let record = load_gc(&txn, &operation.seal.operation_id).await?; + if let Some(record) = &record { + if record.seal != operation.seal { + return Err(integrity( + "qualified GC retry retargeted its immutable sealed identity", + )); + } + self.prove_gc_record(&txn, record).await?; + } + Ok(record) + } + .await; + txn.rollback().await.map_err(|_| uncertain(operation))?; + result.map_err(|error: SnapshotError| { + if error.code == crate::ceres::snapshot::error::SnapshotErrorCode::Internal { + uncertain(operation) + } else { + error + } + }) + } + + async fn claim_gc_candidate( + &self, + candidate: GcSeal, + state: &mut RootedMaintenanceState, + ) -> Result { + self.require_gc_seal(&candidate)?; + let txn = self.transaction().await?; + let result = async { + txn.query_one_raw(sql( + "SELECT mst2_metadata_gc_claim($1,$2,$3,$4::uuid)", + [ + candidate.page_id.to_vec().into(), + candidate.generation.into(), + candidate.primary_scope.clone().into(), + candidate.operation_id.clone().into(), + ], + )) + .await + .map_err(internal)?; + let record = load_gc(&txn, &candidate.operation_id) + .await? + .ok_or_else(|| integrity("qualified claimed GC operation disappeared"))?; + if record.seal != candidate || record.applied { + return Err(integrity( + "qualified claim differs from its exact captured candidate", + )); + } + self.prove_gc_record(&txn, &record).await?; + Ok(record.seal) + } + .await; + let seal = match result { + Ok(seal) => seal, + Err(error) => { + let _ = txn.rollback().await; + return Err(error); + } + }; + let retry = RetryOperation { + seal: seal.clone(), + phase: GcPhase::Claim, + }; + // Retain the hint before COMMIT so task cancellation cannot lose an + // already transmitted commit whose durable outcome is still unknown. + state.retry = Some(retry.clone()); + txn.commit().await.map_err(|_| uncertain(&retry))?; + state.retry = None; + Ok(seal) + } + + async fn apply_gc_seal( + &self, + seal: &GcSeal, + state: &mut RootedMaintenanceState, + ) -> Result { + self.require_gc_seal(seal)?; + let txn = self.transaction().await?; + let result = async { + let record = load_gc(&txn, &seal.operation_id) + .await? + .ok_or_else(|| integrity("qualified committed GC operation is missing"))?; + if &record.seal != seal { + return Err(integrity( + "qualified apply retargeted its exact immutable GC operation", + )); + } + self.prove_gc_record(&txn, &record).await?; + if record.applied { + return Ok(false); + } + txn.query_one_raw(sql( + "SELECT mst2_metadata_gc_apply($1::uuid)", + [seal.operation_id.clone().into()], + )) + .await + .map_err(internal)?; + let applied = load_gc(&txn, &seal.operation_id) + .await? + .ok_or_else(|| integrity("qualified GC apply lost its durable receipt"))?; + if &applied.seal != seal || !applied.applied { + return Err(integrity( + "qualified GC apply did not preserve its sealed APPLIED identity", + )); + } + self.prove_gc_record(&txn, &applied).await?; + Ok(true) + } + .await; + let changed = match result { + Ok(changed) => changed, + Err(error) => { + let _ = txn.rollback().await; + return Err(error); + } + }; + let retry = RetryOperation { + seal: seal.clone(), + phase: GcPhase::Apply, + }; + state.retry = Some(retry.clone()); + txn.commit().await.map_err(|_| uncertain(&retry))?; + state.retry = None; + Ok(changed) + } + + async fn pending_gc_seals(&self, limit: u16) -> Result, SnapshotError> { + let txn = self.transaction().await?; + let result = async { + let rows = txn.query_all_raw(sql( + "SELECT operation_id::text,page_id,generation,primary_scope,graph_domain,metadata_codec, + expected_size,graph_present,had_payload,certificate_digest,state + FROM mst2_metadata_gc_op WHERE state='PENDING' ORDER BY created_at,operation_id LIMIT $1", + [i64::from(limit).into()], + )).await.map_err(internal)?; + rows.iter().map(|row| { + let record=gc_record(row)?; + self.require_gc_seal(&record.seal)?; + Ok(record.seal) + }).collect() + }.await; + let _ = txn.rollback().await; + result + } + + async fn cleanup_gc_owners( + &self, + limit: u16, + work: &mut RootedMaintenanceWork, + ) -> Result<(), SnapshotError> { + let txn = self.transaction().await?; + let result = async { + txn.query_one_raw(sql( + "SELECT * FROM mst2_metadata_gc_owner_cleanup($1::integer)", + [i32::from(limit).into()], + )) + .await + .map_err(internal)? + .ok_or_else(|| integrity("qualified owner cleanup lost its bounded counters")) + } + .await; + let row = sessions::finish(txn, result).await?; + let count = |name: &str| -> Result { + u16::try_from(row.try_get::("", name).map_err(internal)?).map_err(internal) + }; + let examined = count("examined")?; + let readers = count("readers_expired")?; + let leases = count("leases_expired")?; + let prepares = count("prepares_aborted")?; + let handovers = count("handovers_retired")?; + let orphans = count("orphans_retired")?; + if examined > limit || readers + leases + prepares + handovers + orphans > examined { + return Err(integrity( + "qualified owner cleanup counters exceed their shared budget", + )); + } + work.examined += examined; + work.owners_examined += examined; + work.readers_expired += readers; + work.leases_expired += leases; + work.prepares_aborted += prepares; + work.handovers_retired += handovers; + work.orphans_retired += orphans; + Ok(()) + } + + async fn gc_candidates( + &self, + after: Option<([u8; 32], i64)>, + limit: u16, + ) -> Result, SnapshotError> { + let (page, generation) = after.map(|(p, g)| (p.to_vec(), g)).unwrap_or_default(); + let txn = self.transaction().await?; + let result = async { + let rows = txn.query_all_raw(sql( + "SELECT l.page_id,l.generation,l.metadata_codec,l.expected_size, + n.page_id IS NOT NULL AS graph_present,n.certificate_digest, + EXISTS(SELECT 1 FROM mst2_metadata_payload b WHERE b.page_id=l.page_id AND b.generation=l.generation) AS had_payload, + NOT EXISTS(SELECT 1 FROM mst2_metadata_root_anchor a WHERE a.root_page=l.page_id AND a.root_generation=l.generation) + AND NOT EXISTS(SELECT 1 FROM mst2_metadata_graph_root r WHERE r.page_id=l.page_id AND r.generation=l.generation) + AND NOT EXISTS(SELECT 1 FROM mst2_metadata_graph_edge e WHERE e.child_page=l.page_id AND e.child_generation=l.generation) + AND NOT EXISTS(SELECT 1 FROM mst2_metadata_prepare_page m JOIN mst2_metadata_prepare q USING(prepare_id) + WHERE m.page_id=l.page_id AND m.generation=l.generation + AND (q.state='PREPARING' OR q.state='COMMITTED' AND q.coverage_retired_at IS NULL)) + AND NOT EXISTS(SELECT 1 FROM mst2_metadata_prepare_reuse_root r JOIN mst2_metadata_prepare q USING(prepare_id) + WHERE r.root_page=l.page_id AND r.root_generation=l.generation + AND (q.state='PREPARING' OR q.state='COMMITTED' AND q.coverage_retired_at IS NULL)) + AND (l.state='LIVE' OR EXISTS(SELECT 1 FROM mst2_metadata_prepare_page m JOIN mst2_metadata_prepare q USING(prepare_id) + WHERE m.page_id=l.page_id AND m.generation=l.generation AND q.state='ABORTED' AND q.storage_seal IS NOT NULL) + AND NOT EXISTS(SELECT 1 FROM mst2_metadata_prepare_page m JOIN mst2_metadata_prepare q USING(prepare_id) + WHERE m.page_id=l.page_id AND m.generation=l.generation AND q.state<>'ABORTED')) AS eligible + FROM mst2_metadata_lifetime l JOIN mst2_metadata_current c USING(page_id,generation) + LEFT JOIN mst2_metadata_graph_node n USING(page_id,generation) + WHERE l.state IN ('RESERVED','LIVE') AND (l.page_id,l.generation)>($1,$2) + ORDER BY l.page_id,l.generation LIMIT $3", + [page.into(),generation.into(),i64::from(limit).into()], + )).await.map_err(internal)?; + rows.iter().map(|row| { + let certificate:Option>=row.try_get("","certificate_digest").map_err(internal)?; + Ok((GcSeal { + operation_id:uuid::Uuid::new_v4().to_string(), + page_id:digest_column(row,"page_id")?, + generation:row.try_get("","generation").map_err(internal)?, + primary_scope:self.primary_scope.clone(), + metadata_codec:row.try_get("","metadata_codec").map_err(internal)?, + expected_size:row.try_get("","expected_size").map_err(internal)?, + graph_present:row.try_get("","graph_present").map_err(internal)?, + had_payload:row.try_get("","had_payload").map_err(internal)?, + certificate_digest:certificate.map(|bytes|bytes.as_slice().try_into().map_err(internal)).transpose()?, + },row.try_get("","eligible").map_err(internal)?)) + }).collect() + }.await; + let _ = txn.rollback().await; + result + } + + /// A shared owner/page work budget; SQL proof and capacity scans are separately bounded. + pub(crate) async fn maintenance_tick( + &self, + limit: u16, + ) -> Result { + if !(1..=64).contains(&limit) { + return Err(integrity("qualified maintenance limit must be 1..=64")); + } + let mut state = self.maintenance_state.lock().await; + let mut work = RootedMaintenanceWork { + collector_enabled: self.collector_enabled().await?, + ..Default::default() + }; + if let Some(retry) = state.retry.clone() { + work.examined += 1; + work.lifetimes_examined += 1; + match self.observe_gc_retry(&retry).await? { + Some(record) if record.applied => { + state.retry = None; + work.uncertain_receipts_observed += 1; + } + Some(record) if work.collector_enabled => { + let changed = self.apply_gc_seal(&record.seal, &mut state).await?; + work.pending_replayed += 1; + work.record_applied(&record.seal, changed)?; + } + Some(_) => {} + None if retry.phase == GcPhase::Claim => { + state.retry = None; + work.absent_claims_observed += 1; + } + None => { + return Err(integrity( + "qualified committed GC apply lost its sealed operation history", + )); + } + } + } + if work.collector_enabled && work.examined < limit { + for seal in self.pending_gc_seals(limit - work.examined).await? { + work.examined += 1; + work.lifetimes_examined += 1; + let changed = self.apply_gc_seal(&seal, &mut state).await?; + work.pending_replayed += 1; + work.record_applied(&seal, changed)?; + } + } + if work.examined < limit { + let remaining = limit - work.examined; + let owner_budget = if work.collector_enabled { + remaining.div_ceil(2) + } else { + remaining + }; + self.cleanup_gc_owners(owner_budget, &mut work).await?; + } + if work.collector_enabled && work.examined < limit { + let remaining = limit - work.examined; + let candidates = self.gc_candidates(state.cursor, remaining).await?; + let reached_end = candidates.len() < usize::from(remaining); + for (candidate, eligible) in candidates { + work.examined += 1; + work.lifetimes_examined += 1; + state.cursor = Some((candidate.page_id, candidate.generation)); + if eligible { + let seal = self.claim_gc_candidate(candidate, &mut state).await?; + work.gc_claimed += 1; + let changed = self.apply_gc_seal(&seal, &mut state).await?; + work.record_applied(&seal, changed)?; + } + } + if reached_end { + state.cursor = None; + } + } + Ok(work) + } +} + +impl RootedMaintenanceWork { + fn record_applied(&mut self, seal: &GcSeal, changed: bool) -> Result<(), SnapshotError> { + self.gc_applied += 1; + if changed && seal.had_payload { + self.payload_pages_removed += 1; + self.payload_bytes_removed += u64::try_from(seal.expected_size).map_err(internal)?; + } + Ok(()) + } +} diff --git a/src/jupiter/storage/qualified_metadata_gc.sql b/src/jupiter/storage/qualified_metadata_gc.sql new file mode 100644 index 00000000..61276556 --- /dev/null +++ b/src/jupiter/storage/qualified_metadata_gc.sql @@ -0,0 +1,574 @@ +-- Historical certificates and source/session records do not own live graph rows. +ALTER TABLE mst2_metadata_gc_op ADD COLUMN certificate_digest bytea, + ADD CONSTRAINT mst2_gc_certificate_binding CHECK( + (graph_present AND octet_length(certificate_digest)=32 OR NOT graph_present AND certificate_digest IS NULL) IS TRUE); +ALTER TABLE mst2_metadata_prepare ADD COLUMN orphan_expires_at timestamptz NOT NULL; +CREATE INDEX mst2_metadata_prepare_orphan_scan ON mst2_metadata_prepare(orphan_expires_at,prepare_id) + WHERE state='PREPARING' OR state='COMMITTED' AND coverage_retired_at IS NULL; +CREATE INDEX mst2_metadata_current_active_scan ON mst2_metadata_lifetime(page_id,generation) + WHERE state IN ('RESERVED','LIVE'); + +CREATE FUNCTION mst2_metadata_prepare_expiry_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + IF TG_OP='INSERT' THEN NEW.created_at:=clock_timestamp(); NEW.orphan_expires_at:=NEW.created_at+interval '3600 seconds'; + ELSIF NEW.created_at IS DISTINCT FROM OLD.created_at OR NEW.orphan_expires_at IS DISTINCT FROM OLD.orphan_expires_at THEN + RAISE EXCEPTION 'qualified preparation expiry is immutable database evidence'; END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_metadata_prepare_00_expiry_guard BEFORE INSERT OR UPDATE ON mst2_metadata_prepare + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_prepare_expiry_guard(); + +CREATE FUNCTION mst2_metadata_orphan_prepare_eligible(pid text) RETURNS boolean LANGUAGE sql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ + SELECT EXISTS(SELECT 1 FROM mst2_metadata_prepare q WHERE q.prepare_id=$1 AND q.state='COMMITTED' + AND q.plan_kind='ROOTED' AND q.graph_domain='qualified-v1' AND mst2_metadata_scope_matches(q.primary_scope) + AND q.orphan_expires_at=q.created_at+interval '3600 seconds' AND q.orphan_expires_at<=clock_timestamp() + AND NOT EXISTS(SELECT 1 FROM mst2_qualified_session_incarnation s WHERE s.state='READY' + AND (s.prepare_id=q.prepare_id OR s.namespace_uuid='$NAMESPACE_UUID$'::uuid AND s.metadata_root=q.metadata_root + AND s.source_profile=convert_to($CORE_SCHEMA$.mst2_route_profile(q.source_domain,q.tagged_root_tree_oid,q.scope, + q.schema_version,q.metadata_codec,q.materialization_policy,q.fs_semantics,q.access_projection, + q.verification_revision,q.projection_revision)::text,'UTF8') + AND s.root_generation IN (SELECT generation FROM mst2_metadata_prepare_page m + WHERE m.prepare_id=q.prepare_id AND m.page_id=q.metadata_root UNION ALL + SELECT root_generation FROM mst2_metadata_prepare_reuse_root r WHERE r.prepare_id=q.prepare_id AND r.root_page=q.metadata_root))) + AND NOT EXISTS(SELECT 1 FROM mst2_qualified_lease_binding l WHERE l.prepare_id=q.prepare_id AND l.state='ACTIVE') + AND NOT EXISTS(SELECT 1 FROM mst2_metadata_reader_operation r JOIN mst2_qualified_lease_binding l USING(lease_id) + WHERE l.prepare_id=q.prepare_id AND r.state='ACTIVE')) +$$; +CREATE FUNCTION mst2_metadata_orphan_prepare_retired(pid text) RETURNS boolean LANGUAGE sql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ + SELECT mst2_metadata_orphan_prepare_eligible($1) AND EXISTS(SELECT 1 FROM mst2_metadata_prepare q + WHERE q.prepare_id=$1 AND q.coverage_retired_at>=q.orphan_expires_at) + AND NOT EXISTS(SELECT 1 FROM mst2_metadata_root_anchor WHERE prepare_id=$1 AND anchor_kind IN ('PREPARE','REUSE')) +$$; + +CREATE FUNCTION mst2_metadata_gc_enabled() RETURNS boolean LANGUAGE sql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ + SELECT NOT pg_is_in_recovery() AND current_setting('transaction_isolation')='read committed' + AND EXISTS(SELECT 1 FROM $CORE_SCHEMA$.mst2_metadata_namespace n WHERE n.namespace_uuid='$NAMESPACE_UUID$'::uuid + AND n.graph_domain='qualified-v1' AND n.metadata_schema=$Q_LITERAL$ AND n.metadata_schema_oid='$Q_OID$'::oid + AND n.core_schema_oid=$CORE_OID$ AND n.metadata_storage_uuid='$STORAGE_UUID$' + AND n.admission_state='ROOTED_Q_ADMITTED' AND n.collector_state='ENABLED' + AND n.implementation_fingerprint=decode('$IMPLEMENTATION_SHA$','hex') + AND n.catalog_fingerprint=$CORE_SCHEMA$.mst2_route_family_catalog($CORE_OID$,'$Q_OID$'::oid)) +$$; +CREATE FUNCTION mst2_metadata_gc_enter() RETURNS void LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + PERFORM $CORE_SCHEMA$.mst2_route_enter($CORE_LITERAL$); + IF NOT mst2_metadata_gc_enabled() THEN RAISE EXCEPTION 'qualified collector is not independently admitted'; END IF; +END $$; + +CREATE FUNCTION mst2_metadata_assert_uncovered(p bytea,g bigint) RETURNS void LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + IF EXISTS(SELECT 1 FROM mst2_metadata_root_anchor WHERE root_page=p AND root_generation=g) + OR EXISTS(SELECT 1 FROM mst2_metadata_graph_root WHERE page_id=p AND generation=g) + OR EXISTS(SELECT 1 FROM mst2_metadata_graph_edge WHERE child_page=p AND child_generation=g) + OR EXISTS(SELECT 1 FROM mst2_metadata_prepare_page m JOIN mst2_metadata_prepare q USING(prepare_id) + WHERE m.page_id=p AND m.generation=g AND (q.state='PREPARING' OR q.state='COMMITTED' AND q.coverage_retired_at IS NULL)) + OR EXISTS(SELECT 1 FROM mst2_metadata_prepare_reuse_root r JOIN mst2_metadata_prepare q USING(prepare_id) + WHERE r.root_page=p AND r.root_generation=g AND (q.state='PREPARING' OR q.state='COMMITTED' AND q.coverage_retired_at IS NULL)) THEN + RAISE EXCEPTION 'qualified current lifetime still has exact owned coverage or incoming edges'; END IF; +END $$; + +CREATE FUNCTION mst2_metadata_gc_proof(p bytea,g bigint,stage text,graph_present boolean,payload_present boolean,c bytea) +RETURNS void LANGUAGE plpgsql VOLATILE SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE life mst2_metadata_lifetime%ROWTYPE; node mst2_metadata_graph_node%ROWTYPE; + body mst2_metadata_payload%ROWTYPE; certificate mst2_metadata_page_certificate%ROWTYPE; actual boolean; +BEGIN + IF p IS NULL OR octet_length(p)<>32 OR g IS NULL OR g<=0 OR stage IS NULL + OR graph_present IS NULL OR payload_present IS NULL OR graph_present AND c IS NULL THEN + RAISE EXCEPTION 'qualified GC proof has an incomplete exact identity'; END IF; + SELECT * INTO life FROM mst2_metadata_lifetime WHERE page_id=p AND generation=g FOR UPDATE; + IF NOT FOUND OR life.graph_domain<>'qualified-v1' OR life.metadata_codec<>1 OR stage NOT IN ('CLAIM','PENDING','APPLIED') + OR stage='CLAIM' AND life.state NOT IN ('RESERVED','LIVE') OR stage='PENDING' AND life.state<>'DELETING' + OR stage='APPLIED' AND life.state<>'REMOVED' THEN RAISE EXCEPTION 'qualified GC stage is not its exact historical lifetime'; END IF; + IF stage<>'APPLIED' AND NOT EXISTS(SELECT 1 FROM mst2_metadata_current WHERE page_id=p AND generation=g) THEN + RAISE EXCEPTION 'qualified GC does not name its exact current generation'; END IF; + PERFORM mst2_metadata_assert_uncovered(p,g); + SELECT * INTO node FROM mst2_metadata_graph_node WHERE page_id=p AND generation=g FOR UPDATE; + actual:=FOUND; + IF actual IS DISTINCT FROM graph_present OR actual AND (node.incoming_refs<>0 OR node.metadata_codec<>life.metadata_codec + OR node.bytes<>life.expected_size OR node.certificate_digest IS DISTINCT FROM c + OR stage='CLAIM' AND node.state<>'LIVE' OR stage='PENDING' AND node.state<>'DELETING') THEN + RAISE EXCEPTION 'qualified GC graph identity or actual counter changed'; END IF; + SELECT * INTO body FROM mst2_metadata_payload WHERE page_id=p AND generation=g FOR UPDATE; + actual:=FOUND; + IF actual IS DISTINCT FROM payload_present OR actual AND (body.metadata_codec<>life.metadata_codec + OR body.byte_size<>life.expected_size OR octet_length(body.payload)<>body.byte_size + OR sha256(convert_to('mega.mst2.metapage','UTF8')||decode('00','hex')||body.payload)<>p) THEN + RAISE EXCEPTION 'qualified GC durable bytes differ from their exact lifetime and page digest'; END IF; + IF payload_present THEN PERFORM mst2_metadata_decode_local(body.payload); END IF; + IF c IS NOT NULL THEN + SELECT * INTO certificate FROM mst2_metadata_page_certificate WHERE page_id=p AND generation=g AND certificate_digest=c; + IF NOT FOUND OR certificate.namespace_uuid<>'$NAMESPACE_UUID$'::uuid OR certificate.metadata_codec<>life.metadata_codec + OR certificate.byte_size<>life.expected_size OR certificate.proof_revision<>1 + OR (SELECT count(*) FROM (SELECT 1 FROM mst2_metadata_verified_ref WHERE parent_page=p AND parent_generation=g LIMIT 258) refs) + <>jsonb_array_length(certificate.canonical_proof->'references') THEN + RAISE EXCEPTION 'qualified GC lost its immutable canonical certificate and typed reference evidence'; END IF; + ELSIF graph_present OR EXISTS(SELECT 1 FROM mst2_metadata_page_certificate WHERE page_id=p AND generation=g) THEN + RAISE EXCEPTION 'qualified GC cannot omit a certified graph identity'; END IF; + IF graph_present THEN + IF (SELECT count(*) FROM (SELECT 1 FROM mst2_metadata_graph_edge WHERE parent_page=p AND parent_generation=g LIMIT 258) edges)>257 THEN + RAISE EXCEPTION 'qualified canonical node has too many physical outgoing edges'; END IF; + IF payload_present AND (EXISTS((SELECT child_page,child_generation FROM mst2_metadata_verified_ref WHERE parent_page=p AND parent_generation=g) + EXCEPT (SELECT child_page,child_generation FROM mst2_metadata_graph_edge WHERE parent_page=p AND parent_generation=g)) + OR EXISTS((SELECT child_page,child_generation FROM mst2_metadata_graph_edge WHERE parent_page=p AND parent_generation=g) + EXCEPT (SELECT child_page,child_generation FROM mst2_metadata_verified_ref WHERE parent_page=p AND parent_generation=g))) THEN + RAISE EXCEPTION 'qualified GC physical outgoing edges differ from exact typed occurrences'; END IF; + END IF; + IF stage='CLAIM' AND life.state='LIVE' AND NOT (graph_present AND payload_present AND c IS NOT NULL) THEN + RAISE EXCEPTION 'LIVE metadata corruption cannot be collected'; END IF; + IF stage='CLAIM' AND graph_present AND NOT payload_present THEN + RAISE EXCEPTION 'certified RESERVED graph corruption cannot be collected without its durable bytes'; END IF; + IF stage='CLAIM' AND life.state='RESERVED' AND (NOT EXISTS(SELECT 1 FROM mst2_metadata_prepare_page m + JOIN mst2_metadata_prepare q USING(prepare_id) WHERE m.page_id=p AND m.generation=g AND m.expected_size=life.expected_size + AND q.graph_domain='qualified-v1' AND q.state='ABORTED' AND q.storage_seal IS NOT NULL) + OR EXISTS(SELECT 1 FROM mst2_metadata_prepare_page m JOIN mst2_metadata_prepare q USING(prepare_id) + WHERE m.page_id=p AND m.generation=g AND q.state<>'ABORTED')) THEN + RAISE EXCEPTION 'RESERVED metadata requires an explicitly aborted exact origin'; END IF; + IF stage='APPLIED' AND (EXISTS(SELECT 1 FROM mst2_metadata_graph_edge WHERE parent_page=p AND parent_generation=g) + OR EXISTS(SELECT 1 FROM mst2_metadata_reuse_index WHERE root_page=p AND root_generation=g)) THEN + RAISE EXCEPTION 'APPLIED qualified receipt still has old physical edges or reuse hints'; END IF; +END $$; + +CREATE FUNCTION mst2_metadata_gc_op_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE life mst2_metadata_lifetime%ROWTYPE; +BEGIN + IF TG_OP='DELETE' THEN RAISE EXCEPTION 'qualified GC operation history is immutable'; END IF; + PERFORM mst2_metadata_gc_enter(); + IF NOT mst2_metadata_scope_matches(NEW.primary_scope) THEN RAISE EXCEPTION 'qualified GC left its captured primary scope'; END IF; + IF TG_OP='INSERT' THEN + IF NEW.state<>'PENDING' OR NEW.completed_at IS NOT NULL OR NEW.payload_delete_xid IS NOT NULL + OR substr(NEW.operation_id::text,15,1)<>'4' OR substr(NEW.operation_id::text,20,1) NOT IN ('8','9','a','b') THEN + RAISE EXCEPTION 'qualified GC must begin with a fresh UUID and unmodified PENDING proof'; END IF; + PERFORM mst2_metadata_gc_proof(NEW.page_id,NEW.generation,'CLAIM',NEW.graph_present,NEW.had_payload,NEW.certificate_digest); + SELECT * INTO STRICT life FROM mst2_metadata_lifetime WHERE page_id=NEW.page_id AND generation=NEW.generation; + IF ROW(NEW.graph_domain,NEW.metadata_codec,NEW.expected_size) IS DISTINCT FROM + ROW(life.graph_domain,life.metadata_codec,life.expected_size) THEN RAISE EXCEPTION 'qualified GC copied profile is incorrect'; END IF; + NEW.created_at:=clock_timestamp(); + ELSE + IF (to_jsonb(NEW)-ARRAY['state','completed_at','payload_delete_xid']) IS DISTINCT FROM + (to_jsonb(OLD)-ARRAY['state','completed_at','payload_delete_xid']) THEN RAISE EXCEPTION 'qualified GC operation cannot retarget immutable evidence'; END IF; + IF OLD.state='APPLIED' AND NEW IS DISTINCT FROM OLD THEN RAISE EXCEPTION 'qualified GC receipt cannot revive or change'; END IF; + IF NEW.payload_delete_xid IS DISTINCT FROM OLD.payload_delete_xid THEN + IF OLD.state<>'PENDING' OR NEW.state<>'PENDING' OR NOT OLD.had_payload OR OLD.payload_delete_xid IS NOT NULL THEN + RAISE EXCEPTION 'qualified payload deletion has no fresh same-transaction phase'; END IF; + PERFORM mst2_metadata_gc_proof(OLD.page_id,OLD.generation,'PENDING',OLD.graph_present,true,OLD.certificate_digest); + NEW.payload_delete_xid:=txid_current(); + END IF; + IF OLD.state='PENDING' AND NEW.state='APPLIED' THEN + IF OLD.had_payload AND OLD.payload_delete_xid IS DISTINCT FROM txid_current() THEN + RAISE EXCEPTION 'qualified GC completion has no same-transaction payload removal'; END IF; + PERFORM mst2_metadata_gc_proof(OLD.page_id,OLD.generation,'APPLIED',false,false,OLD.certificate_digest); + NEW.completed_at:=clock_timestamp(); + ELSIF NEW.state IS DISTINCT FROM OLD.state OR NEW.completed_at IS DISTINCT FROM OLD.completed_at THEN + RAISE EXCEPTION 'qualified GC state transition is invalid'; END IF; + END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_metadata_gc_op_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_metadata_gc_op + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_gc_op_guard(); + +CREATE FUNCTION mst2_metadata_gc_complete() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE op mst2_metadata_gc_op%ROWTYPE; +BEGIN + SELECT * INTO STRICT op FROM mst2_metadata_gc_op WHERE operation_id=NEW.operation_id; + IF op.state='PENDING' THEN + IF op.payload_delete_xid IS NOT NULL THEN RAISE EXCEPTION 'qualified payload deletion phase cannot escape its atomic apply transaction'; END IF; + PERFORM mst2_metadata_gc_proof(op.page_id,op.generation,'PENDING',op.graph_present,op.had_payload,op.certificate_digest); + ELSE PERFORM mst2_metadata_gc_proof(op.page_id,op.generation,'APPLIED',false,false,op.certificate_digest); END IF; + RETURN NULL; +END $$; +CREATE CONSTRAINT TRIGGER mst2_metadata_gc_complete AFTER INSERT OR UPDATE ON mst2_metadata_gc_op + DEFERRABLE INITIALLY DEFERRED FOR EACH ROW EXECUTE FUNCTION mst2_metadata_gc_complete(); + +CREATE OR REPLACE FUNCTION mst2_metadata_lifetime_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE current_root mst2_metadata_current%ROWTYPE; op mst2_metadata_gc_op%ROWTYPE; previous mst2_metadata_lifetime%ROWTYPE; +BEGIN + IF TG_OP='DELETE' THEN RAISE EXCEPTION 'qualified lifetime watermarks are immutable'; END IF; + IF TG_OP='INSERT' THEN + IF NEW.state<>'RESERVED' THEN RAISE EXCEPTION 'qualified lifetime must begin RESERVED'; END IF; + SELECT * INTO current_root FROM mst2_metadata_current WHERE page_id=NEW.page_id; + IF FOUND THEN + SELECT * INTO previous FROM mst2_metadata_lifetime WHERE page_id=NEW.page_id AND generation=current_root.generation; + SELECT * INTO op FROM mst2_metadata_gc_op WHERE page_id=NEW.page_id AND generation=current_root.generation AND state='APPLIED'; + IF NOT FOUND OR previous.state<>'REMOVED' OR current_root.generation=9223372036854775807 + OR NEW.generation<>current_root.generation+1 OR NEW.metadata_codec<>op.metadata_codec OR NEW.expected_size<>op.expected_size + OR NOT mst2_metadata_scope_matches(op.primary_scope) THEN RAISE EXCEPTION 'qualified reservation needs its exact next-generation APPLIED predecessor'; END IF; + PERFORM mst2_metadata_gc_proof(NEW.page_id,current_root.generation,'APPLIED',false,false,op.certificate_digest); + ELSIF NEW.generation<>1 OR EXISTS(SELECT 1 FROM mst2_metadata_lifetime WHERE page_id=NEW.page_id) + OR EXISTS(SELECT 1 FROM mst2_metadata_payload WHERE page_id=NEW.page_id) + OR EXISTS(SELECT 1 FROM mst2_metadata_graph_node WHERE page_id=NEW.page_id) THEN + RAISE EXCEPTION 'qualified initial reservation cannot adopt history'; END IF; + RETURN NEW; + END IF; + IF (to_jsonb(NEW)-'state') IS DISTINCT FROM (to_jsonb(OLD)-'state') THEN RAISE EXCEPTION 'qualified lifetime identity is immutable'; END IF; + IF NEW.state=OLD.state THEN RETURN NEW; END IF; + IF OLD.state='RESERVED' AND NEW.state='LIVE' THEN + IF NOT EXISTS(SELECT 1 FROM mst2_metadata_graph_node node JOIN mst2_metadata_payload body USING(page_id,generation) + JOIN mst2_metadata_page_certificate certificate USING(page_id,generation) + WHERE node.page_id=OLD.page_id AND node.generation=OLD.generation AND node.state='LIVE' + AND node.certificate_digest=certificate.certificate_digest AND node.metadata_codec=OLD.metadata_codec + AND body.metadata_codec=OLD.metadata_codec AND body.byte_size=OLD.expected_size AND node.bytes=OLD.expected_size) + OR NOT (EXISTS(SELECT 1 FROM mst2_metadata_graph_root root JOIN mst2_metadata_prepare q USING(prepare_id) + WHERE root.page_id=OLD.page_id AND root.generation=OLD.generation AND q.plan_kind='COLD' + AND q.state='COMMITTED' AND root.storage_seal=q.storage_seal AND q.coverage_retired_at IS NULL) + OR EXISTS(SELECT 1 FROM mst2_metadata_prepare_page m JOIN mst2_metadata_prepare q USING(prepare_id) + JOIN mst2_metadata_root_anchor anchor ON anchor.prepare_id=q.prepare_id AND anchor.anchor_kind='PREPARE' + AND anchor.root_page=q.metadata_root AND anchor.owner_key=q.prepare_id + WHERE m.page_id=OLD.page_id AND m.generation=OLD.generation AND q.plan_kind='ROOTED' + AND q.state='COMMITTED' AND q.coverage_retired_at IS NULL)) THEN + RAISE EXCEPTION 'qualified LIVE transition lacks its certified graph and exact owned preparation'; END IF; + RETURN NEW; + END IF; + PERFORM mst2_metadata_gc_enter(); + SELECT * INTO op FROM mst2_metadata_gc_op WHERE page_id=OLD.page_id AND generation=OLD.generation AND state='PENDING'; + IF NOT FOUND OR NOT mst2_metadata_scope_matches(op.primary_scope) THEN RAISE EXCEPTION 'qualified lifecycle transition lacks its exact pending operation'; END IF; + IF OLD.state IN ('RESERVED','LIVE') AND NEW.state='DELETING' THEN + PERFORM mst2_metadata_gc_proof(OLD.page_id,OLD.generation,'CLAIM',op.graph_present,op.had_payload,op.certificate_digest); + ELSIF OLD.state='DELETING' AND NEW.state='REMOVED' THEN + IF op.had_payload AND op.payload_delete_xid IS DISTINCT FROM txid_current() THEN RAISE EXCEPTION 'qualified removal has no atomic payload deletion'; END IF; + PERFORM mst2_metadata_gc_proof(OLD.page_id,OLD.generation,'PENDING',false,false,op.certificate_digest); + IF EXISTS(SELECT 1 FROM mst2_metadata_graph_edge WHERE parent_page=OLD.page_id AND parent_generation=OLD.generation) + OR EXISTS(SELECT 1 FROM mst2_metadata_reuse_index WHERE root_page=OLD.page_id AND root_generation=OLD.generation) THEN + RAISE EXCEPTION 'qualified removal still has old physical references'; END IF; + ELSE RAISE EXCEPTION 'qualified lifecycle transition is invalid'; END IF; + RETURN NEW; +END $$; + +CREATE OR REPLACE FUNCTION mst2_metadata_current_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE op mst2_metadata_gc_op%ROWTYPE; life mst2_metadata_lifetime%ROWTYPE; +BEGIN + IF TG_OP='DELETE' THEN RAISE EXCEPTION 'qualified current watermark cannot be deleted'; END IF; + IF TG_OP='INSERT' THEN + IF NEW.generation<>1 OR EXISTS(SELECT 1 FROM mst2_metadata_lifetime WHERE page_id=NEW.page_id AND generation<>1) + OR EXISTS(SELECT 1 FROM mst2_metadata_payload WHERE page_id=NEW.page_id) + OR EXISTS(SELECT 1 FROM mst2_metadata_graph_node WHERE page_id=NEW.page_id) THEN + RAISE EXCEPTION 'qualified initial current cannot adopt history'; END IF; + RETURN NEW; + END IF; + IF NEW.page_id<>OLD.page_id OR OLD.generation=9223372036854775807 OR NEW.generation<>OLD.generation+1 THEN + RAISE EXCEPTION 'qualified current requires exact next-generation CAS'; END IF; + SELECT * INTO op FROM mst2_metadata_gc_op WHERE page_id=OLD.page_id AND generation=OLD.generation AND state='APPLIED'; + IF NOT FOUND OR NOT mst2_metadata_scope_matches(op.primary_scope) + OR EXISTS(SELECT 1 FROM mst2_metadata_payload WHERE page_id=OLD.page_id) + OR EXISTS(SELECT 1 FROM mst2_metadata_graph_node WHERE page_id=OLD.page_id AND generation=OLD.generation) THEN + RAISE EXCEPTION 'qualified current has no definitive old-generation removal'; END IF; + PERFORM mst2_metadata_gc_proof(OLD.page_id,OLD.generation,'APPLIED',false,false,op.certificate_digest); + SELECT * INTO life FROM mst2_metadata_lifetime WHERE page_id=NEW.page_id AND generation=NEW.generation; + IF NOT FOUND OR life.state<>'RESERVED' OR life.graph_domain<>'qualified-v1' + OR life.metadata_codec<>op.metadata_codec OR life.expected_size<>op.expected_size THEN + RAISE EXCEPTION 'qualified fresh current differs from its immutable page profile'; END IF; + RETURN NEW; +END $$; + +CREATE FUNCTION mst2_metadata_capacity() RETURNS TABLE(resident_pages bigint,resident_bytes bigint) LANGUAGE sql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ + SELECT count(*)::bigint,coalesce(sum(byte_size),0)::bigint FROM (SELECT byte_size FROM mst2_metadata_payload LIMIT 16385) actual +$$; +CREATE FUNCTION mst2_metadata_capacity_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE pages bigint; bytes bigint; actual bigint; +BEGIN + IF TG_TABLE_NAME='mst2_metadata_payload' THEN + SELECT resident_pages,resident_bytes INTO pages,bytes FROM mst2_metadata_capacity(); + IF pages>16384 OR bytes>268435456 THEN RAISE EXCEPTION 'qualified resident payload capacity is exceeded'; END IF; + ELSIF TG_TABLE_NAME='mst2_metadata_source_entry_reference' THEN + SELECT count(*) INTO actual FROM (SELECT 1 FROM mst2_metadata_source_entry_reference LIMIT 262145) bounded; + IF actual>262144 THEN RAISE EXCEPTION 'qualified retained source dictionary capacity is exceeded'; END IF; + RETURN NULL; + ELSIF TG_TABLE_NAME='mst2_qualified_session_incarnation' THEN + SELECT count(*) INTO actual FROM (SELECT 1 FROM mst2_qualified_session_incarnation WHERE state='READY' LIMIT 4097) bounded; + ELSIF TG_TABLE_NAME='mst2_qualified_lease_binding' THEN + SELECT count(*) INTO actual FROM (SELECT 1 FROM mst2_qualified_lease_binding WHERE state='ACTIVE' LIMIT 4097) bounded; + ELSE SELECT count(*) INTO actual FROM (SELECT 1 FROM mst2_metadata_reader_operation WHERE state='ACTIVE' LIMIT 4097) bounded; END IF; + IF actual>4096 THEN RAISE EXCEPTION 'qualified active serving capacity is exceeded'; END IF; + RETURN NULL; +END $$; +DO $$ DECLARE name text; BEGIN + FOREACH name IN ARRAY ARRAY['mst2_metadata_payload','mst2_qualified_session_incarnation', + 'mst2_qualified_lease_binding','mst2_metadata_reader_operation','mst2_metadata_source_entry_reference'] LOOP + EXECUTE format('CREATE TRIGGER mst2_metadata_capacity_guard AFTER INSERT OR UPDATE ON %I FOR EACH STATEMENT EXECUTE FUNCTION mst2_metadata_capacity_guard()',name); + END LOOP; +END $$; + +-- Permanent identity evidence and bounded reader owners have independent +-- resident quotas. Admission reads actual stored rows, never hints. +CREATE FUNCTION mst2_metadata_history_capacity_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE actual bigint; bytes bigint; row_limit integer; size_expression text; +BEGIN + IF TG_TABLE_NAME IN ('mst2_metadata_prepare_page','mst2_metadata_prepare_reuse_root','mst2_metadata_verified_ref') THEN + row_limit:=262144; size_expression:='0'; + ELSIF TG_TABLE_NAME='mst2_metadata_prepare' THEN + row_limit:=16384; + size_expression:='pg_column_size(canonical_plan)::bigint+pg_column_size(canonical_bindings)+pg_column_size(primary_scope)'; + ELSIF TG_TABLE_NAME='mst2_metadata_source_root_attestation' THEN + row_limit:=16384; size_expression:='pg_column_size(source_proof)::bigint+pg_column_size(source_profile)'; + ELSIF TG_TABLE_NAME='mst2_metadata_page_certificate' THEN + row_limit:=16384; size_expression:='pg_column_size(canonical_proof)::bigint'; + ELSIF TG_TABLE_NAME='mst2_metadata_scope_source_reference' THEN + row_limit:=16384; size_expression:='pg_column_size(ancestor_revisions)::bigint'; + ELSIF TG_TABLE_NAME='mst2_qualified_session_incarnation' THEN + row_limit:=65536; size_expression:='pg_column_size(canonical_descriptor)::bigint+pg_column_size(source_profile)'; + ELSIF TG_TABLE_NAME IN ('mst2_metadata_lifetime','mst2_metadata_current','mst2_metadata_gc_op', + 'mst2_qualified_lease_binding','mst2_metadata_reader_operation') THEN + row_limit:=65536; size_expression:='0'; + ELSE RAISE EXCEPTION 'qualified history quota has an unknown physical relation'; END IF; + EXECUTE format('SELECT count(*),coalesce(sum(size),0) FROM (SELECT %s AS size FROM %I.%I LIMIT %s) actual', + size_expression,TG_TABLE_SCHEMA,TG_TABLE_NAME,row_limit+1) INTO actual,bytes; + IF actual>row_limit OR bytes>268435456 THEN + RAISE EXCEPTION 'qualified immutable history capacity is exceeded for %',TG_TABLE_NAME; END IF; + RETURN NULL; +END $$; +DO $$ DECLARE name text; BEGIN + FOREACH name IN ARRAY ARRAY['mst2_metadata_prepare','mst2_metadata_prepare_page','mst2_metadata_prepare_reuse_root', + 'mst2_metadata_verified_ref','mst2_metadata_source_root_attestation','mst2_metadata_page_certificate','mst2_metadata_scope_source_reference', + 'mst2_qualified_session_incarnation','mst2_metadata_lifetime','mst2_metadata_current','mst2_metadata_gc_op', + 'mst2_qualified_lease_binding','mst2_metadata_reader_operation'] LOOP + EXECUTE format('CREATE TRIGGER mst2_metadata_history_capacity_guard AFTER INSERT ON %I FOR EACH STATEMENT EXECUTE FUNCTION mst2_metadata_history_capacity_guard()',name); + END LOOP; +END $$; + +CREATE FUNCTION mst2_metadata_gc_claim(p bytea,g bigint,scope bytea,id uuid) RETURNS uuid LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE life mst2_metadata_lifetime%ROWTYPE; node mst2_metadata_graph_node%ROWTYPE; op mst2_metadata_gc_op%ROWTYPE; + has_graph boolean; has_body boolean; +BEGIN + PERFORM mst2_metadata_gc_enter(); + IF NOT mst2_metadata_scope_matches(scope) THEN RAISE EXCEPTION 'qualified claim scope differs from its captured primary'; END IF; + SELECT * INTO op FROM mst2_metadata_gc_op WHERE operation_id=id; + IF FOUND THEN + IF op.page_id IS DISTINCT FROM p OR op.generation<>g OR op.primary_scope IS DISTINCT FROM scope THEN + RAISE EXCEPTION 'qualified GC operation cannot be reused for a different lifetime'; END IF; + RETURN id; + END IF; + SELECT * INTO STRICT life FROM mst2_metadata_lifetime WHERE page_id=p AND generation=g; + SELECT * INTO node FROM mst2_metadata_graph_node WHERE page_id=p AND generation=g; + has_graph:=FOUND; has_body:=EXISTS(SELECT 1 FROM mst2_metadata_payload WHERE page_id=p AND generation=g); + INSERT INTO mst2_metadata_gc_op(operation_id,page_id,generation,primary_scope,graph_domain,metadata_codec, + expected_size,graph_present,had_payload,certificate_digest,state) + VALUES(id,p,g,scope,'qualified-v1',life.metadata_codec,life.expected_size,has_graph,has_body, + CASE WHEN has_graph THEN node.certificate_digest ELSE NULL END,'PENDING'); + UPDATE mst2_metadata_lifetime SET state='DELETING' WHERE page_id=p AND generation=g; + IF has_graph THEN UPDATE mst2_metadata_graph_node SET state='DELETING' WHERE page_id=p AND generation=g; END IF; + RETURN id; +END $$; + +CREATE OR REPLACE FUNCTION mst2_metadata_gc_apply(id uuid) RETURNS void LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE op mst2_metadata_gc_op%ROWTYPE; +BEGIN + PERFORM mst2_metadata_gc_enter(); + SELECT * INTO STRICT op FROM mst2_metadata_gc_op WHERE operation_id=id FOR UPDATE; + IF NOT mst2_metadata_scope_matches(op.primary_scope) THEN RAISE EXCEPTION 'qualified GC replay left its captured primary'; END IF; + IF op.state='APPLIED' THEN + PERFORM mst2_metadata_gc_proof(op.page_id,op.generation,'APPLIED',false,false,op.certificate_digest); RETURN; + END IF; + PERFORM mst2_metadata_gc_proof(op.page_id,op.generation,'PENDING',op.graph_present,op.had_payload,op.certificate_digest); + IF op.had_payload THEN + UPDATE mst2_metadata_gc_op SET payload_delete_xid=txid_current() WHERE operation_id=id; + DELETE FROM mst2_metadata_payload WHERE page_id=op.page_id AND generation=op.generation; + IF NOT FOUND THEN RAISE EXCEPTION 'qualified atomic payload deletion lost its exact row'; END IF; + END IF; + DELETE FROM mst2_metadata_graph_edge WHERE parent_page=op.page_id AND parent_generation=op.generation; + DELETE FROM mst2_metadata_reuse_index WHERE root_page=op.page_id AND root_generation=op.generation; + DELETE FROM mst2_metadata_graph_node WHERE page_id=op.page_id AND generation=op.generation; + UPDATE mst2_metadata_lifetime SET state='REMOVED' WHERE page_id=op.page_id AND generation=op.generation AND state='DELETING'; + IF NOT FOUND THEN RAISE EXCEPTION 'qualified atomic removal lost its exact lifecycle'; END IF; + UPDATE mst2_metadata_gc_op SET state='APPLIED' WHERE operation_id=id; +END $$; +CREATE OR REPLACE FUNCTION mst2_metadata_gc_finish(id uuid) RETURNS void LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ BEGIN PERFORM mst2_metadata_gc_apply(id); END $$; + +-- Additional graph/payload guards follow; no history relation is collected. +CREATE OR REPLACE FUNCTION mst2_metadata_payload_fenced() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE op mst2_metadata_gc_op%ROWTYPE; +BEGIN + IF TG_OP='UPDATE' THEN RAISE EXCEPTION 'qualified immutable payload cannot be updated'; END IF; + IF TG_OP='DELETE' THEN + PERFORM mst2_metadata_gc_enter(); + SELECT * INTO op FROM mst2_metadata_gc_op WHERE page_id=OLD.page_id AND generation=OLD.generation AND state='PENDING'; + IF NOT FOUND OR NOT op.had_payload OR op.payload_delete_xid IS DISTINCT FROM txid_current() + OR NOT mst2_metadata_scope_matches(op.primary_scope) OR op.expected_size<>OLD.byte_size OR op.metadata_codec<>OLD.metadata_codec THEN + RAISE EXCEPTION 'qualified payload removal needs its exact same-transaction pending evidence'; END IF; + PERFORM mst2_metadata_gc_proof(OLD.page_id,OLD.generation,'PENDING',op.graph_present,true,op.certificate_digest); + RETURN OLD; + END IF; + IF NOT EXISTS(SELECT 1 FROM mst2_metadata_current c JOIN mst2_metadata_lifetime l USING(page_id,generation) + JOIN mst2_metadata_prepare_page m USING(page_id,generation) JOIN mst2_metadata_prepare q USING(prepare_id) + WHERE c.page_id=NEW.page_id AND c.generation=NEW.generation AND l.state IN ('RESERVED','LIVE') + AND l.metadata_codec=NEW.metadata_codec AND l.expected_size=NEW.byte_size AND q.state='PREPARING' + AND q.graph_domain='qualified-v1' AND mst2_metadata_scope_matches(q.primary_scope)) THEN + RAISE EXCEPTION 'qualified payload INSERT crossed its exact active generation'; END IF; + RETURN NEW; +END $$; + +CREATE OR REPLACE FUNCTION mst2_metadata_graph_node_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE life mst2_metadata_lifetime%ROWTYPE; op mst2_metadata_gc_op%ROWTYPE; +BEGIN + IF TG_OP='DELETE' THEN + PERFORM mst2_metadata_gc_enter(); + SELECT * INTO op FROM mst2_metadata_gc_op WHERE page_id=OLD.page_id AND generation=OLD.generation AND state='PENDING'; + IF NOT FOUND OR NOT op.graph_present OR op.certificate_digest IS DISTINCT FROM OLD.certificate_digest + OR OLD.state<>'DELETING' OR NOT mst2_metadata_scope_matches(op.primary_scope) + OR op.had_payload AND op.payload_delete_xid IS DISTINCT FROM txid_current() + OR EXISTS(SELECT 1 FROM mst2_metadata_graph_edge WHERE parent_page=OLD.page_id AND parent_generation=OLD.generation) THEN + RAISE EXCEPTION 'qualified graph removal lacks its atomic byte removal and empty outgoing edges'; END IF; + PERFORM mst2_metadata_gc_proof(OLD.page_id,OLD.generation,'PENDING',true,false,op.certificate_digest); + RETURN OLD; + END IF; + SELECT l.* INTO life FROM mst2_metadata_current c JOIN mst2_metadata_lifetime l USING(page_id,generation) + WHERE c.page_id=NEW.page_id AND c.generation=NEW.generation; + IF NOT FOUND OR life.graph_domain<>'qualified-v1' OR life.metadata_codec<>NEW.metadata_codec OR life.expected_size<>NEW.bytes + OR NEW.incoming_refs<>(SELECT count(*) FROM mst2_metadata_graph_edge WHERE child_page=NEW.page_id AND child_generation=NEW.generation) + OR NOT EXISTS(SELECT 1 FROM mst2_metadata_page_certificate certificate WHERE certificate.page_id=NEW.page_id + AND certificate.generation=NEW.generation AND certificate.certificate_digest=NEW.certificate_digest + AND certificate.metadata_codec=NEW.metadata_codec AND certificate.byte_size=NEW.bytes) THEN + RAISE EXCEPTION 'qualified graph identity, certificate or physical counter changed'; END IF; + IF TG_OP='INSERT' THEN + IF NEW.state<>'LIVE' OR life.state NOT IN ('RESERVED','LIVE') + OR NOT EXISTS(SELECT 1 FROM mst2_metadata_prepare_page m JOIN mst2_metadata_prepare q USING(prepare_id) + WHERE m.page_id=NEW.page_id AND m.generation=NEW.generation AND q.state='PREPARING') THEN + RAISE EXCEPTION 'qualified graph creation requires an active exact preparation'; END IF; + ELSE + IF (to_jsonb(NEW)-ARRAY['state','incoming_refs']) IS DISTINCT FROM (to_jsonb(OLD)-ARRAY['state','incoming_refs']) THEN + RAISE EXCEPTION 'qualified graph identity is immutable'; END IF; + IF NEW.state<>OLD.state THEN + PERFORM mst2_metadata_gc_enter(); + SELECT * INTO op FROM mst2_metadata_gc_op WHERE page_id=OLD.page_id AND generation=OLD.generation AND state='PENDING'; + IF NOT FOUND OR OLD.state<>'LIVE' OR NEW.state<>'DELETING' OR life.state<>'DELETING' OR NEW.incoming_refs<>0 + OR op.certificate_digest IS DISTINCT FROM NEW.certificate_digest OR NOT mst2_metadata_scope_matches(op.primary_scope) THEN + RAISE EXCEPTION 'qualified graph state change has no exact pending claim'; END IF; + PERFORM mst2_metadata_assert_uncovered(OLD.page_id,OLD.generation); + ELSIF NEW.state='LIVE' AND life.state NOT IN ('RESERVED','LIVE') OR NEW.state='DELETING' AND life.state<>'DELETING' THEN + RAISE EXCEPTION 'qualified graph state differs from its current lifecycle'; END IF; + END IF; + RETURN NEW; +END $$; + +CREATE OR REPLACE FUNCTION mst2_metadata_graph_edge_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE op mst2_metadata_gc_op%ROWTYPE; +BEGIN + IF TG_OP='UPDATE' THEN RAISE EXCEPTION 'qualified edge identity is immutable'; END IF; + IF TG_OP='DELETE' THEN + PERFORM mst2_metadata_gc_enter(); + SELECT * INTO op FROM mst2_metadata_gc_op WHERE page_id=OLD.parent_page AND generation=OLD.parent_generation AND state='PENDING'; + IF NOT FOUND OR NOT op.graph_present OR NOT mst2_metadata_scope_matches(op.primary_scope) + OR op.had_payload AND op.payload_delete_xid IS DISTINCT FROM txid_current() + OR EXISTS(SELECT 1 FROM mst2_metadata_payload WHERE page_id=OLD.parent_page AND generation=OLD.parent_generation) + OR NOT EXISTS(SELECT 1 FROM mst2_metadata_graph_node node JOIN mst2_metadata_current cur USING(page_id,generation) + JOIN mst2_metadata_lifetime life USING(page_id,generation) + WHERE node.page_id=OLD.parent_page AND node.generation=OLD.parent_generation AND node.state='DELETING' + AND life.state='DELETING' AND node.incoming_refs=0 AND node.certificate_digest=op.certificate_digest) THEN + RAISE EXCEPTION 'qualified edge removal is outside its exact atomic graph deletion phase'; END IF; + PERFORM mst2_metadata_assert_uncovered(OLD.parent_page,OLD.parent_generation); + RETURN OLD; + END IF; + IF NOT EXISTS(SELECT 1 FROM mst2_metadata_verified_ref ref + JOIN mst2_metadata_page_certificate parent ON parent.page_id=ref.parent_page AND parent.generation=ref.parent_generation + JOIN mst2_metadata_page_certificate child ON child.page_id=ref.child_page AND child.generation=ref.child_generation + WHERE ref.parent_page=NEW.parent_page AND ref.parent_generation=NEW.parent_generation + AND ref.child_page=NEW.child_page AND ref.child_generation=NEW.child_generation + AND ref.child_certificate_digest=child.certificate_digest AND parent.rank>child.rank) THEN + RAISE EXCEPTION 'qualified edge differs from its exact canonical reference or strict certified rank'; END IF; + IF (SELECT count(*) FROM mst2_metadata_graph_node n JOIN mst2_metadata_current c USING(page_id,generation) + JOIN mst2_metadata_lifetime l USING(page_id,generation) + WHERE ((n.page_id=NEW.parent_page AND n.generation=NEW.parent_generation) + OR (n.page_id=NEW.child_page AND n.generation=NEW.child_generation)) + AND n.state='LIVE' AND l.state IN ('RESERVED','LIVE') AND l.graph_domain='qualified-v1')<>2 + OR NOT EXISTS(SELECT 1 FROM mst2_metadata_prepare_page a JOIN mst2_metadata_prepare q USING(prepare_id) + WHERE a.page_id=NEW.parent_page AND a.generation=NEW.parent_generation AND q.state='PREPARING' + AND q.graph_domain='qualified-v1' AND mst2_metadata_scope_matches(q.primary_scope) + AND (EXISTS(SELECT 1 FROM mst2_metadata_prepare_page b WHERE b.prepare_id=a.prepare_id + AND b.page_id=NEW.child_page AND b.generation=NEW.child_generation) + OR EXISTS(SELECT 1 FROM mst2_metadata_prepare_reuse_root r JOIN mst2_metadata_root_anchor anchor + ON anchor.prepare_id=r.prepare_id AND anchor.anchor_kind='REUSE' AND anchor.owner_key=r.prepare_id + AND anchor.root_page=r.root_page AND anchor.root_generation=r.root_generation + WHERE r.prepare_id=a.prepare_id AND r.root_page=NEW.child_page AND r.root_generation=NEW.child_generation))) THEN + RAISE EXCEPTION 'edge creation requires active exact qualified endpoints'; END IF; + RETURN NEW; +END $$; + +CREATE FUNCTION mst2_metadata_edges_removed() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + IF (SELECT count(*) FROM removed_edges)>257 THEN RAISE EXCEPTION 'qualified GC edge deletion exceeds one canonical node'; END IF; + IF EXISTS(SELECT 1 FROM mst2_metadata_graph_node node JOIN (SELECT child_page,child_generation,count(*) AS delta + FROM removed_edges GROUP BY child_page,child_generation) removed + ON removed.child_page=node.page_id AND removed.child_generation=node.generation + WHERE node.incoming_refs-removed.delta<>(SELECT count(*) FROM mst2_metadata_graph_edge edge + WHERE edge.child_page=node.page_id AND edge.child_generation=node.generation)) THEN + RAISE EXCEPTION 'qualified removal discovered physical incoming counter drift'; END IF; + UPDATE mst2_metadata_graph_node node SET incoming_refs=node.incoming_refs-removed.delta + FROM (SELECT child_page,child_generation,count(*) AS delta FROM removed_edges GROUP BY child_page,child_generation) removed + WHERE node.page_id=removed.child_page AND node.generation=removed.child_generation; + RETURN NULL; +END $$; +CREATE TRIGGER mst2_metadata_edges_removed AFTER DELETE ON mst2_metadata_graph_edge + REFERENCING OLD TABLE AS removed_edges FOR EACH STATEMENT EXECUTE FUNCTION mst2_metadata_edges_removed(); + +CREATE FUNCTION mst2_metadata_gc_owner_cleanup(maximum integer) +RETURNS TABLE(examined bigint,readers_expired bigint,leases_expired bigint,prepares_aborted bigint, + handovers_retired bigint,orphans_retired bigint) +LANGUAGE plpgsql VOLATILE SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE bound integer:=maximum; item record; q mst2_metadata_prepare%ROWTYPE; now_unix bigint; +BEGIN + PERFORM $CORE_SCHEMA$.mst2_route_enter($CORE_LITERAL$); + IF maximum IS NULL OR maximum NOT BETWEEN 0 AND 64 THEN RAISE EXCEPTION 'qualified owner cleanup budget must be 0..=64'; END IF; + examined:=0; readers_expired:=0; leases_expired:=0; prepares_aborted:=0; handovers_retired:=0; orphans_retired:=0; + now_unix:=floor(extract(epoch FROM clock_timestamp()))::bigint; + FOR item IN SELECT * FROM ( + (SELECT 'READER'::text AS kind,operation_id::text AS owner,lease_id,hard_deadline_unix AS deadline,reader_issuance + FROM mst2_metadata_reader_operation WHERE state='ACTIVE' AND hard_deadline_unix<=now_unix + ORDER BY hard_deadline_unix,operation_id LIMIT bound) + UNION ALL + (SELECT 'LEASE'::text,lease_id,lease_id,expires_at_unix,NULL::bigint FROM mst2_qualified_lease_binding + WHERE state='ACTIVE' AND expires_at_unix<=now_unix ORDER BY expires_at_unix,lease_id LIMIT bound) + UNION ALL + (SELECT 'PREPARE'::text,prepare_id,NULL::text,floor(extract(epoch FROM orphan_expires_at))::bigint,NULL::bigint + FROM mst2_metadata_prepare WHERE (state='PREPARING' OR state='COMMITTED' AND coverage_retired_at IS NULL) + AND orphan_expires_at<=clock_timestamp() ORDER BY orphan_expires_at,prepare_id LIMIT bound) + ) expired ORDER BY deadline,kind,owner LIMIT bound LOOP + examined:=examined+1; + IF item.kind='READER' THEN + UPDATE mst2_metadata_reader_operation SET state='EXPIRED' WHERE operation_id=item.owner::uuid AND reader_issuance=item.reader_issuance AND state='ACTIVE'; + IF NOT FOUND THEN RAISE EXCEPTION 'qualified expired reader changed behind its mutation barrier'; END IF; + readers_expired:=readers_expired+1; + DELETE FROM mst2_metadata_root_anchor WHERE reader_operation_id=item.owner::uuid AND reader_issuance=item.reader_issuance AND anchor_kind IN ('REQUEST','READER'); + PERFORM mst2_metadata_cleanup_lease(item.lease_id); + ELSIF item.kind='LEASE' THEN + UPDATE mst2_qualified_lease_binding SET state='EXPIRED',lease_epoch=lease_epoch+1 WHERE lease_id=item.owner AND state='ACTIVE'; + IF NOT FOUND THEN RAISE EXCEPTION 'qualified expired lease changed behind its mutation barrier'; END IF; + leases_expired:=leases_expired+1; PERFORM mst2_metadata_cleanup_lease(item.lease_id); + ELSE + SELECT * INTO STRICT q FROM mst2_metadata_prepare WHERE prepare_id=item.owner; + IF q.state='PREPARING' THEN + DELETE FROM mst2_metadata_graph_root WHERE prepare_id=q.prepare_id; + UPDATE mst2_metadata_prepare SET state='ABORTED',aborted_at=clock_timestamp() WHERE prepare_id=q.prepare_id; + DELETE FROM mst2_metadata_root_anchor WHERE prepare_id=q.prepare_id AND anchor_kind IN ('PREPARE','REUSE'); + prepares_aborted:=prepares_aborted+1; + ELSIF mst2_metadata_session_covers_prepare(q.prepare_id) THEN + UPDATE mst2_metadata_prepare SET coverage_retired_at=clock_timestamp() WHERE prepare_id=q.prepare_id; + DELETE FROM mst2_metadata_root_anchor WHERE prepare_id=q.prepare_id AND anchor_kind IN ('PREPARE','REUSE'); + handovers_retired:=handovers_retired+1; + ELSIF mst2_metadata_orphan_prepare_eligible(q.prepare_id) THEN + UPDATE mst2_metadata_prepare SET coverage_retired_at=clock_timestamp() WHERE prepare_id=q.prepare_id; + DELETE FROM mst2_metadata_root_anchor WHERE prepare_id=q.prepare_id AND anchor_kind IN ('PREPARE','REUSE'); + orphans_retired:=orphans_retired+1; + END IF; + END IF; + END LOOP; + RETURN NEXT; +END $$; +DROP TRIGGER mst2_01_family_closed ON mst2_metadata_gc_op; diff --git a/src/jupiter/storage/qualified_metadata_gc_tests.rs b/src/jupiter/storage/qualified_metadata_gc_tests.rs new file mode 100644 index 00000000..346254bf --- /dev/null +++ b/src/jupiter/storage/qualified_metadata_gc_tests.rs @@ -0,0 +1,456 @@ +use std::collections::{BTreeMap, BTreeSet}; + +use super::{canonical_tests::seeded_rooted_plan, *}; +use crate::ceres::snapshot::{ + rooted_metadata_install::RootedMetadataInstallPlan, + rooted_metadata_projection::RootedReuseLookup, +}; + +async fn write_rooted( + writer: &RootedQualifiedMetadataRepository, + operation: &str, + plan: &RootedMetadataInstallPlan, + payload: MetadataPagePayload, +) -> RootedPrepareIntent { + let intent = writer.begin_intent(operation, plan).await.unwrap(); + writer.install_pages(&intent, &[payload]).await.unwrap(); + writer.finalize(&intent).await.unwrap(); + intent +} + +// Fault injection only: expire database evidence in the isolated test schema, +// restore the original guards/catalog, then use normal production maintenance. +async fn age_orphan(q: &DatabaseConnection, namespace: &VerifiedQualifiedNamespace, prepare: &str) { + let txn = q.begin().await.unwrap(); + for trigger in [ + "mst2_00_family_barrier", + "mst2_metadata_prepare_guard", + "mst2_metadata_prepare_00_expiry_guard", + ] { + txn.execute_unprepared(&format!( + "ALTER TABLE mst2_metadata_prepare DISABLE TRIGGER {trigger}" + )) + .await + .unwrap(); + } + txn.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "UPDATE mst2_metadata_prepare SET created_at=clock_timestamp()-interval '3601 seconds', + orphan_expires_at=clock_timestamp()-interval '1 second' WHERE prepare_id=$1", + [prepare.into()], + )) + .await + .unwrap(); + // clock_timestamp() calls differ; preserve the exact database TTL relation. + txn.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "UPDATE mst2_metadata_prepare SET orphan_expires_at=created_at+interval '3600 seconds' WHERE prepare_id=$1", + [prepare.into()])).await.unwrap(); + for trigger in [ + "mst2_00_family_barrier", + "mst2_metadata_prepare_guard", + "mst2_metadata_prepare_00_expiry_guard", + ] { + txn.execute_unprepared(&format!( + "ALTER TABLE mst2_metadata_prepare ENABLE TRIGGER {trigger}" + )) + .await + .unwrap(); + } + txn.commit().await.unwrap(); + assert_eq!( + catalog(q, namespace.core_oid, namespace.schema_oid) + .await + .unwrap(), + namespace.catalog_fingerprint + ); +} + +async fn claim( + q: &DatabaseConnection, + intent: &RootedPrepareIntent, +) -> Result { + let operation = uuid::Uuid::new_v4().to_string(); + q.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "SELECT mst2_metadata_gc_claim($1,$2,(SELECT primary_scope FROM mst2_metadata_prepare WHERE prepare_id=$3),$4::uuid)", + [intent.metadata_root().to_vec().into(),intent.root_generation().into(),intent.prepare_id().into(),operation.clone().into()] + )).await?; + Ok(operation) +} + +#[tokio::test] +async fn admitted_orphan_gc_preserves_replay_identity_and_advances_exact_generation() { + let (config, core, namespace, q, _guard) = fixture().await; + let (plan, payload) = seeded_rooted_plan(&core, 'a', "file", 1).await; + let writer = RootedQualifiedMetadataRepository::open(&core, &config) + .await + .unwrap(); + let first = write_rooted(&writer, "gc-generation-one", &plan, payload.clone()).await; + assert_eq!( + count(&q, "SELECT mst2_metadata_gc_enabled()::bigint").await, + 1 + ); + assert!( + claim(&q, &first) + .await + .unwrap_err() + .to_string() + .contains("owned coverage") + ); + assert_eq!( + count(&q, "SELECT count(*) FROM mst2_metadata_gc_op").await, + 0 + ); + age_orphan(&q, &namespace, first.prepare_id()).await; + let work = writer.maintenance_tick(64).await.unwrap(); + assert!(work.collector_enabled); + assert!(work.examined <= 64); + assert_eq!(work.orphans_retired, 1); + assert_eq!(work.payload_pages_removed, 1); + assert_eq!(work.payload_bytes_removed, payload.size); + assert_eq!( + count(&q, "SELECT count(*) FROM mst2_metadata_payload").await, + 0 + ); + assert_eq!( + count(&q, "SELECT count(*) FROM mst2_metadata_graph_node").await, + 0 + ); + assert_eq!( + count(&q, "SELECT count(*) FROM mst2_metadata_page_certificate").await, + 1 + ); + assert_eq!( + count( + &q, + "SELECT count(*) FROM mst2_metadata_source_root_attestation" + ) + .await, + 1 + ); + assert_eq!( + count( + &q, + "SELECT count(*) FROM mst2_metadata_gc_op WHERE state='APPLIED'" + ) + .await, + 1 + ); + let operation: String = q + .query_one_raw(Statement::from_string( + DbBackend::Postgres, + "SELECT operation_id::text FROM mst2_metadata_gc_op", + )) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap(); + let fresh = write_rooted(&writer, "gc-generation-two", &plan, payload.clone()).await; + assert_eq!(fresh.root_generation(), first.root_generation() + 1); + assert!(writer.recover("gc-generation-one", &plan).await.is_err()); + for _ in 0..2 { + q.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT mst2_metadata_gc_apply($1::uuid)", + [operation.clone().into()], + )) + .await + .unwrap(); + } + assert_eq!( + count( + &q, + "SELECT count(*) FROM mst2_metadata_payload WHERE generation=2" + ) + .await, + 1 + ); + assert_eq!( + count( + &q, + "SELECT count(*) FROM mst2_metadata_lifetime WHERE generation=1 AND state='REMOVED'" + ) + .await, + 1 + ); + assert_eq!( + count( + &q, + "SELECT count(*) FROM mst2_metadata_lifetime WHERE generation=2 AND state='LIVE'" + ) + .await, + 1 + ); + assert_eq!( + count(&q, "SELECT count(*) FROM mst2_metadata_page_certificate").await, + 2 + ); + assert_eq!( + count( + &q, + "SELECT count(*) FROM mst2_metadata_current WHERE generation=2" + ) + .await, + 1 + ); +} + +#[tokio::test] +async fn pending_gc_survives_rebuild_and_payload_stamp_cannot_commit_alone() { + let (config, core, namespace, q, _guard) = fixture().await; + let (plan, payload) = seeded_rooted_plan(&core, 'a', "file", 1).await; + let writer = RootedQualifiedMetadataRepository::open(&core, &config) + .await + .unwrap(); + let intent = write_rooted(&writer, "gc-pending", &plan, payload).await; + age_orphan(&q, &namespace, intent.prepare_id()).await; + assert_eq!(writer.maintenance_tick(1).await.unwrap().orphans_retired, 1); + let operation = claim(&q, &intent).await.unwrap(); + assert_eq!( + count( + &q, + "SELECT count(*) FROM mst2_metadata_gc_op WHERE state='PENDING'" + ) + .await, + 1 + ); + let txn = q.begin().await.unwrap(); + txn.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "UPDATE mst2_metadata_gc_op SET payload_delete_xid=txid_current() WHERE operation_id=$1::uuid", + [operation.into()])).await.unwrap(); + let error = txn.commit().await.unwrap_err(); + assert!( + error.to_string().contains("cannot escape its atomic apply"), + "{error}" + ); + assert_eq!( + count(&q, "SELECT count(*) FROM mst2_metadata_payload").await, + 1 + ); + assert_eq!(count(&q,"SELECT count(*) FROM mst2_metadata_gc_op WHERE state='PENDING' AND payload_delete_xid IS NULL").await,1); + drop(writer); + let rebuilt = RootedQualifiedMetadataRepository::open(&core, &config) + .await + .unwrap(); + let work = rebuilt.maintenance_tick(1).await.unwrap(); + assert_eq!(work.examined, 1); + assert_eq!(work.pending_replayed, 1); + assert_eq!(work.gc_applied, 1); + assert_eq!(work.payload_pages_removed, 1); + assert_eq!( + count(&q, "SELECT count(*) FROM mst2_metadata_payload").await, + 0 + ); + assert_eq!( + count( + &q, + "SELECT count(*) FROM mst2_metadata_gc_op WHERE state='APPLIED'" + ) + .await, + 1 + ); +} + +#[tokio::test] +async fn actual_incoming_edge_blocks_child_gc_until_parent_removal() { + let (config, core, namespace, q, _guard) = fixture().await; + let (child, payload) = seeded_rooted_plan(&core, 'a', "file", 1).await; + let writer = RootedQualifiedMetadataRepository::open(&core, &config) + .await + .unwrap(); + let child_intent = write_rooted(&writer, "gc-edge-child", &child, payload).await; + let hint = writer + .lookup_reuse(&child.identity.tagged_root_tree_oid, &child.identity) + .await + .unwrap() + .unwrap(); + let entries = [ + Entry::dir(b"one", child.root), + Entry::dir(b"two", child.root), + ]; + let bytes = Page::build(&entries).unwrap(); + let root = page_id(&bytes); + let mut body = b"40000 one\0".to_vec(); + body.extend_from_slice(&[0xaa; 20]); + body.extend_from_slice(b"40000 two\0"); + body.extend_from_slice(&[0xaa; 20]); + core.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "INSERT INTO mega_tree(id,tree_id,sub_trees,size,created_at,pack_id,pack_offset,commit_id) + VALUES(2,$1,$2,0,now(),'fixture',0,'fixture')", + ["d".repeat(40).into(), body.into()], + )) + .await + .unwrap(); + let mut identity = child.identity.clone(); + identity.tagged_root_tree_oid = format!("sha1:{}", "d".repeat(40)); + let parent = RootedMetadataInstallPlan::new( + identity.clone(), + root, + BTreeMap::from([(root, bytes.len() as u64)]), + BTreeSet::from([(root, child.root)]), + BTreeMap::from([(child.root, hint.proof)]), + BTreeMap::from([ + (identity.tagged_root_tree_oid, root), + (child.identity.tagged_root_tree_oid.clone(), child.root), + ]), + ) + .unwrap(); + let parent_intent = write_rooted( + &writer, + "gc-edge-parent", + &parent, + MetadataPagePayload { + id: root, + size: bytes.len() as u64, + bytes, + }, + ) + .await; + age_orphan(&q, &namespace, child_intent.prepare_id()).await; + age_orphan(&q, &namespace, parent_intent.prepare_id()).await; + for _ in 0..2 { + assert_eq!(writer.maintenance_tick(1).await.unwrap().orphans_retired, 1); + } + assert_eq!( + count(&q, "SELECT count(*) FROM mst2_metadata_root_anchor").await, + 0 + ); + assert_eq!( + count(&q, "SELECT count(*) FROM mst2_metadata_graph_edge").await, + 1 + ); + assert_eq!( + count( + &q, + "SELECT sum(incoming_refs)::bigint FROM mst2_metadata_graph_node" + ) + .await, + 1 + ); + assert!( + claim(&q, &child_intent) + .await + .unwrap_err() + .to_string() + .contains("incoming edges") + ); + let operation = claim(&q, &parent_intent).await.unwrap(); + q.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT mst2_metadata_gc_apply($1::uuid)", + [operation.into()], + )) + .await + .unwrap(); + assert_eq!( + count(&q, "SELECT count(*) FROM mst2_metadata_graph_edge").await, + 0 + ); + assert_eq!( + count( + &q, + "SELECT sum(incoming_refs)::bigint FROM mst2_metadata_graph_node" + ) + .await, + 0 + ); + let operation = claim(&q, &child_intent).await.unwrap(); + q.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT mst2_metadata_gc_apply($1::uuid)", + [operation.into()], + )) + .await + .unwrap(); + assert_eq!( + count(&q, "SELECT count(*) FROM mst2_metadata_payload").await, + 0 + ); + assert_eq!( + count(&q, "SELECT count(*) FROM mst2_metadata_verified_ref").await, + 2 + ); +} + +#[tokio::test] +async fn restored_guards_refuse_corrupt_payload_size_counter_and_current_bindings() { + for case in 0..4 { + let (config, core, namespace, q, _guard) = fixture().await; + let (plan, payload) = seeded_rooted_plan(&core, 'a', "file", 1).await; + let writer = RootedQualifiedMetadataRepository::open(&core, &config) + .await + .unwrap(); + let intent = write_rooted(&writer, "gc-corruption", &plan, payload).await; + age_orphan(&q, &namespace, intent.prepare_id()).await; + assert_eq!(writer.maintenance_tick(1).await.unwrap().orphans_retired, 1); + let (table, guard, statement) = match case { + 0 => ( + "mst2_metadata_payload", + "mst2_metadata_payload_fenced", + "UPDATE mst2_metadata_payload SET payload=set_byte(payload,20,255)", + ), + 1 => ( + "mst2_metadata_graph_node", + "mst2_metadata_graph_node_guard", + "UPDATE mst2_metadata_graph_node SET bytes=bytes+1", + ), + 2 => ( + "mst2_metadata_graph_node", + "mst2_metadata_graph_node_guard", + "UPDATE mst2_metadata_graph_node SET incoming_refs=1", + ), + _ => ( + "mst2_metadata_current", + "mst2_metadata_current_guard", + "DELETE FROM mst2_metadata_current", + ), + }; + let txn = q.begin().await.unwrap(); + for trigger in ["mst2_00_family_barrier", guard] { + txn.execute_unprepared(&format!("ALTER TABLE {table} DISABLE TRIGGER {trigger}")) + .await + .unwrap(); + } + txn.execute_unprepared(statement).await.unwrap(); + for trigger in ["mst2_00_family_barrier", guard] { + txn.execute_unprepared(&format!("ALTER TABLE {table} ENABLE TRIGGER {trigger}")) + .await + .unwrap(); + } + txn.commit().await.unwrap(); + assert_eq!( + catalog(&q, namespace.core_oid, namespace.schema_oid) + .await + .unwrap(), + namespace.catalog_fingerprint + ); + let error = claim(&q, &intent).await.unwrap_err(); + assert!( + error.to_string().contains(if case == 0 { + "durable bytes" + } else if case < 3 { + "actual counter" + } else { + "current generation" + }), + "{error}" + ); + assert_eq!( + count(&q, "SELECT count(*) FROM mst2_metadata_gc_op").await, + 0 + ); + assert_eq!( + count(&q, "SELECT count(*) FROM mst2_metadata_payload").await, + 1 + ); + assert_eq!( + count( + &q, + "SELECT count(*) FROM mst2_metadata_lifetime WHERE state='LIVE'" + ) + .await, + 1 + ); + } +} diff --git a/src/jupiter/storage/qualified_metadata_reader.rs b/src/jupiter/storage/qualified_metadata_reader.rs new file mode 100644 index 00000000..85b22ff4 --- /dev/null +++ b/src/jupiter/storage/qualified_metadata_reader.rs @@ -0,0 +1,976 @@ +//! Persisted routes traverse exact certified occurrences, including reused roots. + +use std::collections::{BTreeSet, HashMap}; + +use mst2_codec::metapage::{HEADER_LEN, PAGE_MAX_BYTES}; +use serde::Deserialize; + +use super::*; +use crate::{ + ceres::snapshot::{ + retention_dag::MetadataDagLimits, runtime::SnapshotContext, + view::validate_scope_relative_path, + }, + jupiter::storage::native_snapshot_session::{ + MetadataRouteRequest, PersistedMetadataReadWork, PersistedMetadataRouteBatch, + }, +}; + +type ProofPages = Vec<([u8; 32], Vec)>; + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +struct Binding { + generation: i64, + certificate: [u8; 32], +} + +#[cfg(test)] +tokio::task_local! { + static READER_ADMISSION_BARRIERS: (Arc, Arc); + static SOURCE_FACT_BARRIERS: (Arc, Arc); + pub(super) static READER_TEMP_SOURCE_SHADOW: bool; +} + +#[cfg(test)] +pub(crate) async fn with_rooted_source_fact_barriers( + admitted: Arc, + resume: Arc, + future: F, +) -> F::Output { + SOURCE_FACT_BARRIERS.scope((admitted, resume), future).await +} + +#[cfg(test)] +pub(crate) async fn with_rooted_source_temporary_shadow( + future: F, +) -> F::Output { + READER_TEMP_SOURCE_SHADOW.scope(true, future).await +} + +#[cfg(test)] +pub(crate) async fn with_rooted_reader_barriers( + admitted: Arc, + resume: Arc, + future: F, +) -> F::Output { + READER_ADMISSION_BARRIERS + .scope((admitted, resume), future) + .await +} + +#[derive(Debug, Deserialize)] +struct Reference { + kind: String, + child: String, + generation: i64, + certificate: String, + name: Option, + label: Option, + count: Option, +} + +fn digest_hex(value: &str) -> Result<[u8; 32], SnapshotError> { + hex::decode(value) + .map_err(internal)? + .as_slice() + .try_into() + .map_err(internal) +} +fn limit(message: &str) -> SnapshotError { + SnapshotError::new( + crate::ceres::snapshot::error::SnapshotErrorCode::LimitExceeded, + message, + ) +} +fn absent(message: &str) -> SnapshotError { + SnapshotError::new( + crate::ceres::snapshot::error::SnapshotErrorCode::PathNotFound, + message, + ) +} + +struct StoredPage { + bytes: Vec, + page: Page, + entries: u64, +} + +pub(crate) struct RootedDirectoryWindow { + pub directory_root: [u8; 32], + pub entry_count: u64, + pub entries: Vec, + pub has_more: bool, + pub proof_pages: ProofPages, + pub ancestors: Vec<(String, [u8; 32])>, +} + +pub(crate) enum RootedLookupStatus { + Directory([u8; 32]), + File { + entry: mst2_codec::metapage::Entry, + git_oid: String, + }, + Absent, + NotDirectory { + symlink: bool, + }, +} +pub(crate) struct RootedLookupBatch { + pub results: Vec, + pub proof_pages: Vec<([u8; 32], Vec)>, +} +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +struct ReaderIdentity { + operation_id: uuid::Uuid, + issuance: i64, +} + +struct Reader<'a> { + txn: &'a DatabaseTransaction, + operation: ReaderIdentity, + bindings: HashMap<[u8; 32], Binding>, + cache: HashMap<[u8; 32], StoredPage>, + work: PersistedMetadataReadWork, + limits: MetadataDagLimits, + directories: BTreeMap, + source_ids: BTreeMap, + file_oids: HashMap<(uuid::Uuid, Vec), String>, +} + +impl RootedQualifiedMetadataRepository { + pub(crate) async fn fixed_path_metadata( + &self, + pinned: &SnapshotContext, + path: &str, + ) -> Result { + validate_scope_relative_path(path)?; + let absolute = if pinned.built.descriptor.scope == "/" { + path.to_owned() + } else if path == "/" { + pinned.built.descriptor.scope.clone() + } else { + format!("{}{}", pinned.built.descriptor.scope, path) + }; + validate_scope_relative_path(&absolute)?; + let (operation, binding, source) = self.admit_reader(pinned).await?; + let result = async { + let txn = self.read_transaction().await?; + let root = pinned.built.descriptor.metadata_root; + let mut reader = Reader::new(&txn, operation, root, binding, source); + let result = reader.lookup(root, path).await; + sessions::finish(txn, result).await + } + .await; + self.finish_reader(operation, result).await + } + + pub(crate) async fn directory_window( + &self, + pinned: &SnapshotContext, + path: &str, + after: Option<&str>, + count: usize, + ) -> Result { + validate_scope_relative_path(path)?; + if !(1..=256).contains(&count) { + return Err(limit("directory limit must be 1..256")); + } + let absolute = if pinned.built.descriptor.scope == "/" { + path.to_owned() + } else if path == "/" { + pinned.built.descriptor.scope.clone() + } else { + format!("{}{}", pinned.built.descriptor.scope, path) + }; + validate_scope_relative_path(&absolute)?; + let (operation, binding, source) = self.admit_reader(pinned).await?; + let result = async { + let txn = self.read_transaction().await?; + let read = async { + let root = pinned.built.descriptor.metadata_root; + let mut reader = Reader::new(&txn, operation, root, binding, source); + let directory = reader.directory(root, path).await?; + reader.load(directory).await?; + let total = reader.cache[&directory].entries; + let mut entries = Vec::with_capacity(count + 1); + reader + .range(directory, after.map(str::as_bytes), count + 1, &mut entries) + .await?; + let has_more = entries.len() > count; + entries.truncate(count); + reader + .source_entries(reader.source_ids[path], directory, &entries) + .await?; + Ok(RootedDirectoryWindow { + directory_root: directory, + entry_count: total, + entries, + has_more, + proof_pages: reader.proofs()?, + ancestors: reader.directories.into_iter().collect(), + }) + } + .await; + sessions::finish(txn, read).await + } + .await; + self.finish_reader(operation, result).await + } + + pub(crate) async fn lookup_metadata( + &self, + pinned: &SnapshotContext, + paths: &[String], + ) -> Result { + if paths.len() > 128 { + return Err(limit("at most 128 paths per lookup")); + } + for path in paths { + validate_scope_relative_path(path)?; + let absolute = if pinned.built.descriptor.scope == "/" { + path.clone() + } else if path == "/" { + pinned.built.descriptor.scope.clone() + } else { + format!("{}{}", pinned.built.descriptor.scope, path) + }; + validate_scope_relative_path(&absolute)?; + } + let (operation, binding, source) = self.admit_reader(pinned).await?; + let result = async { + let txn = self.read_transaction().await?; + let read = async { + let root = pinned.built.descriptor.metadata_root; + let mut reader = Reader::new(&txn, operation, root, binding, source); + let mut results = Vec::with_capacity(paths.len()); + for path in paths { + results.push(reader.lookup(root, path).await?); + } + Ok(RootedLookupBatch { + results, + proof_pages: reader.proofs()?, + }) + } + .await; + sessions::finish(txn, read).await + } + .await; + self.finish_reader(operation, result).await + } + pub(crate) async fn metadata_routes( + &self, + pinned: &SnapshotContext, + requests: &[MetadataRouteRequest<'_>], + ) -> Result { + if requests.is_empty() || requests.len() > 64 { + return Err(limit("items must hold 1..64 entries")); + } + for request in requests { + validate_scope_relative_path(request.directory_path)?; + let scope = &pinned.built.descriptor.scope; + let absolute = if scope == "/" { + request.directory_path.to_owned() + } else if request.directory_path == "/" { + scope.clone() + } else { + format!("{scope}{}", request.directory_path) + }; + validate_scope_relative_path(&absolute)?; + } + // Acquire REQUEST and READER ownership atomically, then release the + // mutation barrier before fetching payloads. Cleanup never precedes + // ownership of the returned byte buffers. + let (operation, binding, source) = self.admit_reader(pinned).await?; + let result = self + .read_routes(pinned, requests, operation, binding, source) + .await; + self.finish_reader(operation, result).await + } + + async fn admit_reader( + &self, + pinned: &SnapshotContext, + ) -> Result<(ReaderIdentity, Binding, uuid::Uuid), SnapshotError> { + let txn = self.transaction().await?; + let admitted = async { + let row = self + .session_row( + &txn, + &pinned.built.snapshot_id, + &pinned.lease_id, + &pinned.built.instance_id, + ) + .await?; + let current = sessions::context( + &row, + &pinned.built.snapshot_id, + &pinned.lease_id, + &pinned.built.instance_id, + )?; + if current.built.descriptor != pinned.built.descriptor + || current.commit_oid != pinned.commit_oid + || current.root_tree_oid != pinned.root_tree_oid + || current.authorization_epoch != pinned.authorization_epoch + { + return Err(integrity( + "qualified fixed session changed during reader admission", + )); + } + let reader = txn + .query_one_raw(sql( + "SELECT operation_id::text,reader_issuance,root_generation,certificate_digest + FROM mst2_metadata_begin_reader($1,$2,$3)", + [ + pinned.built.snapshot_id.clone().into(), + pinned.lease_id.clone().into(), + pinned.built.instance_id.clone().into(), + ], + )) + .await + .map_err(database_error)? + .ok_or_else(|| unavailable("qualified reader admission returned no owned root"))?; + Ok(( + ReaderIdentity { + operation_id: uuid::Uuid::parse_str( + &reader + .try_get::("", "operation_id") + .map_err(internal)?, + ) + .map_err(internal)?, + issuance: reader.try_get("", "reader_issuance").map_err(internal)?, + }, + Binding { + generation: reader.try_get("", "root_generation").map_err(internal)?, + certificate: digest_column(&reader, "certificate_digest")?, + }, + row.try_get("", "attestation_id").map_err(internal)?, + )) + } + .await; + let owned = sessions::finish(txn, admitted).await?; + #[cfg(test)] + if let Ok((admitted, resume)) = READER_ADMISSION_BARRIERS.try_with(Clone::clone) { + admitted.wait().await; + resume.wait().await; + } + Ok(owned) + } + + async fn finish_reader( + &self, + operation: ReaderIdentity, + result: Result, + ) -> Result { + let cleanup = self.transaction().await; + let cleanup = match cleanup { + Ok(txn) => { + let finished = txn + .execute_raw(sql( + "SELECT mst2_metadata_finish_reader($1::uuid,$2::bigint)", + [ + operation.operation_id.to_string().into(), + operation.issuance.into(), + ], + )) + .await + .map_err(internal) + .map(|_| ()); + sessions::finish(txn, finished).await + } + Err(error) => Err(error), + }; + // A failed cleanup leaves durable roots until the bounded hard deadline. + // Report the failure instead of pretending that a protected operation + // has been definitively finished. + match (result, cleanup) { + (Err(error), _) => Err(error), + (Ok(_), Err(error)) => Err(error), + (Ok(value), Ok(())) => Ok(value), + } + } + + async fn read_routes( + &self, + pinned: &SnapshotContext, + requests: &[MetadataRouteRequest<'_>], + operation: ReaderIdentity, + binding: Binding, + source: uuid::Uuid, + ) -> Result { + let txn = self.read_transaction().await?; + let result = async { + let root = pinned.built.descriptor.metadata_root; + let mut reader = Reader::new(&txn, operation, root, binding, source); + let mut seen = BTreeSet::new(); + let mut pages = Vec::new(); + for request in requests { + let directory = reader.directory(root, request.directory_path).await?; + let route = reader.route(directory, request.route).await?; + let reached = route + .last() + .ok_or_else(|| integrity("qualified metadata route is empty"))?; + if let Some(expected) = request.expected_digest + && expected != format!("sha256:{}", hex::encode(reached)) + { + return Err(SnapshotError::new( + crate::ceres::snapshot::error::SnapshotErrorCode::DigestMismatch, + "qualified metadata route does not reach expected_digest", + )); + } + for page in route { + if seen.insert(page) { + pages.push((page, reader.cache[&page].bytes.clone())); + } + } + } + Ok(PersistedMetadataRouteBatch { + pages, + work: reader.work, + }) + } + .await; + sessions::finish(txn, result).await + } +} + +impl Reader<'_> { + fn new( + txn: &DatabaseTransaction, + operation: ReaderIdentity, + root: [u8; 32], + binding: Binding, + source: uuid::Uuid, + ) -> Reader<'_> { + Reader { + txn, + operation, + bindings: HashMap::from([(root, binding)]), + cache: HashMap::new(), + work: PersistedMetadataReadWork::default(), + limits: MetadataDagLimits::default(), + directories: BTreeMap::from([("/".into(), root)]), + source_ids: BTreeMap::from([("/".into(), source)]), + file_oids: HashMap::new(), + } + } + + async fn source_entries( + &mut self, + source: uuid::Uuid, + directory: [u8; 32], + entries: &[mst2_codec::metapage::Entry], + ) -> Result, uuid::Uuid>, SnapshotError> { + let binding = self + .bindings + .get(&directory) + .ok_or_else(|| integrity("qualified source directory has no certified binding"))?; + let names: Vec<_> = entries + .iter() + .map(|entry| hex::encode(&entry.name)) + .collect(); + let rows = self.txn.query_all_raw(sql("SELECT * FROM mst2_metadata_read_source_entries($1::uuid,$2::bigint,$3::uuid,$4,$5,$6,$7::jsonb)", + [self.operation.operation_id.to_string().into(), self.operation.issuance.into(), source.to_string().into(), directory.to_vec().into(), + binding.generation.into(), binding.certificate.to_vec().into(),json!(names).into()])).await.map_err(database_error)?; + #[cfg(test)] + if entries.iter().any(|entry| !entry.is_dir()) + && let Ok((admitted, resume)) = SOURCE_FACT_BARRIERS.try_with(Clone::clone) + { + admitted.wait().await; + resume.wait().await; + } + if rows.len() != entries.len() { + return Err(integrity( + "qualified source-name index differs from selected actual page entries", + )); + } + let mut children = HashMap::new(); + for row in rows { + let name: Vec = row.try_get("", "name").map_err(internal)?; + let entry = entries + .iter() + .find(|entry| entry.name == name) + .ok_or_else(|| { + integrity("qualified source-name read returned an unrequested occurrence") + })?; + let state: String = row.try_get("", "fact_state").map_err(internal)?; + match state.as_str() { + "READY" => {} + "MISSING" => { + return Err(SnapshotError::new( + crate::ceres::snapshot::error::SnapshotErrorCode::MetadataNotReady, + "selected fixed file has no current verified object metadata", + )); + } + "SOURCE_UNAVAILABLE" => { + return Err(unavailable( + "selected fixed directory has no current exact source attestation", + )); + } + _ => { + return Err(integrity( + "selected fixed file has invalid or different current verified object metadata", + )); + } + } + let kind: i16 = row.try_get("", "kind").map_err(internal)?; + if kind as u8 != entry.kind as u8 { + return Err(integrity( + "qualified selected occurrence changed its fixed filesystem kind", + )); + } + if entry.is_dir() { + let child = self.bindings.get(&entry.child_root).ok_or_else(|| { + integrity("qualified source child has no certified occurrence") + })?; + let root = digest_column(&row, "child_root")?; + if root != entry.child_root + || child.generation + != row + .try_get::("", "child_generation") + .map_err(internal)? + || child.certificate != digest_column(&row, "child_certificate_digest")? + { + return Err(integrity( + "qualified source-name directory differs from its exact certified lifetime", + )); + } + children.insert( + name.clone(), + row.try_get("", "child_attestation_id").map_err(internal)?, + ); + } else if row.try_get::("", "byte_size").map_err(internal)? as u64 != entry.size + || digest_column(&row, "content_digest")? != entry.content_id + { + return Err(integrity( + "qualified source-name file differs from its certified actual page", + )); + } + if !entry.is_dir() { + self.file_oids.insert( + (source, name), + row.try_get("", "git_oid").map_err(internal)?, + ); + } + } + Ok(children) + } + + fn proofs(&self) -> Result { + let bytes: usize = self.cache.values().map(|page| page.bytes.len()).sum(); + if bytes > 1_048_576 { + return Err(SnapshotError::new( + crate::ceres::snapshot::error::SnapshotErrorCode::ProofBudgetExceeded, + "qualified proof pages exceed the response budget; use metadata/pages", + )); + } + let mut pages: Vec<_> = self + .cache + .iter() + .map(|(id, page)| (*id, page.bytes.clone())) + .collect(); + pages.sort_by_key(|item| item.0); + Ok(pages) + } + + async fn range( + &mut self, + id: [u8; 32], + after: Option<&[u8]>, + count: usize, + out: &mut Vec, + ) -> Result<(), SnapshotError> { + if out.len() >= count { + return Ok(()); + } + self.load(id).await?; + match self.cache[&id].page.clone() { + Page::Leaf { entries } => { + let start = after + .map(|name| entries.partition_point(|entry| entry.name.as_slice() <= name)) + .unwrap_or(0); + out.extend(entries.into_iter().skip(start).take(count - out.len())); + } + Page::Branch { + prefix, + terminal, + children, + } => { + if let Some(entry) = terminal + && after.is_none_or(|name| entry.name.as_slice() > name) + { + out.push(entry); + } + for child in children { + if out.len() >= count { + break; + } + let mut partition = prefix.clone(); + partition.push(child.label); + if after.is_some_and(|name| { + partition.as_slice() < name && !name.starts_with(&partition) + }) { + continue; + } + self.descend(id, child.label).await?; + Box::pin(self.range(child.child_page_id, after, count, out)).await?; + } + } + } + Ok(()) + } + + async fn find_entry( + &mut self, + mut id: [u8; 32], + name: &[u8], + ) -> Result, SnapshotError> { + loop { + self.load(id).await?; + match &self.cache[&id].page { + Page::Leaf { entries } => { + return Ok(entries.iter().find(|entry| entry.name == name).cloned()); + } + Page::Branch { + prefix, + terminal, + children, + } => { + if name == prefix { + return Ok(terminal.clone()); + } + if !name.starts_with(prefix) { + return Ok(None); + } + let Some(label) = name.get(prefix.len()).copied() else { + return Ok(None); + }; + if !children.iter().any(|child| child.label == label) { + return Ok(None); + } + id = self.descend(id, label).await?; + } + } + } + } + + async fn lookup( + &mut self, + mut root: [u8; 32], + path: &str, + ) -> Result { + if path == "/" { + self.load(root).await?; + return Ok(RootedLookupStatus::Directory(root)); + } + let mut prefix = String::new(); + let mut source = self.source_ids["/"]; + let mut parts = path[1..].split('/').peekable(); + while let Some(name) = parts.next() { + let Some(entry) = self.find_entry(root, name.as_bytes()).await? else { + return Ok(RootedLookupStatus::Absent); + }; + if !entry.is_dir() { + if parts.peek().is_some() { + return Ok(RootedLookupStatus::NotDirectory { + symlink: entry.kind == mst2_codec::metapage::EntryKind::Symlink, + }); + } + self.source_entries(source, root, std::slice::from_ref(&entry)) + .await?; + let git_oid = self + .file_oids + .get(&(source, entry.name.clone())) + .cloned() + .ok_or_else(|| { + integrity("selected fixed file has no exact source OID binding") + })?; + return Ok(RootedLookupStatus::File { entry, git_oid }); + } + let children = self + .source_entries(source, root, std::slice::from_ref(&entry)) + .await?; + source = *children.get(&entry.name).ok_or_else(|| { + integrity("qualified named directory has no independently derived source child") + })?; + root = entry.child_root; + prefix.push('/'); + prefix.push_str(name); + self.directories.insert(prefix.clone(), root); + self.source_ids.insert(prefix.clone(), source); + } + self.load(root).await?; + Ok(RootedLookupStatus::Directory(root)) + } + async fn load(&mut self, id: [u8; 32]) -> Result<(), SnapshotError> { + self.work.walk_visits += 1; + if self.work.walk_visits > self.limits.prepare_entry_visits as u64 { + return Err(limit("qualified route work budget exceeded")); + } + if self.cache.contains_key(&id) { + return Ok(()); + } + if self.cache.len() >= self.limits.nodes { + return Err(limit("qualified route page budget exceeded")); + } + let binding = *self.bindings.get(&id).ok_or_else(|| { + integrity("qualified page was not reached through a certified occurrence") + })?; + self.work.page_queries += 1; + let row = self + .txn + .query_one_raw(sql( + PAGE_SQL, + [ + id.to_vec().into(), + binding.generation.into(), + binding.certificate.to_vec().into(), + self.operation.operation_id.to_string().into(), + self.operation.issuance.into(), + ], + )) + .await + .map_err(internal)? + .ok_or_else(|| { + unavailable("qualified certified page or reader ownership is unavailable") + })?; + let bytes: Vec = row.try_get("", "payload").map_err(internal)?; + let size: i32 = row.try_get("", "byte_size").map_err(internal)?; + if !(HEADER_LEN..=PAGE_MAX_BYTES).contains(&bytes.len()) + || bytes.len() != size as usize + || page_id(&bytes) != id + { + return Err(integrity( + "qualified durable metadata bytes differ from their certified page", + )); + } + let (page, entries) = Page::decode(&bytes).map_err(internal)?; + self.work.payload_bytes = self + .work + .payload_bytes + .checked_add(bytes.len() as u64) + .filter(|value| *value <= self.limits.payload_bytes) + .ok_or_else(|| limit("qualified route byte budget exceeded"))?; + let raw_refs: serde_json::Value = row.try_get("", "references").map_err(internal)?; + let refs: Vec = serde_json::from_value(raw_refs).map_err(internal)?; + if refs.len() > 257 { + return Err(integrity("qualified canonical reference count is invalid")); + } + let mut expected = Vec::new(); + let direct = match &page { + Page::Leaf { entries } => entries.as_slice(), + Page::Branch { + terminal, children, .. + } => { + for child in children { + expected.push(( + "RADIX", + child.child_page_id, + None, + Some(child.label), + Some(child.subtree_entries), + )); + } + terminal.as_slice() + } + }; + // Reference ordering is direct DIRECTORY occurrences followed by RADIX, + // as defined by the independent canonical parser. Compare occurrences + // as a set here; SQL validates the certificate's ordinal completeness. + for entry in direct.iter().filter(|entry| entry.is_dir()) { + expected.push(( + "DIRECTORY", + entry.child_root, + Some(hex::encode(&entry.name)), + None, + None, + )); + } + let mut actual = BTreeSet::new(); + for reference in refs { + let child = digest_hex(&reference.child)?; + let binding = Binding { + generation: reference.generation, + certificate: digest_hex(&reference.certificate)?, + }; + if binding.generation <= 0 + || self + .bindings + .get(&child) + .is_some_and(|known| *known != binding) + { + return Err(integrity( + "qualified occurrence retargeted its exact lifetime or certificate", + )); + } + self.bindings.insert(child, binding); + if !actual.insert(( + reference.kind, + child, + reference.name, + reference.label, + reference.count, + )) { + return Err(integrity("qualified typed occurrence has a duplicate")); + } + } + let expected: BTreeSet<_> = expected + .into_iter() + .map(|(kind, child, name, label, count)| (kind.to_owned(), child, name, label, count)) + .collect(); + if actual != expected { + return Err(integrity( + "qualified certificate references differ from its actual durable page", + )); + } + self.work.edge_references_checked += actual.len() as u64; + if self.work.edge_references_checked > self.limits.edges as u64 { + return Err(limit("qualified route reference budget exceeded")); + } + self.work.pages_loaded += 1; + self.cache.insert( + id, + StoredPage { + bytes, + page, + entries, + }, + ); + Ok(()) + } + + async fn descend(&mut self, parent: [u8; 32], label: u8) -> Result<[u8; 32], SnapshotError> { + self.load(parent).await?; + let (prefix, child) = match &self.cache[&parent].page { + Page::Branch { + prefix, children, .. + } => ( + prefix.clone(), + children + .iter() + .find(|child| child.label == label) + .ok_or_else(|| absent("route label is absent in fixed qualified metadata"))? + .clone(), + ), + Page::Leaf { .. } => return Err(absent("route descends past a qualified leaf")), + }; + let id = child.child_page_id; + self.load(id).await?; + let received = &self.cache[&id]; + let mut partition = prefix; + partition.push(label); + let valid = match &received.page { + Page::Leaf { entries } => entries + .iter() + .all(|entry| entry.name.starts_with(&partition)), + Page::Branch { prefix, .. } => prefix.starts_with(&partition), + }; + if !valid || received.entries != child.subtree_entries { + return Err(integrity( + "qualified radix partition differs from its exact child", + )); + } + Ok(id) + } + + async fn directory( + &mut self, + mut root: [u8; 32], + path: &str, + ) -> Result<[u8; 32], SnapshotError> { + if path == "/" { + return Ok(root); + } + let mut directory_path = String::new(); + let mut source = self.source_ids["/"]; + for name in path[1..].split('/') { + let mut id = root; + loop { + self.load(id).await?; + let entry = match &self.cache[&id].page { + Page::Leaf { entries } => entries + .iter() + .find(|entry| entry.name == name.as_bytes()) + .cloned(), + Page::Branch { + prefix, terminal, .. + } => { + if name.as_bytes() == prefix { + terminal.clone() + } else { + if !name.as_bytes().starts_with(prefix) { + return Err(absent("name is absent in qualified directory")); + } + let label = + name.as_bytes().get(prefix.len()).copied().ok_or_else(|| { + absent("name is absent in qualified directory") + })?; + id = self.descend(id, label).await?; + continue; + } + } + } + .ok_or_else(|| absent("name is absent in qualified directory"))?; + if !entry.is_dir() { + return Err(SnapshotError::new( + crate::ceres::snapshot::error::SnapshotErrorCode::NotDirectory, + "fixed qualified path is not a directory", + )); + } + let children = self + .source_entries(source, root, std::slice::from_ref(&entry)) + .await?; + source = *children.get(&entry.name).ok_or_else(|| { + integrity("qualified directory route lost its exact source-name binding") + })?; + root = entry.child_root; + directory_path.push('/'); + directory_path.push_str(name); + self.directories.insert(directory_path.clone(), root); + self.source_ids.insert(directory_path.clone(), source); + break; + } + } + Ok(root) + } + async fn route( + &mut self, + root: [u8; 32], + labels: &[u8], + ) -> Result, SnapshotError> { + self.load(root).await?; + let mut pages = vec![root]; + let mut current = root; + for label in labels { + current = self.descend(current, *label).await?; + pages.push(current); + } + Ok(pages) + } +} + +const PAGE_SQL:&str="SELECT body.payload,body.byte_size,proof.canonical_proof->'references' AS references + FROM mst2_metadata_page_certificate proof JOIN mst2_metadata_current cur USING(page_id,generation) + JOIN mst2_metadata_lifetime life USING(page_id,generation) JOIN mst2_metadata_graph_node node USING(page_id,generation) + JOIN mst2_metadata_payload body USING(page_id,generation) + WHERE proof.page_id=$1 AND proof.generation=$2 AND proof.certificate_digest=$3 + AND proof.namespace_uuid=(SELECT namespace_uuid FROM mst2_metadata_family_identity WHERE singleton=1) + AND life.state='LIVE' AND life.graph_domain='qualified-v1' AND node.state='LIVE' + AND node.certificate_digest=proof.certificate_digest AND node.bytes=proof.byte_size + AND life.expected_size=proof.byte_size AND body.byte_size=proof.byte_size AND body.metadata_codec=1 + AND octet_length(body.payload)=proof.byte_size AND NOT EXISTS(SELECT 1 FROM mst2_metadata_gc_op gc + WHERE gc.page_id=proof.page_id AND gc.generation=proof.generation) + AND EXISTS(SELECT 1 FROM mst2_metadata_reader_operation reader JOIN mst2_metadata_root_anchor anchor + ON anchor.reader_operation_id=reader.operation_id AND anchor.reader_issuance=reader.reader_issuance AND anchor.anchor_kind='READER' + AND anchor.root_page=reader.root_page AND anchor.root_generation=reader.root_generation + WHERE reader.operation_id=$4::uuid AND reader.reader_issuance=$5::bigint AND reader.state='ACTIVE' + AND reader.hard_deadline_unix>floor(extract(epoch FROM clock_timestamp()))::bigint) + AND (SELECT count(*) FROM mst2_metadata_verified_ref ref WHERE ref.parent_page=proof.page_id + AND ref.parent_generation=proof.generation)=jsonb_array_length(proof.canonical_proof->'references') + AND NOT EXISTS((SELECT ref.child_page,ref.child_generation FROM mst2_metadata_verified_ref ref + WHERE ref.parent_page=proof.page_id AND ref.parent_generation=proof.generation) EXCEPT + (SELECT edge.child_page,edge.child_generation FROM mst2_metadata_graph_edge edge + WHERE edge.parent_page=proof.page_id AND edge.parent_generation=proof.generation)) + AND NOT EXISTS((SELECT edge.child_page,edge.child_generation FROM mst2_metadata_graph_edge edge + WHERE edge.parent_page=proof.page_id AND edge.parent_generation=proof.generation) EXCEPT + (SELECT ref.child_page,ref.child_generation FROM mst2_metadata_verified_ref ref + WHERE ref.parent_page=proof.page_id AND ref.parent_generation=proof.generation))"; diff --git a/src/jupiter/storage/qualified_metadata_reader_lifecycle.sql b/src/jupiter/storage/qualified_metadata_reader_lifecycle.sql new file mode 100644 index 00000000..058c9a84 --- /dev/null +++ b/src/jupiter/storage/qualified_metadata_reader_lifecycle.sql @@ -0,0 +1,110 @@ +CREATE FUNCTION mst2_metadata_next_reader_issuance(issued bigint) RETURNS bigint LANGUAGE plpgsql IMMUTABLE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + IF issued IS NULL OR issued<0 OR issued=9223372036854775807 THEN + RAISE EXCEPTION 'qualified reader issuance is exhausted or invalid'; END IF; + RETURN issued+1; +END $$; + +CREATE FUNCTION mst2_metadata_reader_issuance_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + IF TG_OP<>'UPDATE' THEN RAISE EXCEPTION 'qualified reader issuance cannot be reset or removed'; END IF; + IF NEW.singleton IS DISTINCT FROM OLD.singleton OR NEW.high_water IS DISTINCT FROM mst2_metadata_next_reader_issuance(OLD.high_water) THEN + RAISE EXCEPTION 'qualified reader issuance must advance exactly once'; END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_metadata_reader_issuance_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_metadata_reader_issuance + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_reader_issuance_guard(); + +CREATE FUNCTION mst2_metadata_reader_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE l mst2_qualified_lease_binding%ROWTYPE; issued bigint; + now_unix bigint:=floor(extract(epoch FROM clock_timestamp()))::bigint; +BEGIN + IF TG_OP='DELETE' THEN + IF OLD.state NOT IN ('FINISHED','EXPIRED') OR OLD.terminal_xid IS NULL OR OLD.terminal_xid=txid_current() + OR OLD.state='EXPIRED' AND OLD.hard_deadline_unix>now_unix + OR EXISTS(SELECT 1 FROM mst2_metadata_root_anchor WHERE reader_operation_id=OLD.operation_id) THEN + RAISE EXCEPTION 'qualified reader still has its active owner, deferred completion, or owned roots'; END IF; + RETURN OLD; + END IF; + IF TG_OP='UPDATE' THEN + IF (to_jsonb(NEW)-'state') IS DISTINCT FROM (to_jsonb(OLD)-'state') + OR OLD.state<>'ACTIVE' AND NEW.state<>OLD.state OR NEW.state NOT IN ('ACTIVE','FINISHED','EXPIRED') + OR NEW.state='EXPIRED' AND OLD.hard_deadline_unix>now_unix THEN + RAISE EXCEPTION 'qualified reader identity cannot change or be prematurely expired'; END IF; + IF OLD.state<>'ACTIVE' THEN RETURN NULL; END IF; + IF NEW.state<>'ACTIVE' THEN NEW.terminal_xid:=txid_current(); END IF; + RETURN NEW; + END IF; + SELECT * INTO l FROM mst2_qualified_lease_binding WHERE lease_id=NEW.lease_id AND state='ACTIVE' AND expires_at_unix>now_unix; + IF NOT FOUND OR NEW.state<>'ACTIVE' OR NEW.terminal_xid IS NOT NULL OR NEW.reader_issuance<=0 + OR substr(NEW.operation_id::text,15,1)<>'4' OR substr(NEW.operation_id::text,20,1) NOT IN ('8','9','a','b') + OR ROW(NEW.snapshot_id,NEW.session_incarnation,NEW.root_page,NEW.root_generation,NEW.lease_epoch) IS DISTINCT FROM + ROW(l.snapshot_id,l.session_incarnation,l.metadata_root,l.root_generation,l.lease_epoch) + OR NEW.hard_deadline_unix<=now_unix OR NEW.hard_deadline_unix>least(l.expires_at_unix,now_unix+60) + OR NOT mst2_metadata_incarnation_proof(l.snapshot_id,l.session_incarnation,true) THEN + RAISE EXCEPTION 'qualified reader lacks its exact active lease and bounded deadline'; END IF; + SELECT high_water INTO STRICT issued FROM mst2_metadata_reader_issuance WHERE singleton=1; + IF NEW.reader_issuance<>mst2_metadata_next_reader_issuance(issued) THEN RAISE EXCEPTION 'qualified reader cannot replay an issued identity'; END IF; + UPDATE mst2_metadata_reader_issuance SET high_water=NEW.reader_issuance WHERE singleton=1; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_metadata_reader_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_metadata_reader_operation + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_reader_guard(); + +CREATE FUNCTION mst2_metadata_prune_readers(maximum integer) RETURNS bigint LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE removed bigint; now_unix bigint:=floor(extract(epoch FROM clock_timestamp()))::bigint; +BEGIN + PERFORM $CORE_SCHEMA$.mst2_route_enter($CORE_LITERAL$); + IF maximum IS NULL OR maximum NOT BETWEEN 0 AND 64 THEN RAISE EXCEPTION 'qualified reader prune budget must be 0..=64'; END IF; + WITH candidates AS ( + SELECT r.operation_id,r.reader_issuance FROM mst2_metadata_reader_operation r + WHERE r.state IN ('FINISHED','EXPIRED') AND r.terminal_xid (usize, usize) { + let start = source.find(&format!("CREATE FUNCTION {name}(")).unwrap(); + let f = &source[start..]; + let body = f.find("AS $").unwrap() + 3; + let tag_end = f[body + 1..].find('$').unwrap() + body + 2; + let tag = &f[body..tag_end]; + let end = f[tag_end..].find(tag).unwrap() + tag_end + tag.len() + 1; + assert_eq!(f.as_bytes()[end - 1], b';'); + (start, start + end) +} + +fn render_previous( + core: &str, + core_oid: i64, + q: &str, + q_oid: i64, + namespace: &str, + storage: &str, +) -> String { + crate::jupiter::migration::qualified_native_runtime_upgrade::render_prior_family( + core, core_oid, q, q_oid, namespace, storage, PREVIOUS, + ) + .unwrap() +} + +struct Table { + name: String, + rows: Value, + dependencies: BTreeSet, +} + +async fn fingerprint(txn: &DatabaseTransaction, core_oid: i64, q_oid: i64, exempt: i64) -> Vec { + catalog_with_exemption(txn, core_oid, q_oid, exempt) + .await + .unwrap() +} + +/// Recreate Q relations from captured old DDL within an owned test schema. +/// Preserve its namespace OID, scope and real source/certificate/session rows. +pub(crate) async fn restore_previous(core: &DatabaseConnection, q_schema: &str, keep_q: bool) { + let txn = core.begin().await.unwrap(); + let (core_schema, core_oid) = captured_core(&txn).await.unwrap(); + assert!(core_schema.starts_with("mega2_test_")); + assert!(q_schema.starts_with("mst2q_")); + let c = identifier(&core_schema); + let q = identifier(q_schema); + txn.execute_unprepared(&format!( + "SELECT {c}.mst2_route_enter({})", + literal(&core_schema) + )) + .await + .unwrap(); + let registry=txn.query_one_raw(Statement::from_sql_and_values(DbBackend::Postgres, + format!("SELECT metadata_schema_oid::bigint,namespace_uuid::text,metadata_storage_uuid FROM {c}.mst2_metadata_namespace WHERE metadata_schema=$1 AND graph_domain='qualified-v1'"), + [q_schema.into()])).await.unwrap().unwrap(); + let q_oid: i64 = registry.try_get("", "metadata_schema_oid").unwrap(); + let namespace: String = registry.try_get("", "namespace_uuid").unwrap(); + let storage: String = registry.try_get("", "metadata_storage_uuid").unwrap(); + let table_rows=txn.query_all_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "SELECT relname FROM pg_catalog.pg_class WHERE relnamespace=$1::bigint::oid AND relkind='r' ORDER BY relname",[q_oid.into()])).await.unwrap(); + let mut tables = Vec::new(); + let mut all_names = Vec::new(); + for row in table_rows { + let name: String = row.try_get("", "relname").unwrap(); + all_names.push(name.clone()); + if [ + "mst2_metadata_storage_scope", + "mst2_metadata_family_identity", + "mst2_metadata_reader_issuance", + ] + .contains(&name.as_str()) + { + continue; + } + let rows:Value=txn.query_one_raw(Statement::from_string(DbBackend::Postgres,format!( + "SELECT coalesce(jsonb_agg(to_jsonb(data)),'[]'::jsonb) AS rows FROM {q}.{} data",identifier(&name)))) + .await.unwrap().unwrap().try_get("","rows").unwrap(); + if !keep_q { + assert_eq!( + rows, + json!([]), + "Q-less fixture must have no history to discard" + ); + } + let dependencies=txn.query_all_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "SELECT target.relname FROM pg_catalog.pg_constraint fk JOIN pg_catalog.pg_class source ON source.oid=fk.conrelid + JOIN pg_catalog.pg_class target ON target.oid=fk.confrelid WHERE source.relnamespace=$1::bigint::oid + AND target.relnamespace=source.relnamespace AND source.relname=$2 AND target.relname<>source.relname AND fk.contype='f'", + [q_oid.into(),name.clone().into()])).await.unwrap().into_iter() + .map(|row|row.try_get("","relname").unwrap()).collect(); + tables.push(Table { + name, + rows, + dependencies, + }); + } + txn.execute_unprepared(&format!( + "DROP TABLE {} CASCADE", + all_names + .iter() + .map(|name| format!("{q}.{}", identifier(name))) + .collect::>() + .join(",") + )) + .await + .unwrap(); + let funcs=txn.query_all_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "SELECT proname,pg_catalog.pg_get_function_identity_arguments(oid) AS args FROM pg_catalog.pg_proc WHERE pronamespace=$1::bigint::oid",[q_oid.into()])).await.unwrap(); + for row in funcs { + let name: String = row.try_get("", "proname").unwrap(); + let args: String = row.try_get("", "args").unwrap(); + txn.execute_unprepared(&format!( + "DROP FUNCTION IF EXISTS {q}.{}({args}) CASCADE", + identifier(&name) + )) + .await + .unwrap(); + } + txn.execute_unprepared(&render_previous( + &core_schema, + core_oid, + q_schema, + q_oid, + &namespace, + &storage, + )) + .await + .unwrap(); + let old_tables=txn.query_all_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "SELECT relname FROM pg_catalog.pg_class WHERE relnamespace=$1::bigint::oid AND relkind='r'",[q_oid.into()])).await.unwrap(); + let old_names: Vec = old_tables + .into_iter() + .map(|row| row.try_get("", "relname").unwrap()) + .collect(); + for name in &old_names { + txn.execute_unprepared(&format!( + "ALTER TABLE {q}.{} DISABLE TRIGGER USER", + identifier(name) + )) + .await + .unwrap(); + } + let mut restored = BTreeSet::from([ + "mst2_metadata_storage_scope".to_owned(), + "mst2_metadata_family_identity".to_owned(), + ]); + while !tables.is_empty() { + let index = tables + .iter() + .position(|table| table.dependencies.is_subset(&restored)) + .expect("captured Q foreign keys are acyclic"); + let table = tables.remove(index); + txn.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres,format!( + "INSERT INTO {q}.{t} SELECT * FROM jsonb_populate_recordset(NULL::{q}.{t},$1::jsonb)",t=identifier(&table.name)),[table.rows.into()])).await.unwrap(); + restored.insert(table.name); + } + for name in &old_names { + txn.execute_unprepared(&format!( + "ALTER TABLE {q}.{} ENABLE TRIGGER USER", + identifier(name) + )) + .await + .unwrap(); + } + txn.execute_unprepared(&format!("SET LOCAL search_path={c},pg_catalog,pg_temp")) + .await + .unwrap(); + let capture = CAPTURE.replace("\r\n", "\n"); + let (a, z) = function(&capture, "mst2_route_family_registration_guard"); + let registration = capture[a..z] + .replacen("CREATE FUNCTION", "CREATE OR REPLACE FUNCTION", 1) + .replace("$CORE_SCHEMA$", &c) + .replace("$IMPLEMENTATION_SHA$", PREVIOUS); + txn.execute_unprepared(®istration).await.unwrap(); + let old_shape: Vec = txn + .query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + format!( + "SELECT {c}.mst2_route_family_shape($1::bigint::oid,$2::uuid,$3) AS fingerprint" + ), + [q_oid.into(), namespace.clone().into(), storage.into()], + )) + .await + .unwrap() + .unwrap() + .try_get("", "fingerprint") + .unwrap(); + txn.execute_unprepared(&format!("SET LOCAL search_path={c},pg_catalog,pg_temp")) + .await + .unwrap(); + if !keep_q { + txn.execute_unprepared(&format!( + "ALTER TABLE {c}.mst2_metadata_namespace DISABLE TRIGGER mst2_route_immutable" + )) + .await + .unwrap(); + txn.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + format!("DELETE FROM {c}.mst2_metadata_namespace WHERE namespace_uuid=$1::uuid"), + [namespace.clone().into()], + )) + .await + .unwrap(); + txn.execute_unprepared(&format!("ALTER TABLE {c}.mst2_metadata_namespace ENABLE TRIGGER mst2_route_immutable; DROP SCHEMA {q} CASCADE")).await.unwrap(); + } + let authority = fingerprint(&txn, core_oid, 0, if keep_q { q_oid } else { 0 }).await; + let full = if keep_q { + Some(fingerprint(&txn, core_oid, q_oid, 0).await) + } else { + None + }; + txn.execute_unprepared(&format!("ALTER TABLE {c}.mst2_qualified_family_policy DISABLE TRIGGER mst2_route_family_policy_immutable")).await.unwrap(); + txn.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres,format!( + "UPDATE {c}.mst2_qualified_family_policy SET implementation_fingerprint=$1,expected_shape=$2,authority_catalog=$3 WHERE singleton=1"), + [hex::decode(PREVIOUS).unwrap().into(),old_shape.into(),authority.clone().into()])).await.unwrap(); + txn.execute_unprepared(&format!("ALTER TABLE {c}.mst2_qualified_family_policy ENABLE TRIGGER mst2_route_family_policy_immutable")).await.unwrap(); + if let Some(full) = &full { + txn.execute_unprepared(&format!( + "ALTER TABLE {c}.mst2_metadata_namespace DISABLE TRIGGER mst2_route_immutable" + )) + .await + .unwrap(); + txn.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres,format!( + "UPDATE {c}.mst2_metadata_namespace SET implementation_fingerprint=$1,catalog_fingerprint=$2 WHERE namespace_uuid=$3::uuid"), + [hex::decode(PREVIOUS).unwrap().into(),full.clone().into(),namespace.into()])).await.unwrap(); + txn.execute_unprepared(&format!( + "ALTER TABLE {c}.mst2_metadata_namespace ENABLE TRIGGER mst2_route_immutable" + )) + .await + .unwrap(); + } + assert_eq!( + fingerprint(&txn, core_oid, 0, if keep_q { q_oid } else { 0 }).await, + authority + ); + if let Some(full) = full { + assert_eq!(fingerprint(&txn, core_oid, q_oid, 0).await, full); + } + txn.commit().await.unwrap(); +} + +/// Restore the exact cc90 functions and stamps in an explicitly owned schema. +/// Existing reader generations, rows, relations and function OIDs are retained. +pub(crate) async fn restore_retention( + core: &DatabaseConnection, + q_schema: &str, + keep_q: bool, + legacy_shape: bool, +) { + restore_reader_family( + core, + q_schema, + keep_q, + legacy_shape, + crate::jupiter::migration::qualified_native_runtime_upgrade::RETENTION_IMPLEMENTATION, + ) + .await; +} + +/// Restore the exact 87b runtime family without rebuilding any owned history. +pub(crate) async fn restore_native_runtime( + core: &DatabaseConnection, + q_schema: &str, + keep_q: bool, +) { + restore_reader_family( + core, + q_schema, + keep_q, + false, + crate::jupiter::migration::qualified_native_runtime_upgrade::NATIVE_RUNTIME_IMPLEMENTATION, + ) + .await; +} + +async fn restore_reader_family( + core: &DatabaseConnection, + q_schema: &str, + keep_q: bool, + legacy_shape: bool, + implementation: &str, +) { + use crate::jupiter::migration::qualified_native_runtime_upgrade::render_prior_family; + + assert!(!legacy_shape || !keep_q); + let txn = core.begin().await.unwrap(); + let (core_schema, core_oid) = captured_core(&txn).await.unwrap(); + assert!(core_schema.starts_with("mega2_test_")); + assert!(q_schema.starts_with("mst2q_")); + let c = identifier(&core_schema); + let q = identifier(q_schema); + txn.execute_unprepared(&format!( + "SELECT {c}.mst2_route_enter({})", + literal(&core_schema) + )) + .await + .unwrap(); + let registry = txn.query_one_raw(Statement::from_sql_and_values(DbBackend::Postgres, + format!("SELECT metadata_schema_oid::bigint,namespace_uuid::text,metadata_storage_uuid + FROM {c}.mst2_metadata_namespace WHERE metadata_schema=$1 AND graph_domain='qualified-v1'"), + [q_schema.into()])).await.unwrap().unwrap(); + let oid: i64 = registry.try_get("", "metadata_schema_oid").unwrap(); + let namespace: String = registry.try_get("", "namespace_uuid").unwrap(); + let storage: String = registry.try_get("", "metadata_storage_uuid").unwrap(); + let rendered = render_prior_family( + &core_schema, + core_oid, + q_schema, + oid, + &namespace, + &storage, + implementation, + ) + .unwrap(); + txn.execute_unprepared(&format!("SET LOCAL search_path={q},pg_catalog,pg_temp")) + .await + .unwrap(); + for name in [ + "mst2_metadata_decode_rooted_plan", + "mst2_metadata_descriptor", + "mst2_metadata_dml_barrier", + "mst2_metadata_gc_enabled", + ] { + let (a, z) = function(&rendered, name); + txn.execute_unprepared(&rendered[a..z].replacen( + "CREATE FUNCTION", + "CREATE OR REPLACE FUNCTION", + 1, + )) + .await + .unwrap(); + } + txn.execute_unprepared(&format!("SET LOCAL search_path={c},pg_catalog,pg_temp")) + .await + .unwrap(); + let capture = CAPTURE.replace("\r\n", "\n"); + let (a, z) = function(&capture, "mst2_route_family_registration_guard"); + txn.execute_unprepared( + &capture[a..z] + .replacen("CREATE FUNCTION", "CREATE OR REPLACE FUNCTION", 1) + .replace("$CORE_SCHEMA$", &c) + .replace("$IMPLEMENTATION_SHA$", implementation), + ) + .await + .unwrap(); + let old_shape: Vec = if legacy_shape { + txn.execute_unprepared(&format!("SET LOCAL search_path={q},pg_catalog,pg_temp")) + .await + .unwrap(); + txn.query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + include_str!("qualified_family_shape.sql") + .replace("q_oid", "$1::bigint::oid") + .replace("n_uuid", "$2::uuid") + .replace("s_uuid", "$3::text"), + [oid.into(), namespace.clone().into(), storage.clone().into()], + )) + .await + .unwrap() + .unwrap() + .try_get("", "fingerprint") + .unwrap() + } else { + txn.query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + format!( + "SELECT {c}.mst2_route_family_shape($1::bigint::oid,$2::uuid,$3) AS fingerprint" + ), + [oid.into(), namespace.clone().into(), storage.into()], + )) + .await + .unwrap() + .unwrap() + .try_get("", "fingerprint") + .unwrap() + }; + txn.execute_unprepared(&format!("SET LOCAL search_path={c},pg_catalog,pg_temp")) + .await + .unwrap(); + if !keep_q { + let tables = txn.query_all_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "SELECT relname FROM pg_catalog.pg_class WHERE relnamespace=$1::bigint::oid AND relkind='r'", + [oid.into()])).await.unwrap(); + for row in tables { + let name: String = row.try_get("", "relname").unwrap(); + let count: i64 = txn + .query_one_raw(Statement::from_string( + DbBackend::Postgres, + format!("SELECT count(*) FROM {q}.{}", identifier(&name)), + )) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap(); + assert_eq!( + count, + if [ + "mst2_metadata_storage_scope", + "mst2_metadata_family_identity", + "mst2_metadata_reader_issuance" + ] + .contains(&name.as_str()) + { + 1 + } else { + 0 + }, + "Q-less captured runtime fixture must have no history to discard: {name}" + ); + } + let water: i64 = txn + .query_one_raw(Statement::from_string( + DbBackend::Postgres, + format!("SELECT high_water FROM {q}.mst2_metadata_reader_issuance"), + )) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap(); + assert_eq!(water, 0); + txn.execute_unprepared(&format!( + "ALTER TABLE {c}.mst2_metadata_namespace DISABLE TRIGGER mst2_route_immutable" + )) + .await + .unwrap(); + txn.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + format!("DELETE FROM {c}.mst2_metadata_namespace WHERE namespace_uuid=$1::uuid"), + [namespace.clone().into()], + )) + .await + .unwrap(); + txn.execute_unprepared(&format!("ALTER TABLE {c}.mst2_metadata_namespace ENABLE TRIGGER mst2_route_immutable; DROP SCHEMA {q} CASCADE")).await.unwrap(); + } + let authority = fingerprint(&txn, core_oid, 0, if keep_q { oid } else { 0 }).await; + let full = if keep_q { + Some(fingerprint(&txn, core_oid, oid, 0).await) + } else { + None + }; + txn.execute_unprepared(&format!("ALTER TABLE {c}.mst2_qualified_family_policy DISABLE TRIGGER mst2_route_family_policy_immutable")).await.unwrap(); + txn.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres, + format!("UPDATE {c}.mst2_qualified_family_policy SET implementation_fingerprint=$1,expected_shape=$2,authority_catalog=$3 WHERE singleton=1"), + [hex::decode(implementation).unwrap().into(), old_shape.into(), authority.clone().into()])).await.unwrap(); + txn.execute_unprepared(&format!("ALTER TABLE {c}.mst2_qualified_family_policy ENABLE TRIGGER mst2_route_family_policy_immutable")).await.unwrap(); + if let Some(full) = &full { + txn.execute_unprepared(&format!( + "ALTER TABLE {q}.mst2_metadata_family_identity DISABLE TRIGGER mst2_00_family_barrier; + ALTER TABLE {q}.mst2_metadata_family_identity DISABLE TRIGGER mst2_metadata_identity_immutable; + ALTER TABLE {c}.mst2_metadata_namespace DISABLE TRIGGER mst2_00_route_statement_barrier; + ALTER TABLE {c}.mst2_metadata_namespace DISABLE TRIGGER mst2_route_immutable")).await.unwrap(); + txn.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres, + format!("UPDATE {q}.mst2_metadata_family_identity SET implementation_fingerprint=$1 WHERE singleton=1"), + [hex::decode(implementation).unwrap().into()])).await.unwrap(); + txn.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres, + format!("UPDATE {c}.mst2_metadata_namespace SET implementation_fingerprint=$1,catalog_fingerprint=$2 WHERE namespace_uuid=$3::uuid"), + [hex::decode(implementation).unwrap().into(), full.clone().into(), namespace.into()])).await.unwrap(); + txn.execute_unprepared(&format!( + "ALTER TABLE {q}.mst2_metadata_family_identity ENABLE TRIGGER mst2_00_family_barrier; + ALTER TABLE {q}.mst2_metadata_family_identity ENABLE TRIGGER mst2_metadata_identity_immutable; + ALTER TABLE {c}.mst2_metadata_namespace ENABLE TRIGGER mst2_00_route_statement_barrier; + ALTER TABLE {c}.mst2_metadata_namespace ENABLE TRIGGER mst2_route_immutable")).await.unwrap(); + assert_eq!(fingerprint(&txn, core_oid, oid, 0).await, *full); + } + assert_eq!( + fingerprint(&txn, core_oid, 0, if keep_q { oid } else { 0 }).await, + authority + ); + txn.commit().await.unwrap(); +} diff --git a/src/jupiter/storage/qualified_metadata_rooted.rs b/src/jupiter/storage/qualified_metadata_rooted.rs new file mode 100644 index 00000000..05527117 --- /dev/null +++ b/src/jupiter/storage/qualified_metadata_rooted.rs @@ -0,0 +1,766 @@ +use std::{collections::BTreeMap, sync::Arc}; + +use mst2_codec::metapage::{Page, page_id}; +use sea_orm::{QueryResult, Value}; +use serde_json::json; + +use super::*; +use crate::ceres::snapshot::{ + metadata_install::MetadataInstallIdentity, + rooted_metadata_install::{RootedMetadataInstallPlan, RootedReuseRoot}, + rooted_metadata_projection::{CertifiedReusableDirectory, RootedReuseLookup}, +}; + +#[path = "qualified_metadata_session.rs"] +mod sessions; + +#[path = "qualified_metadata_gc.rs"] +mod gc; +#[path = "qualified_metadata_reader.rs"] +mod reader; +pub(crate) use reader::RootedLookupStatus; +#[cfg(test)] +pub(crate) use reader::{ + with_rooted_reader_barriers, with_rooted_source_fact_barriers, + with_rooted_source_temporary_shadow, +}; + +#[derive(Debug, Clone, PartialEq, Eq)] +pub(crate) struct RootedPrepareIntent { + prepare_id: String, + operation_id: String, + plan: Arc, + bindings: BTreeMap<[u8; 32], (i64, u64)>, + manifest_digest: [u8; 32], + bindings_digest: [u8; 32], + primary_scope: Vec, + storage_seal: [u8; 32], + root_generation: i64, +} + +#[derive(Debug, Clone, PartialEq, Eq)] +pub(crate) struct RootedMetadataReceipt { + intent: RootedPrepareIntent, + certificate_digest: [u8; 32], + attestation_id: uuid::Uuid, + attestation_digest: [u8; 32], +} + +impl RootedPrepareIntent { + #[cfg(test)] + pub(crate) fn prepare_id(&self) -> &str { + &self.prepare_id + } + #[cfg(test)] + pub(crate) fn metadata_root(&self) -> [u8; 32] { + self.plan.root + } + #[cfg(test)] + pub(crate) fn root_generation(&self) -> i64 { + self.root_generation + } +} +impl RootedMetadataReceipt { + pub(crate) fn metadata_root(&self) -> [u8; 32] { + self.intent.plan.root + } +} + +pub(crate) struct RootedQualifiedMetadataRepository { + connection: DatabaseConnection, + namespace: VerifiedQualifiedNamespace, + primary_scope: Vec, + maintenance_state: tokio::sync::Mutex, +} + +#[async_trait::async_trait] +impl RootedReuseLookup for RootedQualifiedMetadataRepository { + async fn lookup_reuse( + &self, + tree_oid: &str, + identity: &MetadataInstallIdentity, + ) -> Result, SnapshotError> { + // This is a bounded projection hint. It acquires no writer/retention + // lock and grants no installation authority: begin_intent and finalize + // independently bind the exact attestation and current lifetime. + let kind = identity + .tagged_root_tree_oid + .split_once(':') + .ok_or_else(|| integrity("rooted source identity has no tagged hash kind"))? + .0; + let profile = json!({"source_domain": identity.source_domain,"hash_kind": kind, + "schema_version": identity.schema_version,"metadata_codec": identity.metadata_codec, + "materialization_policy": identity.materialization_policy,"fs_semantics": identity.fs_semantics, + "access_projection": identity.access_projection,"verification_revision": identity.verification_revision, + "projection_revision": identity.projection_revision}); + let q = identifier(&self.namespace.schema); + let c = identifier(&self.namespace.core_schema); + let row = self.connection.query_one_raw(sql(format!( + "SELECT a.root_page,a.root_generation,a.attestation_id::text,a.attestation_digest, + p.certificate_digest,p.relative_path_bytes,p.relative_components,p.closure_nodes_upper, + p.closure_edges_upper,p.closure_bytes_upper,p.closure_entries_upper + FROM {q}.mst2_metadata_reuse_index i JOIN {q}.mst2_metadata_source_root_attestation a + ON a.attestation_id=i.attestation_id AND a.root_page=i.root_page AND a.root_generation=i.root_generation + AND a.attestation_digest=i.attestation_digest + JOIN {q}.mst2_metadata_page_certificate p ON p.page_id=a.root_page AND p.generation=a.root_generation + AND p.certificate_digest=a.root_certificate_digest + JOIN {q}.mst2_metadata_current cur ON cur.page_id=p.page_id AND cur.generation=p.generation + JOIN {q}.mst2_metadata_lifetime life ON life.page_id=p.page_id AND life.generation=p.generation + JOIN {q}.mst2_metadata_graph_node node ON node.page_id=p.page_id AND node.generation=p.generation + JOIN {q}.mst2_metadata_payload body ON body.page_id=p.page_id AND body.generation=p.generation + JOIN {q}.mst2_metadata_prepare origin ON origin.prepare_id=a.origin_prepare_id + JOIN {c}.mega_tree source ON source.tree_id=split_part(a.tagged_tree_oid,':',2) + WHERE i.tagged_tree_oid=$1 AND i.profile_digest=sha256(convert_to('mega.mst2.native-profile.v1','UTF8') + ||decode('00','hex')||convert_to($2::jsonb::text,'UTF8')) AND a.source_profile=$2::jsonb + AND a.namespace_uuid=$3::uuid AND {c}.mst2_route_source_tree_matches(split_part(a.tagged_tree_oid,':',2),a.source_revision,a.source_body_digest) + AND life.state='LIVE' AND life.graph_domain='qualified-v1' AND node.state='LIVE' + AND node.certificate_digest=p.certificate_digest AND node.bytes=p.byte_size + AND body.byte_size=p.byte_size AND body.metadata_codec=p.metadata_codec + AND life.expected_size=p.byte_size AND origin.state='COMMITTED' AND NOT pg_is_in_recovery() + AND NOT EXISTS(SELECT 1 FROM {q}.mst2_metadata_gc_op gc WHERE gc.page_id=p.page_id AND gc.generation=p.generation)" + ),[tree_oid.into(),profile.into(),self.namespace.namespace_uuid.clone().into()])) + .await.map_err(database_error)?; + let Some(row) = row else { + return Ok(None); + }; + let count = |name: &str| -> Result { + usize::try_from(row.try_get::("", name).map_err(internal)?).map_err(internal) + }; + let wide = |name: &str| -> Result { + u64::try_from(row.try_get::("", name).map_err(internal)?).map_err(internal) + }; + Ok(Some(CertifiedReusableDirectory { + page_id: digest_column(&row, "root_page")?, + proof: RootedReuseRoot { + generation: row.try_get("", "root_generation").map_err(internal)?, + attestation_id: uuid::Uuid::parse_str( + &row.try_get::("", "attestation_id") + .map_err(internal)?, + ) + .map_err(internal)?, + attestation_digest: digest_column(&row, "attestation_digest")?, + certificate_digest: digest_column(&row, "certificate_digest")?, + }, + relative_path_bytes: count("relative_path_bytes")?, + relative_components: count("relative_components")?, + closure_nodes_upper: count("closure_nodes_upper")?, + closure_edges_upper: count("closure_edges_upper")?, + closure_bytes_upper: wide("closure_bytes_upper")?, + closure_entries_upper: usize::try_from(wide("closure_entries_upper")?) + .map_err(internal)?, + })) + } +} + +fn sql(text: impl Into, values: impl IntoIterator) -> Statement { + Statement::from_sql_and_values(DbBackend::Postgres, text, values) +} +fn internal(error: impl std::fmt::Display) -> SnapshotError { + SnapshotError::new( + crate::ceres::snapshot::error::SnapshotErrorCode::Internal, + error.to_string(), + ) +} +pub(super) fn is_lock_unavailable(error: &sea_orm::DbErr) -> bool { + let runtime = match error { + sea_orm::DbErr::Exec(runtime) | sea_orm::DbErr::Query(runtime) => runtime, + _ => return false, + }; + let sea_orm::RuntimeErr::SqlxError(sqlx_error) = runtime else { + return false; + }; + let sea_orm::sqlx::Error::Database(database_error) = sqlx_error.as_ref() else { + return false; + }; + database_error.code().as_deref() == Some("55P03") +} +fn database_error(error: sea_orm::DbErr) -> SnapshotError { + if is_lock_unavailable(&error) { + return SnapshotError::new( + crate::ceres::snapshot::error::SnapshotErrorCode::TemporaryUnavailable, + "qualified source is being updated; retry the operation", + ); + } + internal(error) +} +fn integrity(message: &str) -> SnapshotError { + SnapshotError::new( + crate::ceres::snapshot::error::SnapshotErrorCode::IntegrityError, + message, + ) +} +fn unavailable(message: &str) -> SnapshotError { + SnapshotError::new( + crate::ceres::snapshot::error::SnapshotErrorCode::ObjectUnavailable, + message, + ) +} +fn digest_column(row: &QueryResult, name: &str) -> Result<[u8; 32], SnapshotError> { + let bytes: Vec = row.try_get("", name).map_err(internal)?; + bytes.as_slice().try_into().map_err(internal) +} +fn encode_bindings(bindings: &BTreeMap<[u8; 32], (i64, u64)>) -> Vec { + let mut bytes = Vec::with_capacity(12 + 48 * bindings.len()); + bytes.extend_from_slice(b"MST2GEN1"); + bytes.extend_from_slice(&(bindings.len() as u32).to_be_bytes()); + for (page, (generation, size)) in bindings { + bytes.extend_from_slice(page); + bytes.extend_from_slice(&generation.to_be_bytes()); + bytes.extend_from_slice(&size.to_be_bytes()); + } + bytes +} +fn decode_bindings( + bytes: &[u8], + plan: &RootedMetadataInstallPlan, +) -> Result, SnapshotError> { + if bytes.len() != 12 + 48 * plan.delta.len() || bytes.get(..8) != Some(b"MST2GEN1") { + return Err(integrity( + "rooted delta binding encoding differs from its plan", + )); + } + let count = u32::from_be_bytes(bytes[8..12].try_into().map_err(internal)?) as usize; + if count != plan.delta.len() { + return Err(integrity( + "rooted delta binding count differs from its plan", + )); + } + let mut bindings = BTreeMap::new(); + for ((expected, size), record) in plan.delta.iter().zip(bytes[12..].as_chunks::<48>().0) { + let page: [u8; 32] = record[..32].try_into().map_err(internal)?; + let generation = i64::from_be_bytes(record[32..40].try_into().map_err(internal)?); + let recorded_size = u64::from_be_bytes(record[40..48].try_into().map_err(internal)?); + if page != *expected || recorded_size != *size || generation <= 0 { + return Err(integrity("rooted exact delta lifetime binding is invalid")); + } + bindings.insert(page, (generation, recorded_size)); + } + Ok(bindings) +} + +async fn committed( + txn: DatabaseTransaction, + result: Result, + operation: &str, + digest: [u8; 32], + phase: super::super::native_metadata_install::MetadataCommitPhase, +) -> Result { + match result { + Err(error) => { + let _ = txn.rollback().await; + Err(error.into()) + } + Ok(value) => match txn.commit().await { + Ok(()) => Ok(value), + Err(_) => Err(MetadataInstallError::CommitUncertain { + operation_id: operation.into(), + manifest_digest: digest, + phase, + }), + }, + } +} + +impl RootedQualifiedMetadataRepository { + pub(crate) async fn open( + core: &DatabaseConnection, + config: &DbConfig, + ) -> Result { + let captured = captured_core(core).await?; + let namespace = registered(core, &captured) + .await? + .ok_or_else(|| rejected("rooted qualified family is not provisioned"))?; + let mut q_config = config.clone(); + q_config.db_url = pool_url(&config.db_url, &namespace)?; + q_config.max_connection = q_config.max_connection.clamp(1, 4); + q_config.min_connection = 1; + let connection = postgres_connection(&q_config).await?; + let txn = connection + .begin_with_config(Some(IsolationLevel::ReadCommitted), None) + .await?; + namespace + .enter(&txn) + .await + .map_err(|e| rejected(&e.to_string()))?; + let scope:serde_json::Value=txn.query_one_raw(sql( + "SELECT jsonb_build_array(s.storage_uuid,current_database(),d.oid::bigint,current_schema(),n.oid::bigint, + inet_server_addr()::text,inet_server_port()) AS scope FROM mst2_metadata_storage_scope s + JOIN pg_catalog.pg_database d ON d.datname=current_database() + JOIN pg_catalog.pg_namespace n ON n.nspname=current_schema() WHERE s.singleton=1 AND NOT pg_is_in_recovery()",[], + )).await?.ok_or_else(||rejected("rooted captured primary scope is missing"))?.try_get("","scope")?; + let primary_scope = serde_json::to_vec(&scope).map_err(|e| rejected(&e.to_string()))?; + txn.commit().await?; + Ok(Self { + connection, + namespace, + primary_scope, + maintenance_state: tokio::sync::Mutex::default(), + }) + } + + async fn transaction(&self) -> Result { + let txn = self + .connection + .begin_with_config(Some(IsolationLevel::ReadCommitted), None) + .await + .map_err(database_error)?; + self.namespace.enter(&txn).await?; + let valid: bool = txn + .query_one_raw(sql( + "SELECT mst2_metadata_scope_matches($1) AS valid", + [self.primary_scope.clone().into()], + )) + .await + .map_err(database_error)? + .ok_or_else(|| integrity("rooted primary scope is missing"))? + .try_get("", "valid") + .map_err(internal)?; + if !valid { + return Err(integrity("rooted writer left its captured primary scope")); + } + Ok(txn) + } + + async fn load_intent( + &self, + db: &C, + operation: &str, + plan: &RootedMetadataInstallPlan, + ) -> Result, SnapshotError> { + let Some(row)=db.query_one_raw(sql("SELECT prepare_id,plan_kind,state,manifest_digest,canonical_plan,canonical_bindings, + bindings_digest,primary_scope,storage_seal FROM mst2_metadata_prepare WHERE operation_id=$1",[operation.into()])) + .await.map_err(internal)? else {return Ok(None);}; + let digest = plan.digest()?; + if row.try_get::("", "plan_kind").map_err(internal)? != "ROOTED" + || digest_column(&row, "manifest_digest")? != digest + { + return Err(SnapshotError::new( + crate::ceres::snapshot::error::SnapshotErrorCode::Conflict, + "operation is bound to a different rooted manifest", + )); + } + let stored: Vec = row.try_get("", "canonical_plan").map_err(internal)?; + if RootedMetadataInstallPlan::decode(&stored, &digest)? != *plan { + return Err(integrity( + "rooted stored plan differs from its exact source identity", + )); + } + let encoded_bindings: Vec = row.try_get("", "canonical_bindings").map_err(internal)?; + let bindings_digest: [u8; 32] = Sha256::digest(&encoded_bindings).into(); + if bindings_digest != digest_column(&row, "bindings_digest")? { + return Err(integrity( + "rooted stored lifetime binding digest is invalid", + )); + } + let primary_scope: Vec = row.try_get("", "primary_scope").map_err(internal)?; + if primary_scope != self.primary_scope { + return Err(integrity( + "rooted operation belongs to another physical primary", + )); + } + let bindings = decode_bindings(&encoded_bindings, plan)?; + let root_generation = bindings + .get(&plan.root) + .map(|b| b.0) + .or_else(|| plan.reused.get(&plan.root).map(|b| b.generation)) + .ok_or_else(|| integrity("rooted root lifetime is missing"))?; + Ok(Some(( + RootedPrepareIntent { + prepare_id: row.try_get("", "prepare_id").map_err(internal)?, + operation_id: operation.into(), + plan: Arc::new(plan.clone()), + bindings, + manifest_digest: digest, + bindings_digest, + primary_scope, + storage_seal: digest_column(&row, "storage_seal")?, + root_generation, + }, + row.try_get("", "state").map_err(internal)?, + ))) + } + + pub(crate) async fn begin_intent( + &self, + operation: &str, + plan: &RootedMetadataInstallPlan, + ) -> Result { + plan.validate()?; + if operation.is_empty() || operation.len() > 255 || operation.contains('\0') { + return Err(integrity("invalid rooted operation ID").into()); + } + let manifest_digest = plan.digest()?; + let txn = self.transaction().await?; + let result=async { + if let Some((intent,state))=self.load_intent(&txn,operation,plan).await? { + if state=="ABORTED" {return Err(unavailable("rooted preparation was definitively aborted"));} + return Ok(intent); + } + let pages:Vec<_>=plan.delta.iter().map(|(page,size)|json!({"page":hex::encode(page),"size":size})).collect(); + let encoded=serde_json::to_string(&pages).map_err(internal)?; + txn.execute_raw(sql("INSERT INTO mst2_metadata_lifetime(page_id,node_id,generation,state,metadata_codec,expected_size,graph_domain) + SELECT decode(p.page,'hex'),'page:sha256:'||p.page,coalesce(cur.generation+1,1),'RESERVED',1,p.size,'qualified-v1' + FROM jsonb_to_recordset($1::jsonb) p(page text,size integer) + LEFT JOIN mst2_metadata_current cur ON cur.page_id=decode(p.page,'hex') + LEFT JOIN mst2_metadata_lifetime previous ON previous.page_id=cur.page_id AND previous.generation=cur.generation + WHERE (cur.page_id IS NULL AND NOT EXISTS(SELECT 1 FROM mst2_metadata_lifetime life WHERE life.page_id=decode(p.page,'hex'))) + OR (previous.state='REMOVED' AND cur.generation<9223372036854775807 + AND EXISTS(SELECT 1 FROM mst2_metadata_gc_op proof WHERE proof.page_id=cur.page_id + AND proof.generation=cur.generation AND proof.state='APPLIED'))", + [encoded.clone().into()])).await.map_err(internal)?; + txn.execute_raw(sql("INSERT INTO mst2_metadata_current(page_id,generation) + SELECT life.page_id,life.generation FROM jsonb_to_recordset($1::jsonb) p(page text,size integer) + JOIN mst2_metadata_lifetime life ON life.page_id=decode(p.page,'hex') AND life.generation=1 AND life.state='RESERVED' + WHERE NOT EXISTS(SELECT 1 FROM mst2_metadata_current cur WHERE cur.page_id=life.page_id)", + [encoded.clone().into()])).await.map_err(internal)?; + txn.execute_raw(sql("UPDATE mst2_metadata_current cur SET generation=fresh.generation + FROM jsonb_to_recordset($1::jsonb) p(page text,size integer),mst2_metadata_lifetime previous, + mst2_metadata_lifetime fresh + WHERE cur.page_id=decode(p.page,'hex') AND previous.page_id=cur.page_id AND previous.generation=cur.generation + AND previous.state='REMOVED' AND cur.generation<9223372036854775807 + AND fresh.page_id=cur.page_id AND fresh.generation=cur.generation+1 AND fresh.state='RESERVED' + AND fresh.expected_size=p.size AND fresh.metadata_codec=1 AND fresh.graph_domain='qualified-v1' + AND EXISTS(SELECT 1 FROM mst2_metadata_gc_op proof WHERE proof.page_id=cur.page_id + AND proof.generation=cur.generation AND proof.state='APPLIED')",[encoded.clone().into()])) + .await.map_err(internal)?; + let rows=txn.query_all_raw(sql("SELECT cur.page_id,cur.generation,life.expected_size,life.metadata_codec,life.graph_domain,life.state, + n.state AS graph_state,n.certificate_digest,body.byte_size FROM jsonb_to_recordset($1::jsonb) p(page text,size integer) + JOIN mst2_metadata_current cur ON cur.page_id=decode(p.page,'hex') JOIN mst2_metadata_lifetime life USING(page_id,generation) + LEFT JOIN mst2_metadata_graph_node n USING(page_id,generation) LEFT JOIN mst2_metadata_payload body USING(page_id,generation) + WHERE NOT EXISTS(SELECT 1 FROM mst2_metadata_gc_op gc WHERE gc.page_id=cur.page_id AND gc.generation=cur.generation) + ORDER BY cur.page_id",[encoded.into()])).await.map_err(internal)?; + let mut bindings=BTreeMap::new(); + for row in rows { + let page=digest_column(&row,"page_id")?; let size=row.try_get::("","expected_size").map_err(internal)? as u64; + let state:String=row.try_get("","state").map_err(internal)?; + if plan.delta.get(&page)!=Some(&size) || row.try_get::("","metadata_codec").map_err(internal)?!=1 + || row.try_get::("","graph_domain").map_err(internal)?!="qualified-v1" || !matches!(state.as_str(),"RESERVED"|"LIVE") + || state=="LIVE" && (row.try_get::>("","graph_state").map_err(internal)?.as_deref()!=Some("LIVE") + || row.try_get::>("","byte_size").map_err(internal)?!=Some(size as i32)) { + return Err(unavailable("rooted requested delta lifetime is not exactly installable")); + } + bindings.insert(page,(row.try_get("","generation").map_err(internal)?,size)); + } + if bindings.len()!=plan.delta.len() {return Err(unavailable("rooted delta lifetime coverage is incomplete"));} + let canonical_bindings=encode_bindings(&bindings); let bindings_digest:[u8;32]=Sha256::digest(&canonical_bindings).into(); + let prepare_id=uuid::Uuid::new_v4().to_string(); + let root_generation=bindings.get(&plan.root).map(|b|b.0).or_else(||plan.reused.get(&plan.root).map(|b|b.generation)) + .ok_or_else(||integrity("rooted root lifetime is missing"))?; + let mut seal=Sha256::new(); seal.update(b"mega.mst2.rooted-storage-seal.v1\0"); + seal.update(prepare_id.as_bytes()); seal.update(manifest_digest); seal.update(bindings_digest); + seal.update((self.primary_scope.len() as u64).to_be_bytes()); seal.update(&self.primary_scope); + seal.update(plan.root); seal.update(root_generation.to_be_bytes()); + let storage_seal:[u8;32]=seal.finalize().into(); let identity=&plan.identity; + txn.execute_raw(sql("INSERT INTO mst2_metadata_prepare(prepare_id,operation_id,manifest_digest,canonical_plan,plan_kind, + source_domain,tagged_root_tree_oid,scope,schema_version,metadata_codec,materialization_policy,fs_semantics,access_projection, + verification_revision,projection_revision,metadata_root,node_count,edge_count,total_bytes,state,canonical_bindings,bindings_digest, + primary_scope,storage_seal,graph_domain) VALUES($1,$2,$3,$4,'ROOTED',$5,$6,$7,$8,$9,$10,$11,$12,$13,$14,$15,$16,$17,$18,'PREPARING',$19,$20,$21,$22,'qualified-v1')", + [prepare_id.clone().into(),operation.into(),manifest_digest.to_vec().into(),plan.encode()?.into(),identity.source_domain.clone().into(), + identity.tagged_root_tree_oid.clone().into(),identity.scope.clone().into(),(identity.schema_version as i16).into(), + (identity.metadata_codec as i16).into(),(identity.materialization_policy as i16).into(),(identity.fs_semantics as i16).into(), + (identity.access_projection as i16).into(),identity.verification_revision.into(),(identity.projection_revision as i16).into(), + plan.root.to_vec().into(),(plan.delta.len() as i32).into(),(plan.edges.len() as i32).into(),(plan.delta_bytes()? as i64).into(), + canonical_bindings.into(),bindings_digest.to_vec().into(),self.primary_scope.clone().into(),storage_seal.to_vec().into()])).await.map_err(internal)?; + let delta:Vec<_>=bindings.iter().map(|(page,(generation,size))|json!({"page":hex::encode(page),"generation":generation,"size":size})).collect(); + txn.execute_raw(sql("INSERT INTO mst2_metadata_prepare_page(prepare_id,page_id,generation,expected_size) + SELECT $1,decode(p.page,'hex'),p.generation,p.size FROM jsonb_to_recordset($2::jsonb) p(page text,generation bigint,size integer)", + [prepare_id.clone().into(),serde_json::to_string(&delta).map_err(internal)?.into()])).await.map_err(internal)?; + let reused:Vec<_>=plan.reused.iter().map(|(page,proof)|json!({"page":hex::encode(page),"generation":proof.generation, + "attestation_id":proof.attestation_id.to_string(),"attestation_digest":hex::encode(proof.attestation_digest)})).collect(); + txn.execute_raw(sql("INSERT INTO mst2_metadata_prepare_reuse_root(prepare_id,root_page,root_generation,attestation_id,attestation_digest) + SELECT $1,decode(r.page,'hex'),r.generation,r.attestation_id::uuid,decode(r.attestation_digest,'hex') + FROM jsonb_to_recordset($2::jsonb) r(page text,generation bigint,attestation_id text,attestation_digest text)", + [prepare_id.clone().into(),serde_json::to_string(&reused).map_err(internal)?.into()])).await.map_err(internal)?; + txn.execute_raw(sql("INSERT INTO mst2_metadata_root_anchor(anchor_id,anchor_kind,owner_key,prepare_id,root_page,root_generation,root_certificate_digest) + SELECT gen_random_uuid(),'REUSE',$1,$1,r.root_page,r.root_generation,a.root_certificate_digest + FROM mst2_metadata_prepare_reuse_root r JOIN mst2_metadata_source_root_attestation a USING(attestation_id) WHERE r.prepare_id=$1", + [prepare_id.clone().into()])).await.map_err(internal)?; + Ok(RootedPrepareIntent {prepare_id,operation_id:operation.into(),plan:Arc::new(plan.clone()),bindings,manifest_digest, + bindings_digest,primary_scope:self.primary_scope.clone(),storage_seal,root_generation}) + }.await; + committed( + txn, + result, + operation, + manifest_digest, + super::super::native_metadata_install::MetadataCommitPhase::Intent, + ) + .await + } + + async fn require_intent( + &self, + db: &C, + expected: &RootedPrepareIntent, + ) -> Result { + let (stored, state) = self + .load_intent(db, &expected.operation_id, &expected.plan) + .await? + .ok_or_else(|| unavailable("rooted preparation is missing"))?; + if stored != *expected { + return Err(integrity( + "rooted preparation crossed its immutable physical binding", + )); + } + if state == "ABORTED" { + return Err(unavailable("rooted preparation was definitively aborted")); + } + Ok(state) + } + + pub(crate) async fn install_pages( + &self, + intent: &RootedPrepareIntent, + payloads: &[MetadataPagePayload], + ) -> Result<(), MetadataInstallError> { + if payloads.is_empty() || payloads.len() > 64 { + return Err(integrity("rooted payload batch requires 1..=64 pages").into()); + } + let mut pages = BTreeMap::new(); + for payload in payloads { + let &(generation, size) = intent + .bindings + .get(&payload.id) + .ok_or_else(|| integrity("rooted payload is outside its fixed delta"))?; + if size != payload.size + || payload.bytes.len() as u64 != size + || page_id(&payload.bytes) != payload.id + { + return Err(integrity( + "rooted delta payload differs from its exact page digest or size", + ) + .into()); + } + Page::decode(&payload.bytes).map_err(internal)?; + if pages + .insert( + payload.id, + json!({"page":hex::encode(payload.id),"generation":generation,"size":size, + "payload":hex::encode(&payload.bytes)}), + ) + .is_some() + { + return Err(integrity("rooted payload batch has duplicate pages").into()); + } + } + let encoded = + serde_json::to_string(&pages.into_values().collect::>()).map_err(internal)?; + let txn = self.transaction().await?; + let result=async { + let state=self.require_intent(&txn,intent).await?; + if state=="PREPARING" { + txn.execute_raw(sql("INSERT INTO mst2_metadata_payload(page_id,generation,metadata_codec,byte_size,payload) + SELECT decode(p.page,'hex'),p.generation,1,p.size,decode(p.payload,'hex') + FROM jsonb_to_recordset($1::jsonb) p(page text,generation bigint,size integer,payload text) + WHERE NOT EXISTS(SELECT 1 FROM mst2_metadata_payload body WHERE body.page_id=decode(p.page,'hex')) + ON CONFLICT(page_id) DO NOTHING",[encoded.clone().into()])).await.map_err(internal)?; + } + if txn.query_one_raw(sql("SELECT p.page FROM jsonb_to_recordset($1::jsonb) p(page text,generation bigint,size integer,payload text) + LEFT JOIN mst2_metadata_payload body ON body.page_id=decode(p.page,'hex') + LEFT JOIN mst2_metadata_current cur ON cur.page_id=body.page_id AND cur.generation=body.generation + WHERE body.page_id IS NULL OR cur.page_id IS NULL OR body.generation<>p.generation OR body.metadata_codec<>1 + OR body.byte_size<>p.size OR body.payload<>decode(p.payload,'hex') LIMIT 1",[encoded.into()])) + .await.map_err(internal)?.is_some() {return Err(integrity("rooted durable delta payload conflicts with its exact incarnation"));} + Ok(()) + }.await; + committed( + txn, + result, + &intent.operation_id, + intent.manifest_digest, + super::super::native_metadata_install::MetadataCommitPhase::Payload, + ) + .await + } + + pub(crate) async fn finalize( + &self, + intent: &RootedPrepareIntent, + ) -> Result { + // Rehash/decode the actual delta outside the core route lock. A cold + // preparation additionally keeps the full Rust DAG validator as oracle. + let payloads = self.read_delta(intent).await?; + if intent.plan.reused.is_empty() { + use crate::ceres::snapshot::retention_dag::{ + MetadataDagCandidate, MetadataDagLimits, ValidatedMetadataDag, + }; + ValidatedMetadataDag::validate( + MetadataDagCandidate { + metadata_codec: intent.plan.identity.metadata_codec, + root: intent.plan.root, + pages: payloads, + edges: intent.plan.edges.iter().copied().collect(), + }, + MetadataDagLimits::default(), + )?; + } + let txn = self.transaction().await?; + let result=async { + txn.query_one_raw(sql("SELECT prepare_id FROM mst2_metadata_prepare WHERE prepare_id=$1 FOR UPDATE", + [intent.prepare_id.clone().into()])).await.map_err(internal)?; + let state=self.require_intent(&txn,intent).await?; + if state=="COMMITTED" {return self.receipt(&txn,intent).await;} + let ordered:Vec<_>=intent.plan.child_first_delta()?.iter().map(|page|json!({"page":hex::encode(page), + "generation":intent.bindings[page].0})).collect(); + if !ordered.is_empty() { + let count:i32=txn.query_one_raw(sql("SELECT mst2_metadata_certify_batch($1,$2::jsonb) AS certified", + [intent.prepare_id.clone().into(),serde_json::to_string(&ordered).map_err(internal)?.into()])) + .await.map_err(internal)?.ok_or_else(||integrity("rooted certification result is missing"))?.try_get("","certified").map_err(internal)?; + if count as usize!=ordered.len() {return Err(integrity("rooted certification did not cover its exact delta"));} + } + txn.execute_raw(sql("INSERT INTO mst2_metadata_root_anchor(anchor_id,anchor_kind,owner_key,prepare_id,root_page,root_generation,root_certificate_digest) + SELECT gen_random_uuid(),'PREPARE',$1,$1,c.page_id,c.generation,c.certificate_digest + FROM mst2_metadata_page_certificate c WHERE c.page_id=$2 AND c.generation=$3 + ON CONFLICT(anchor_kind,owner_key,root_page,root_generation) DO NOTHING", + [intent.prepare_id.clone().into(),intent.plan.root.to_vec().into(),intent.root_generation.into()])).await.map_err(internal)?; + let sources:Vec<_>=intent.plan.source_roots.iter().map(|(tree,page)|json!({"tree":tree,"page":hex::encode(page)})).collect(); + let sources=txn.query_all_raw(sql("SELECT source.tree,cur.page_id FROM jsonb_to_recordset($1::jsonb) source(tree text,page text) + JOIN mst2_metadata_current cur ON cur.page_id=decode(source.page,'hex') + JOIN mst2_metadata_page_certificate proof USING(page_id,generation) ORDER BY proof.rank,source.tree", + [serde_json::to_string(&sources).map_err(internal)?.into()])).await.map_err(internal)?; + if sources.len()!=intent.plan.source_roots.len() {return Err(unavailable("rooted source certificate order is incomplete"));} + for source in sources { + let tree:String=source.try_get("","tree").map_err(internal)?; + let page=digest_column(&source,"page_id")?; + let generation=intent.bindings.get(&page).map(|b|b.0).or_else(||intent.plan.reused.get(&page).map(|b|b.generation)) + .ok_or_else(||integrity("rooted directory source has no exact lifetime"))?; + let reused=txn.query_one_raw(sql("SELECT a.attestation_id FROM mst2_metadata_prepare_reuse_root r + JOIN mst2_metadata_source_root_attestation boundary ON boundary.attestation_id=r.attestation_id + JOIN mst2_metadata_source_root_attestation a ON a.root_page=r.root_page AND a.root_generation=r.root_generation + AND a.root_certificate_digest=boundary.root_certificate_digest + JOIN mst2_metadata_prepare origin ON origin.prepare_id=a.origin_prepare_id + JOIN mst2_metadata_current cur ON cur.page_id=a.root_page AND cur.generation=a.root_generation + JOIN mst2_metadata_lifetime life USING(page_id,generation) + JOIN $CORE$.mega_tree source ON source.tree_id=split_part(a.tagged_tree_oid,':',2) + WHERE r.prepare_id=$1 AND r.root_page=$2 AND r.root_generation=$3 AND a.tagged_tree_oid=$4 + AND origin.state='COMMITTED' AND life.state='LIVE' AND a.source_profile=mst2_metadata_native_profile($1) + AND $CORE$.mst2_route_source_tree_matches(split_part(a.tagged_tree_oid,':',2),a.source_revision,a.source_body_digest) + ORDER BY a.attestation_id LIMIT 1".replace("$CORE$",&identifier(&self.namespace.core_schema)), + [intent.prepare_id.clone().into(),page.to_vec().into(),generation.into(),tree.clone().into()])).await.map_err(internal)?; + if reused.is_some() {continue;} + txn.execute_raw(sql("INSERT INTO mst2_metadata_source_root_attestation(attestation_id,namespace_uuid,origin_prepare_id,tagged_tree_oid, + source_profile,profile_digest,source_body_digest,root_page,root_generation,root_certificate_digest,source_proof,attestation_digest) + SELECT gen_random_uuid(),(proof->>'namespace')::uuid,$1,$2,proof->'source_profile',decode(proof->>'profile_digest','hex'), + decode(proof->>'source_body_digest','hex'),$3,$4,decode(proof->>'root_certificate','hex'),proof,decode(proof->>'attestation','hex') + FROM (SELECT mst2_metadata_compute_source_proof($1,$2,$3,$4) AS proof) input", + [intent.prepare_id.clone().into(),tree.clone().into(),page.to_vec().into(),generation.into()])).await.map_err(internal)?; + } + let changed=txn.execute_raw(sql("UPDATE mst2_metadata_prepare SET state='COMMITTED',committed_at=clock_timestamp() + WHERE prepare_id=$1 AND state='PREPARING' AND storage_seal=$2", + [intent.prepare_id.clone().into(),intent.storage_seal.to_vec().into()])).await.map_err(internal)?; + if changed.rows_affected()!=1 {return Err(integrity("rooted finalize lost its exact prepare transition"));} + txn.execute_raw(sql("UPDATE mst2_metadata_lifetime life SET state='LIVE' FROM mst2_metadata_prepare_page member + WHERE member.prepare_id=$1 AND life.page_id=member.page_id AND life.generation=member.generation AND life.state='RESERVED'", + [intent.prepare_id.clone().into()])).await.map_err(internal)?; + txn.execute_raw(sql("INSERT INTO mst2_metadata_reuse_index(profile_digest,tagged_tree_oid,attestation_id,root_page,root_generation,attestation_digest) + SELECT a.profile_digest,a.tagged_tree_oid,a.attestation_id,a.root_page,a.root_generation,a.attestation_digest + FROM mst2_metadata_source_root_attestation a WHERE a.origin_prepare_id=$1 + ON CONFLICT(profile_digest,tagged_tree_oid) DO NOTHING",[intent.prepare_id.clone().into()])).await.map_err(internal)?; + self.receipt(&txn,intent).await + }.await; + committed( + txn, + result, + &intent.operation_id, + intent.manifest_digest, + super::super::native_metadata_install::MetadataCommitPhase::Finalize, + ) + .await + } + + async fn read_delta( + &self, + intent: &RootedPrepareIntent, + ) -> Result, SnapshotError> { + let rows=self.connection.query_all_raw(sql("SELECT member.page_id,member.generation,member.expected_size,body.metadata_codec, + body.byte_size,body.payload FROM mst2_metadata_prepare_page member JOIN mst2_metadata_payload body USING(page_id,generation) + JOIN mst2_metadata_current cur USING(page_id,generation) WHERE member.prepare_id=$1 ORDER BY member.page_id LIMIT 4097", + [intent.prepare_id.clone().into()])).await.map_err(internal)?; + if rows.len() != intent.bindings.len() { + return Err(unavailable("rooted durable delta is incomplete")); + } + let mut payloads = Vec::with_capacity(rows.len()); + for row in rows { + let page = digest_column(&row, "page_id")?; + let &(generation, size) = intent + .bindings + .get(&page) + .ok_or_else(|| integrity("rooted durable delta has an extra member"))?; + let bytes: Vec = row.try_get("", "payload").map_err(internal)?; + if row.try_get::("", "generation").map_err(internal)? != generation + || row.try_get::("", "expected_size").map_err(internal)? as u64 != size + || row.try_get::("", "byte_size").map_err(internal)? as u64 != size + || row.try_get::("", "metadata_codec").map_err(internal)? != 1 + || bytes.len() as u64 != size + || page_id(&bytes) != page + { + return Err(integrity( + "rooted durable delta crossed its fixed byte and lifetime binding", + )); + } + Page::decode(&bytes).map_err(internal)?; + payloads.push(MetadataPagePayload { + id: page, + size, + bytes, + }); + } + Ok(payloads) + } + + async fn receipt( + &self, + db: &C, + intent: &RootedPrepareIntent, + ) -> Result { + let row=db.query_one_raw(sql("SELECT c.certificate_digest,a.attestation_id::text,a.attestation_digest + FROM mst2_metadata_prepare q JOIN mst2_metadata_current cur ON cur.page_id=q.metadata_root AND cur.generation=$2 + JOIN mst2_metadata_lifetime life USING(page_id,generation) JOIN mst2_metadata_graph_node n USING(page_id,generation) + JOIN mst2_metadata_page_certificate c USING(page_id,generation) + JOIN mst2_metadata_source_root_attestation a ON a.root_page=c.page_id AND a.root_generation=c.generation + JOIN mst2_metadata_prepare origin ON origin.prepare_id=a.origin_prepare_id + JOIN $CORE$.mega_tree source ON source.tree_id=split_part(a.tagged_tree_oid,':',2) + WHERE q.prepare_id=$1 AND q.plan_kind='ROOTED' AND q.state='COMMITTED' AND life.state='LIVE' AND n.state='LIVE' + AND n.certificate_digest=c.certificate_digest AND a.root_certificate_digest=c.certificate_digest + AND origin.state='COMMITTED' AND a.source_profile=mst2_metadata_native_profile($1) + AND a.tagged_tree_oid=mst2_metadata_rooted_scope_tree($1) + AND $CORE$.mst2_route_source_tree_matches(split_part(a.tagged_tree_oid,':',2),a.source_revision,a.source_body_digest) + AND NOT EXISTS(SELECT 1 FROM mst2_metadata_gc_op gc WHERE gc.page_id=c.page_id AND gc.generation=c.generation) + AND (q.coverage_retired_at IS NULL AND EXISTS(SELECT 1 FROM mst2_metadata_root_anchor anchor + WHERE anchor.anchor_kind='PREPARE' AND anchor.prepare_id=q.prepare_id AND anchor.owner_key=q.prepare_id + AND anchor.root_page=c.page_id AND anchor.root_generation=c.generation + AND anchor.root_certificate_digest=c.certificate_digest) OR mst2_metadata_session_covers_prepare(q.prepare_id)) + ORDER BY (a.origin_prepare_id=q.prepare_id) DESC,a.attestation_id LIMIT 1".replace("$CORE$",&identifier(&self.namespace.core_schema)), + [intent.prepare_id.clone().into(),intent.root_generation.into()])).await.map_err(internal)? + .ok_or_else(||unavailable("rooted definitive receipt has no exact active canonical source root"))?; + Ok(RootedMetadataReceipt { + intent: intent.clone(), + certificate_digest: digest_column(&row, "certificate_digest")?, + attestation_id: uuid::Uuid::parse_str( + &row.try_get::("", "attestation_id") + .map_err(internal)?, + ) + .map_err(internal)?, + attestation_digest: digest_column(&row, "attestation_digest")?, + }) + } + + pub(crate) async fn recover( + &self, + operation: &str, + plan: &RootedMetadataInstallPlan, + ) -> Result, SnapshotError> { + let txn = self.transaction().await?; + let Some((intent, state)) = self.load_intent(&txn, operation, plan).await? else { + txn.commit().await.map_err(internal)?; + return Ok(None); + }; + let receipt = if state == "COMMITTED" { + Some(self.receipt(&txn, &intent).await?) + } else { + None + }; + txn.commit().await.map_err(internal)?; + Ok(receipt) + } +} diff --git a/src/jupiter/storage/qualified_metadata_rooted.sql b/src/jupiter/storage/qualified_metadata_rooted.sql new file mode 100644 index 00000000..fdea989c --- /dev/null +++ b/src/jupiter/storage/qualified_metadata_rooted.sql @@ -0,0 +1,358 @@ +CREATE FUNCTION mst2_metadata_read_be(b bytea,p integer,w integer) RETURNS numeric +LANGUAGE plpgsql IMMUTABLE STRICT SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE v numeric:=0; i integer; +BEGIN + IF w NOT IN (2,4,8) OR p<0 OR p>octet_length(b)-w THEN RAISE EXCEPTION 'rooted integer is out of bounds'; END IF; + FOR i IN 0..w-1 LOOP v:=v*256+get_byte(b,p+i); END LOOP; + RETURN v; +END $$; + +CREATE FUNCTION mst2_metadata_rooted_string(b bytea,p integer,maximum integer) RETURNS jsonb +LANGUAGE plpgsql IMMUTABLE STRICT SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE n numeric; value text; +BEGIN + n:=mst2_metadata_read_be(b,p,4); p:=p+4; + IF n>maximum OR n>octet_length(b)-p THEN RAISE EXCEPTION 'rooted string exceeds its exact byte boundary'; END IF; + value:=convert_from(substring(b FROM p+1 FOR n::integer),'UTF8'); + RETURN jsonb_build_object('end',p+n::integer,'text',value); +END $$; + +CREATE FUNCTION mst2_metadata_decode_rooted_plan(b bytea) RETURNS jsonb +LANGUAGE plpgsql IMMUTABLE STRICT SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE domain bytea:=convert_to('mega.mst2.rooted-install.v1','UTF8')||decode('00','hex'); + cursor_pos integer; part jsonb; source_domain text; tree_oid text; scope text; hash_kind text; identity jsonb; + root bytea; count numeric; i integer; page bytea; parent bytea; child bytea; previous bytea; previous_tree text; + size numeric; generation numeric; attestation bytea; attestation_digest bytea; certificate_digest bytea; + delta_rows jsonb[]:=ARRAY[]::jsonb[]; edge_rows jsonb[]:=ARRAY[]::jsonb[]; + reuse_rows jsonb[]:=ARRAY[]::jsonb[]; source_rows jsonb[]:=ARRAY[]::jsonb[]; total_bytes bigint:=0; + delta_index jsonb; node_index jsonb; adjacency jsonb; +BEGIN + IF octet_length(b)>2097152 OR substring(b FROM 1 FOR octet_length(domain))<>domain THEN + RAISE EXCEPTION 'rooted preparation domain or byte budget is invalid'; + END IF; + cursor_pos:=octet_length(domain); + IF mst2_metadata_read_be(b,cursor_pos,2)<>1 THEN RAISE EXCEPTION 'rooted preparation version is unsupported'; END IF; + cursor_pos:=cursor_pos+2; + part:=mst2_metadata_rooted_string(b,cursor_pos,64); cursor_pos:=(part->>'end')::integer; source_domain:=part->>'text'; + part:=mst2_metadata_rooted_string(b,cursor_pos,128); cursor_pos:=(part->>'end')::integer; tree_oid:=part->>'text'; + part:=mst2_metadata_rooted_string(b,cursor_pos,4096); cursor_pos:=(part->>'end')::integer; scope:=part->>'text'; + IF source_domain<>'native-git' OR tree_oid !~ '^(sha1:[0-9a-f]{40}|sha256:[0-9a-f]{64}|blake3:[0-9a-f]{64})$' + OR scope NOT LIKE '/%' OR scope<>'/' AND (scope LIKE '%/' OR scope LIKE '%//%' OR EXISTS( + SELECT 1 FROM unnest(string_to_array(substring(scope FROM 2),'/')) name WHERE name IN ('','.','..') + OR octet_length(name)>255) OR cardinality(string_to_array(substring(scope FROM 2),'/'))>256) THEN + RAISE EXCEPTION 'rooted source identity or scope is invalid'; END IF; + hash_kind:=split_part(tree_oid,':',1); + identity:=jsonb_build_object('source_domain',source_domain,'tagged_root_tree_oid',tree_oid,'scope',scope, + 'schema_version',mst2_metadata_read_be(b,cursor_pos,2), + 'metadata_codec',mst2_metadata_read_be(b,cursor_pos+2,2), + 'materialization_policy',mst2_metadata_read_be(b,cursor_pos+4,2), + 'fs_semantics',mst2_metadata_read_be(b,cursor_pos+6,2), + 'access_projection',mst2_metadata_read_be(b,cursor_pos+8,2), + 'verification_revision',mst2_metadata_read_be(b,cursor_pos+10,4), + 'projection_revision',mst2_metadata_read_be(b,cursor_pos+14,2)); + cursor_pos:=cursor_pos+16; + IF (identity->>'schema_version')::integer<>2 OR (identity->>'metadata_codec')::integer<>1 + OR (identity->>'materialization_policy')::integer<>1 OR (identity->>'fs_semantics')::integer<>1 + OR (identity->>'access_projection')::integer<>0 OR (identity->>'verification_revision')::integer<>2 + OR (identity->>'projection_revision')::integer<>1 THEN RAISE EXCEPTION 'rooted source profile is not current native'; END IF; + IF cursor_pos>octet_length(b)-32 THEN RAISE EXCEPTION 'rooted metadata root is truncated'; END IF; + root:=substring(b FROM cursor_pos+1 FOR 32); cursor_pos:=cursor_pos+32; + count:=mst2_metadata_read_be(b,cursor_pos,4); cursor_pos:=cursor_pos+4; + IF count>4096 THEN RAISE EXCEPTION 'rooted delta exceeds its node budget'; END IF; + IF count>0 THEN FOR i IN 1..count::integer LOOP + IF cursor_pos>octet_length(b)-40 THEN RAISE EXCEPTION 'rooted delta member is truncated'; END IF; + page:=substring(b FROM cursor_pos+1 FOR 32); size:=mst2_metadata_read_be(b,cursor_pos+32,8); cursor_pos:=cursor_pos+40; + IF previous IS NOT NULL AND previous>=page OR size NOT BETWEEN 20 AND 16384 THEN + RAISE EXCEPTION 'rooted delta is not exactly ordered or has invalid size'; END IF; + previous:=page; total_bytes:=total_bytes+size::bigint; + IF total_bytes>67108864 THEN RAISE EXCEPTION 'rooted delta exceeds its metadata byte budget'; END IF; + delta_rows:=array_append(delta_rows,jsonb_build_object('page',encode(page,'hex'),'size',size)); + END LOOP; END IF; + previous:=NULL; count:=mst2_metadata_read_be(b,cursor_pos,4); cursor_pos:=cursor_pos+4; + IF count>16384 THEN RAISE EXCEPTION 'rooted delta exceeds its edge budget'; END IF; + IF count>0 THEN FOR i IN 1..count::integer LOOP + IF cursor_pos>octet_length(b)-64 THEN RAISE EXCEPTION 'rooted edge is truncated'; END IF; + parent:=substring(b FROM cursor_pos+1 FOR 32); child:=substring(b FROM cursor_pos+33 FOR 32); cursor_pos:=cursor_pos+64; + IF previous IS NOT NULL AND previous>=parent||child OR parent=child THEN RAISE EXCEPTION 'rooted edges are not unique and ordered'; END IF; + previous:=parent||child; + edge_rows:=array_append(edge_rows,jsonb_build_object('parent',encode(parent,'hex'),'child',encode(child,'hex'))); + END LOOP; END IF; + previous:=NULL; count:=mst2_metadata_read_be(b,cursor_pos,4); cursor_pos:=cursor_pos+4; + IF count>4096 OR coalesce(array_length(delta_rows,1),0)+count>4096 THEN RAISE EXCEPTION 'rooted delta and boundaries exceed node budget'; END IF; + IF count>0 THEN FOR i IN 1..count::integer LOOP + IF cursor_pos>octet_length(b)-120 THEN RAISE EXCEPTION 'rooted reuse boundary is truncated'; END IF; + page:=substring(b FROM cursor_pos+1 FOR 32); generation:=mst2_metadata_read_be(b,cursor_pos+32,8); + attestation:=substring(b FROM cursor_pos+41 FOR 16); attestation_digest:=substring(b FROM cursor_pos+57 FOR 32); + certificate_digest:=substring(b FROM cursor_pos+89 FOR 32); cursor_pos:=cursor_pos+120; + IF previous IS NOT NULL AND previous>=page OR generation NOT BETWEEN 1 AND 9223372036854775807 THEN + RAISE EXCEPTION 'rooted reuse boundaries are not exact positive ordered lifetimes'; END IF; + previous:=page; + reuse_rows:=array_append(reuse_rows,jsonb_build_object('page',encode(page,'hex'),'generation',generation, + 'attestation_id',encode(attestation,'hex')::uuid,'attestation_digest',encode(attestation_digest,'hex'), + 'certificate_digest',encode(certificate_digest,'hex'))); + END LOOP; END IF; + count:=mst2_metadata_read_be(b,cursor_pos,4); cursor_pos:=cursor_pos+4; + IF count NOT BETWEEN 1 AND 4096 THEN RAISE EXCEPTION 'rooted source-root budget is invalid'; END IF; + FOR i IN 1..count::integer LOOP + part:=mst2_metadata_rooted_string(b,cursor_pos,128); cursor_pos:=(part->>'end')::integer; tree_oid:=part->>'text'; + IF cursor_pos>octet_length(b)-32 THEN RAISE EXCEPTION 'rooted source root is truncated'; END IF; + page:=substring(b FROM cursor_pos+1 FOR 32); cursor_pos:=cursor_pos+32; + IF tree_oid !~ '^(sha1:[0-9a-f]{40}|sha256:[0-9a-f]{64}|blake3:[0-9a-f]{64})$' + OR split_part(tree_oid,':',1)<>hash_kind OR previous_tree IS NOT NULL AND convert_to(previous_tree,'UTF8')>=convert_to(tree_oid,'UTF8') THEN + RAISE EXCEPTION 'rooted source roots are not unique ordered same-profile identities'; END IF; + previous_tree:=tree_oid; source_rows:=array_append(source_rows,jsonb_build_object('tree_oid',tree_oid,'page',encode(page,'hex'))); + END LOOP; + IF cursor_pos<>octet_length(b) THEN RAISE EXCEPTION 'rooted preparation has trailing bytes'; END IF; + SELECT coalesce(jsonb_object_agg(value->>'page',true),'{}'::jsonb) INTO delta_index FROM unnest(delta_rows) d(value); + SELECT coalesce(jsonb_object_agg(value->>'page',true),'{}'::jsonb) INTO node_index FROM ( + SELECT d.value FROM unnest(delta_rows) d(value) UNION ALL SELECT r.value FROM unnest(reuse_rows) r(value)) nodes; + IF coalesce(array_length(delta_rows,1),0)+coalesce(array_length(reuse_rows,1),0)=0 + OR EXISTS(SELECT 1 FROM unnest(reuse_rows) r(value) WHERE delta_index ? (r.value->>'page')) + OR NOT node_index ? encode(root,'hex') + OR EXISTS(SELECT 1 FROM unnest(edge_rows) e(value) WHERE NOT delta_index ? (e.value->>'parent') + OR NOT node_index ? (e.value->>'child')) + OR EXISTS(SELECT 1 FROM unnest(source_rows) s(value) WHERE NOT node_index ? (s.value->>'page')) + OR NOT EXISTS(SELECT 1 FROM unnest(source_rows) s(value) WHERE value->>'page'=encode(root,'hex')) THEN + RAISE EXCEPTION 'rooted plan has overlap or an unbound graph/source endpoint'; + END IF; + SELECT coalesce(jsonb_object_agg(grouped.parent,grouped.children),'{}'::jsonb) INTO adjacency FROM ( + SELECT value->>'parent' AS parent,jsonb_agg(value->>'child') AS children FROM unnest(edge_rows) e(value) + GROUP BY value->>'parent') grouped; + IF EXISTS(WITH RECURSIVE reached(page) AS (SELECT encode(root,'hex') UNION + SELECT children.child FROM reached r CROSS JOIN LATERAL jsonb_array_elements_text(adjacency->r.page) children(child)) + SELECT 1 FROM (SELECT d.value FROM unnest(delta_rows) d(value) UNION ALL SELECT r.value FROM unnest(reuse_rows) r(value)) nodes + WHERE NOT EXISTS(SELECT 1 FROM reached r WHERE r.page=nodes.value->>'page')) THEN + RAISE EXCEPTION 'rooted plan includes members outside its bounded delta and boundary closure'; + END IF; + RETURN identity||jsonb_build_object('root',encode(root,'hex'),'delta',to_jsonb(delta_rows),'edges',to_jsonb(edge_rows), + 'reused',to_jsonb(reuse_rows),'source_roots',to_jsonb(source_rows),'total_delta_bytes',total_bytes); +END $$; + +CREATE FUNCTION mst2_metadata_decode_delta_bindings(b bytea,plan jsonb) RETURNS jsonb +LANGUAGE plpgsql IMMUTABLE STRICT SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE count numeric; cursor_pos integer:=12; i integer; page bytea; generation numeric; size numeric; + previous bytea; binding_rows jsonb[]:=ARRAY[]::jsonb[]; +BEGIN + IF octet_length(b) NOT BETWEEN 12 AND 196620 OR substring(b FROM 1 FOR 8)<>convert_to('MST2GEN1','UTF8') THEN + RAISE EXCEPTION 'rooted generation binding encoding is invalid'; END IF; + count:=mst2_metadata_read_be(b,8,4); + IF count<>jsonb_array_length(plan->'delta') OR octet_length(b)<>12+48*count THEN + RAISE EXCEPTION 'rooted delta binding count differs from its immutable plan'; END IF; + IF count>0 THEN FOR i IN 1..count::integer LOOP + page:=substring(b FROM cursor_pos+1 FOR 32); generation:=mst2_metadata_read_be(b,cursor_pos+32,8); + size:=mst2_metadata_read_be(b,cursor_pos+40,8); cursor_pos:=cursor_pos+48; + IF previous IS NOT NULL AND previous>=page OR generation NOT BETWEEN 1 AND 9223372036854775807 + OR encode(page,'hex') IS DISTINCT FROM plan->'delta'->(i-1)->>'page' + OR size IS DISTINCT FROM (plan->'delta'->(i-1)->>'size')::numeric THEN + RAISE EXCEPTION 'rooted generation binding is not its exact positive ordered delta member'; END IF; + previous:=page; binding_rows:=array_append(binding_rows,jsonb_build_object('page',encode(page,'hex'),'generation',generation,'size',size)); + END LOOP; END IF; + RETURN to_jsonb(binding_rows); +END $$; + +CREATE FUNCTION mst2_metadata_rooted_manifest(q mst2_metadata_prepare) RETURNS jsonb +LANGUAGE plpgsql IMMUTABLE STRICT SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE plan jsonb:=mst2_metadata_decode_rooted_plan(q.canonical_plan); bindings jsonb; +BEGIN + bindings:=mst2_metadata_decode_delta_bindings(q.canonical_bindings,plan); + IF q.plan_kind<>'ROOTED' OR q.source_domain IS DISTINCT FROM plan->>'source_domain' + OR q.tagged_root_tree_oid IS DISTINCT FROM plan->>'tagged_root_tree_oid' OR q.scope IS DISTINCT FROM plan->>'scope' + OR q.schema_version IS DISTINCT FROM (plan->>'schema_version')::smallint + OR q.metadata_codec IS DISTINCT FROM (plan->>'metadata_codec')::smallint + OR q.materialization_policy IS DISTINCT FROM (plan->>'materialization_policy')::smallint + OR q.fs_semantics IS DISTINCT FROM (plan->>'fs_semantics')::smallint + OR q.access_projection IS DISTINCT FROM (plan->>'access_projection')::smallint + OR q.verification_revision IS DISTINCT FROM (plan->>'verification_revision')::integer + OR q.projection_revision IS DISTINCT FROM (plan->>'projection_revision')::smallint + OR q.metadata_root IS DISTINCT FROM decode(plan->>'root','hex') + OR q.node_count<>jsonb_array_length(plan->'delta') OR q.edge_count<>jsonb_array_length(plan->'edges') + OR q.total_bytes<>(plan->>'total_delta_bytes')::bigint THEN + RAISE EXCEPTION 'rooted preparation fields differ from its independently decoded manifest'; END IF; + RETURN plan||jsonb_build_object('bindings',bindings); +END $$; + +CREATE FUNCTION mst2_metadata_check_rooted_members(pid text) RETURNS jsonb +LANGUAGE plpgsql VOLATILE STRICT SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE q mst2_metadata_prepare%ROWTYPE; plan jsonb; +BEGIN + SELECT * INTO STRICT q FROM mst2_metadata_prepare WHERE prepare_id=pid AND plan_kind='ROOTED'; + plan:=mst2_metadata_rooted_manifest(q); + IF EXISTS((SELECT decode(value->>'page','hex'),(value->>'generation')::bigint,(value->>'size')::integer + FROM jsonb_array_elements(plan->'bindings')) EXCEPT + (SELECT page_id,generation,expected_size FROM mst2_metadata_prepare_page WHERE prepare_id=pid)) + OR EXISTS((SELECT page_id,generation,expected_size FROM mst2_metadata_prepare_page WHERE prepare_id=pid) EXCEPT + (SELECT decode(value->>'page','hex'),(value->>'generation')::bigint,(value->>'size')::integer + FROM jsonb_array_elements(plan->'bindings'))) + OR EXISTS((SELECT decode(value->>'page','hex'),(value->>'generation')::bigint,(value->>'attestation_id')::uuid, + decode(value->>'attestation_digest','hex'),decode(value->>'certificate_digest','hex') + FROM jsonb_array_elements(plan->'reused')) EXCEPT + (SELECT r.root_page,r.root_generation,r.attestation_id,r.attestation_digest,a.root_certificate_digest + FROM mst2_metadata_prepare_reuse_root r JOIN mst2_metadata_source_root_attestation a USING(attestation_id) + WHERE r.prepare_id=pid)) + OR EXISTS((SELECT r.root_page,r.root_generation,r.attestation_id,r.attestation_digest,a.root_certificate_digest + FROM mst2_metadata_prepare_reuse_root r JOIN mst2_metadata_source_root_attestation a USING(attestation_id) + WHERE r.prepare_id=pid) EXCEPT + (SELECT decode(value->>'page','hex'),(value->>'generation')::bigint,(value->>'attestation_id')::uuid, + decode(value->>'attestation_digest','hex'),decode(value->>'certificate_digest','hex') + FROM jsonb_array_elements(plan->'reused'))) THEN + RAISE EXCEPTION 'rooted preparation has missing or extra exact delta/reuse lifetime bindings'; END IF; + RETURN plan; +END $$; + +CREATE FUNCTION mst2_metadata_rooted_members_complete() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + IF EXISTS(SELECT 1 FROM mst2_metadata_prepare WHERE prepare_id=NEW.prepare_id AND plan_kind='ROOTED') THEN + PERFORM mst2_metadata_check_rooted_members(NEW.prepare_id); + END IF; + RETURN NULL; +END $$; +CREATE CONSTRAINT TRIGGER mst2_metadata_rooted_prepare_complete AFTER INSERT OR UPDATE OF bindings_revision ON mst2_metadata_prepare + DEFERRABLE INITIALLY DEFERRED FOR EACH ROW EXECUTE FUNCTION mst2_metadata_rooted_members_complete(); + +CREATE FUNCTION mst2_metadata_rooted_members_added() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + -- Every actual membership statement queues one authoritative commit check + -- per affected prepare, including a late raw-SQL addition. No caller state + -- can disable the decoded exact-coverage proof. + UPDATE mst2_metadata_prepare q SET bindings_revision=q.bindings_revision+1 + WHERE q.plan_kind='ROOTED' AND q.prepare_id IN (SELECT DISTINCT prepare_id FROM added_members); + RETURN NULL; +END $$; +CREATE TRIGGER mst2_metadata_rooted_delta_added AFTER INSERT ON mst2_metadata_prepare_page + REFERENCING NEW TABLE AS added_members FOR EACH STATEMENT EXECUTE FUNCTION mst2_metadata_rooted_members_added(); +CREATE TRIGGER mst2_metadata_rooted_reuse_added AFTER INSERT ON mst2_metadata_prepare_reuse_root + REFERENCING NEW TABLE AS added_members FOR EACH STATEMENT EXECUTE FUNCTION mst2_metadata_rooted_members_added(); + +CREATE TABLE mst2_metadata_scope_source_reference ( + prepare_id text PRIMARY KEY REFERENCES mst2_metadata_prepare(prepare_id), + scope_tree_oid text NOT NULL CHECK(scope_tree_oid ~ '^(sha1:[0-9a-f]{40}|sha256:[0-9a-f]{64}|blake3:[0-9a-f]{64})$'), + ancestor_revisions jsonb NOT NULL CHECK((jsonb_typeof(ancestor_revisions)='array' AND jsonb_array_length(ancestor_revisions)<=256) IS TRUE) +); +CREATE FUNCTION mst2_metadata_derive_scope_source(pid text) RETURNS jsonb LANGUAGE plpgsql VOLATILE STRICT +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE q mst2_metadata_prepare%ROWTYPE; tree_oid text; component text; body bytea; item jsonb; scanned_bytes bigint:=0; + digest bytea; revision uuid; ancestors jsonb[]:=ARRAY[]::jsonb[]; +BEGIN + SELECT * INTO STRICT q FROM mst2_metadata_prepare WHERE prepare_id=pid AND plan_kind='ROOTED' AND state='PREPARING'; + tree_oid:=q.tagged_root_tree_oid; + IF q.scope='/' THEN RETURN jsonb_build_object('scope_tree_oid',tree_oid,'ancestor_revisions','[]'::jsonb); END IF; + FOREACH component IN ARRAY string_to_array(substring(q.scope FROM 2),'/') LOOP + SELECT sub_trees INTO body FROM $CORE_SCHEMA$.mega_tree WHERE tree_id=split_part(tree_oid,':',2); + IF NOT FOUND THEN RAISE EXCEPTION 'rooted source scope has a missing fixed ancestor'; END IF; + scanned_bytes:=scanned_bytes+octet_length(body); + IF scanned_bytes>67108864 THEN RAISE EXCEPTION 'rooted source-scope walk exceeds its fixed byte-work budget'; END IF; + digest:=sha256(body); + revision:=$CORE_SCHEMA$.mst2_route_capture_source_tree(split_part(tree_oid,':',2),digest); + ancestors:=array_append(ancestors,jsonb_build_object('tree_oid',tree_oid,'revision',revision::text,'body_digest',encode(digest,'hex'))); + SELECT value INTO item FROM jsonb_array_elements(mst2_metadata_decode_git_tree(body,split_part(tree_oid,':',1))) + WHERE decode(value->>'name','hex')=convert_to(component,'UTF8'); + IF NOT FOUND OR (item->>'kind')::integer<>4 THEN RAISE EXCEPTION 'rooted source scope is not its exact fixed directory'; END IF; + tree_oid:=item->>'oid'; + END LOOP; + RETURN jsonb_build_object('scope_tree_oid',tree_oid,'ancestor_revisions',to_jsonb(ancestors)); +END $$; +CREATE FUNCTION mst2_metadata_scope_source_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE proof jsonb; +BEGIN + IF TG_OP<>'INSERT' THEN RAISE EXCEPTION 'rooted scope source proof is immutable'; END IF; + proof:=mst2_metadata_derive_scope_source(NEW.prepare_id); + IF NEW.scope_tree_oid IS DISTINCT FROM proof->>'scope_tree_oid' + OR NEW.ancestor_revisions IS DISTINCT FROM proof->'ancestor_revisions' THEN + RAISE EXCEPTION 'rooted scope source proof was not independently derived from actual core ancestors'; END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_metadata_scope_source_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_metadata_scope_source_reference + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_scope_source_guard(); + +CREATE FUNCTION mst2_metadata_rooted_scope_tree(pid text) RETURNS text LANGUAGE plpgsql VOLATILE STRICT +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE bound mst2_metadata_scope_source_reference%ROWTYPE; proof jsonb; ancestor jsonb; +BEGIN + SELECT * INTO bound FROM mst2_metadata_scope_source_reference WHERE prepare_id=pid; + IF NOT FOUND THEN + proof:=mst2_metadata_derive_scope_source(pid); + INSERT INTO mst2_metadata_scope_source_reference(prepare_id,scope_tree_oid,ancestor_revisions) + VALUES(pid,proof->>'scope_tree_oid',proof->'ancestor_revisions') RETURNING * INTO bound; + END IF; + FOR ancestor IN SELECT value FROM jsonb_array_elements(bound.ancestor_revisions) LOOP + IF NOT $CORE_SCHEMA$.mst2_route_source_tree_matches(split_part(ancestor->>'tree_oid',':',2), + (ancestor->>'revision')::uuid,decode(ancestor->>'body_digest','hex')) THEN + RAISE EXCEPTION 'rooted fixed scope ancestor lost its exact current source revision'; END IF; + END LOOP; + RETURN bound.scope_tree_oid; +END $$; + +CREATE FUNCTION mst2_metadata_rooted_finalize_proof(pid text) RETURNS jsonb LANGUAGE plpgsql VOLATILE STRICT +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE q mst2_metadata_prepare%ROWTYPE; plan jsonb; selected_root_generation bigint; root_certificate bytea; + profile jsonb; source jsonb; source_generation bigint; scoped_tree text; + generation_index jsonb; +BEGIN + SELECT * INTO STRICT q FROM mst2_metadata_prepare WHERE prepare_id=pid AND plan_kind='ROOTED' AND state='PREPARING'; + plan:=mst2_metadata_check_rooted_members(pid); profile:=mst2_metadata_native_profile(pid); + SELECT jsonb_object_agg(value->>'page',value->'generation') INTO generation_index FROM ( + SELECT value FROM jsonb_array_elements(plan->'bindings') UNION ALL + SELECT value FROM jsonb_array_elements(plan->'reused')) members; + selected_root_generation:=(generation_index->>(plan->>'root'))::bigint; + SELECT c.certificate_digest INTO root_certificate FROM mst2_metadata_page_certificate c + JOIN mst2_metadata_graph_node n USING(page_id,generation) JOIN mst2_metadata_current cur USING(page_id,generation) + JOIN mst2_metadata_lifetime life USING(page_id,generation) + WHERE c.page_id=q.metadata_root AND c.generation=selected_root_generation AND n.state='LIVE' + AND n.certificate_digest=c.certificate_digest AND life.state IN ('RESERVED','LIVE') AND life.graph_domain='qualified-v1' + AND c.relative_path_bytes+CASE WHEN q.scope='/' THEN 0 ELSE octet_length(q.scope) END<=4096 + AND c.relative_components+CASE WHEN q.scope='/' THEN 0 ELSE cardinality(string_to_array(substring(q.scope FROM 2),'/')) END<=256 + AND NOT EXISTS(SELECT 1 FROM mst2_metadata_gc_op gc WHERE gc.page_id=c.page_id AND gc.generation=c.generation); + IF NOT FOUND OR NOT EXISTS(SELECT 1 FROM mst2_metadata_root_anchor a WHERE a.anchor_kind='PREPARE' + AND a.owner_key=pid AND a.prepare_id=pid AND a.root_page=q.metadata_root AND a.root_generation=selected_root_generation + AND a.root_certificate_digest=root_certificate) THEN RAISE EXCEPTION 'rooted finalize lacks its independently certified owned root'; END IF; + IF EXISTS(SELECT 1 FROM mst2_metadata_prepare_page m + LEFT JOIN mst2_metadata_current cur USING(page_id,generation) LEFT JOIN mst2_metadata_lifetime life USING(page_id,generation) + LEFT JOIN mst2_metadata_payload body USING(page_id,generation) LEFT JOIN mst2_metadata_graph_node n USING(page_id,generation) + LEFT JOIN mst2_metadata_page_certificate c USING(page_id,generation) + WHERE m.prepare_id=pid AND (cur.page_id IS NULL OR life.page_id IS NULL OR body.page_id IS NULL OR n.page_id IS NULL OR c.page_id IS NULL + OR life.state NOT IN ('RESERVED','LIVE') OR life.graph_domain<>'qualified-v1' OR life.metadata_codec<>q.metadata_codec + OR body.metadata_codec<>q.metadata_codec OR n.metadata_codec<>q.metadata_codec OR c.metadata_codec<>q.metadata_codec + OR life.expected_size<>m.expected_size OR body.byte_size<>m.expected_size OR n.bytes<>m.expected_size OR c.byte_size<>m.expected_size + OR n.state<>'LIVE' OR n.certificate_digest<>c.certificate_digest + OR life.state='RESERVED' AND c.origin_prepare_id<>pid + OR n.incoming_refs<>(SELECT count(*) FROM mst2_metadata_graph_edge e WHERE e.child_page=m.page_id AND e.child_generation=m.generation) + OR EXISTS(SELECT 1 FROM mst2_metadata_gc_op gc WHERE gc.page_id=m.page_id AND gc.generation=m.generation))) THEN + RAISE EXCEPTION 'rooted finalize lost its exact durable canonical delta graph'; END IF; + IF EXISTS((SELECT decode(value->>'parent','hex'),decode(value->>'child','hex') FROM jsonb_array_elements(plan->'edges')) EXCEPT + (SELECT e.parent_page,e.child_page FROM mst2_metadata_prepare_page m JOIN mst2_metadata_graph_edge e + ON e.parent_page=m.page_id AND e.parent_generation=m.generation WHERE m.prepare_id=pid)) + OR EXISTS((SELECT e.parent_page,e.child_page FROM mst2_metadata_prepare_page m JOIN mst2_metadata_graph_edge e + ON e.parent_page=m.page_id AND e.parent_generation=m.generation WHERE m.prepare_id=pid) EXCEPT + (SELECT decode(value->>'parent','hex'),decode(value->>'child','hex') FROM jsonb_array_elements(plan->'edges'))) + OR EXISTS(SELECT 1 FROM mst2_metadata_prepare_reuse_root r WHERE r.prepare_id=pid AND NOT EXISTS( + SELECT 1 FROM mst2_metadata_root_anchor a WHERE a.prepare_id=pid AND a.owner_key=pid AND a.anchor_kind='REUSE' + AND a.root_page=r.root_page AND a.root_generation=r.root_generation)) THEN + RAISE EXCEPTION 'rooted finalize differs from its exact delta edges or owned reuse boundaries'; END IF; + scoped_tree:=mst2_metadata_rooted_scope_tree(pid); + IF NOT EXISTS(SELECT 1 FROM jsonb_array_elements(plan->'source_roots') binding(value) + WHERE value->>'tree_oid'=scoped_tree AND value->>'page'=plan->>'root') THEN + RAISE EXCEPTION 'rooted metadata root is not bound to its independently selected fixed source scope'; END IF; + FOR source IN SELECT value FROM jsonb_array_elements(plan->'source_roots') LOOP + source_generation:=(generation_index->>(source->>'page'))::bigint; + IF NOT EXISTS(SELECT 1 FROM mst2_metadata_source_root_attestation a + JOIN mst2_metadata_current cur ON cur.page_id=a.root_page AND cur.generation=a.root_generation + JOIN mst2_metadata_lifetime life USING(page_id,generation) JOIN mst2_metadata_graph_node n USING(page_id,generation) + JOIN mst2_metadata_prepare origin ON origin.prepare_id=a.origin_prepare_id + JOIN $CORE_SCHEMA$.mega_tree t ON t.tree_id=split_part(a.tagged_tree_oid,':',2) + WHERE a.tagged_tree_oid=source->>'tree_oid' AND a.root_page=decode(source->>'page','hex') AND a.root_generation=source_generation + AND a.source_profile=profile + AND $CORE_SCHEMA$.mst2_route_source_tree_matches(split_part(a.tagged_tree_oid,':',2),a.source_revision,a.source_body_digest) AND n.state='LIVE' + AND n.certificate_digest=a.root_certificate_digest + AND (a.origin_prepare_id=pid AND origin.state='PREPARING' AND life.state IN ('RESERVED','LIVE') + OR origin.state='COMMITTED' AND life.state='LIVE' AND EXISTS(SELECT 1 FROM mst2_metadata_prepare_reuse_root r + WHERE r.prepare_id=pid AND r.root_page=a.root_page AND r.root_generation=a.root_generation + AND EXISTS(SELECT 1 FROM mst2_metadata_source_root_attestation boundary + WHERE boundary.attestation_id=r.attestation_id AND boundary.root_certificate_digest=a.root_certificate_digest)))) THEN + RAISE EXCEPTION 'rooted source binding lacks its independently attested exact current directory'; END IF; + END LOOP; + RETURN jsonb_build_object('root',plan->>'root','generation',selected_root_generation,'certificate',encode(root_certificate,'hex'), + 'scoped_tree_oid',scoped_tree,'delta_nodes',q.node_count,'delta_edges',q.edge_count,'delta_bytes',q.total_bytes); +END $$; diff --git a/src/jupiter/storage/qualified_metadata_serving.sql b/src/jupiter/storage/qualified_metadata_serving.sql new file mode 100644 index 00000000..18ad4663 --- /dev/null +++ b/src/jupiter/storage/qualified_metadata_serving.sql @@ -0,0 +1,525 @@ +-- Root ownership is derived from durable identities, never an application flag. +CREATE FUNCTION mst2_metadata_publication_valid(instance text,commit_id text,tree_id text, + sequence_id bigint,epoch_id bigint,receipt_id bigint,current_head boolean DEFAULT false) +RETURNS boolean LANGUAGE sql VOLATILE SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ + SELECT EXISTS(SELECT 1 FROM $CORE_SCHEMA$.mst2_native_publication c + JOIN $CORE_SCHEMA$.mst2_publication p ON p.id=c.receipt_id + JOIN $CORE_SCHEMA$.mst2_publication_outbox o ON o.operation_id=p.operation_id + JOIN $CORE_SCHEMA$.mega_commit source ON source.commit_id=c.root_commit AND source.tree=c.root_tree + JOIN $CORE_SCHEMA$.mega_commit previous ON previous.commit_id=c.old_root_commit AND previous.tree=c.old_root_tree + JOIN $CORE_SCHEMA$.mega_commit path_source ON path_source.commit_id=c.path_commit AND path_source.tree=c.path_tree + WHERE c.receipt_id=$6 AND c.namespace='/' AND c.instance_id=$1 + AND c.root_commit=$2 AND c.root_tree=$3 AND c.sequence=$4 AND c.writer_epoch=$5 + AND $4>0 AND $5>0 AND p.writer_epoch=$5 AND p.writer_kind='trunk_push' + AND p.native_certificate_version=1 AND p.request_digest_version=1 + AND p.request_digest ~ '^sha256:[0-9a-f]{64}$' AND c.origin_ref='refs/heads/main' + AND c.origin_path=p.namespace AND c.path_commit=p.new_oid AND c.old_root_commit=p.old_oid + AND c.old_path_commit IS DISTINCT FROM c.path_commit AND o.namespace=p.namespace AND o.sequence=p.sequence + AND (NOT $7 OR (EXISTS(SELECT 1 FROM $CORE_SCHEMA$.mst2_native_head h + WHERE h.namespace='/' AND h.state='READY' AND h.instance_id=$1 AND h.root_commit=$2 + AND h.root_tree=$3 AND h.sequence=$4 AND h.writer_epoch=$5 + AND h.certificate_receipt_id=$6) + AND (SELECT count(*) FROM $CORE_SCHEMA$.mega_refs r + WHERE r.path='/' AND r.ref_name='refs/heads/main' AND NOT r.is_cl)=1 + AND EXISTS(SELECT 1 FROM $CORE_SCHEMA$.mega_refs r WHERE r.path='/' + AND r.ref_name='refs/heads/main' AND NOT r.is_cl AND r.ref_commit_hash=$2 AND r.ref_tree_hash=$3)))) +$$; + +CREATE FUNCTION mst2_metadata_descriptor(pid text,instance text,commit_id text,tree_id text) +RETURNS bytea LANGUAGE plpgsql VOLATILE STRICT SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE p record; scope_bytes bytea; view_digest bytea; instance_bytes bytea; +BEGIN + SELECT tagged_root_tree_oid,scope,metadata_root INTO p FROM mst2_metadata_prepare WHERE prepare_id=pid AND plan_kind='ROOTED' + AND state='COMMITTED' AND graph_domain='qualified-v1' AND mst2_metadata_scope_matches(primary_scope); + IF NOT FOUND OR instance IS DISTINCT FROM (instance::uuid)::text + OR split_part(p.tagged_root_tree_oid,':',2) IS DISTINCT FROM tree_id + OR NOT EXISTS(SELECT 1 FROM $CORE_SCHEMA$.mega_commit c WHERE c.commit_id=$3 AND c.tree=$4) + OR octet_length(commit_id)<>octet_length(tree_id) + OR commit_id !~ '^([0-9a-f]{40}|[0-9a-f]{64})$' THEN + RAISE EXCEPTION 'qualified descriptor has no exact canonical instance and fixed source'; + END IF; + PERFORM mst2_metadata_native_profile(pid); + scope_bytes:=convert_to(p.scope,'UTF8'); + IF p.scope !~ '^/' OR octet_length(scope_bytes)>4096 OR p.scope<>'/' AND + (p.scope ~ '/$|//|/(\.|\.\.)(/|$)' OR cardinality(string_to_array(substring(p.scope FROM 2),'/'))>256) + OR EXISTS(SELECT 1 FROM unnest(string_to_array(substring(p.scope FROM 2),'/')) component + WHERE octet_length(component)>255) THEN + RAISE EXCEPTION 'qualified descriptor scope is not canonical'; + END IF; + instance_bytes:=decode(replace(instance,'-',''),'hex'); + view_digest:=sha256(convert_to('mega.mst2.namespaceview','UTF8')||decode('00','hex')||convert_to(commit_id,'UTF8')); + RETURN convert_to('MSD2','UTF8')||decode('02000100','hex')||instance_bytes||view_digest + ||mst2_metadata_write_le(octet_length(scope_bytes),2)||scope_bytes + ||decode('0100010000000000','hex')||p.metadata_root; +END $$; + +CREATE FUNCTION mst2_metadata_root_live(p bytea,g bigint,c bytea) RETURNS boolean +LANGUAGE sql VOLATILE STRICT SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ + SELECT EXISTS(SELECT 1 FROM mst2_metadata_current cur JOIN mst2_metadata_lifetime life USING(page_id,generation) + JOIN mst2_metadata_graph_node node USING(page_id,generation) + JOIN mst2_metadata_page_certificate proof USING(page_id,generation) + JOIN mst2_metadata_payload body USING(page_id,generation) + WHERE cur.page_id=p AND cur.generation=g AND life.state='LIVE' AND life.graph_domain='qualified-v1' + AND life.metadata_codec=1 AND node.state='LIVE' AND node.metadata_codec=1 AND body.metadata_codec=1 + AND node.certificate_digest=c AND proof.certificate_digest=c AND proof.proof_revision=1 + AND proof.namespace_uuid='$NAMESPACE_UUID$'::uuid AND life.expected_size=proof.byte_size + AND node.bytes=proof.byte_size AND body.byte_size=proof.byte_size AND octet_length(body.payload)=proof.byte_size + AND p=sha256(convert_to('mega.mst2.metapage','UTF8')||decode('00','hex')||body.payload) + AND NOT EXISTS(SELECT 1 FROM mst2_metadata_gc_op gc WHERE gc.page_id=p AND gc.generation=g)) +$$; + +CREATE FUNCTION mst2_metadata_serving_source(pid text,a_id uuid,p bytea,g bigint,c bytea) +RETURNS boolean LANGUAGE plpgsql VOLATILE STRICT SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE a record; scope text; certificate record; +BEGIN + SELECT proof.tagged_tree_oid,proof.source_revision,proof.source_body_digest,proof.profile_digest,proof.source_profile + INTO a FROM mst2_metadata_source_root_attestation proof + JOIN mst2_metadata_prepare origin ON origin.prepare_id=proof.origin_prepare_id + WHERE proof.attestation_id=a_id AND proof.namespace_uuid='$NAMESPACE_UUID$'::uuid + AND proof.root_page=p AND proof.root_generation=g AND proof.root_certificate_digest=c + AND origin.state='COMMITTED' AND proof.source_profile=mst2_metadata_native_profile(pid) + AND proof.tagged_tree_oid=mst2_metadata_rooted_scope_tree(pid); + IF NOT FOUND THEN RETURN false; END IF; + SELECT relative_path_bytes,relative_components INTO certificate FROM mst2_metadata_page_certificate WHERE page_id=p AND generation=g AND certificate_digest=c; + IF NOT FOUND THEN RETURN false; END IF; + SELECT q.scope INTO STRICT scope FROM mst2_metadata_prepare q WHERE q.prepare_id=pid; + IF (CASE WHEN scope='/' THEN 0 ELSE octet_length(scope) END)::bigint+certificate.relative_path_bytes>4096 + OR (CASE WHEN scope='/' THEN 0 ELSE cardinality(string_to_array(substring(scope FROM 2),'/')) END)::bigint + +certificate.relative_components>256 THEN RETURN false; END IF; + RETURN $CORE_SCHEMA$.mst2_route_source_tree_matches(split_part(a.tagged_tree_oid,':',2),a.source_revision,a.source_body_digest) + AND a.profile_digest=sha256(convert_to('mega.mst2.native-profile.v1','UTF8')||decode('00','hex') + ||convert_to(a.source_profile::text,'UTF8')); +END $$; + +CREATE FUNCTION mst2_metadata_snapshot_candidate(sid text,descriptor bytea,instance text,commit_id text, + tree_id text,root bytea,profile jsonb) RETURNS boolean LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE p record; a record; +BEGIN + IF sid IS DISTINCT FROM 'sha256:'||encode(sha256(convert_to('mega.mst2.descriptor','UTF8') + ||decode('00','hex')||descriptor),'hex') THEN RETURN false; END IF; + FOR p IN SELECT q.prepare_id FROM mst2_metadata_prepare q WHERE q.metadata_root=root AND q.state='COMMITTED' + AND q.plan_kind='ROOTED' AND q.graph_domain='qualified-v1' AND q.coverage_retired_at IS NULL + AND q.tagged_root_tree_oid=profile->>'tagged_root_tree_oid' AND q.scope=profile->>'scope' + AND $CORE_SCHEMA$.mst2_route_profile(q.source_domain,q.tagged_root_tree_oid,q.scope,q.schema_version, + q.metadata_codec,q.materialization_policy,q.fs_semantics,q.access_projection, + q.verification_revision,q.projection_revision)=profile ORDER BY q.prepare_id LIMIT 1 LOOP + IF descriptor IS DISTINCT FROM mst2_metadata_descriptor(p.prepare_id,instance,commit_id,tree_id) THEN CONTINUE; END IF; + FOR a IN SELECT proof.attestation_id,proof.root_generation,proof.root_certificate_digest FROM mst2_metadata_source_root_attestation proof + JOIN mst2_metadata_current current_root ON current_root.page_id=proof.root_page AND current_root.generation=proof.root_generation + WHERE proof.root_page=root AND proof.tagged_tree_oid=mst2_metadata_rooted_scope_tree(p.prepare_id) + AND proof.source_profile=mst2_metadata_native_profile(p.prepare_id) + AND $CORE_SCHEMA$.mst2_route_source_tree_matches(split_part(proof.tagged_tree_oid,':',2),proof.source_revision,proof.source_body_digest) + ORDER BY proof.attestation_id LIMIT 1 LOOP + IF mst2_metadata_serving_source(p.prepare_id,a.attestation_id,root,a.root_generation,a.root_certificate_digest) + AND mst2_metadata_root_live(root,a.root_generation,a.root_certificate_digest) THEN RETURN true; END IF; + END LOOP; + END LOOP; + RETURN false; +END $$; + +CREATE FUNCTION mst2_metadata_incarnation_proof(sid text,inc uuid,require_live boolean DEFAULT false) +RETURNS boolean LANGUAGE plpgsql VOLATILE SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE s mst2_qualified_session_incarnation%ROWTYPE; p record; + a record; profile jsonb; +BEGIN + SELECT * INTO s FROM mst2_qualified_session_incarnation WHERE snapshot_id=sid AND session_incarnation=inc; + IF NOT FOUND OR s.namespace_uuid<>'$NAMESPACE_UUID$'::uuid OR s.authorization_epoch<>1 THEN RETURN false; END IF; + SELECT prepare_id,source_domain,tagged_root_tree_oid,scope,schema_version,metadata_codec, + materialization_policy,fs_semantics,access_projection,verification_revision,projection_revision + INTO p FROM mst2_metadata_prepare WHERE prepare_id=s.prepare_id AND state='COMMITTED' + AND plan_kind='ROOTED' AND storage_seal=s.storage_seal AND metadata_root=s.metadata_root; + IF NOT FOUND THEN RETURN false; END IF; + profile:=$CORE_SCHEMA$.mst2_route_profile(p.source_domain,p.tagged_root_tree_oid,p.scope,p.schema_version, + p.metadata_codec,p.materialization_policy,p.fs_semantics,p.access_projection,p.verification_revision,p.projection_revision); + IF s.source_profile IS DISTINCT FROM convert_to(profile::text,'UTF8') + OR s.canonical_descriptor IS DISTINCT FROM mst2_metadata_descriptor(p.prepare_id,s.instance_id,s.commit_oid,s.root_tree_oid) + OR s.snapshot_id IS DISTINCT FROM 'sha256:'||encode(sha256(convert_to('mega.mst2.descriptor','UTF8') + ||decode('00','hex')||s.canonical_descriptor),'hex') + OR NOT EXISTS(SELECT 1 FROM $CORE_SCHEMA$.mst2_snapshot_storage_route r + WHERE r.snapshot_id=s.snapshot_id AND r.namespace_uuid=s.namespace_uuid AND r.canonical_descriptor=s.canonical_descriptor + AND r.instance_id=s.instance_id AND r.commit_oid=s.commit_oid AND r.root_tree_oid=s.root_tree_oid + AND r.metadata_root=s.metadata_root AND r.source_profile=profile) + OR NOT mst2_metadata_publication_valid(s.instance_id,s.commit_oid,s.root_tree_oid, + s.publication_sequence,s.writer_epoch,s.certificate_receipt_id,false) THEN RETURN false; END IF; + SELECT attestation_id,root_certificate_digest INTO a FROM mst2_metadata_source_root_attestation WHERE attestation_id=s.attestation_id + AND attestation_digest=s.attestation_digest AND root_page=s.metadata_root AND root_generation=s.root_generation; + IF NOT FOUND OR NOT mst2_metadata_serving_source(s.prepare_id,a.attestation_id,s.metadata_root, + s.root_generation,a.root_certificate_digest) THEN RETURN false; END IF; + IF require_live AND (s.state<>'READY' OR NOT mst2_metadata_root_live(s.metadata_root,s.root_generation,a.root_certificate_digest) + OR NOT EXISTS(SELECT 1 FROM mst2_metadata_root_anchor anchor WHERE anchor.anchor_kind='SESSION' + AND anchor.owner_key=s.snapshot_id||':'||s.session_incarnation::text AND anchor.snapshot_id=s.snapshot_id + AND anchor.session_incarnation=s.session_incarnation AND anchor.root_page=s.metadata_root + AND anchor.root_generation=s.root_generation AND anchor.root_certificate_digest=a.root_certificate_digest)) THEN RETURN false; END IF; + RETURN true; +END $$; + +CREATE FUNCTION mst2_metadata_snapshot_route_proof(sid text,descriptor bytea,instance text,commit_id text, + tree_id text,root bytea,profile jsonb) RETURNS boolean LANGUAGE sql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ + SELECT EXISTS(SELECT 1 FROM mst2_qualified_session_incarnation s WHERE s.snapshot_id=sid + AND s.canonical_descriptor=descriptor AND s.instance_id=instance AND s.commit_oid=commit_id AND s.root_tree_oid=tree_id + AND s.metadata_root=root AND s.source_profile=convert_to(profile::text,'UTF8') + AND mst2_metadata_incarnation_proof(s.snapshot_id,s.session_incarnation,false)) +$$; + +CREATE FUNCTION mst2_metadata_lease_route_proof(lid text,sid text,namespace uuid,inc uuid,pid text,root bytea, + auth bigint,publication bigint,epoch bigint,receipt bigint) RETURNS boolean LANGUAGE sql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ + SELECT EXISTS(SELECT 1 FROM mst2_qualified_lease_binding l JOIN mst2_qualified_session_incarnation s + ON s.snapshot_id=l.snapshot_id AND s.session_incarnation=l.session_incarnation + WHERE l.lease_id=lid AND l.snapshot_id=sid AND l.namespace_uuid=namespace AND namespace='$NAMESPACE_UUID$'::uuid + AND l.session_incarnation=inc AND l.prepare_id=pid AND l.metadata_root=root + AND l.authorization_epoch=auth AND l.publication_sequence=publication AND l.writer_epoch=epoch + AND l.certificate_receipt_id=receipt AND l.storage_seal=s.storage_seal AND l.root_generation=s.root_generation + AND ROW(l.authorization_epoch,l.publication_sequence,l.writer_epoch,l.certificate_receipt_id) + IS NOT DISTINCT FROM ROW(s.authorization_epoch,s.publication_sequence,s.writer_epoch,s.certificate_receipt_id) + AND mst2_metadata_incarnation_proof(s.snapshot_id,s.session_incarnation,false)) +$$; + +CREATE UNIQUE INDEX mst2_qualified_snapshot_ready ON mst2_qualified_session_incarnation(snapshot_id) WHERE state='READY'; +CREATE INDEX mst2_metadata_committed_source ON mst2_metadata_prepare(metadata_root,tagged_root_tree_oid,scope,prepare_id) + WHERE state='COMMITTED' AND plan_kind='ROOTED'; +CREATE INDEX mst2_qualified_lease_expiry ON mst2_qualified_lease_binding(expires_at_unix,lease_id) WHERE state='ACTIVE'; +CREATE FUNCTION mst2_metadata_session_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE p record; a record; profile jsonb; +BEGIN + IF TG_OP='DELETE' THEN RAISE EXCEPTION 'qualified session incarnation history is immutable'; END IF; + IF TG_OP='UPDATE' THEN + IF (to_jsonb(NEW)-'state') IS DISTINCT FROM (to_jsonb(OLD)-'state') OR OLD.state='RETIRED' AND NEW.state<>'RETIRED' + OR NEW.state NOT IN ('READY','RETIRED') THEN RAISE EXCEPTION 'qualified fixed incarnation cannot change or revive'; END IF; + IF NEW.state='RETIRED' AND EXISTS(SELECT 1 FROM mst2_qualified_lease_binding l + WHERE l.snapshot_id=OLD.snapshot_id AND l.session_incarnation=OLD.session_incarnation AND l.state='ACTIVE') THEN + RAISE EXCEPTION 'qualified ready session still has active lease identities'; END IF; + RETURN NEW; + END IF; + SELECT prepare_id,source_domain,tagged_root_tree_oid,scope,schema_version,metadata_codec, + materialization_policy,fs_semantics,access_projection,verification_revision,projection_revision + INTO p FROM mst2_metadata_prepare WHERE prepare_id=NEW.prepare_id AND state='COMMITTED' + AND plan_kind='ROOTED' AND storage_seal=NEW.storage_seal AND metadata_root=NEW.metadata_root AND coverage_retired_at IS NULL; + IF NOT FOUND OR NEW.namespace_uuid<>'$NAMESPACE_UUID$'::uuid OR NEW.state<>'READY' OR NEW.authorization_epoch<>1 + OR substr(NEW.session_incarnation::text,15,1)<>'4' OR substr(NEW.session_incarnation::text,20,1) NOT IN ('8','9','a','b') + OR EXISTS(SELECT 1 FROM mst2_qualified_session_incarnation WHERE snapshot_id=NEW.snapshot_id AND state='READY') THEN + RAISE EXCEPTION 'qualified incarnation requires a fresh server identity and definitive rooted preparation'; END IF; + profile:=$CORE_SCHEMA$.mst2_route_profile(p.source_domain,p.tagged_root_tree_oid,p.scope,p.schema_version, + p.metadata_codec,p.materialization_policy,p.fs_semantics,p.access_projection,p.verification_revision,p.projection_revision); + SELECT attestation_id,root_certificate_digest INTO a FROM mst2_metadata_source_root_attestation WHERE attestation_id=NEW.attestation_id + AND attestation_digest=NEW.attestation_digest AND root_page=NEW.metadata_root AND root_generation=NEW.root_generation; + IF NOT FOUND OR NEW.source_profile IS DISTINCT FROM convert_to(profile::text,'UTF8') + OR NEW.canonical_descriptor IS DISTINCT FROM mst2_metadata_descriptor(p.prepare_id,NEW.instance_id,NEW.commit_oid,NEW.root_tree_oid) + OR NEW.snapshot_id IS DISTINCT FROM 'sha256:'||encode(sha256(convert_to('mega.mst2.descriptor','UTF8') + ||decode('00','hex')||NEW.canonical_descriptor),'hex') + OR NOT mst2_metadata_serving_source(p.prepare_id,a.attestation_id,NEW.metadata_root,NEW.root_generation,a.root_certificate_digest) + OR NOT mst2_metadata_root_live(NEW.metadata_root,NEW.root_generation,a.root_certificate_digest) + OR NOT EXISTS(SELECT 1 FROM mst2_metadata_root_anchor anchor WHERE anchor.anchor_kind='PREPARE' + AND anchor.prepare_id=NEW.prepare_id AND anchor.owner_key=NEW.prepare_id AND anchor.root_page=NEW.metadata_root + AND anchor.root_generation=NEW.root_generation AND anchor.root_certificate_digest=a.root_certificate_digest) + OR NOT mst2_metadata_publication_valid(NEW.instance_id,NEW.commit_oid,NEW.root_tree_oid, + NEW.publication_sequence,NEW.writer_epoch,NEW.certificate_receipt_id,true) + OR NOT EXISTS(SELECT 1 FROM $CORE_SCHEMA$.mst2_snapshot_storage_route r WHERE r.snapshot_id=NEW.snapshot_id + AND r.namespace_uuid=NEW.namespace_uuid AND r.canonical_descriptor=NEW.canonical_descriptor + AND r.instance_id=NEW.instance_id AND r.commit_oid=NEW.commit_oid AND r.root_tree_oid=NEW.root_tree_oid + AND r.metadata_root=NEW.metadata_root AND r.source_profile=profile) THEN + RAISE EXCEPTION 'qualified incarnation differs from its independently derived source and permanent route'; END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_metadata_session_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_qualified_session_incarnation + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_session_guard(); + +CREATE FUNCTION mst2_metadata_lease_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE s mst2_qualified_session_incarnation%ROWTYPE; now_unix bigint:=floor(extract(epoch FROM clock_timestamp()))::bigint; +BEGIN + IF TG_OP='DELETE' THEN RAISE EXCEPTION 'qualified lease binding history is immutable'; END IF; + IF TG_OP='UPDATE' THEN + IF (to_jsonb(NEW)-ARRAY['state','expires_at_unix','lease_epoch']) IS DISTINCT FROM + (to_jsonb(OLD)-ARRAY['state','expires_at_unix','lease_epoch']) OR OLD.state<>'ACTIVE' AND NEW IS DISTINCT FROM OLD THEN + RAISE EXCEPTION 'qualified fixed lease cannot change or revive'; END IF; + IF NEW.state='ACTIVE' THEN + IF OLD.expires_at_unix<=now_unix OR NEW.expires_at_unix<=now_unix OR NEW.expires_at_unix>now_unix+3600 + OR NEW.lease_epoch<>OLD.lease_epoch THEN RAISE EXCEPTION 'qualified renewal requires its still-active bounded lease'; END IF; + ELSIF NEW.state IN ('RELEASED','EXPIRED') THEN + IF NEW.expires_at_unix<>OLD.expires_at_unix OR NEW.lease_epoch<>OLD.lease_epoch+1 + OR NEW.state='EXPIRED' AND OLD.expires_at_unix>now_unix THEN + RAISE EXCEPTION 'qualified terminal lease differs from its exact deadline and epoch'; END IF; + ELSE RAISE EXCEPTION 'qualified lease state is unsupported'; END IF; + RETURN NEW; + END IF; + SELECT * INTO s FROM mst2_qualified_session_incarnation + WHERE snapshot_id=NEW.snapshot_id AND session_incarnation=NEW.session_incarnation AND state='READY'; + IF NOT FOUND OR NEW.namespace_uuid<>'$NAMESPACE_UUID$'::uuid OR NEW.namespace_uuid<>s.namespace_uuid + OR NEW.state<>'ACTIVE' OR NEW.lease_epoch<>1 OR NEW.expires_at_unix<=now_unix OR NEW.expires_at_unix>now_unix+3600 + OR NEW.lease_id IS DISTINCT FROM (NEW.lease_id::uuid)::text OR substr(NEW.lease_id,15,1)<>'4' + OR substr(NEW.lease_id,20,1) NOT IN ('8','9','a','b') + OR ROW(NEW.prepare_id,NEW.storage_seal,NEW.metadata_root,NEW.root_generation,NEW.authorization_epoch, + NEW.publication_sequence,NEW.writer_epoch,NEW.certificate_receipt_id) IS DISTINCT FROM + ROW(s.prepare_id,s.storage_seal,s.metadata_root,s.root_generation,s.authorization_epoch, + s.publication_sequence,s.writer_epoch,s.certificate_receipt_id) + OR NOT mst2_metadata_incarnation_proof(s.snapshot_id,s.session_incarnation,true) THEN + RAISE EXCEPTION 'qualified lease differs from its exact active incarnation and source'; END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_metadata_lease_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_qualified_lease_binding + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_lease_guard(); + +$READER_LIFECYCLE_SQL$ + +CREATE FUNCTION mst2_metadata_serving_complete() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE s mst2_qualified_session_incarnation%ROWTYPE; l mst2_qualified_lease_binding%ROWTYPE; + r mst2_metadata_reader_operation%ROWTYPE; +BEGIN + IF TG_TABLE_NAME='mst2_qualified_session_incarnation' THEN + SELECT * INTO STRICT s FROM mst2_qualified_session_incarnation + WHERE snapshot_id=NEW.snapshot_id AND session_incarnation=NEW.session_incarnation; + IF NOT mst2_metadata_incarnation_proof(s.snapshot_id,s.session_incarnation,s.state='READY') + OR s.state='READY' AND NOT EXISTS(SELECT 1 FROM mst2_qualified_lease_binding lease + WHERE lease.snapshot_id=s.snapshot_id AND lease.session_incarnation=s.session_incarnation AND lease.state='ACTIVE') + OR s.state='RETIRED' AND EXISTS(SELECT 1 FROM mst2_metadata_root_anchor a + WHERE a.anchor_kind='SESSION' AND a.snapshot_id=s.snapshot_id AND a.session_incarnation=s.session_incarnation) THEN + RAISE EXCEPTION 'qualified session cannot commit without its exact final serving roots'; END IF; + ELSIF TG_TABLE_NAME='mst2_qualified_lease_binding' THEN + SELECT * INTO STRICT l FROM mst2_qualified_lease_binding WHERE lease_id=NEW.lease_id; + IF NOT EXISTS(SELECT 1 FROM $CORE_SCHEMA$.mst2_lease_storage_route route + WHERE route.lease_id=l.lease_id AND route.namespace_uuid=l.namespace_uuid AND route.snapshot_id=l.snapshot_id + AND route.session_incarnation=l.session_incarnation AND route.prepare_id=l.prepare_id AND route.metadata_root=l.metadata_root + AND route.authorization_epoch=l.authorization_epoch AND route.publication_sequence=l.publication_sequence + AND route.writer_epoch=l.writer_epoch AND route.certificate_receipt_id=l.certificate_receipt_id) + OR l.state='ACTIVE' AND (l.expires_at_unix<=floor(extract(epoch FROM clock_timestamp()))::bigint + OR NOT mst2_metadata_incarnation_proof(l.snapshot_id,l.session_incarnation,true) + OR NOT EXISTS(SELECT 1 FROM mst2_metadata_root_anchor a WHERE a.anchor_kind='LEASE' AND a.owner_key=l.lease_id + AND a.lease_id=l.lease_id AND a.root_page=l.metadata_root AND a.root_generation=l.root_generation)) + OR l.state<>'ACTIVE' AND NOT EXISTS(SELECT 1 FROM mst2_metadata_reader_operation operation + WHERE operation.lease_id=l.lease_id AND operation.state='ACTIVE') + AND EXISTS(SELECT 1 FROM mst2_metadata_root_anchor a WHERE a.anchor_kind='LEASE' AND a.lease_id=l.lease_id) THEN + RAISE EXCEPTION 'qualified lease cannot commit without its exact final route and owned protection'; END IF; + ELSE + SELECT * INTO STRICT r FROM mst2_metadata_reader_operation WHERE operation_id=NEW.operation_id AND reader_issuance=NEW.reader_issuance; + IF r.state='ACTIVE' AND (r.hard_deadline_unix<=floor(extract(epoch FROM clock_timestamp()))::bigint + OR (SELECT count(*) FROM mst2_metadata_root_anchor a WHERE a.reader_operation_id=r.operation_id AND a.reader_issuance=r.reader_issuance + AND a.anchor_kind IN ('REQUEST','READER') AND a.owner_key=r.operation_id::text + AND a.lease_id=r.lease_id AND a.snapshot_id=r.snapshot_id AND a.session_incarnation=r.session_incarnation + AND a.root_page=r.root_page AND a.root_generation=r.root_generation)<>2) + OR r.state<>'ACTIVE' AND EXISTS(SELECT 1 FROM mst2_metadata_root_anchor a WHERE a.reader_operation_id=r.operation_id) THEN + RAISE EXCEPTION 'qualified reader cannot commit without both exact owned roots or definitive cleanup'; END IF; + END IF; + RETURN NULL; +END $$; +CREATE CONSTRAINT TRIGGER mst2_metadata_session_complete AFTER INSERT OR UPDATE ON mst2_qualified_session_incarnation + DEFERRABLE INITIALLY DEFERRED FOR EACH ROW EXECUTE FUNCTION mst2_metadata_serving_complete(); +CREATE CONSTRAINT TRIGGER mst2_metadata_lease_complete AFTER INSERT OR UPDATE ON mst2_qualified_lease_binding + DEFERRABLE INITIALLY DEFERRED FOR EACH ROW EXECUTE FUNCTION mst2_metadata_serving_complete(); +CREATE CONSTRAINT TRIGGER mst2_metadata_reader_complete AFTER INSERT OR UPDATE ON mst2_metadata_reader_operation + DEFERRABLE INITIALLY DEFERRED FOR EACH ROW EXECUTE FUNCTION mst2_metadata_serving_complete(); + +CREATE FUNCTION mst2_metadata_session_row(sid text,lid text,instance text) +RETURNS TABLE(canonical_descriptor bytea,commit_oid text,root_tree_oid text,expires_at_unix bigint, + authorization_epoch bigint,session_incarnation uuid,root_generation bigint,certificate_digest bytea,attestation_id uuid) +LANGUAGE plpgsql VOLATILE SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE l mst2_qualified_lease_binding%ROWTYPE; s mst2_qualified_session_incarnation%ROWTYPE; + c bytea; +BEGIN + SELECT * INTO l FROM mst2_qualified_lease_binding lease WHERE lease.lease_id=lid AND lease.snapshot_id=sid + AND lease.state='ACTIVE' AND lease.expires_at_unix>floor(extract(epoch FROM clock_timestamp()))::bigint; + IF NOT FOUND THEN RETURN; END IF; + SELECT * INTO s FROM mst2_qualified_session_incarnation session WHERE session.snapshot_id=sid + AND session.session_incarnation=l.session_incarnation AND session.state='READY' AND session.instance_id=instance; + IF NOT FOUND THEN RETURN; END IF; + SELECT proof.root_certificate_digest INTO c FROM mst2_metadata_source_root_attestation proof + WHERE proof.attestation_id=s.attestation_id AND proof.attestation_digest=s.attestation_digest; + IF NOT FOUND OR NOT mst2_metadata_incarnation_proof(sid,s.session_incarnation,true) + OR NOT $CORE_SCHEMA$.mst2_route_qualified_namespace_valid('$NAMESPACE_UUID$'::uuid) + OR NOT EXISTS(SELECT 1 FROM $CORE_SCHEMA$.mst2_lease_storage_route route + WHERE route.lease_id=lid AND route.snapshot_id=sid AND route.namespace_uuid=l.namespace_uuid + AND route.session_incarnation=l.session_incarnation AND route.prepare_id=l.prepare_id AND route.metadata_root=l.metadata_root + AND route.authorization_epoch=l.authorization_epoch AND route.publication_sequence=l.publication_sequence + AND route.writer_epoch=l.writer_epoch AND route.certificate_receipt_id=l.certificate_receipt_id) + OR l.namespace_uuid IS DISTINCT FROM s.namespace_uuid + OR ROW(l.prepare_id,l.storage_seal,l.metadata_root,l.root_generation,l.authorization_epoch, + l.publication_sequence,l.writer_epoch,l.certificate_receipt_id) IS DISTINCT FROM + ROW(s.prepare_id,s.storage_seal,s.metadata_root,s.root_generation,s.authorization_epoch, + s.publication_sequence,s.writer_epoch,s.certificate_receipt_id) + OR NOT EXISTS(SELECT 1 FROM mst2_metadata_root_anchor anchor WHERE anchor.anchor_kind='LEASE' AND anchor.owner_key=lid + AND anchor.lease_id=lid AND anchor.snapshot_id=sid AND anchor.session_incarnation=l.session_incarnation + AND anchor.root_page=l.metadata_root AND anchor.root_generation=l.root_generation AND anchor.root_certificate_digest=c) THEN + RAISE EXCEPTION 'qualified durable session route, source, current lifetime or root ownership changed'; END IF; + RETURN QUERY SELECT s.canonical_descriptor,s.commit_oid,s.root_tree_oid,l.expires_at_unix, + l.authorization_epoch,s.session_incarnation,s.root_generation,c,s.attestation_id; +END $$; + +CREATE FUNCTION mst2_metadata_handoff(request jsonb) RETURNS TABLE(lease_id text,expires_at_unix bigint) +LANGUAGE plpgsql VOLATILE SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE sid text:=request->>'snapshot_id'; descriptor bytea:=decode(request->>'canonical_descriptor','hex'); + instance text:=request->>'instance_id'; commit_id text:=request->>'commit_oid'; tree_id text:=request->>'root_tree_oid'; + lid text:=request->>'lease_id'; inc uuid:=(request->>'session_incarnation')::uuid; pid text:=request->>'prepare_id'; + p record; s mst2_qualified_session_incarnation%ROWTYPE; + a record; profile jsonb; now_unix bigint; deadline bigint; +BEGIN + PERFORM $CORE_SCHEMA$.mst2_route_enter($CORE_LITERAL$); + PERFORM mst2_metadata_cleanup_expired(64); + now_unix:=floor(extract(epoch FROM clock_timestamp()))::bigint; + deadline:=now_unix+least(3600,greatest(1,(request->>'lease_seconds')::bigint)); + IF request->>'authorization_epoch' IS DISTINCT FROM '1' OR lid IS NULL OR inc IS NULL OR descriptor IS NULL + OR NOT mst2_metadata_publication_valid(instance,commit_id,tree_id,(request->>'publication_sequence')::bigint, + (request->>'writer_epoch')::bigint,(request->>'certificate_receipt_id')::bigint,true) THEN + RAISE EXCEPTION 'qualified handoff differs from the independently observed current publication'; END IF; + SELECT * INTO s FROM mst2_qualified_session_incarnation WHERE snapshot_id=sid AND state='READY'; + IF FOUND THEN + IF s.state<>'READY' OR s.canonical_descriptor IS DISTINCT FROM descriptor OR s.instance_id IS DISTINCT FROM instance + OR s.commit_oid IS DISTINCT FROM commit_id OR s.root_tree_oid IS DISTINCT FROM tree_id + OR ROW(s.publication_sequence,s.writer_epoch,s.certificate_receipt_id) IS DISTINCT FROM + ROW((request->>'publication_sequence')::bigint,(request->>'writer_epoch')::bigint,(request->>'certificate_receipt_id')::bigint) + OR NOT mst2_metadata_incarnation_proof(sid,s.session_incarnation,true) THEN + RAISE EXCEPTION 'qualified handoff cannot revive or replace a historical incarnation'; END IF; + ELSE + SELECT metadata_root,storage_seal,source_domain,tagged_root_tree_oid,scope,schema_version,metadata_codec, + materialization_policy,fs_semantics,access_projection,verification_revision,projection_revision + INTO p FROM mst2_metadata_prepare WHERE prepare_id=pid AND state='COMMITTED' AND plan_kind='ROOTED' + AND storage_seal=decode(request->>'storage_seal','hex') AND coverage_retired_at IS NULL; + IF NOT FOUND THEN RAISE EXCEPTION 'qualified handoff has no exact definitive rooted receipt'; END IF; + SELECT attestation_id,attestation_digest,root_generation,root_certificate_digest + INTO a FROM mst2_metadata_source_root_attestation WHERE attestation_id=(request->>'attestation_id')::uuid + AND attestation_digest=decode(request->>'attestation_digest','hex') AND root_page=p.metadata_root + AND root_generation=(request->>'root_generation')::bigint AND root_certificate_digest=decode(request->>'certificate_digest','hex'); + IF NOT FOUND OR descriptor IS DISTINCT FROM mst2_metadata_descriptor(pid,instance,commit_id,tree_id) + OR NOT mst2_metadata_serving_source(pid,a.attestation_id,p.metadata_root,a.root_generation,a.root_certificate_digest) + OR NOT mst2_metadata_root_live(p.metadata_root,a.root_generation,a.root_certificate_digest) + OR NOT EXISTS(SELECT 1 FROM mst2_metadata_root_anchor anchor WHERE anchor.anchor_kind='PREPARE' + AND anchor.prepare_id=pid AND anchor.owner_key=pid AND anchor.root_page=p.metadata_root + AND anchor.root_generation=a.root_generation AND anchor.root_certificate_digest=a.root_certificate_digest) THEN + RAISE EXCEPTION 'qualified handoff lacks its continuously owned exact canonical source root'; END IF; + profile:=$CORE_SCHEMA$.mst2_route_profile(p.source_domain,p.tagged_root_tree_oid,p.scope,p.schema_version, + p.metadata_codec,p.materialization_policy,p.fs_semantics,p.access_projection,p.verification_revision,p.projection_revision); + INSERT INTO $CORE_SCHEMA$.mst2_snapshot_storage_route(snapshot_id,namespace_uuid,canonical_descriptor,instance_id, + commit_oid,root_tree_oid,metadata_root,source_profile) + VALUES(sid,'$NAMESPACE_UUID$'::uuid,descriptor,instance,commit_id,tree_id,p.metadata_root,profile) + ON CONFLICT(snapshot_id) DO NOTHING; + INSERT INTO mst2_qualified_session_incarnation(snapshot_id,session_incarnation,namespace_uuid,prepare_id,storage_seal, + metadata_root,root_generation,source_profile,instance_id,commit_oid,root_tree_oid,authorization_epoch, + publication_sequence,writer_epoch,certificate_receipt_id,canonical_descriptor,attestation_id,attestation_digest,state) + VALUES(sid,inc,'$NAMESPACE_UUID$'::uuid,pid,p.storage_seal,p.metadata_root,a.root_generation,convert_to(profile::text,'UTF8'), + instance,commit_id,tree_id,1,(request->>'publication_sequence')::bigint,(request->>'writer_epoch')::bigint, + (request->>'certificate_receipt_id')::bigint,descriptor,a.attestation_id,a.attestation_digest,'READY') RETURNING * INTO s; + INSERT INTO mst2_metadata_root_anchor(anchor_id,anchor_kind,owner_key,root_page,root_generation,root_certificate_digest, + snapshot_id,session_incarnation) VALUES(gen_random_uuid(),'SESSION',sid||':'||inc::text, + s.metadata_root,s.root_generation,a.root_certificate_digest,sid,inc); + END IF; + SELECT root_certificate_digest INTO STRICT a FROM mst2_metadata_source_root_attestation WHERE attestation_id=s.attestation_id; + INSERT INTO mst2_qualified_lease_binding(lease_id,snapshot_id,session_incarnation,prepare_id,storage_seal,metadata_root, + root_generation,namespace_uuid,authorization_epoch,publication_sequence,writer_epoch,certificate_receipt_id, + expires_at_unix,state,lease_epoch) VALUES(lid,sid,s.session_incarnation,s.prepare_id,s.storage_seal,s.metadata_root, + s.root_generation,s.namespace_uuid,s.authorization_epoch,s.publication_sequence,s.writer_epoch,s.certificate_receipt_id, + deadline,'ACTIVE',1); + INSERT INTO $CORE_SCHEMA$.mst2_lease_storage_route(lease_id,snapshot_id,namespace_uuid,session_incarnation,prepare_id, + metadata_root,authorization_epoch,publication_sequence,writer_epoch,certificate_receipt_id) + VALUES(lid,sid,s.namespace_uuid,s.session_incarnation,s.prepare_id,s.metadata_root,s.authorization_epoch, + s.publication_sequence,s.writer_epoch,s.certificate_receipt_id); + INSERT INTO mst2_metadata_root_anchor(anchor_id,anchor_kind,owner_key,root_page,root_generation,root_certificate_digest, + snapshot_id,session_incarnation,lease_id) VALUES(gen_random_uuid(),'LEASE',lid,s.metadata_root,s.root_generation, + a.root_certificate_digest,sid,s.session_incarnation,lid); + IF pid IS NOT NULL THEN + SELECT coverage_retired_at INTO p FROM mst2_metadata_prepare WHERE prepare_id=pid AND state='COMMITTED' AND plan_kind='ROOTED' + AND storage_seal=decode(request->>'storage_seal','hex') AND metadata_root=s.metadata_root; + IF NOT FOUND OR s.root_generation IS DISTINCT FROM (request->>'root_generation')::bigint + OR a.root_certificate_digest IS DISTINCT FROM decode(request->>'certificate_digest','hex') + OR descriptor IS DISTINCT FROM mst2_metadata_descriptor(pid,instance,commit_id,tree_id) + OR NOT mst2_metadata_session_covers_prepare(pid) + OR NOT EXISTS(SELECT 1 FROM mst2_metadata_source_root_attestation receipt + WHERE receipt.attestation_id=(request->>'attestation_id')::uuid + AND receipt.attestation_digest=decode(request->>'attestation_digest','hex') + AND receipt.root_page=s.metadata_root AND receipt.root_generation=s.root_generation + AND receipt.root_certificate_digest=a.root_certificate_digest + AND mst2_metadata_serving_source(pid,receipt.attestation_id,s.metadata_root,s.root_generation,a.root_certificate_digest)) THEN + RAISE EXCEPTION 'qualified losing preparation cannot retire a different serving identity'; END IF; + IF p.coverage_retired_at IS NULL THEN + UPDATE mst2_metadata_prepare SET coverage_retired_at=clock_timestamp() WHERE prepare_id=pid; + DELETE FROM mst2_metadata_root_anchor WHERE prepare_id=pid AND anchor_kind IN ('PREPARE','REUSE'); + END IF; + END IF; + RETURN QUERY SELECT lid,deadline; +END $$; + +CREATE FUNCTION mst2_metadata_cleanup_lease(lid text) RETURNS void LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE l mst2_qualified_lease_binding%ROWTYPE; +BEGIN + SELECT * INTO l FROM mst2_qualified_lease_binding WHERE lease_id=lid; + IF NOT FOUND THEN RETURN; END IF; + IF l.state<>'ACTIVE' AND NOT EXISTS(SELECT 1 FROM mst2_metadata_reader_operation WHERE lease_id=lid AND state='ACTIVE') THEN + DELETE FROM mst2_metadata_root_anchor WHERE lease_id=lid AND anchor_kind='LEASE'; + END IF; + IF NOT EXISTS(SELECT 1 FROM mst2_qualified_lease_binding lease WHERE lease.snapshot_id=l.snapshot_id + AND lease.session_incarnation=l.session_incarnation AND lease.state='ACTIVE') THEN + UPDATE mst2_qualified_session_incarnation SET state='RETIRED' + WHERE snapshot_id=l.snapshot_id AND session_incarnation=l.session_incarnation AND state='READY'; + DELETE FROM mst2_metadata_root_anchor WHERE snapshot_id=l.snapshot_id + AND session_incarnation=l.session_incarnation AND anchor_kind='SESSION'; + END IF; +END $$; + +CREATE FUNCTION mst2_metadata_cleanup_expired(maximum integer DEFAULT 64) RETURNS void LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE item record; now_unix bigint; bound integer:=least(64,greatest(0,maximum)); +BEGIN + PERFORM $CORE_SCHEMA$.mst2_route_enter($CORE_LITERAL$); + now_unix:=floor(extract(epoch FROM clock_timestamp()))::bigint; + FOR item IN SELECT * FROM ( + (SELECT 'READER'::text AS kind,operation_id::text AS owner,lease_id,hard_deadline_unix AS deadline,reader_issuance + FROM mst2_metadata_reader_operation WHERE state='ACTIVE' AND hard_deadline_unix<=now_unix + ORDER BY hard_deadline_unix,operation_id LIMIT bound) + UNION ALL + (SELECT 'LEASE'::text,lease_id,lease_id,expires_at_unix,NULL::bigint + FROM mst2_qualified_lease_binding WHERE state='ACTIVE' AND expires_at_unix<=now_unix + ORDER BY expires_at_unix,lease_id LIMIT bound) + ) expired ORDER BY deadline,kind,owner LIMIT bound LOOP + IF item.kind='READER' THEN + UPDATE mst2_metadata_reader_operation SET state='EXPIRED' WHERE operation_id=item.owner::uuid AND reader_issuance=item.reader_issuance AND state='ACTIVE'; + DELETE FROM mst2_metadata_root_anchor WHERE reader_operation_id=item.owner::uuid AND reader_issuance=item.reader_issuance AND anchor_kind IN ('REQUEST','READER'); + ELSE + UPDATE mst2_qualified_lease_binding SET state='EXPIRED',lease_epoch=lease_epoch+1 WHERE lease_id=item.owner AND state='ACTIVE'; + END IF; + PERFORM mst2_metadata_cleanup_lease(item.lease_id); + END LOOP; +END $$; + +CREATE FUNCTION mst2_metadata_renew_lease(lid text,seconds bigint,instance text) +RETURNS TABLE(snapshot_id text,expires_at_unix bigint) LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE l mst2_qualified_lease_binding%ROWTYPE; now_unix bigint; deadline bigint; +BEGIN + PERFORM $CORE_SCHEMA$.mst2_route_enter($CORE_LITERAL$); + PERFORM mst2_metadata_cleanup_expired(64); + now_unix:=floor(extract(epoch FROM clock_timestamp()))::bigint; + SELECT * INTO l FROM mst2_qualified_lease_binding WHERE lease_id=lid AND state='ACTIVE'; + IF NOT FOUND THEN RETURN; END IF; + IF l.expires_at_unix<=now_unix THEN + UPDATE mst2_qualified_lease_binding SET state='EXPIRED',lease_epoch=lease_epoch+1 WHERE lease_id=lid; + PERFORM mst2_metadata_cleanup_lease(lid); RETURN; + END IF; + IF NOT EXISTS(SELECT 1 FROM mst2_metadata_session_row(l.snapshot_id,lid,instance)) THEN RETURN; END IF; + deadline:=greatest(l.expires_at_unix,now_unix+least(3600,greatest(1,seconds))); + UPDATE mst2_qualified_lease_binding lease SET expires_at_unix=deadline WHERE lease.lease_id=lid; + RETURN QUERY SELECT l.snapshot_id,deadline; +END $$; + +CREATE FUNCTION mst2_metadata_release_lease(lid text) RETURNS boolean LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE l mst2_qualified_lease_binding%ROWTYPE; changed boolean; +BEGIN + PERFORM $CORE_SCHEMA$.mst2_route_enter($CORE_LITERAL$); + PERFORM mst2_metadata_cleanup_expired(64); + SELECT * INTO l FROM mst2_qualified_lease_binding WHERE lease_id=lid; + IF NOT FOUND THEN RETURN false; END IF; + IF NOT mst2_metadata_lease_route_proof(lid,l.snapshot_id,l.namespace_uuid,l.session_incarnation,l.prepare_id,l.metadata_root, + l.authorization_epoch,l.publication_sequence,l.writer_epoch,l.certificate_receipt_id) THEN + RAISE EXCEPTION 'qualified release lost its fixed historical source identity'; END IF; + changed:=l.state='ACTIVE'; + IF changed THEN UPDATE mst2_qualified_lease_binding SET state='RELEASED',lease_epoch=lease_epoch+1 WHERE lease_id=lid; END IF; + PERFORM mst2_metadata_cleanup_lease(lid); + RETURN changed; +END $$; + +DROP TRIGGER mst2_01_family_closed ON mst2_qualified_session_incarnation; +DROP TRIGGER mst2_01_family_closed ON mst2_qualified_lease_binding; +DROP TRIGGER mst2_01_family_closed ON mst2_metadata_reader_operation; diff --git a/src/jupiter/storage/qualified_metadata_session.rs b/src/jupiter/storage/qualified_metadata_session.rs new file mode 100644 index 00000000..8e29b301 --- /dev/null +++ b/src/jupiter/storage/qualified_metadata_session.rs @@ -0,0 +1,291 @@ +//! Rooted session handoff and lease operations over the captured physical Q. + +use mst2_codec::descriptor::ServingDescriptor; + +use super::*; +use crate::{ + ceres::snapshot::{ + descriptor::BuiltDescriptor, + rooted_metadata_projection::PreparedRootedNativeMetadata, + runtime::{LeaseRenewed, SnapshotContext}, + view::{SnapshotView, hex}, + }, + jupiter::storage::native_publication_storage::NativePublicationHead, +}; + +fn gone() -> SnapshotError { + SnapshotError::new( + crate::ceres::snapshot::error::SnapshotErrorCode::LeaseExpired, + "lease is not active for this snapshot; re-resolve", + ) +} + +pub(super) async fn finish( + txn: DatabaseTransaction, + result: Result, +) -> Result { + match result { + Ok(value) => { + txn.commit().await.map_err(|_| { + SnapshotError::new( + crate::ceres::snapshot::error::SnapshotErrorCode::TemporaryUnavailable, + "qualified transaction outcome unknown; retry the original operation", + ) + })?; + Ok(value) + } + Err(error) => { + let _ = txn.rollback().await; + Err(error) + } + } +} + +pub(super) fn context( + row: &QueryResult, + sid: &str, + lease: &str, + instance: &str, +) -> Result { + let bytes: Vec = row.try_get("", "canonical_descriptor").map_err(internal)?; + let descriptor = ServingDescriptor::decode(&bytes).map_err(internal)?; + let commit: String = row.try_get("", "commit_oid").map_err(internal)?; + let tree: String = row.try_get("", "root_tree_oid").map_err(internal)?; + let view = SnapshotView::from_commit(&commit, &tree); + if format!( + "sha256:{}", + hex(&descriptor.snapshot_id().map_err(internal)?) + ) != sid + || uuid::Uuid::from_bytes(descriptor.instance_uuid).to_string() != instance + || format!("sha256:{}", hex(&descriptor.namespace_view_id)) != view.view_id + { + return Err(integrity( + "qualified durable descriptor differs from its fixed source", + )); + } + let deadline = u64::try_from( + row.try_get::("", "expires_at_unix") + .map_err(internal)?, + ) + .map_err(internal)?; + let epoch = u64::try_from( + row.try_get::("", "authorization_epoch") + .map_err(internal)?, + ) + .map_err(internal)?; + let metadata_root = format!("sha256:{}", hex(&descriptor.metadata_root)); + Ok(SnapshotContext { + built: BuiltDescriptor { + descriptor, + instance_id: instance.into(), + snapshot_id: sid.into(), + metadata_root, + }, + commit_oid: commit, + root_tree_oid: tree, + lease_id: lease.into(), + lease_expires_at_unix: deadline, + authorization_epoch: epoch, + }) +} + +impl RootedQualifiedMetadataRepository { + pub(crate) async fn install( + &self, + built: &BuiltDescriptor, + prepared: &PreparedRootedNativeMetadata, + ) -> Result { + if prepared.plan.root != built.descriptor.metadata_root + || prepared.plan.identity.scope != built.descriptor.scope + { + return Err(integrity("rooted projection differs from the serving descriptor").into()); + } + // Each attempt has its own durable identity. A previous retired + // incarnation may have the same cold plan but a different generation; + // its historical receipt cannot install or resurrect this attempt. + let operation = format!("rooted-http:{}:{}", built.snapshot_id, uuid::Uuid::new_v4()); + let intent = self.begin_intent(&operation, &prepared.plan).await?; + for pages in prepared.payloads.chunks(64) { + self.install_pages(&intent, pages).await?; + } + self.finalize(&intent).await + } + + pub(crate) async fn open_session( + &self, + expected: &NativePublicationHead, + built: &BuiltDescriptor, + receipt: Option<&RootedMetadataReceipt>, + seconds: u64, + ) -> Result, SnapshotError> { + let txn = self.transaction().await?; + let result = async { + if receipt.is_none() { + txn.execute_raw(sql("SELECT mst2_metadata_cleanup_expired(64)",[])).await.map_err(database_error)?; + } + let present = txn.query_one_raw(sql("SELECT session_incarnation FROM mst2_qualified_session_incarnation + WHERE snapshot_id=$1 AND state='READY'", [built.snapshot_id.clone().into()])).await.map_err(database_error)?; + if present.is_none() && receipt.is_none() { return Ok(None); } + if let Some(receipt) = receipt { + self.require_intent(&txn, &receipt.intent).await?; + if self.receipt(&txn, &receipt.intent).await? != *receipt + || receipt.metadata_root() != built.descriptor.metadata_root + { return Err(integrity("qualified handoff receipt differs from its definitive canonical root")); } + } + let request = json!({ + "snapshot_id":built.snapshot_id,"canonical_descriptor":hex::encode(built.descriptor.encode().map_err(internal)?), + "instance_id":expected.instance_id,"commit_oid":expected.root.commit,"root_tree_oid":expected.root.tree, + "publication_sequence":expected.token.sequence,"writer_epoch":expected.token.epoch, + "certificate_receipt_id":expected.token.certificate,"authorization_epoch":1, + "lease_id":uuid::Uuid::new_v4().to_string(),"lease_seconds":seconds.clamp(1,3600), + "session_incarnation":uuid::Uuid::new_v4().to_string(), + "prepare_id":receipt.map(|r|r.intent.prepare_id.as_str()), + "storage_seal":receipt.map(|r|hex::encode(r.intent.storage_seal)), + "root_generation":receipt.map(|r|r.intent.root_generation), + "attestation_id":receipt.map(|r|r.attestation_id.to_string()), + "attestation_digest":receipt.map(|r|hex::encode(r.attestation_digest)), + "certificate_digest":receipt.map(|r|hex::encode(r.certificate_digest)), + }); + let row = txn.query_one_raw(sql("SELECT * FROM mst2_metadata_handoff($1::jsonb)", [request.into()])) + .await.map_err(database_error)?.ok_or_else(|| integrity("qualified handoff returned no lease"))?; + let lease: String = row.try_get("", "lease_id").map_err(internal)?; + let row = self.session_row(&txn, &built.snapshot_id, &lease, &expected.instance_id).await?; + let context = context(&row, &built.snapshot_id, &lease, &expected.instance_id)?; + if context.built.descriptor != built.descriptor || context.commit_oid != expected.root.commit + || context.root_tree_oid != expected.root.tree { return Err(integrity("qualified handoff fixed source changed")); } + Ok(Some(context)) + }.await; + finish(txn, result).await + } + + pub(super) async fn session_row( + &self, + db: &C, + sid: &str, + lease: &str, + instance: &str, + ) -> Result { + db.query_one_raw(sql( + "SELECT * FROM mst2_metadata_session_row($1,$2,$3)", + [sid.into(), lease.into(), instance.into()], + )) + .await + .map_err(database_error)? + .ok_or_else(gone) + } + + pub(crate) async fn context( + &self, + sid: &str, + lease: &str, + instance: &str, + ) -> Result { + // Read-only revalidation grants no mutable authority. The helper checks + // the exact route, incarnation, publication, LIVE root and owned roots. + let txn = self.read_transaction().await?; + let result = async { + let row = self.session_row(&txn, sid, lease, instance).await?; + context(&row, sid, lease, instance) + } + .await; + finish(txn, result).await + } + + pub(crate) async fn renew( + &self, + lease: &str, + seconds: u64, + instance: &str, + ) -> Result { + let txn = self.transaction().await?; + let result = async { + let row = txn + .query_one_raw(sql( + "SELECT * FROM mst2_metadata_renew_lease($1,$2,$3)", + [ + lease.into(), + (seconds.clamp(1, 3600) as i64).into(), + instance.into(), + ], + )) + .await + .map_err(database_error)? + .ok_or_else(gone)?; + Ok(LeaseRenewed { + lease_id: lease.into(), + snapshot_id: row.try_get("", "snapshot_id").map_err(internal)?, + expires_at_unix: u64::try_from( + row.try_get::("", "expires_at_unix") + .map_err(internal)?, + ) + .map_err(internal)?, + }) + } + .await; + finish(txn, result).await + } + + pub(crate) async fn release(&self, lease: &str) -> Result { + let txn = self.transaction().await?; + let result = async { + txn.query_one_raw(sql( + "SELECT mst2_metadata_release_lease($1) AS released", + [lease.into()], + )) + .await + .map_err(database_error)? + .ok_or_else(|| integrity("qualified release result missing"))? + .try_get("", "released") + .map_err(internal) + } + .await; + finish(txn, result).await + } + + pub(super) async fn read_transaction(&self) -> Result { + let txn = self + .connection + .begin_with_config(Some(IsolationLevel::ReadCommitted), None) + .await + .map_err(database_error)?; + if registered( + &txn, + &(self.namespace.core_schema.clone(), self.namespace.core_oid), + ) + .await + .map_err(|error| match error { + crate::common::errors::MegaError::Db(error) => database_error(error), + error => internal(error), + })? + .as_ref() + != Some(&self.namespace) + { + return Err(integrity( + "qualified read physical namespace or authority catalog changed", + )); + } + let valid: bool = txn + .query_one_raw(sql( + "SELECT mst2_metadata_scope_matches($1) AS valid", + [self.primary_scope.clone().into()], + )) + .await + .map_err(database_error)? + .ok_or_else(|| integrity("qualified read scope missing"))? + .try_get("", "valid") + .map_err(internal)?; + if !valid { + return Err(integrity("qualified read left its captured primary scope")); + } + #[cfg(test)] + if super::reader::READER_TEMP_SOURCE_SHADOW + .try_with(|shadow| *shadow) + .unwrap_or(false) + { + txn.execute_unprepared("CREATE TEMP TABLE mega_tree(id bigint,tree_id text,sub_trees bytea) ON COMMIT DROP; + CREATE TEMP TABLE mst2_rooted_source_tree_revision(tree_id text,tree_row_id bigint,revision uuid,body_digest bytea,valid boolean) ON COMMIT DROP") + .await.map_err(database_error)?; + } + Ok(txn) + } +} diff --git a/src/jupiter/storage/qualified_metadata_source_read.sql b/src/jupiter/storage/qualified_metadata_source_read.sql new file mode 100644 index 00000000..f2442e1b --- /dev/null +++ b/src/jupiter/storage/qualified_metadata_source_read.sql @@ -0,0 +1,132 @@ +-- Indexed fixed-source names are independently derived by the attestation +-- proof. Selected windows revalidate current file facts without source scans. +CREATE TABLE mst2_metadata_source_entry_reference ( + attestation_id uuid NOT NULL REFERENCES mst2_metadata_source_root_attestation(attestation_id), + name bytea NOT NULL CHECK(octet_length(name) BETWEEN 1 AND 255), + git_oid text NOT NULL CHECK(git_oid ~ '^(sha1:[0-9a-f]{40}|sha256:[0-9a-f]{64}|blake3:[0-9a-f]{64})$'), + kind smallint NOT NULL CHECK(kind BETWEEN 1 AND 4), + byte_size bigint,content_digest bytea,child_root bytea,child_generation bigint,child_certificate_digest bytea, + PRIMARY KEY(attestation_id,name), + CHECK(((kind=4 AND byte_size IS NULL AND content_digest IS NULL AND octet_length(child_root)=32 + AND child_generation>0 AND octet_length(child_certificate_digest)=32) + OR (kind<>4 AND byte_size BETWEEN 0 AND 8796093022208 AND octet_length(content_digest)=32 + AND child_root IS NULL AND child_generation IS NULL AND child_certificate_digest IS NULL)) IS TRUE), + CHECK((kind<>3 OR byte_size BETWEEN 1 AND 4095) IS TRUE) +); + +CREATE FUNCTION mst2_metadata_source_entry_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + RAISE EXCEPTION 'qualified fixed source entry history is immutable'; +END $$; +CREATE TRIGGER mst2_metadata_source_entry_guard BEFORE UPDATE OR DELETE ON mst2_metadata_source_entry_reference + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_source_entry_guard(); + +CREATE FUNCTION mst2_metadata_source_entries_proof() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE source_id uuid; expected jsonb; supplied jsonb; +BEGIN + FOR source_id IN SELECT DISTINCT attestation_id FROM added_source_entries LOOP + SELECT source_proof->'source_entries' INTO expected FROM mst2_metadata_source_root_attestation + WHERE attestation_id=source_id; + SELECT jsonb_object_agg(encode(reference.name,'hex'), + jsonb_build_object('kind',reference.kind,'git_oid',reference.git_oid)||CASE WHEN reference.kind=4 + THEN jsonb_build_object('child_root',encode(reference.child_root,'hex'), + 'child_generation',reference.child_generation,'child_certificate',encode(reference.child_certificate_digest,'hex')) + ELSE jsonb_build_object('size',reference.byte_size,'content_digest',encode(reference.content_digest,'hex')) END) + INTO supplied FROM added_source_entries reference WHERE reference.attestation_id=source_id; + IF expected IS NULL OR expected IS DISTINCT FROM supplied THEN + RAISE EXCEPTION 'fixed source entry batch differs from independently derived complete body and fact proof'; END IF; + END LOOP; + RETURN NULL; +END $$; +CREATE TRIGGER mst2_metadata_source_entries_proof AFTER INSERT ON mst2_metadata_source_entry_reference + REFERENCING NEW TABLE AS added_source_entries FOR EACH STATEMENT EXECUTE FUNCTION mst2_metadata_source_entries_proof(); + +CREATE FUNCTION mst2_metadata_source_entries_install() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + INSERT INTO mst2_metadata_source_entry_reference(attestation_id,name,git_oid,kind,byte_size,content_digest, + child_root,child_generation,child_certificate_digest) + SELECT NEW.attestation_id,decode(key,'hex'),value->>'git_oid',(value->>'kind')::smallint, + (value->>'size')::bigint,decode(value->>'content_digest','hex'),decode(value->>'child_root','hex'), + (value->>'child_generation')::bigint,decode(value->>'child_certificate','hex') + FROM jsonb_each(NEW.source_proof->'source_entries'); + RETURN NULL; +END $$; +CREATE TRIGGER mst2_metadata_source_entries_install AFTER INSERT ON mst2_metadata_source_root_attestation + FOR EACH ROW EXECUTE FUNCTION mst2_metadata_source_entries_install(); + +CREATE FUNCTION mst2_metadata_source_entries_complete() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + IF (SELECT count(*) FROM mst2_metadata_source_entry_reference WHERE attestation_id=NEW.attestation_id) + IS DISTINCT FROM (NEW.source_proof->>'source_entry_count')::bigint THEN + RAISE EXCEPTION 'source attestation cannot commit with incomplete exact name references'; END IF; + RETURN NULL; +END $$; +CREATE CONSTRAINT TRIGGER mst2_metadata_source_entries_complete AFTER INSERT ON mst2_metadata_source_root_attestation + DEFERRABLE INITIALLY DEFERRED FOR EACH ROW EXECUTE FUNCTION mst2_metadata_source_entries_complete(); + +CREATE FUNCTION mst2_metadata_read_source_entries(op uuid,issuance bigint,source_id uuid,p bytea,g bigint,c bytea,names jsonb) +RETURNS TABLE(name bytea,git_oid text,kind smallint,byte_size bigint,content_digest bytea,child_root bytea, + child_generation bigint,child_certificate_digest bytea,child_attestation_id uuid,fact_state text) +LANGUAGE plpgsql VOLATILE SET search_path=$Q_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE a record; profile jsonb; +BEGIN + IF jsonb_typeof(names)<>'array' OR jsonb_array_length(names)>256 + OR EXISTS(SELECT 1 FROM jsonb_array_elements_text(names) wanted + WHERE wanted !~ '^([0-9a-f]{2}){1,255}$') THEN + RAISE EXCEPTION 'qualified source-name read exceeds its bounded exact request'; END IF; + SELECT source.namespace_uuid,source.source_profile,source.tagged_tree_oid,source.source_body_digest,source.source_revision, + source.root_page,source.root_generation,source.root_certificate_digest + INTO a FROM mst2_metadata_source_root_attestation source + JOIN mst2_metadata_prepare origin ON origin.prepare_id=source.origin_prepare_id + WHERE source.attestation_id=source_id AND origin.state='COMMITTED' + AND source.root_page=p AND source.root_generation=g AND source.root_certificate_digest=c + AND source.namespace_uuid='$NAMESPACE_UUID$'::uuid; + IF NOT FOUND THEN RAISE EXCEPTION 'qualified selected directory has no definitive source attestation'; END IF; + SELECT mst2_metadata_native_profile(session.prepare_id) INTO profile + FROM mst2_metadata_reader_operation reader JOIN mst2_qualified_session_incarnation session + ON session.snapshot_id=reader.snapshot_id AND session.session_incarnation=reader.session_incarnation + WHERE reader.operation_id=op AND reader.reader_issuance=issuance AND reader.state='ACTIVE' + AND reader.hard_deadline_unix>floor(extract(epoch FROM clock_timestamp()))::bigint + AND EXISTS(SELECT 1 FROM mst2_metadata_root_anchor anchor WHERE anchor.reader_operation_id=reader.operation_id AND anchor.reader_issuance=reader.reader_issuance + AND anchor.anchor_kind='READER' AND anchor.root_page=reader.root_page AND anchor.root_generation=reader.root_generation); + IF NOT FOUND OR profile IS DISTINCT FROM a.source_profile + OR NOT mst2_metadata_root_live(a.root_page,a.root_generation,a.root_certificate_digest) THEN + RAISE EXCEPTION 'qualified source read lost its exact reader profile and current directory'; END IF; + IF NOT $CORE_SCHEMA$.mst2_route_source_tree_matches(split_part(a.tagged_tree_oid,':',2),a.source_revision,a.source_body_digest) THEN + RAISE EXCEPTION 'qualified selected directory source body changed'; END IF; + RETURN QUERY SELECT reference.name,reference.git_oid,reference.kind,reference.byte_size,reference.content_digest, + reference.child_root,reference.child_generation,reference.child_certificate_digest,child.attestation_id, + CASE WHEN reference.kind=4 THEN CASE WHEN child.attestation_id IS NULL THEN 'SOURCE_UNAVAILABLE' ELSE 'READY' END + WHEN fact.git_oid IS NULL THEN 'MISSING' + WHEN fact.state<>'VERIFIED' OR fact.verification_version NOT IN (1,2) + OR fact.size NOT BETWEEN 0 AND 8796093022208 OR octet_length(fact.raw_sha256)<>32 THEN 'INVALID' + WHEN fact.verification_version=1 THEN 'MISSING' + WHEN fact.size IS DISTINCT FROM reference.byte_size OR reference.kind=3 AND fact.size NOT BETWEEN 1 AND 4095 + OR CASE WHEN octet_length(fact.raw_sha256)=32 THEN fact.raw_sha256 ELSE NULL END + IS DISTINCT FROM reference.content_digest THEN 'INVALID' + ELSE 'READY' END + FROM (SELECT DISTINCT decode(value,'hex') AS name FROM jsonb_array_elements_text(names)) wanted + JOIN mst2_metadata_source_entry_reference reference ON reference.attestation_id=source_id AND reference.name=wanted.name + LEFT JOIN LATERAL ( + SELECT verified.git_oid,verified.state,verified.verification_version,verified.size,verified.raw_sha256 + FROM $CORE_SCHEMA$.mst2_verified_object verified WHERE reference.kind<>4 AND verified.storage_domain='git' + AND verified.object_kind='blob' AND verified.git_oid=split_part(reference.git_oid,':',2) + FOR SHARE OF verified NOWAIT + ) fact ON true + LEFT JOIN LATERAL ( + SELECT candidate.attestation_id FROM mst2_metadata_source_root_attestation candidate + JOIN mst2_metadata_prepare origin ON origin.prepare_id=candidate.origin_prepare_id + JOIN $CORE_SCHEMA$.mega_tree source_tree ON source_tree.tree_id=split_part(candidate.tagged_tree_oid,':',2) + WHERE reference.kind=4 AND candidate.namespace_uuid=a.namespace_uuid AND origin.state='COMMITTED' + AND candidate.tagged_tree_oid=reference.git_oid AND candidate.source_profile=a.source_profile + AND candidate.root_page=reference.child_root AND candidate.root_generation=reference.child_generation + AND candidate.root_certificate_digest=reference.child_certificate_digest + AND $CORE_SCHEMA$.mst2_route_source_tree_matches(split_part(candidate.tagged_tree_oid,':',2),candidate.source_revision,candidate.source_body_digest) + AND mst2_metadata_root_live(candidate.root_page,candidate.root_generation,candidate.root_certificate_digest) + ORDER BY candidate.attestation_id LIMIT 1 + ) child ON true ORDER BY reference.name; +END $$; diff --git a/src/jupiter/storage/qualified_metadata_source_revision_tests.rs b/src/jupiter/storage/qualified_metadata_source_revision_tests.rs new file mode 100644 index 00000000..9bbb6c54 --- /dev/null +++ b/src/jupiter/storage/qualified_metadata_source_revision_tests.rs @@ -0,0 +1,329 @@ +use super::{canonical_tests::seeded_rooted_plan, *}; +use crate::ceres::snapshot::rooted_metadata_projection::RootedReuseLookup; + +async fn revision(core: &DatabaseConnection, oid: &str) -> String { + core.query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT revision::text FROM mst2_rooted_source_tree_revision WHERE tree_id=$1", + [oid.into()], + )) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap() +} + +#[tokio::test] +async fn source_revision_rejects_forgery_preserves_unchanged_writes_and_requires_fresh_attestation() +{ + let (config, core, namespace, q, _guard) = fixture().await; + let (plan, payload) = seeded_rooted_plan(&core, 'a', "file", 1).await; + let writer = RootedQualifiedMetadataRepository::open(&core, &config) + .await + .unwrap(); + let intent = writer + .begin_intent("source-revision-first", &plan) + .await + .unwrap(); + writer + .install_pages(&intent, std::slice::from_ref(&payload)) + .await + .unwrap(); + writer.finalize(&intent).await.unwrap(); + let oid = "a".repeat(40); + let original = revision(&core, &oid).await; + for statement in [ + "UPDATE mst2_rooted_source_tree_revision SET revision=gen_random_uuid()", + "UPDATE mst2_rooted_source_tree_revision SET body_digest=decode(repeat('11',32),'hex')", + "UPDATE mst2_rooted_source_tree_revision SET valid=false", + "DELETE FROM mst2_rooted_source_tree_revision", + "TRUNCATE mst2_rooted_source_tree_revision", + ] { + assert!( + core.execute_unprepared(statement).await.is_err(), + "{statement}" + ); + } + assert_eq!(revision(&core, &oid).await, original); + core.execute_unprepared( + "CREATE TABLE source_revision_spoof(id integer); + CREATE FUNCTION source_revision_spoof_trigger() RETURNS trigger LANGUAGE plpgsql AS $$ + BEGIN UPDATE mst2_rooted_source_tree_revision SET valid=false; RETURN NEW; END $$; + CREATE TRIGGER source_revision_spoof AFTER INSERT ON source_revision_spoof + FOR EACH ROW EXECUTE FUNCTION source_revision_spoof_trigger()", + ) + .await + .unwrap(); + assert!( + core.execute_unprepared("INSERT INTO source_revision_spoof VALUES(1)") + .await + .is_err() + ); + assert_eq!(revision(&core, &oid).await, original); + core.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "UPDATE mega_tree SET sub_trees=sub_trees,pack_offset=pack_offset WHERE tree_id=$1", + [oid.clone().into()], + )) + .await + .unwrap(); + assert_eq!(revision(&core, &oid).await, original); + assert!( + writer + .lookup_reuse(&plan.identity.tagged_root_tree_oid, &plan.identity) + .await + .unwrap() + .is_some() + ); + core.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "UPDATE mega_tree SET sub_trees=set_byte(sub_trees,7,120) WHERE tree_id=$1", + [oid.clone().into()], + )) + .await + .unwrap(); + assert_ne!(revision(&core, &oid).await, original); + assert_eq!( + count( + &core, + "SELECT count(*) FROM mst2_rooted_source_tree_revision WHERE valid" + ) + .await, + 0 + ); + assert!( + writer + .lookup_reuse(&plan.identity.tagged_root_tree_oid, &plan.identity) + .await + .unwrap() + .is_none() + ); + let rejected=core.execute_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "SELECT mst2_route_capture_source_tree($1,(SELECT source_body_digest FROM {q}.mst2_metadata_source_root_attestation LIMIT 1))" + .replace("{q}",&identifier(&namespace.schema)),[oid.clone().into()])).await; + assert!(rejected.is_err()); + // Restoring the same old bytes does not silently resurrect the old revision. + core.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "UPDATE mega_tree SET sub_trees=set_byte(sub_trees,7,102) WHERE tree_id=$1", + [oid.clone().into()], + )) + .await + .unwrap(); + assert_eq!( + count( + &core, + "SELECT count(*) FROM mst2_rooted_source_tree_revision WHERE valid" + ) + .await, + 0 + ); + let fresh = writer + .begin_intent("source-revision-reattest", &plan) + .await + .unwrap(); + writer.install_pages(&fresh, &[payload]).await.unwrap(); + writer.finalize(&fresh).await.unwrap(); + assert_eq!( + count( + &core, + "SELECT count(*) FROM mst2_rooted_source_tree_revision WHERE valid" + ) + .await, + 1 + ); + assert_ne!(revision(&core, &oid).await, original); + assert_eq!( + count( + &q, + "SELECT count(*) FROM mst2_metadata_source_root_attestation" + ) + .await, + 2 + ); + assert_eq!(count(&q,&format!("SELECT count(*) FROM mst2_metadata_source_root_attestation a WHERE NOT {}.mst2_route_source_tree_matches( + split_part(a.tagged_tree_oid,':',2),a.source_revision,a.source_body_digest)",identifier(&namespace.core_schema))).await,1); +} + +#[tokio::test] +async fn deleted_and_reinserted_source_oid_requires_new_independent_revision() { + let (config, core, namespace, q, _guard) = fixture().await; + let (plan, payload) = seeded_rooted_plan(&core, 'a', "file", 1).await; + let writer = RootedQualifiedMetadataRepository::open(&core, &config) + .await + .unwrap(); + let intent = writer + .begin_intent("source-delete-original", &plan) + .await + .unwrap(); + writer + .install_pages(&intent, std::slice::from_ref(&payload)) + .await + .unwrap(); + writer.finalize(&intent).await.unwrap(); + let oid = "a".repeat(40); + let original = revision(&core, &oid).await; + let saved: String = core + .query_one_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "SELECT row_to_json(t)::text FROM mega_tree t WHERE tree_id=$1", + [oid.clone().into()], + )) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap(); + core.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "DELETE FROM mega_tree WHERE tree_id=$1", + [oid.clone().into()], + )) + .await + .unwrap(); + let deleted = revision(&core, &oid).await; + assert_ne!(deleted, original); + assert_eq!( + count( + &core, + "SELECT count(*) FROM mst2_rooted_source_tree_revision WHERE valid" + ) + .await, + 0 + ); + assert!( + writer + .lookup_reuse(&plan.identity.tagged_root_tree_oid, &plan.identity) + .await + .unwrap() + .is_none() + ); + // Even an identical OID, row ID, and byte body cannot revive the old stamp. + core.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "INSERT INTO mega_tree SELECT (json_populate_record(NULL::mega_tree,$1::json)).*", + [saved.into()], + )) + .await + .unwrap(); + assert_eq!(revision(&core, &oid).await, deleted); + assert_eq!( + count( + &core, + "SELECT count(*) FROM mst2_rooted_source_tree_revision WHERE valid" + ) + .await, + 0 + ); + assert!( + writer + .lookup_reuse(&plan.identity.tagged_root_tree_oid, &plan.identity) + .await + .unwrap() + .is_none() + ); + let fresh = writer + .begin_intent("source-delete-reattest", &plan) + .await + .unwrap(); + writer.install_pages(&fresh, &[payload]).await.unwrap(); + writer.finalize(&fresh).await.unwrap(); + assert_ne!(revision(&core, &oid).await, deleted); + assert_eq!( + count( + &core, + "SELECT count(*) FROM mst2_rooted_source_tree_revision WHERE valid" + ) + .await, + 1 + ); + assert_eq!(count(&q,&format!("SELECT count(*) FROM mst2_metadata_source_root_attestation a WHERE NOT {}.mst2_route_source_tree_matches( + split_part(a.tagged_tree_oid,':',2),a.source_revision,a.source_body_digest)",identifier(&namespace.core_schema))).await,1); +} + +#[tokio::test] +async fn current_source_revision_uses_captured_core_relations_under_temp_shadow() { + let (config, core, namespace, q, _guard) = fixture().await; + let (plan, payload) = seeded_rooted_plan(&core, 'a', "file", 1).await; + let writer = RootedQualifiedMetadataRepository::open(&core, &config) + .await + .unwrap(); + let intent = writer + .begin_intent("source-temp-shadow", &plan) + .await + .unwrap(); + writer.install_pages(&intent, &[payload]).await.unwrap(); + writer.finalize(&intent).await.unwrap(); + q.execute_unprepared("CREATE TEMP TABLE mega_tree(id bigint,tree_id text,sub_trees bytea); + CREATE TEMP TABLE mst2_rooted_source_tree_revision(tree_id text,tree_row_id bigint,revision uuid,body_digest bytea,valid boolean)") + .await.unwrap(); + assert_eq!(count(&q,&format!("SELECT {}.mst2_route_source_tree_matches(split_part(tagged_tree_oid,':',2),source_revision,source_body_digest)::bigint + FROM mst2_metadata_source_root_attestation",identifier(&namespace.core_schema))).await,1); +} + +#[tokio::test] +async fn current_source_revision_share_fence_orders_real_source_update_after_read() { + let (config, core, namespace, q, _guard) = fixture().await; + let (plan, payload) = seeded_rooted_plan(&core, 'a', "file", 1).await; + let writer = RootedQualifiedMetadataRepository::open(&core, &config) + .await + .unwrap(); + let intent = writer + .begin_intent("source-share-fence", &plan) + .await + .unwrap(); + writer.install_pages(&intent, &[payload]).await.unwrap(); + writer.finalize(&intent).await.unwrap(); + let read = q.begin().await.unwrap(); + assert_eq!(count(&read,&format!("SELECT {}.mst2_route_source_tree_matches(split_part(tagged_tree_oid,':',2),source_revision,source_body_digest)::bigint + FROM mst2_metadata_source_root_attestation",identifier(&namespace.core_schema))).await,1); + let core_writer = core.clone(); + let (ready, received) = tokio::sync::oneshot::channel(); + let pending = tokio::spawn(async move { + let txn = core_writer.begin().await.unwrap(); + let pid: i32 = txn + .query_one_raw(Statement::from_string( + DbBackend::Postgres, + "SELECT pg_backend_pid()", + )) + .await + .unwrap() + .unwrap() + .try_get_by_index(0) + .unwrap(); + ready.send(pid).unwrap(); + txn.execute_raw(Statement::from_sql_and_values( + DbBackend::Postgres, + "UPDATE mega_tree SET sub_trees=set_byte(sub_trees,7,120) WHERE tree_id=$1", + ["a".repeat(40).into()], + )) + .await + .unwrap(); + txn.commit().await + }); + let pid = received.await.unwrap(); + tokio::time::timeout(Duration::from_secs(4),async { + loop { + let waiting:bool=core.query_one_raw(Statement::from_sql_and_values(DbBackend::Postgres, + "SELECT coalesce(wait_event_type='Lock',false) FROM pg_stat_activity WHERE pid=$1",[pid.into()])) + .await.unwrap().unwrap().try_get_by_index(0).unwrap(); + if waiting {break;} tokio::task::yield_now().await; + } + }).await.unwrap(); + assert!(!pending.is_finished()); + read.commit().await.unwrap(); + tokio::time::timeout(Duration::from_secs(4), pending) + .await + .unwrap() + .unwrap() + .unwrap(); + assert_eq!( + count( + &core, + "SELECT count(*) FROM mst2_rooted_source_tree_revision WHERE valid" + ) + .await, + 0 + ); +} diff --git a/src/jupiter/storage/qualified_source_revision.sql b/src/jupiter/storage/qualified_source_revision.sql new file mode 100644 index 00000000..855112ce --- /dev/null +++ b/src/jupiter/storage/qualified_source_revision.sql @@ -0,0 +1,124 @@ +-- This records only trees actually attested by Q, not the global Git inventory. +CREATE TABLE mst2_rooted_source_tree_revision ( + tree_id text PRIMARY KEY CHECK(tree_id ~ '^([0-9a-f]{40}|[0-9a-f]{64})$'), + tree_row_id bigint NOT NULL,revision uuid NOT NULL,body_digest bytea NOT NULL CHECK(octet_length(body_digest)=32), + valid boolean NOT NULL, + CHECK(substr(revision::text,15,1)='4' AND substr(revision::text,20,1) IN ('8','9','a','b')) +); + +CREATE FUNCTION mst2_route_source_tree_revision_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$CORE_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE source_id bigint; body bytea; +BEGIN + IF TG_OP='DELETE' OR TG_OP='TRUNCATE' THEN RAISE EXCEPTION 'rooted source revision watermarks are immutable'; END IF; + IF pg_is_in_recovery() OR current_setting('transaction_isolation')<>'read committed' + OR TG_RELID<>'mst2_rooted_source_tree_revision'::regclass THEN + RAISE EXCEPTION 'rooted source revision requires its captured primary relation'; END IF; + IF TG_OP='UPDATE' THEN + IF NEW.tree_id IS DISTINCT FROM OLD.tree_id THEN RAISE EXCEPTION 'rooted source revision cannot retarget its OID'; END IF; + IF NEW.valid IS FALSE THEN + -- A nested caller cannot invent invalidation: independently observe the + -- actual core mutation after it happened in this still-uncommitted writer. + IF pg_trigger_depth()<2 OR NOT OLD.valid OR NEW.tree_row_id IS DISTINCT FROM OLD.tree_row_id + OR NEW.body_digest IS DISTINCT FROM OLD.body_digest OR NEW.revision IS DISTINCT FROM OLD.revision THEN + RAISE EXCEPTION 'rooted source invalidation is outside its actual source writer'; END IF; + PERFORM mst2_route_enter($CORE_LITERAL$); + SELECT id,sub_trees INTO source_id,body FROM mega_tree WHERE tree_id=OLD.tree_id; + IF FOUND AND source_id=OLD.tree_row_id AND octet_length(body)<=67108864 + AND sha256(body)=OLD.body_digest THEN + RAISE EXCEPTION 'rooted source invalidation has no actual changed or deleted core bytes'; END IF; + NEW.revision:=gen_random_uuid(); RETURN NEW; + END IF; + IF OLD.valid OR NEW.revision IS DISTINCT FROM OLD.revision OR NEW.valid IS DISTINCT FROM true THEN + RAISE EXCEPTION 'ordinary DML cannot rewrite a valid rooted source revision'; END IF; + END IF; + PERFORM mst2_route_enter($CORE_LITERAL$); + SELECT id,sub_trees INTO source_id,body FROM mega_tree WHERE tree_id=NEW.tree_id FOR SHARE; + IF NOT FOUND OR octet_length(body)>67108864 THEN RAISE EXCEPTION 'rooted source capture has no bounded actual core tree'; END IF; + IF NEW.body_digest IS DISTINCT FROM sha256(body) THEN + RAISE EXCEPTION 'rooted source revision was not independently derived from its actual core bytes'; END IF; + NEW.tree_row_id:=source_id; NEW.revision:=gen_random_uuid(); NEW.valid:=true; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_route_source_tree_revision_guard BEFORE INSERT OR UPDATE OR DELETE ON mst2_rooted_source_tree_revision + FOR EACH ROW EXECUTE FUNCTION mst2_route_source_tree_revision_guard(); +CREATE TRIGGER mst2_route_source_tree_revision_truncate BEFORE TRUNCATE ON mst2_rooted_source_tree_revision + FOR EACH STATEMENT EXECUTE FUNCTION mst2_route_source_tree_revision_guard(); + +CREATE FUNCTION mst2_route_source_tree_inventory_guard() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$CORE_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + IF (SELECT count(*) FROM (SELECT 1 FROM mst2_rooted_source_tree_revision LIMIT 65537) bounded)>65536 THEN + RAISE EXCEPTION 'rooted attested source inventory capacity is exceeded'; END IF; + RETURN NULL; +END $$; +CREATE TRIGGER mst2_route_source_tree_inventory_guard AFTER INSERT ON mst2_rooted_source_tree_revision + FOR EACH STATEMENT EXECUTE FUNCTION mst2_route_source_tree_inventory_guard(); + +CREATE FUNCTION mst2_route_source_tree_writer_enter() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$CORE_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + IF TG_RELID<>'mega_tree'::regclass THEN RAISE EXCEPTION 'rooted source writer left its captured core relation'; END IF; + IF TG_OP='TRUNCATE' THEN RAISE EXCEPTION 'core tree truncation cannot bypass rooted source revision invalidation'; END IF; + PERFORM mst2_route_enter($CORE_LITERAL$); + RETURN NULL; +END $$; +CREATE TRIGGER mst2_route_source_tree_writer_enter BEFORE INSERT OR UPDATE OR DELETE OR TRUNCATE ON mega_tree + FOR EACH STATEMENT EXECUTE FUNCTION mst2_route_source_tree_writer_enter(); + +CREATE FUNCTION mst2_route_source_tree_invalidate() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$CORE_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + IF TG_RELID<>'mega_tree'::regclass THEN RAISE EXCEPTION 'rooted invalidation left its captured core relation'; END IF; + IF TG_OP='UPDATE' AND NEW.tree_id IS NOT DISTINCT FROM OLD.tree_id AND NEW.id IS NOT DISTINCT FROM OLD.id + AND NEW.sub_trees IS NOT DISTINCT FROM OLD.sub_trees THEN RETURN NEW; END IF; + UPDATE mst2_rooted_source_tree_revision SET valid=false WHERE tree_id=OLD.tree_id AND valid; + IF TG_OP='DELETE' THEN RETURN OLD; END IF; + RETURN NEW; +END $$; +CREATE TRIGGER mst2_route_source_tree_invalidate AFTER UPDATE OR DELETE ON mega_tree + FOR EACH ROW EXECUTE FUNCTION mst2_route_source_tree_invalidate(); +CREATE FUNCTION mst2_route_source_tree_inserted() RETURNS trigger LANGUAGE plpgsql VOLATILE +SET search_path=$CORE_SCHEMA$,pg_catalog,pg_temp AS $$ +BEGIN + IF TG_RELID<>'mega_tree'::regclass THEN RAISE EXCEPTION 'rooted insertion left its captured core relation'; END IF; + UPDATE mst2_rooted_source_tree_revision SET valid=false WHERE tree_id=NEW.tree_id AND valid; + RETURN NULL; +END $$; +CREATE TRIGGER mst2_route_source_tree_inserted AFTER INSERT ON mega_tree + FOR EACH ROW EXECUTE FUNCTION mst2_route_source_tree_inserted(); + +CREATE FUNCTION mst2_route_capture_source_tree(oid_text text,expected_digest bytea) RETURNS uuid LANGUAGE plpgsql VOLATILE +SET search_path=$CORE_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE captured mst2_rooted_source_tree_revision%ROWTYPE; actual_id bigint; +BEGIN + PERFORM mst2_route_enter($CORE_LITERAL$); + SELECT id INTO actual_id FROM mega_tree WHERE tree_id=oid_text FOR SHARE; + IF NOT FOUND OR expected_digest IS NULL OR octet_length(expected_digest)<>32 THEN + RAISE EXCEPTION 'rooted source capture has no exact actual row and body digest'; END IF; + SELECT * INTO captured FROM mst2_rooted_source_tree_revision WHERE tree_id=oid_text FOR UPDATE; + IF FOUND THEN + IF captured.valid THEN + IF captured.tree_row_id<>actual_id OR captured.body_digest IS DISTINCT FROM expected_digest THEN + RAISE EXCEPTION 'rooted source bytes changed behind their exact immutable revision'; END IF; + RETURN captured.revision; + END IF; + UPDATE mst2_rooted_source_tree_revision SET valid=true,body_digest=expected_digest WHERE tree_id=oid_text + RETURNING revision INTO captured.revision; + ELSE + INSERT INTO mst2_rooted_source_tree_revision(tree_id,tree_row_id,revision,body_digest,valid) + VALUES(oid_text,actual_id,gen_random_uuid(),expected_digest,true) RETURNING revision INTO captured.revision; + END IF; + RETURN captured.revision; +END $$; + +CREATE FUNCTION mst2_route_source_tree_matches(oid_text text,exact_revision uuid,exact_digest bytea) +RETURNS boolean LANGUAGE plpgsql VOLATILE STRICT SET search_path=$CORE_SCHEMA$,pg_catalog,pg_temp AS $$ +DECLARE captured mst2_rooted_source_tree_revision%ROWTYPE; +BEGIN + IF pg_is_in_recovery() OR current_setting('transaction_isolation')<>'read committed' THEN RETURN false; END IF; + SELECT revision.* INTO captured FROM mst2_rooted_source_tree_revision revision + JOIN mega_tree source ON source.id=revision.tree_row_id AND source.tree_id=revision.tree_id + WHERE revision.tree_id=oid_text FOR SHARE OF revision NOWAIT; + RETURN FOUND AND captured.valid AND captured.revision=exact_revision AND captured.body_digest=exact_digest; +END $$; diff --git a/src/jupiter/storage/view_ops_tests.rs b/src/jupiter/storage/view_ops_tests.rs index e0cfa446..bc691179 100644 --- a/src/jupiter/storage/view_ops_tests.rs +++ b/src/jupiter/storage/view_ops_tests.rs @@ -15,7 +15,6 @@ use crate::{ ceres::view::filter::parse_for_registration, config::testing::isolated_config, jupiter::{ - migration::apply_migrations, service::{ view_metrics::ViewMetrics, view_projection_service::{CatchUpOutcome, ViewProjectionService}, @@ -23,6 +22,7 @@ use crate::{ storage::{ Storage, git_db_storage::fu18_support::single_connection, + init::database_connection, object_storage::mock_object_storage, view_root_chain::RootChainOutcome, view_storage::ViewLockMode, @@ -31,7 +31,7 @@ use crate::{ seed_unrelated_root_history, }, }, - tests::test_db_connection, + tests::{TestSchemaGuard, test_db_config}, }, }; @@ -58,6 +58,7 @@ struct OpsFixture { filter_a: i64, filter_b: i64, filter_a_id: String, + _schema: TestSchemaGuard, } fn fixed_time() -> chrono::NaiveDateTime { @@ -222,9 +223,10 @@ async fn insert_warming_filter( async fn ops_fixture() -> OpsFixture { let temp = tempfile::tempdir().unwrap(); - let db = test_db_connection(temp.path()).await; - apply_migrations(&db, true).await.unwrap(); - let config = isolated_config(temp.path().join("config")); + let (database, schema) = test_db_config(temp.path()).await; + let mut config = isolated_config(temp.path().join("config")); + config.database = database; + let db = database_connection(&config.database).await.unwrap(); let storage = Storage::new_with_connection( Arc::new(config), Arc::new(db.clone()), @@ -406,6 +408,7 @@ async fn ops_fixture() -> OpsFixture { filter_a: filter_a.id, filter_b: filter_b.id, filter_a_id: filter_a.filter_id, + _schema: schema, } } diff --git a/src/jupiter/tests.rs b/src/jupiter/tests.rs index ca684f5d..aa3dd08d 100644 --- a/src/jupiter/tests.rs +++ b/src/jupiter/tests.rs @@ -223,6 +223,45 @@ fn drop_test_schema(admin_url: &str, schema: &str) { [Value::from(schema)], )) .await?; + // A bootstrap-created physical Q family belongs to this exact test + // core schema. Drop the pair even on panic, before its core FK targets. + let registry = admin + .query_one_raw(Statement::from_sql_and_values( + DatabaseBackend::Postgres, + "SELECT pg_catalog.to_regclass($1) IS NOT NULL AS present", + [format!("{schema}.mst2_metadata_namespace").into()], + )) + .await? + .unwrap() + .try_get::("", "present")?; + if registry { + let families = admin + .query_all_raw(Statement::from_string( + DatabaseBackend::Postgres, + format!( + "SELECT n.nspname FROM {schema}.mst2_metadata_namespace q + JOIN pg_catalog.pg_namespace n ON n.oid=q.metadata_schema_oid + JOIN pg_catalog.pg_namespace c ON c.oid=q.core_schema_oid + WHERE q.graph_domain='qualified-v1' AND c.nspname='{schema}' + AND q.core_schema='{schema}' AND q.family_identity='v3-rooted-qualified-1'" + ), + )) + .await?; + for family in families { + let q_schema: String = family.try_get("", "nspname")?; + if !q_schema.starts_with("mst2q_") { + return Err(sea_orm::DbErr::Custom( + "test Q schema escaped its generated prefix".into(), + )); + } + let quoted = format!("\"{}\"", q_schema.replace('"', "\"\"")); + admin + .execute_unprepared(&format!( + "SET lock_timeout='120s'; DROP SCHEMA {quoted} CASCADE" + )) + .await?; + } + } // The timeout turns a lock held by anything else into a leaked schema // and a message, not a hung test run. admin @@ -359,7 +398,8 @@ pub async fn test_storage_with_config(temp_dir: impl AsRef, config: Config view_storage: ViewStorage::new(base.clone()), conversation_storage: ConversationStorage { base: base.clone() }, commit_binding_storage: CommitBindingStorage { base: base.clone() }, - push_queue_storage: PushQueueStorage::new(base.clone()), + push_queue_storage: PushQueueStorage::new(base.clone()) + .with_native_publication(config.mst2.publication_enabled), buck_storage: BuckStorage { base: base.clone() }, bots_storage: BotsStorage { base: base.clone() }, webhook_storage: WebhookStorage { base: base.clone() }, @@ -376,6 +416,12 @@ pub async fn test_storage_with_config(temp_dir: impl AsRef, config: Config Storage { app_service: Arc::new(svc), + native_projection_cache: Arc::default(), + native_snapshot_sessions: Arc::default(), + native_chunk_maps: Arc::default(), + shadow_qualified_metadata: Arc::default(), + rooted_qualified_metadata: Arc::default(), + projection_observation_sink: None, cl_service: CLService::mock(), push_queue_service: PushQueueService::new( base.clone(), @@ -383,7 +429,8 @@ pub async fn test_storage_with_config(temp_dir: impl AsRef, config: Config ) .with_view_signal(view_runtime.signal()) .with_timeouts(Duration::from_secs(30), Duration::from_millis(20)) - .with_max_push_commits(config.monorepo.max_push_commits), + .with_max_push_commits(config.monorepo.max_push_commits) + .with_native_publication(config.mst2.publication_enabled), artifact_service: ArtifactService::new(base.clone(), mock_object_storage()), buck_service: BuckService::mock(), config_handle: ConfigHandle::from_arc(config.clone()), diff --git a/src/jupiter/utils/converter.rs b/src/jupiter/utils/converter.rs index 40f3502f..697816c4 100644 --- a/src/jupiter/utils/converter.rs +++ b/src/jupiter/utils/converter.rs @@ -1,4 +1,4 @@ -use std::{cell::RefCell, collections::HashMap}; +use std::{cell::RefCell, collections::HashMap, str::FromStr}; use git_internal::{ hash::{HashKind, ObjectHash, get_hash_kind, set_hash_kind}, @@ -663,6 +663,21 @@ pub struct MegaModelConverter { pub refs: mega_refs::ActiveModel, } +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub(crate) struct BootstrapCommitTime(u32); + +impl FromStr for BootstrapCommitTime { + type Err = String; + + fn from_str(value: &str) -> Result { + let invalid = || "commit time must be Unix seconds in 0..=4294967295".to_string(); + if value.is_empty() || !value.bytes().all(|byte| byte.is_ascii_digit()) { + return Err(invalid()); + } + value.parse::().map(Self).map_err(|_| invalid()) + } +} + struct InitializationHashKindGuard { previous: HashKind, } @@ -725,11 +740,32 @@ impl MegaModelConverter { } pub fn init(mono_config: &MonoConfig) -> Result { + Self::init_with_commit_time(mono_config, None) + } + + pub(crate) fn init_with_commit_time( + mono_config: &MonoConfig, + commit_time: Option, + ) -> Result { let hash_kind = mono_config.object_hash_kind()?; - Ok(with_initialization_hash_kind(hash_kind, || { + with_initialization_hash_kind(hash_kind, || { let (tree_maps, blob_maps, root_tree) = init_trees(mono_config); - let commit = Commit::from_tree_id(root_tree.id, vec![], "\nInit Mega Directory"); + let commit = match commit_time { + Some(BootstrapCommitTime(seconds)) => Commit::new_with_kind( + hash_kind, + Signature::from_data( + format!("author mega {seconds} +0800").into_bytes(), + )?, + Signature::from_data( + format!("committer mega {seconds} +0800").into_bytes(), + )?, + root_tree.id, + vec![], + "\nInit Mega Directory", + )?, + None => Commit::from_tree_id(root_tree.id, vec![], "\nInit Mega Directory"), + }; let mega_ref = mega_refs::Model { id: generate_id(), @@ -753,8 +789,8 @@ impl MegaModelConverter { refs: mega_ref.into(), }; converter.traverse_from_root(); - converter - })) + Ok(converter) + }) } } diff --git a/src/orbit/adapter/log.rs b/src/orbit/adapter/log.rs index 4d116744..6dd1d78f 100644 --- a/src/orbit/adapter/log.rs +++ b/src/orbit/adapter/log.rs @@ -8,6 +8,7 @@ impl LogStorage for ObjectStoreAdapter { data: ObjectByteStream, _meta: ObjectMeta, ) -> OrbitResult<()> { + super::object::reject_receipt_mutation(key)?; key.validate()?; // Fast path for single writer/single thread (optimized here): // - No conditional writes, no retries, no cleanup (no concurrent write conflicts) @@ -187,6 +188,7 @@ impl LogStorage for ObjectStoreAdapter { data: ObjectByteStream, _meta: ObjectMeta, ) -> OrbitResult<()> { + super::object::reject_receipt_mutation(key)?; key.validate()?; // The local backend has no conditional-write (CAS) support, so this // method cannot be made safe under contention there (it would fall back diff --git a/src/orbit/adapter/object.rs b/src/orbit/adapter/object.rs index ca1751b3..14c3d175 100644 --- a/src/orbit/adapter/object.rs +++ b/src/orbit/adapter/object.rs @@ -6,12 +6,17 @@ impl MegaObjectStorage for ObjectStoreAdapter { matches!(&self.store, BackendStore::S3(_) | BackendStore::Gcs(_)) } + fn supports_chunk_map_retention(&self) -> bool { + matches!(&self.store, BackendStore::Local(_)) + } + async fn put_stream( &self, key: &ObjectKey, data: ObjectByteStream, _meta: ObjectMeta, ) -> OrbitResult<()> { + reject_receipt_mutation(key)?; let path = Self::checked_path(key)?; // Artifacts are keyed by UUID and must not depend on LFS/Git upload_strategy // (e.g. S3 often uses `Multipart` for LFS while we still need create-if-absent semantics). @@ -35,6 +40,7 @@ impl MegaObjectStorage for ObjectStoreAdapter { data: ObjectByteStream, _meta: ObjectMeta, ) -> OrbitResult<()> { + reject_receipt_mutation(key)?; let path = Self::checked_path(key)?; self.put_multipart(&path, data).await } @@ -45,6 +51,7 @@ impl MegaObjectStorage for ObjectStoreAdapter { bytes: Bytes, _meta: ObjectMeta, ) -> OrbitResult<()> { + reject_receipt_mutation(key)?; use crate::orbit_api::object_storage::MAX_METADATA_ATOMIC_BYTES; if bytes.len() > MAX_METADATA_ATOMIC_BYTES { return Err(IoOrbitError::Other(format!( @@ -61,6 +68,32 @@ impl MegaObjectStorage for ObjectStoreAdapter { Ok(()) } + async fn put_metadata_atomic_create( + &self, + key: &ObjectKey, + bytes: Bytes, + _meta: ObjectMeta, + ) -> OrbitResult<()> { + if bytes.len() > crate::orbit_api::object_storage::MAX_METADATA_ATOMIC_BYTES { + return Err(IoOrbitError::Other( + "immutable atomic metadata exceeds its byte limit".into(), + )); + } + let path = Self::checked_path(key)?; + match self + .to_store() + .put_opts( + &path, + PutPayload::from_bytes(bytes), + PutOptions::from(PutMode::Create), + ) + .await + { + Ok(_) | Err(object_store::Error::AlreadyExists { .. }) => Ok(()), + Err(error) => Err(IoOrbitError::from(error)), + } + } + async fn get_stream(&self, key: &ObjectKey) -> OrbitResult<(ObjectByteStream, ObjectMeta)> { let path = Self::checked_path(key)?; @@ -72,7 +105,115 @@ impl MegaObjectStorage for ObjectStoreAdapter { let meta = build_object_meta(&res.meta); let stream = res.into_stream().map_err(std::io::Error::other); - Ok((Box::pin(stream), meta)) + Ok(( + crate::orbit_api::object_storage::fragment_object_stream(Box::pin(stream)), + meta, + )) + } + + async fn chunk_map_receipt_inventory( + &self, + ) -> OrbitResult { + use crate::orbit_api::object_storage::{ + ChunkMapReceiptInventory, MAX_CHUNK_MAP_RECEIPT_BYTES, MAX_CHUNK_MAP_RECEIPTS, + }; + // object_store 0.14 cannot inventory/delete retained cloud versions. + // A current-object listing is not a physical backing-byte quota. + if !self.supports_chunk_map_retention() { + return Err(IoOrbitError::ChunkMapRetentionUnsupported); + } + let prefix = object_store::path::Path::from("chunk-map-receipt"); + let mut listing = self.to_store().list(Some(&prefix)); + let mut inventory = ChunkMapReceiptInventory { + objects: Vec::new(), + bytes: 0, + }; + while let Some(entry) = listing.next().await { + let meta = entry.map_err(IoOrbitError::from)?; + if inventory.objects.len() == MAX_CHUNK_MAP_RECEIPTS { + return Err(IoOrbitError::ChunkMapRetentionCapacityExceeded); + } + inventory.bytes = inventory + .bytes + .checked_add(meta.size) + .filter(|bytes| *bytes <= MAX_CHUNK_MAP_RECEIPT_BYTES) + .ok_or(IoOrbitError::ChunkMapRetentionCapacityExceeded)?; + let location = meta.location.as_ref(); + let parts: Vec<_> = location.split('/').collect(); + if parts.len() != 5 + || parts[0] != "chunk-map-receipt" + || parts[1..4].iter().any(|part| part.len() != 2) + || parts[4].len() > 128 + { + return Err(IoOrbitError::Other( + "invalid physical chunk-map receipt path".into(), + )); + } + let key = ObjectKey { + namespace: ObjectNamespace::ChunkMapReceipt, + key: parts[1..].concat(), + }; + key.validate()?; + if key.default_sharding() != location { + return Err(IoOrbitError::Other( + "noncanonical physical chunk-map receipt path".into(), + )); + } + inventory.objects.push((key, meta.size)); + } + Ok(inventory) + } + + async fn delete_chunk_map_receipt( + &self, + authority: &crate::orbit_api::object_storage::ChunkMapReceiptDeletion, + ) -> OrbitResult { + if !self.supports_chunk_map_retention() { + return Err(IoOrbitError::ChunkMapRetentionUnsupported); + } + let key = authority.key(); + if key.namespace != ObjectNamespace::ChunkMapReceipt { + return Err(IoOrbitError::Other( + "invalid sealed receipt namespace".into(), + )); + } + let (mut stream, meta) = match self.get_stream(key).await { + Ok(object) => object, + Err(error) if error.is_not_found() => return Ok(false), + Err(error) => return Err(error), + }; + let expected = authority.expected_bytes(); + if meta.size != expected.len() as i64 { + return Err(IoOrbitError::Other( + "retired receipt body size changed".into(), + )); + } + let mut offset: usize = 0; + while let Some(part) = stream.next().await { + let bytes = part?; + let end = offset + .checked_add(bytes.len()) + .filter(|end| *end <= expected.len()) + .ok_or_else(|| { + IoOrbitError::Other("retired receipt body exceeds its exact profile".into()) + })?; + if expected[offset..end] != bytes[..] { + return Err(IoOrbitError::Other("retired receipt body changed".into())); + } + offset = end; + } + if offset != expected.len() { + return Err(IoOrbitError::Other( + "retired receipt body is truncated".into(), + )); + } + // Physical generation keys are never reused. A late create can only + // restore this same retired body; persistent history will reconcile it. + match self.to_store().delete(&Self::checked_path(key)?).await { + Ok(()) => Ok(true), + Err(object_store::Error::NotFound { .. }) => Ok(false), + Err(error) => Err(IoOrbitError::from(error)), + } } async fn get_range_stream( @@ -106,12 +247,54 @@ impl MegaObjectStorage for ObjectStoreAdapter { Ok((Box::pin(stream), meta)) } + async fn get_range_stream_exact( + &self, + key: &ObjectKey, + start: u64, + end: u64, + ) -> OrbitResult> { + if start >= end { + return Err(std::io::Error::new( + std::io::ErrorKind::InvalidInput, + "invalid exact object range", + ) + .into()); + } + let path = Self::checked_path(key)?; + let res = self + .to_store() + .get_opts( + &path, + object_store::GetOptions { + range: Some(object_store::GetRange::Bounded(start..end)), + ..Default::default() + }, + ) + .await + .map_err(IoOrbitError::from)?; + if res.range != (start..end) || end > res.meta.size { + return Err(std::io::Error::new( + std::io::ErrorKind::InvalidData, + "backend returned a different object range", + ) + .into()); + } + let meta = build_object_meta(&res.meta); + let stream = res.into_stream().map_err(std::io::Error::other); + Ok(Some((Box::pin(stream), meta))) + } + async fn signed_url( &self, key: &ObjectKey, method: Method, expires_in: Duration, ) -> OrbitResult> { + if key.namespace == ObjectNamespace::ChunkMapReceipt && method != Method::GET { + return Err(IoOrbitError::Other( + "immutable chunk map receipts do not permit presigned mutation".into(), + )); + } let path = Self::checked_path(key)?; let url = match &self.store { @@ -141,6 +324,7 @@ impl MegaObjectStorage for ObjectStoreAdapter { } async fn delete(&self, key: &ObjectKey) -> OrbitResult<()> { + reject_receipt_mutation(key)?; let path = Self::checked_path(key)?; self.to_store() .delete(&path) @@ -149,3 +333,164 @@ impl MegaObjectStorage for ObjectStoreAdapter { Ok(()) } } + +pub(super) fn reject_receipt_mutation(key: &ObjectKey) -> OrbitResult<()> { + if key.namespace == ObjectNamespace::ChunkMapReceipt { + Err(IoOrbitError::Other( + "chunk map receipts require immutable atomic creation".into(), + )) + } else { + Ok(()) + } +} + +#[cfg(test)] +mod exact_range_tests { + use super::*; + + #[tokio::test] + async fn backend_chunk_retention_capabilities_match_cold_admission_without_cloud_io() { + let s3 = object_store::aws::AmazonS3Builder::new() + .with_bucket_name("capabilities-test") + .with_region("us-east-1") + .with_access_key_id("fixture-access") + .with_secret_access_key("fixture-secret") + .with_skip_signature(true) + .build() + .unwrap(); + let gcs = object_store::gcp::GoogleCloudStorageBuilder::new() + .with_bucket_name("capabilities-test") + .with_credentials(Arc::new(object_store::StaticCredentialProvider::new( + object_store::gcp::GcpCredential { + bearer: "fixture-bearer".into(), + }, + ))) + .with_skip_signature(true) + .build() + .unwrap(); + for store in [ + BackendStore::S3(Arc::new(s3)), + BackendStore::Gcs(Arc::new(gcs)), + ] { + let backend = crate::orbit_api::factory::MegaObjectStorageWrapper::new(Arc::new( + ObjectStoreAdapter { + store, + upload_strategy: UploadStrategy::SinglePut, + presign_store: None, + }, + )); + assert!(!backend.supports_chunk_map_retention()); + assert!(matches!( + backend.inner.chunk_map_receipt_inventory().await, + Err(IoOrbitError::ChunkMapRetentionUnsupported) + )); + } + } + + #[tokio::test] + async fn local_whole_stream_keeps_large_file_size_and_digest_with_bounded_fragments() { + use sha2::{Digest, Sha256}; + + let directory = tempfile::tempdir().unwrap(); + let adapter = ObjectStoreAdapter { + store: BackendStore::Local(Arc::new( + LocalFileSystem::new_with_prefix(directory.path()).unwrap(), + )), + upload_strategy: UploadStrategy::SinglePut, + presign_store: None, + }; + let key = ObjectKey { + namespace: ObjectNamespace::Git, + key: "ab".repeat(20), + }; + let raw = Bytes::from(vec![ + 0xff; + crate::orbit_api::object_storage::OBJECT_STREAM_ITEM_BYTES + + 113 + ]); + let expected: [u8; 32] = Sha256::digest(&raw).into(); + let size = raw.len(); + adapter + .put_stream( + &key, + Box::pin(stream::iter([Ok(raw)])), + ObjectMeta::default(), + ) + .await + .unwrap(); + let (mut input, meta) = adapter.get_stream(&key).await.unwrap(); + assert_eq!(meta.size, size as i64); + assert!(adapter.supports_chunk_map_retention()); + let mut received = 0; + let mut hashed = Sha256::new(); + while let Some(part) = input.next().await { + let part = part.unwrap(); + assert!(part.len() <= crate::orbit_api::object_storage::OBJECT_STREAM_ITEM_BYTES); + received += part.len(); + hashed.update(&part); + } + assert_eq!(received, size); + assert_eq!(<[u8; 32]>::from(hashed.finalize()), expected); + } + + #[tokio::test] + async fn local_chunk_receipts_reject_all_mutation_routes() { + let directory = tempfile::tempdir().unwrap(); + let adapter = ObjectStoreAdapter { + store: BackendStore::Local(Arc::new( + LocalFileSystem::new_with_prefix(directory.path()).unwrap(), + )), + upload_strategy: UploadStrategy::SinglePut, + presign_store: None, + }; + crate::jupiter::storage::object_storage::assert_immutable_chunk_receipt_contract(&adapter) + .await; + } + + #[tokio::test] + async fn local_exact_range_returns_selected_bytes_and_full_object_size_without_clipping() { + let directory = tempfile::tempdir().unwrap(); + let adapter = ObjectStoreAdapter { + store: BackendStore::Local(Arc::new( + LocalFileSystem::new_with_prefix(directory.path()).unwrap(), + )), + upload_strategy: UploadStrategy::SinglePut, + presign_store: None, + }; + let key = ObjectKey { + namespace: ObjectNamespace::Git, + key: "abcdef1234567890".to_string(), + }; + adapter + .put_stream( + &key, + Box::pin(stream::iter([Ok(Bytes::from_static(b"0123456789"))])), + ObjectMeta::default(), + ) + .await + .unwrap(); + let (input, meta) = adapter + .get_range_stream_exact(&key, 3, 7) + .await + .unwrap() + .unwrap(); + assert_eq!(meta.size, 10); + assert_eq!( + ObjectStoreAdapter::buffer_stream(input, 4).await.unwrap(), + b"3456".as_slice() + ); + let (input, meta) = adapter + .get_range_stream_exact(&key, 9, 10) + .await + .unwrap() + .unwrap(); + assert_eq!(meta.size, 10); + assert_eq!( + ObjectStoreAdapter::buffer_stream(input, 1).await.unwrap(), + b"9".as_slice() + ); + assert!(adapter.get_range_stream_exact(&key, 9, 11).await.is_err()); + assert!(adapter.get_range_stream_exact(&key, 5, 5).await.is_err()); + assert!(adapter.get_range_stream_exact(&key, 10, 11).await.is_err()); + } +} diff --git a/src/orbit_api/error.rs b/src/orbit_api/error.rs index 365d2e4d..9de51a0c 100644 --- a/src/orbit_api/error.rs +++ b/src/orbit_api/error.rs @@ -17,6 +17,12 @@ pub enum IoOrbitError { #[error("write manifest precondition failed")] WriteManifestPreconditionFailed, + #[error("chunk-map retention is not supported by this storage backend")] + ChunkMapRetentionUnsupported, + + #[error("chunk-map backing receipts exceed the fixed retention quota")] + ChunkMapRetentionCapacityExceeded, + #[error("other error: {0}")] Other(String), } diff --git a/src/orbit_api/factory.rs b/src/orbit_api/factory.rs index c1f8e86e..f6891a18 100644 --- a/src/orbit_api/factory.rs +++ b/src/orbit_api/factory.rs @@ -135,6 +135,10 @@ impl MegaObjectStorageWrapper { pub fn supports_presigned_urls(&self) -> bool { MegaObjectStorage::supports_presigned_urls(&*self.inner) } + + pub fn supports_chunk_map_retention(&self) -> bool { + MegaObjectStorage::supports_chunk_map_retention(&*self.inner) + } } #[cfg(test)] diff --git a/src/orbit_api/object_storage.rs b/src/orbit_api/object_storage.rs index 00e47dda..e3898424 100644 --- a/src/orbit_api/object_storage.rs +++ b/src/orbit_api/object_storage.rs @@ -122,6 +122,8 @@ pub enum ObjectNamespace { Media, /// Agent Capture objects (`docs/refactoring/agent-capture.md`). Agent, + /// Immutable receipts minted by the full-stream chunk-map verifier. + ChunkMapReceipt, } impl ObjectNamespace { @@ -135,6 +137,7 @@ impl ObjectNamespace { ObjectNamespace::Oci => "oci", ObjectNamespace::Media => "media", ObjectNamespace::Agent => "agent", + ObjectNamespace::ChunkMapReceipt => "chunk-map-receipt", } } } @@ -165,9 +168,66 @@ pub struct ObjectMeta { /// - The stream must be fully consumed by the caller. pub type ObjectByteStream = Pin> + Send>>; +pub(crate) const OBJECT_STREAM_ITEM_BYTES: usize = 8 * 1024 * 1024; + +/// Split visible items without copying or prebuilding a fragment list. Bytes +/// retain their original owner; this does not bound backend backing buffers. +pub(crate) fn fragment_object_stream(input: ObjectByteStream) -> ObjectByteStream { + Box::pin(futures::stream::unfold( + (input, Bytes::new()), + |(mut input, pending)| async move { + let next = if pending.is_empty() { + input.next().await? + } else { + Ok(pending) + }; + match next { + Ok(mut bytes) => { + let part = bytes.split_to(bytes.len().min(OBJECT_STREAM_ITEM_BYTES)); + Some((Ok(part), (input, bytes))) + } + Err(error) => Some((Err(error), (input, Bytes::new()))), + } + }, + )) +} + /// Upper bound for [`MegaObjectStorage::put_metadata_atomic`] (ADR-MF-05). pub const MAX_METADATA_ATOMIC_BYTES: usize = 1024 * 1024; +pub const MAX_CHUNK_MAP_RECEIPTS: usize = 16_384; +pub const MAX_CHUNK_MAP_RECEIPT_BYTES: u64 = 64 * 1024 * 1024; + +/// Actual backing objects, including uncommitted and late-created receipts. +/// A complete inventory is returned only while both fixed quotas hold. +pub struct ChunkMapReceiptInventory { + pub objects: Vec<(ObjectKey, u64)>, + pub bytes: u64, +} + +/// Sealed by the primary retention repository after claiming one exact +/// retired generation and independently reading its immutable backend body. +/// Ordinary object callers cannot construct a receipt deletion authority. +pub struct ChunkMapReceiptDeletion { + claim: crate::jupiter::storage::native_chunk_map::retention::VerifiedReceiptDeletionClaim, +} + +impl ChunkMapReceiptDeletion { + pub(crate) fn from_claim( + claim: crate::jupiter::storage::native_chunk_map::retention::VerifiedReceiptDeletionClaim, + ) -> Self { + Self { claim } + } + + pub(crate) fn key(&self) -> &ObjectKey { + self.claim.key() + } + + pub(crate) fn expected_bytes(&self) -> &[u8] { + self.claim.expected_bytes() + } +} + /// A streaming source of multiple objects. /// /// Each item yields: @@ -196,6 +256,13 @@ pub trait MegaObjectStorage: Send + Sync { false } + /// Complete receipt inventory and sealed deletion are available without + /// uncounted retained versions. This is a static backend contract, not an + /// I/O health or quota check; cold requests still perform real admission. + fn supports_chunk_map_retention(&self) -> bool { + false + } + /// Upload a single object to the storage backend. /// /// # Parameters @@ -260,6 +327,37 @@ pub trait MegaObjectStorage: Send + Sync { )) } + /// Create a complete <=1 MiB immutable metadata object. An existing key + /// must retain its original bytes. The caller must read and compare the + /// existing object before treating an idempotent replay as success. + async fn put_metadata_atomic_create( + &self, + _key: &ObjectKey, + _bytes: Bytes, + _meta: ObjectMeta, + ) -> OrbitResult<()> { + Err(IoOrbitError::Other( + "immutable atomic metadata creation is not supported by this storage backend" + .to_string(), + )) + } + + /// Enumerate the real receipt namespace with fixed count/byte bounds. + /// Unsupported stores fail before a cold source body can be opened. + async fn chunk_map_receipt_inventory(&self) -> OrbitResult { + Err(IoOrbitError::ChunkMapRetentionUnsupported) + } + + /// Delete only the independently checked body of a sealed retired + /// physical generation. The key is never reused by a later installation. + /// Missing is idempotent; transport/authentication errors remain errors. + async fn delete_chunk_map_receipt( + &self, + _authority: &ChunkMapReceiptDeletion, + ) -> OrbitResult { + Err(IoOrbitError::ChunkMapRetentionUnsupported) + } + /// Retrieve a single object from the storage backend. /// /// # Returns @@ -306,6 +404,20 @@ pub trait MegaObjectStorage: Send + Sync { end: Option, ) -> OrbitResult<(ObjectByteStream, ObjectMeta)>; + /// Exact raw byte range, with no full-download fallback. `None` means + /// unsupported and must be returned without source I/O. Implementations + /// must validate the backend's actual start/end before exposing the stream; + /// metadata describes the complete object. Consumers still verify EOF, + /// length and content hashes. This does not promise bounded producer RSS. + async fn get_range_stream_exact( + &self, + _key: &ObjectKey, + _start: u64, + _end: u64, + ) -> OrbitResult> { + Ok(None) + } + /// Check whether an object exists. async fn exists(&self, key: &ObjectKey) -> OrbitResult; @@ -513,6 +625,7 @@ mod tests { ObjectNamespace::Oci, ObjectNamespace::Media, ObjectNamespace::Agent, + ObjectNamespace::ChunkMapReceipt, ] { let key = ObjectKey { namespace: ns, @@ -536,6 +649,10 @@ mod tests { assert_eq!(ObjectNamespace::Oci.to_string(), "oci"); assert_eq!(ObjectNamespace::Media.to_string(), "media"); assert_eq!(ObjectNamespace::Agent.to_string(), "agent"); + assert_eq!( + ObjectNamespace::ChunkMapReceipt.to_string(), + "chunk-map-receipt" + ); } #[test] @@ -590,6 +707,80 @@ mod tests { struct UnsupportedBoundedStore; + struct SharedStreamOwner { + bytes: Vec, + drops: std::sync::Arc, + } + + impl AsRef<[u8]> for SharedStreamOwner { + fn as_ref(&self) -> &[u8] { + &self.bytes + } + } + + impl Drop for SharedStreamOwner { + fn drop(&mut self) { + self.drops.fetch_add(1, std::sync::atomic::Ordering::SeqCst); + } + } + + #[tokio::test] + async fn fragmented_source_is_lazy_zero_copy_and_keeps_original_owner_until_last_clone() { + use std::sync::{ + Arc, + atomic::{AtomicUsize, Ordering}, + }; + + let drops = Arc::new(AtomicUsize::new(0)); + let input_polls = Arc::new(AtomicUsize::new(0)); + let bytes = Bytes::from_owner(SharedStreamOwner { + bytes: vec![0xa5; 2 * OBJECT_STREAM_ITEM_BYTES + 17], + drops: drops.clone(), + }); + let pointer = bytes.as_ptr(); + let polls = input_polls.clone(); + let input = + futures::stream::iter([Ok(bytes), Err(std::io::Error::other("late backend error"))]) + .inspect(move |_| { + polls.fetch_add(1, Ordering::SeqCst); + }); + let mut fragmented = fragment_object_stream(Box::pin(input)); + assert_eq!(input_polls.load(Ordering::SeqCst), 0); + let first = fragmented.next().await.unwrap().unwrap(); + assert_eq!(first.len(), OBJECT_STREAM_ITEM_BYTES); + assert_eq!(first.as_ptr(), pointer); + let transport = first.clone(); + drop(first); + let second = fragmented.next().await.unwrap().unwrap(); + assert_eq!(second.len(), OBJECT_STREAM_ITEM_BYTES); + assert_eq!( + second.as_ptr() as usize, + pointer as usize + OBJECT_STREAM_ITEM_BYTES + ); + drop(second); + let tail = fragmented.next().await.unwrap().unwrap(); + assert_eq!(tail.as_ref(), &[0xa5; 17]); + drop(tail); + assert_eq!(input_polls.load(Ordering::SeqCst), 1); + assert!(fragmented.next().await.unwrap().is_err()); + assert_eq!(input_polls.load(Ordering::SeqCst), 2); + assert!(fragmented.next().await.is_none()); + drop(fragmented); + assert_eq!(drops.load(Ordering::SeqCst), 0); + drop(transport); + assert_eq!(drops.load(Ordering::SeqCst), 1); + } + + #[tokio::test] + async fn default_backend_does_not_advertise_or_infer_chunk_retention() { + let backend = UnsupportedBoundedStore; + assert!(!backend.supports_chunk_map_retention()); + assert!(matches!( + backend.chunk_map_receipt_inventory().await, + Err(IoOrbitError::ChunkMapRetentionUnsupported) + )); + } + #[async_trait::async_trait] impl MegaObjectStorage for UnsupportedBoundedStore { async fn put_stream( @@ -682,4 +873,22 @@ mod tests { .contains("atomic metadata put is not supported") ); } + + #[tokio::test] + async fn exact_range_default_is_unsupported_without_calling_fallback() { + let store = UnsupportedBoundedStore; + let key = ObjectKey { + namespace: ObjectNamespace::Git, + key: "abcdef".to_string(), + }; + // Its general range getter returns an error; the new default must + // return typed absence without invoking that method or full get. + assert!( + store + .get_range_stream_exact(&key, 1, 2) + .await + .unwrap() + .is_none() + ); + } } diff --git a/src/server/http_server.rs b/src/server/http_server.rs index 04be7f77..777e732a 100644 --- a/src/server/http_server.rs +++ b/src/server/http_server.rs @@ -380,6 +380,35 @@ fn spawn_view_worker_task( spawn_view_worker_with_round(storage, token, production_round()) } +fn spawn_chunk_map_retention_task( + storage: Storage, + token: CancellationToken, + available: bool, +) -> Option> { + if !available { + return None; + } + Some(tokio::spawn(async move { + let mut interval = tokio::time::interval(std::time::Duration::from_secs(30)); + interval.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Skip); + loop { + tokio::select! { + _=token.cancelled()=>break, + _=interval.tick()=>{ + // Cancellation drops the real maintenance future. Claims + // are durable, so interrupted backend operations replay. + tokio::select! { + _=token.cancelled()=>break, + result=async { + storage.chunk_maps().await?.maintain(&storage.git_service.obj_storage,64).await + }=>if let Err(error)=result {tracing::warn!(?error,"bounded chunk-map retention tick failed");} + } + } + } + } + })) +} + /// Returns a future that completes when the cancellation token is triggered. async fn shutdown_signal(token: CancellationToken) { token.cancelled().await; @@ -450,6 +479,46 @@ pub async fn start_http(ctx: AppContext, options: CommonHttpOptions) -> MegaResu // TP-15 / 4.1 ②③⑥: open CLs, last_policy vs non-terminal rows, watermark reset. ctx.storage.prepare_push_policy_startup().await?; ctx.storage.check_view_reserved_startup().await?; + let chunk_map_retention_available = if ctx.storage.config().mst2.enabled { + // Detect the actual backend capability. Unsupported CHUNK retention + // must not disable other snapshot objects or generic HTTP serving. + match tokio::time::timeout( + std::time::Duration::from_secs(30), + ctx.storage + .git_service + .obj_storage + .inner + .chunk_map_receipt_inventory(), + ) + .await + { + Ok(Err(crate::orbit_api::error::IoOrbitError::ChunkMapRetentionUnsupported)) => { + tracing::info!( + "CHUNK cold admission and maintenance unavailable for this backing store; other snapshot routes remain available" + ); + false + } + _ => { + let result = async { + ctx.storage + .chunk_maps() + .await? + .maintain(&ctx.storage.git_service.obj_storage, 64) + .await + } + .await; + if let Err(error) = result { + tracing::warn!( + ?error, + "initial CHUNK retention check failed; cold admission remains fail closed" + ); + } + true + } + } + } else { + false + }; // First-build the shared authorization snapshot before the listener binds // (UN-02). `off` is a no-op; a failed first build fails startup. @@ -479,6 +548,11 @@ pub async fn start_http(ctx: AppContext, options: CommonHttpOptions) -> MegaResu spawn_artifact_gc_task(shutdown_context.clone(), shutdown_token.clone())?; let view_worker_handle = spawn_view_worker_task(shutdown_context.storage.clone(), shutdown_token.clone())?; + let chunk_map_retention_handle = spawn_chunk_map_retention_task( + shutdown_context.storage.clone(), + shutdown_token.clone(), + chunk_map_retention_available, + ); let notification_shutdown = shutdown_context.notification_shutdown.clone(); let server_token = shutdown_token.clone(); @@ -541,7 +615,7 @@ pub async fn start_http(ctx: AppContext, options: CommonHttpOptions) -> MegaResu }, }; - let (cleanup_result, artifact_gc_result, view_worker_result) = tokio::join!( + let (cleanup_result, artifact_gc_result, view_worker_result, chunk_map_retention_result) = tokio::join!( async { if let Some(handle) = cleanup_handle { match tokio::time::timeout(std::time::Duration::from_secs(30), handle).await { @@ -612,11 +686,34 @@ pub async fn start_http(ctx: AppContext, options: CommonHttpOptions) -> MegaResu } else { Ok(()) } + }, + async { + if let Some(mut handle) = chunk_map_retention_handle { + match tokio::time::timeout(std::time::Duration::from_secs(30), &mut handle).await { + Ok(Ok(())) => Ok(()), + Ok(Err(error)) => { + tracing::error!(%error,"chunk-map retention task panicked"); + Err(()) + } + Err(_) => { + handle.abort(); + let _ = handle.await; + tracing::error!( + "chunk-map retention task exceeded shutdown deadline and was aborted" + ); + Err(()) + } + } + } else { + Ok(()) + } } ); - let shutdown_failed = - cleanup_result.is_err() || artifact_gc_result.is_err() || view_worker_result.is_err(); + let shutdown_failed = cleanup_result.is_err() + || artifact_gc_result.is_err() + || view_worker_result.is_err() + || chunk_map_retention_result.is_err(); match (shutdown_failed, &server_result) { (false, Ok(())) => { tracing::info!("Graceful shutdown completed successfully"); diff --git a/tests/common/git_cli.rs b/tests/common/git_cli.rs index 9f5078a6..b2fa7c2a 100644 --- a/tests/common/git_cli.rs +++ b/tests/common/git_cli.rs @@ -1283,12 +1283,7 @@ pub fn ensure_git_cli_passwd_for_runtime_uid() { "unexpected git-cli id output: {id_pair}" ); - let script = format!( - "if getent passwd {uid} >/dev/null 2>&1; then exit 0; fi; \ - if ! getent group {gid} >/dev/null 2>&1; then addgroup -g {gid} gitcliruntime || true; fi; \ - group_name=\"$(getent group {gid} | cut -d: -f1)\"; \ - adduser -D -u {uid} -G \"$group_name\" -h /home/gitcli -s /bin/sh gitcliruntime" - ); + let script = git_cli_runtime_passwd_script(&uid, &gid); let mut command = docker_exec_base(); command .arg("-u") @@ -1303,6 +1298,81 @@ pub fn ensure_git_cli_passwd_for_runtime_uid() { ); } +fn git_cli_runtime_passwd_script(uid: &str, gid: &str) -> String { + assert!( + !uid.is_empty() + && !gid.is_empty() + && uid.bytes().all(|byte| byte.is_ascii_digit()) + && gid.bytes().all(|byte| byte.is_ascii_digit()), + "git-cli runtime uid/gid must be numeric" + ); + format!( + r#"passwd_matches() {{ + getent passwd {uid} | awk -F: -v uid={uid} -v gid={gid} \ + '$3 == uid && $4 == gid {{ matched++ }} END {{ exit matched != 1 }}' + }} + if getent passwd {uid} >/dev/null 2>&1; then passwd_matches; exit $?; fi + if ! getent group {gid} >/dev/null 2>&1; then addgroup -g {gid} gitcliruntime || true; fi + group_name="$(getent group {gid} | cut -d: -f1)" + test -n "$group_name" || exit 1 + if adduser -D -u {uid} -G "$group_name" -h /home/gitcli -s /bin/sh gitcliruntime; then + passwd_matches + else + status=$? + # Another SSH test can win the check/create race. Accept only the + # exact runtime identity, never an unrelated name or primary GID. + if passwd_matches; then exit 0; fi + exit "$status" + fi"# + ) +} + +#[cfg(target_os = "linux")] +#[test] +fn git_cli_passwd_creation_race_requires_exact_runtime_identity() { + let fixture = r#" + getent() { + if [ "$1" = group ]; then printf 'gitcliruntime:x:1001:\n'; return 0; fi + if [ "$SCENARIO" = existing ]; then printf 'gitcliruntime:x:1001:1001::/home/gitcli:/bin/sh\n'; return 0; fi + if [ "$SCENARIO" = existing_wrong_gid ]; then printf 'gitcliruntime:x:1001:1000::/home/gitcli:/bin/sh\n'; return 0; fi + if [ ! -f "$MOCK_STATE" ]; then return 2; fi + cat "$MOCK_STATE" + } + adduser() { + case "$SCENARIO" in + raced|created) printf 'gitcliruntime:x:1001:1001::/home/gitcli:/bin/sh\n' > "$MOCK_STATE" ;; + wrong_uid) printf 'gitcliruntime:x:1000:1001::/home/gitcli:/bin/sh\n' > "$MOCK_STATE" ;; + wrong_gid|created_wrong_gid) printf 'gitcliruntime:x:1001:1000::/home/gitcli:/bin/sh\n' > "$MOCK_STATE" ;; + esac + if [ "$SCENARIO" = created ] || [ "$SCENARIO" = created_wrong_gid ]; then return 0; fi + return 17 + } + "#; + let temp = tempfile::tempdir().expect("passwd race fixture"); + let script = format!( + "{fixture}\n{}", + git_cli_runtime_passwd_script("1001", "1001") + ); + for (scenario, expected) in [ + ("existing", 0), + ("existing_wrong_gid", 1), + ("raced", 0), + ("created", 0), + ("missing", 17), + ("wrong_uid", 17), + ("wrong_gid", 17), + ("created_wrong_gid", 1), + ] { + let mut command = Command::new("sh"); + command + .args(["-c", &script]) + .env("SCENARIO", scenario) + .env("MOCK_STATE", temp.path().join(scenario)); + let output = output_with_timeout(command, Duration::from_secs(5), "passwd race fixture"); + assert_eq!(output.status.code(), Some(expected), "{scenario}"); + } +} + /// Build ADR-GM-05 `GIT_SSH_COMMAND` for the selected runner (container paths under `/work`). #[allow( dead_code, diff --git a/tests/integration_git_cli.rs b/tests/integration_git_cli.rs index f0c326b4..1e3cd312 100644 --- a/tests/integration_git_cli.rs +++ b/tests/integration_git_cli.rs @@ -20,7 +20,6 @@ use std::{ collections::BTreeMap, fs, io::{Read, Write}, - net::TcpStream, path::{Path, PathBuf}, process::{Child, Command, ExitStatus, Stdio}, sync::atomic::{AtomicUsize, Ordering}, @@ -204,34 +203,51 @@ impl ServiceProcess { service } - fn wait_until_listening( + fn wait_until_openapi_ready( &mut self, port: u16, timeout: Duration, stdout_path: &Path, stderr_path: &Path, ) { + let client = reqwest::blocking::Client::builder() + .timeout(Duration::from_secs(5)) + .redirect(reqwest::redirect::Policy::none()) + .build() + .expect("build HTTP readiness client"); + let url = format!("http://127.0.0.1:{port}/api/openapi.json"); let deadline = Instant::now() + timeout; loop { - if TcpStream::connect(("127.0.0.1", port)).is_ok() { - return; - } if let Some(status) = self.child.try_wait().expect("poll service") { self.reaped = true; panic!( - "service exited before binding port {port} (status {status})\nstdout:\n{}\nstderr:\n{}", + "service exited before OpenAPI was ready on port {port} (status {status})\nstdout:\n{}\nstderr:\n{}", read_log(stdout_path), read_log(stderr_path), ); } - if Instant::now() >= deadline { + let remaining = deadline.saturating_duration_since(Instant::now()); + if remaining.is_zero() { panic!( - "service did not bind port {port} within {timeout:?}\nstdout:\n{}\nstderr:\n{}", + "service OpenAPI not ready on port {port} within {timeout:?}\nstdout:\n{}\nstderr:\n{}", read_log(stdout_path), read_log(stderr_path), ); } - sleep(Duration::from_millis(200)); + if let Ok(response) = client + .get(&url) + .timeout(remaining.min(Duration::from_secs(5))) + .send() + && response.status() == reqwest::StatusCode::OK + && Instant::now() <= deadline + { + return; + } + sleep( + deadline + .saturating_duration_since(Instant::now()) + .min(Duration::from_millis(200)), + ); } } @@ -345,7 +361,7 @@ fn boot_service_http_with_env( .stderr(Stdio::from(create_log_file(&stderr_path))); let mut service = ServiceProcess::spawn(command); - service.wait_until_listening(port, Duration::from_secs(90), &stdout_path, &stderr_path); + service.wait_until_openapi_ready(port, Duration::from_secs(90), &stdout_path, &stderr_path); (service, port, stdout_path, stderr_path) } diff --git a/tests/integration_storage_events_runtime.rs b/tests/integration_storage_events_runtime.rs index 9443a443..ab8c4908 100644 --- a/tests/integration_storage_events_runtime.rs +++ b/tests/integration_storage_events_runtime.rs @@ -363,12 +363,6 @@ fn integration_storage_events_runtime_occupied_port_exits_through_cleanup() { // evidence that the cleanup tail ran. let receipt = wait_for_shutdown_receipt(&stdout_path, &stderr_path, Duration::from_secs(60)); assert_shutdown_receipt_sanitized(&receipt); - // The failure must be the port bind, not some unrelated startup error. - let combined = format!("{}\n{}", read_log(&stdout_path), read_log(&stderr_path)); - assert!( - combined.contains("failed to bind HTTP listener"), - "occupied-port case must fail at the HTTP bind:\n{combined}" - ); let status = service .wait_for_exit(Duration::from_secs(60)) .expect("occupied-port startup failure must exit on its own"); @@ -378,6 +372,13 @@ fn integration_storage_events_runtime_occupied_port_exits_through_cleanup() { read_log(&stdout_path), read_log(&stderr_path), ); + // main prints the startup error after cleanup returns; wait for exit + // before reading the final diagnostic that must identify the bind failure. + let combined = format!("{}\n{}", read_log(&stdout_path), read_log(&stderr_path)); + assert!( + combined.contains("failed to bind HTTP listener"), + "occupied-port case must fail at the HTTP bind:\n{combined}" + ); } #[test]