From 8d2bbae6fc9df763cd08cea6fb4d2e1763458f49 Mon Sep 17 00:00:00 2001 From: River Date: Tue, 29 Sep 2026 18:20:21 +0000 Subject: [PATCH 1/6] libsql-server: create migration targets in quarantine Add `NamespaceStore::create_target_quarantined` (and route the admin `CreateTargetQuarantined` command through it), which creates a namespace as a quarantined migration target atomically with namespace creation: - A name the server already knows (config in memory, or a namespace cache entry) is refused with FENCE_PRECONDITION_FAILED/namespace_exists without touching its gate. - Otherwise the controller publishes an in-memory target-creation gate before the metastore transaction writes the marker, config row, record and receipt. The gate refuses every class but maintenance and observability, and `check_available` refuses the name, so create, fork and `with()` neither store nor set anything up for it. - The committed record's quarantine gate replaces it; only then is the config published into the in-memory map (from the durable row, with the record's own block values) and the namespace loaded, so its first connection maker is created behind the quarantine gate. - The command runs on its own task under the transition lock; a replay returns the stored result and completes the publication and load of a commit that was not acknowledged, or of a creation interrupted between its marker and its commit. `CreateTargetRequest` is the typed entry point for the admin route and bulk import. The target's log id is not written back into the record, to keep the marker rule for same-revision records intact (documented). Co-authored-by: Tomasz Szymczyszyn --- .../src/namespace/fence/controller.rs | 80 ++- libsql-server/src/namespace/fence/mod.rs | 6 +- libsql-server/src/namespace/fence/registry.rs | 6 +- libsql-server/src/namespace/fence/target.rs | 546 ++++++++++++++++++ libsql-server/src/namespace/meta_store.rs | 52 +- libsql-server/src/namespace/store.rs | 159 ++++- 6 files changed, 838 insertions(+), 11 deletions(-) create mode 100644 libsql-server/src/namespace/fence/target.rs diff --git a/libsql-server/src/namespace/fence/controller.rs b/libsql-server/src/namespace/fence/controller.rs index ff946554c6..04149a66b4 100644 --- a/libsql-server/src/namespace/fence/controller.rs +++ b/libsql-server/src/namespace/fence/controller.rs @@ -58,6 +58,11 @@ pub struct GateSnapshot { /// allows. Never persisted, and cleared by every publication of a commit. It does not move /// the write generation: write admission is already closed wherever a read fence can be set. pub closing_reads: Option, + /// The in-memory gate of a `CreateTargetQuarantined` that is being persisted (section + /// 10.1): the name is becoming a quarantined target, so everything but maintenance and + /// observability is refused with `MIGRATION_TARGET_QUARANTINED`, and the namespace is not + /// set up. Never persisted; replaced by the record the command's commit publishes. + pub creating_target: Option, } impl GateSnapshot { @@ -68,6 +73,7 @@ impl GateSnapshot { indeterminate: None, installing: None, closing_reads: None, + creating_target: None, } } @@ -104,6 +110,20 @@ impl GateSnapshot { .with_detail(FenceDetail::IndeterminateCommit)); } } + if let Some((operation_id, command_id)) = self.creating_target { + if !matches!( + class, + OperationClass::Maintenance | OperationClass::Observability + ) { + return Err(FenceError::new( + FenceOutcome::MigrationTargetQuarantined, + format!( + "{class:?} is not permitted: fence command {command_id} of operation \ + {operation_id} is creating this namespace as a migration target" + ), + )); + } + } self.fence.permits(class)?; if let Some((operation_id, command_id)) = self.installing { if matches!( @@ -141,6 +161,11 @@ impl GateSnapshot { self.installing.is_some() } + /// Whether the namespace is being created as a quarantined target. + pub fn is_creating_target(&self) -> bool { + self.creating_target.is_some() + } + /// Normal write admission. pub fn write(&self) -> Admission { Admission::from( @@ -507,14 +532,31 @@ impl FenceController { } /// Publish a new gate. `fence: None` keeps the published fence. The write generation moves - /// whenever the state, the owning operation, the indeterminate flag or the installing gate - /// changes. Every publication removes the read-closing gate: the commit that follows it - /// either persists the read fence or proves that nothing changed. + /// whenever the state, the owning operation, the indeterminate flag, the installing gate or + /// the target-creation gate changes. Every publication removes the read-closing gate: the + /// commit that follows it either persists the read fence or proves that nothing changed. + /// A publication with a fence also removes the target-creation gate, which the committed + /// record replaces; one without keeps it. fn publish( &self, fence: Option, indeterminate: Option, installing: Option, + ) { + let creating_target = if fence.is_some() { + None + } else { + self.gate.borrow().creating_target + }; + self.publish_gate(fence, indeterminate, installing, creating_target); + } + + fn publish_gate( + &self, + fence: Option, + indeterminate: Option, + installing: Option, + creating_target: Option, ) { let mut generation_changed = false; self.gate.send_modify(|gate| { @@ -523,10 +565,12 @@ impl FenceController { || fence.record().map(|r| r.operation_id) != gate.fence.record().map(|r| r.operation_id) || indeterminate != gate.indeterminate - || installing != gate.installing; + || installing != gate.installing + || creating_target != gate.creating_target; gate.fence = fence; gate.indeterminate = indeterminate; gate.installing = installing; + gate.creating_target = creating_target; gate.closing_reads = None; if changed { gate.write_generation += 1; @@ -542,6 +586,7 @@ impl FenceController { write_generation = gate.write_generation, indeterminate = gate.indeterminate.is_some(), installing = gate.installing.is_some(), + creating_target = gate.creating_target.is_some(), "published namespace fence gate" ); } @@ -586,6 +631,33 @@ impl Transition { } } + /// Publish the in-memory target-creation gate for `CreateTargetQuarantined` `key` (section + /// 10.1): until the command's commit publishes the quarantined record, every class but + /// maintenance and observability is refused and the namespace is not set up. It is removed + /// with [`remove_creating_target`](Self::remove_creating_target) when the command is proven + /// not to have committed, and kept (with the indeterminate flag) when its outcome is + /// unknown. + pub fn install_creating_target(&mut self, key: CommandKey) { + let (indeterminate, installing) = { + let gate = self.controller.gate.borrow(); + (gate.indeterminate, gate.installing) + }; + self.controller + .publish_gate(None, indeterminate, installing, Some(key)); + } + + /// Remove the target-creation gate of a command that was proven not to have committed. + pub fn remove_creating_target(&mut self) { + let (indeterminate, installing, creating) = { + let gate = self.controller.gate.borrow(); + (gate.indeterminate, gate.installing, gate.creating_target) + }; + if creating.is_some() { + self.controller + .publish_gate(None, indeterminate, installing, None); + } + } + /// Publish the in-memory read-closing gate for `SetSourceReadFence` `key` (section 9, /// step 2): new SQL programs, dumps, replication calls and ATTACHes of the namespace are /// refused with `MIGRATION_READ_FENCED`. A read lease is only ever taken after checking the diff --git a/libsql-server/src/namespace/fence/mod.rs b/libsql-server/src/namespace/fence/mod.rs index 74f227cf23..fd8810e0db 100644 --- a/libsql-server/src/namespace/fence/mod.rs +++ b/libsql-server/src/namespace/fence/mod.rs @@ -10,8 +10,9 @@ //! ([`store`], driven by `MetaStore::apply_fence_command`), and the in-memory authority built //! on them: the per-namespace [`controller`] with its gate and read leases, the positive write //! [`drain`], the source [`read`] fence and its -//! [`stream`] leases for dump and replication, the [`registry`] that holds the controllers outside the -//! namespace cache, and the test [`hooks`] on their paths. +//! [`stream`] leases for dump and replication, quarantined migration [`target`]s, the +//! [`registry`] that holds the controllers outside the namespace cache, and the test [`hooks`] +//! on their paths. // The persistence, controller and protocol layers that consume these types land in the // following commits of this series; until then most of the module is unused by the rest of @@ -29,6 +30,7 @@ pub mod registry; pub mod state; pub mod store; pub mod stream; +pub mod target; pub mod transition; #[cfg(test)] diff --git a/libsql-server/src/namespace/fence/registry.rs b/libsql-server/src/namespace/fence/registry.rs index 610189677b..7c03102cd1 100644 --- a/libsql-server/src/namespace/fence/registry.rs +++ b/libsql-server/src/namespace/fence/registry.rs @@ -66,13 +66,13 @@ impl FenceRegistry { self.controllers.lock().remove(namespace) } - /// Refuse a namespace whose fence state is `UNKNOWN_UNAVAILABLE`, before any work is done - /// to serve it. + /// Refuse a namespace whose fence state is `UNKNOWN_UNAVAILABLE`, or that is being created + /// as a quarantined target, before any work is done to serve it. pub fn check_available(&self, namespace: &NamespaceName) -> Result<(), FenceError> { match self.get(namespace) { Some(controller) => { let gate = controller.gate(); - if gate.is_unavailable() { + if gate.is_unavailable() || gate.is_creating_target() { gate.permits(OperationClass::NormalRead) } else { Ok(()) diff --git a/libsql-server/src/namespace/fence/target.rs b/libsql-server/src/namespace/fence/target.rs new file mode 100644 index 0000000000..3b69b7efe5 --- /dev/null +++ b/libsql-server/src/namespace/fence/target.rs @@ -0,0 +1,546 @@ +//! Migration targets (`docs/NAMESPACE_FENCE.md` sections 10 and 11). +//! +//! A target namespace is created by the operation that will fill it, already in +//! `TARGET_QUARANTINED`, and it is quarantined from the first instant anything else could +//! observe it: the in-memory target-creation gate is installed before the metastore transaction +//! that writes its marker, config row, record and receipt; the committed record replaces that +//! gate; and only then is the config put where `exists()` and `lookup()` find it and the +//! namespace loaded, so its first connection maker is created behind the quarantine gate. +//! +//! [`CreateTargetRequest`] is the typed entry point that the admin route and bulk import both +//! use; `NamespaceStore::create_target_quarantined` runs it. `AbortQuarantinedTarget` needs +//! nothing of its own: it is an ordinary transition, and `TARGET_ABORTED` denies every +//! normal class exactly as the quarantine does. + +use uuid::Uuid; + +use crate::namespace::NamespaceName; + +use super::command::{FenceCommand, FenceRequest, TargetConfig}; +use super::outcome::{FenceDetail, FenceError, FenceOutcome}; +use super::state::FenceState; + +/// `CreateTargetQuarantined` for `namespace`, by `operation_id`. The expectation is always +/// `ABSENT` at revision 0, so it is not part of the request. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct CreateTargetRequest { + pub namespace: NamespaceName, + pub operation_id: Uuid, + /// Idempotency key: replaying the same command returns its stored result, and completes a + /// creation that was interrupted between its marker and its commit. + pub command_id: Uuid, + pub config: TargetConfig, +} + +impl From for FenceRequest { + fn from(req: CreateTargetRequest) -> Self { + FenceRequest { + namespace: req.namespace, + operation_id: req.operation_id, + command_id: req.command_id, + expected_state: FenceState::Absent, + expected_revision: 0, + command: FenceCommand::CreateTargetQuarantined { config: req.config }, + } + } +} + +/// The refusal of a target name that the server already knows, in memory or in the namespace +/// cache, although the metastore may not hold it yet (a create or fork in flight, or one the +/// fence refused after it had published its config in memory). +pub(crate) fn name_in_use(namespace: &NamespaceName) -> FenceError { + FenceError::new( + FenceOutcome::FencePreconditionFailed, + format!("namespace `{namespace}` already exists on this server"), + ) + .with_detail(FenceDetail::NamespaceExists) +} + +#[cfg(test)] +pub(crate) mod tests { + use std::sync::Arc; + + use libsql_replication::rpc::replication::replication_log_server::ReplicationLog; + use libsql_replication::rpc::replication::{HelloRequest, NAMESPACE_METADATA_KEY}; + use tempfile::{tempdir, TempDir}; + use tonic::metadata::BinaryMetadataValue; + + use super::*; + use crate::auth::Authenticated; + use crate::connection::config::DatabaseConfig; + use crate::connection::program::Program; + use crate::connection::{Connection as _, RequestContext}; + use crate::error::Error; + use crate::namespace::fence::command::FenceCommand; + use crate::namespace::fence::controller::FenceController; + use crate::namespace::fence::drain::tests::{raw, PROMPT}; + use crate::namespace::fence::hooks::{HookAction, HookPoint}; + use crate::namespace::fence::record::ServerIdentity; + use crate::namespace::fence::state::OperationClass; + use crate::namespace::meta_store::{metastore_connection_maker, FenceCommit, FenceCommitKind}; + use crate::namespace::store::fence_tests::open_store; + use crate::namespace::store::NamespaceStore; + use crate::namespace::RestoreOption; + use crate::query_result_builder::test::TestBuilder; + use crate::rpc::replication::replication_log::ReplicationLogService; + + pub(crate) const OP: Uuid = Uuid::from_u128(0xa); + const OTHER_OP: Uuid = Uuid::from_u128(0xb); + + pub(crate) fn server() -> ServerIdentity { + ServerIdentity { + build: "test".into(), + instance_id: Uuid::from_u128(0x99), + } + } + + pub(crate) fn create_request(ns: &'static str, command_id: u128) -> CreateTargetRequest { + CreateTargetRequest { + namespace: ns.into(), + operation_id: OP, + command_id: Uuid::from_u128(command_id), + config: TargetConfig { + max_db_size: Some(4096 * 1000), + ..Default::default() + }, + } + } + + /// Run `CreateTargetQuarantined` through the store on a task of its own. + pub(crate) fn create( + store: &NamespaceStore, + req: CreateTargetRequest, + ) -> tokio::task::JoinHandle> { + let store = store.clone(); + tokio::spawn(async move { store.create_target_quarantined(req, server()).await }) + } + + fn fence_error(e: &Error) -> &FenceError { + match e { + Error::NamespaceFence(f) => f, + other => panic!("expected a fence error, got {other:?}"), + } + } + + /// Denied by the fence, or not there at all: never served. + fn assert_not_served(what: &str, r: &crate::Result) { + match r { + Err(Error::NamespaceDoesntExist(_)) => (), + Err(Error::NamespaceFence(_)) => (), + other => panic!("{what}: expected a denial, got {other:?}"), + } + } + + fn assert_quarantined(fence: &FenceController) { + for class in [ + OperationClass::NormalRead, + OperationClass::NormalWrite, + OperationClass::Stream, + OperationClass::Lifecycle, + OperationClass::Vacuum, + ] { + let e = fence.permits(class).unwrap_err(); + assert_eq!( + e.outcome(), + FenceOutcome::MigrationTargetQuarantined, + "{class:?}" + ); + } + } + + async fn controller(store: &NamespaceStore, ns: &'static str) -> Arc { + store + .fence_gate(&ns.into()) + .await + .unwrap() + .expect("the target has a controller") + } + + /// The loaded target's controller, and a connection to it. + async fn loaded( + store: &NamespaceStore, + ns: &'static str, + ) -> (Arc, Arc) { + let (fence, maker) = store + .with(ns.into(), |ns| { + (ns.fence().clone(), ns.db.connection_maker()) + }) + .await + .unwrap(); + (fence, Arc::new(maker.create().await.unwrap())) + } + + async fn replication_hello(store: &NamespaceStore, ns: &'static str) -> tonic::Status { + let service = ReplicationLogService::new(store.clone(), None, None, false, false, true); + let mut req = tonic::Request::new(HelloRequest { + handshake_version: Some(1), + }); + req.metadata_mut().insert_bin( + NAMESPACE_METADATA_KEY, + BinaryMetadataValue::from_bytes(ns.as_bytes()), + ); + service.hello(req).await.unwrap_err() + } + + /// Every way of reaching `ns` other than the fence commands: none of them is served. + async fn attempt_everything(store: &NamespaceStore, ns: &'static str) { + let r = store + .with(ns.into(), |ns| ns.db.connection_maker()) + .await + .map(|_| ()); + assert_not_served("SQL connection", &r); + let r = store.stats(ns.into()).await.map(|_| ()); + assert_not_served("stats", &r); + // The dump route and the replication service reach the namespace the same way. + let hello = replication_hello(store, ns).await; + assert_ne!(hello.code(), tonic::Code::Ok); + assert_ne!(hello.code(), tonic::Code::Unavailable, "{hello:?}"); + let r = store + .create(ns.into(), RestoreOption::Latest, DatabaseConfig::default()) + .await; + assert_not_served("create", &r); + let r = store.destroy(ns.into(), false).await; + assert!(r.is_err(), "delete: {r:?}"); + let r = store + .fork("src".into(), ns.into(), DatabaseConfig::default(), None) + .await; + assert!(r.is_err(), "fork: {r:?}"); + } + + async fn store_with_source(dir: &TempDir) -> NamespaceStore { + let store = open_store(dir.path()).await; + store + .create("src".into(), RestoreOption::Latest, Default::default()) + .await + .unwrap(); + store + } + + /// Parked after its rows are committed and its quarantine gate published, and before its + /// config is published and the namespace loaded, the target is never observable: SQL, + /// dump/replication, create, delete and fork of the name are all denied or find nothing. + /// Afterwards it is loaded behind the quarantine gate. + #[tokio::test(flavor = "multi_thread")] + async fn create_race_never_observable() { + let dir = tempdir().unwrap(); + let store = store_with_source(&dir).await; + let fence = store.fence_controller(&"tgt".into()); + let paused = fence.hooks().pause_at(HookPoint::AfterTargetRowsCommitted); + let creating = create(&store, create_request("tgt", 1)); + paused.reached().await; + + assert_eq!(fence.gate().state(), FenceState::TargetQuarantined); + assert!(!store.exists(&"tgt".into()).await); + attempt_everything(&store, "tgt").await; + + paused.resume(); + let commit = creating.await.unwrap().unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + assert!(store.exists(&"tgt".into()).await); + + // Loaded behind the gate it was created with. + let (loaded_fence, conn) = loaded(&store, "tgt").await; + assert!(Arc::ptr_eq(&fence, &loaded_fence)); + assert_quarantined(&fence); + let ctx = RequestContext::new( + Authenticated::FullAccess, + "tgt".into(), + store.meta_store().clone(), + ); + let e = conn + .execute_program( + Program::seq(&["select 1"]), + ctx, + TestBuilder::default(), + None, + ) + .await + .map(|_| ()) + .unwrap_err(); + assert_eq!( + fence_error(&e).outcome(), + FenceOutcome::MigrationTargetQuarantined + ); + crate::namespace::fence::drain::tests::assert_fenced( + raw(&conn, "create table t (x)").await, + ); + // The in-memory config is the logical one; the stored row carries the legacy mirror. + let config = store.config_store("tgt".into()).await.unwrap().get(); + assert_eq!(config.max_db_pages, 1000); + assert!(!config.block_reads && !config.block_writes); + // Everything but the commands is still refused once it is loaded. + attempt_everything_loaded(&store, "tgt").await; + } + + /// Once loaded, the lifecycle paths are refused by the fence. + async fn attempt_everything_loaded(store: &NamespaceStore, ns: &'static str) { + let r = store + .create(ns.into(), RestoreOption::Latest, DatabaseConfig::default()) + .await; + assert!(r.is_err(), "create: {r:?}"); + let r = store.destroy(ns.into(), false).await; + assert_eq!( + fence_error(&r.unwrap_err()).outcome(), + FenceOutcome::MigrationTargetQuarantined + ); + let r = store + .fork("src".into(), ns.into(), DatabaseConfig::default(), None) + .await; + assert!(r.is_err(), "fork: {r:?}"); + let hello = replication_hello(store, ns).await; + assert_eq!( + FenceError::outcome_from_grpc_status(&hello), + Some(FenceOutcome::MigrationTargetQuarantined), + "{hello:?}" + ); + assert!(store.exists(&ns.into()).await); + } + + /// Before its commit, while the target-creation gate is in place, the name is refused + /// before any setup work. + #[tokio::test(flavor = "multi_thread")] + async fn creating_gate_refuses_before_commit() { + let dir = tempdir().unwrap(); + let store = store_with_source(&dir).await; + let fence = store.fence_controller(&"tgt".into()); + let paused = fence.hooks().pause_at(HookPoint::BeforeMetastoreCommit); + let creating = create(&store, create_request("tgt", 1)); + paused.reached().await; + + assert!(fence.gate().is_creating_target()); + assert_quarantined(&fence); + attempt_everything(&store, "tgt").await; + // No database was set up under the name. + assert!(!dir.path().join("dbs").join("tgt").join("data").exists()); + + paused.resume(); + creating.await.unwrap().unwrap(); + assert!(!fence.gate().is_creating_target()); + assert_eq!(fence.gate().state(), FenceState::TargetQuarantined); + } + + /// A creation interrupted between its marker and its commit leaves the name + /// `UNKNOWN_UNAVAILABLE` after a restart; only the same command completes it, and the + /// completed target is loaded quarantined. + #[tokio::test(flavor = "multi_thread")] + async fn create_replay_completes_interrupted_creation() { + let dir = tempdir().unwrap(); + { + let store = store_with_source(&dir).await; + create(&store, create_request("tgt", 1)) + .await + .unwrap() + .unwrap(); + store.shutdown().await.unwrap(); + } + // What a crash after the marker and before the commit leaves: the marker alone. + { + let (maker, _) = metastore_connection_maker(None, dir.path()).await.unwrap(); + let conn = maker().unwrap(); + for sql in [ + "DELETE FROM namespace_fence_receipts WHERE namespace = 'tgt'", + "DELETE FROM namespace_fences WHERE namespace = 'tgt'", + "DELETE FROM namespace_configs WHERE namespace = 'tgt'", + ] { + conn.execute(sql, ()).unwrap(); + } + } + std::fs::remove_file(dir.path().join("dbs").join("tgt").join("data")).ok(); + + let store = open_store(dir.path()).await; + let fence = store.fence_controller(&"tgt".into()); + assert!(fence.gate().is_unavailable()); + let r = store.with("tgt".into(), |_| ()).await; + assert_eq!( + fence_error(&r.unwrap_err()).detail(), + Some(FenceDetail::IncompleteTargetCreation) + ); + // Another command does not complete it. + let r = create(&store, create_request("tgt", 2)).await.unwrap(); + assert_eq!( + fence_error(&r.unwrap_err()).outcome(), + FenceOutcome::FenceStateUnavailable + ); + + let commit = create(&store, create_request("tgt", 1)) + .await + .unwrap() + .unwrap(); + assert_eq!(commit.kind, FenceCommitKind::Committed); + assert_eq!(fence.gate().state(), FenceState::TargetQuarantined); + let (loaded_fence, _) = loaded(&store, "tgt").await; + assert!(Arc::ptr_eq(&fence, &loaded_fence)); + assert_quarantined(&fence); + let config = store.config_store("tgt".into()).await.unwrap().get(); + assert!(!config.block_reads && !config.block_writes); + } + + /// The caller going away does not stop a creation: it is published and loaded, and a replay + /// returns the stored result. + #[tokio::test(flavor = "multi_thread")] + async fn create_completes_when_the_caller_goes_away() { + let dir = tempdir().unwrap(); + let store = store_with_source(&dir).await; + let fence = store.fence_controller(&"tgt".into()); + let paused = fence.hooks().pause_at(HookPoint::AfterTargetRowsCommitted); + let creating = create(&store, create_request("tgt", 1)); + paused.reached().await; + creating.abort(); + let _ = creating.await; + paused.resume(); + + // The creation carries on without its caller; the replay waits for it on the + // transition lock and then answers from the receipt. + let replay = tokio::time::timeout(PROMPT, create(&store, create_request("tgt", 1))) + .await + .unwrap() + .unwrap() + .unwrap(); + assert_eq!(replay.kind, FenceCommitKind::Replayed); + assert_eq!(replay.receipt.outcome, FenceOutcome::Applied); + assert!(store.exists(&"tgt".into()).await); + let (loaded_fence, _) = loaded(&store, "tgt").await; + assert!(Arc::ptr_eq(&fence, &loaded_fence)); + } + + /// A commit whose acknowledgement is lost keeps the name closed; the replay reconciles it + /// from the durable rows and publishes and loads the target. + #[tokio::test(flavor = "multi_thread")] + async fn indeterminate_create_is_completed_by_replay() { + let dir = tempdir().unwrap(); + let store = store_with_source(&dir).await; + let fence = store.fence_controller(&"tgt".into()); + fence + .hooks() + .arm(HookPoint::AfterMetastoreCommit, HookAction::Indeterminate); + let r = create(&store, create_request("tgt", 1)).await.unwrap(); + assert_eq!( + fence_error(&r.unwrap_err()).outcome(), + FenceOutcome::FenceCommitIndeterminate + ); + // Still refused before any setup, and not published. + assert!(fence.gate().is_creating_target()); + assert!(!store.exists(&"tgt".into()).await); + assert_not_served("SQL", &store.with("tgt".into(), |_| ()).await); + + let replay = create(&store, create_request("tgt", 1)) + .await + .unwrap() + .unwrap(); + assert_eq!(replay.kind, FenceCommitKind::Replayed); + assert!(!fence.gate().is_creating_target()); + assert!(fence.gate().indeterminate.is_none()); + assert_eq!(fence.gate().state(), FenceState::TargetQuarantined); + assert!(store.exists(&"tgt".into()).await); + loaded(&store, "tgt").await; + assert_quarantined(&fence); + } + + /// A name the server already has is refused, and its traffic is not disturbed; a name that + /// is refused keeps no creation gate. + #[tokio::test(flavor = "multi_thread")] + async fn create_rejects_existing_name() { + let dir = tempdir().unwrap(); + let store = store_with_source(&dir).await; + let src_conn = { + let maker = store + .with("src".into(), |ns| ns.db.connection_maker()) + .await + .unwrap(); + Arc::new(maker.create().await.unwrap()) + }; + raw(&src_conn, "create table t (x)").await.unwrap(); + let generation = store.fence_controller(&"src".into()).write_generation(); + + let r = create(&store, create_request("src", 1)).await.unwrap(); + let e = r.unwrap_err(); + assert_eq!( + fence_error(&e).outcome(), + FenceOutcome::FencePreconditionFailed + ); + assert_eq!(fence_error(&e).detail(), Some(FenceDetail::NamespaceExists)); + // The source's gate never moved, and it still takes writes. + let src_fence = store.fence_controller(&"src".into()); + assert_eq!(src_fence.write_generation(), generation); + assert!(!src_fence.gate().is_creating_target()); + raw(&src_conn, "insert into t values (1)").await.unwrap(); + + // A name that exists in the metastore but is not loaded is refused the same way. + store + .create("cold".into(), RestoreOption::Latest, Default::default()) + .await + .unwrap(); + let r = create(&store, create_request("cold", 2)).await.unwrap(); + assert_eq!( + fence_error(&r.unwrap_err()).detail(), + Some(FenceDetail::NamespaceExists) + ); + + // A second target of the same name by another operation is refused, and the first is + // untouched. + create(&store, create_request("tgt", 3)) + .await + .unwrap() + .unwrap(); + let mut other = create_request("tgt", 4); + other.operation_id = OTHER_OP; + let r = create(&store, other).await.unwrap(); + assert!(r.is_err()); + let fence = store.fence_controller(&"tgt".into()); + assert_eq!(fence.gate().operation_id(), Some(OP)); + assert!(!fence.gate().is_creating_target()); + } + + /// `AbortQuarantinedTarget` finishes the operation and keeps every normal class denied; + /// the target is not deletable by the generic lifecycle either. + #[tokio::test(flavor = "multi_thread")] + async fn abort_keeps_traffic_denied() { + let dir = tempdir().unwrap(); + let store = store_with_source(&dir).await; + create(&store, create_request("tgt", 1)) + .await + .unwrap() + .unwrap(); + let abort = FenceRequest { + namespace: "tgt".into(), + operation_id: OP, + command_id: Uuid::from_u128(2), + expected_state: FenceState::TargetQuarantined, + expected_revision: 1, + command: FenceCommand::AbortQuarantinedTarget, + }; + let commit = store + .execute_fence_command(abort.clone(), server()) + .await + .unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + let fence = controller(&store, "tgt").await; + assert_eq!(fence.gate().state(), FenceState::TargetAborted); + assert_quarantined(&fence); + let (_, conn) = loaded(&store, "tgt").await; + crate::namespace::fence::drain::tests::assert_fenced( + raw(&conn, "create table t (x)").await, + ); + let r = store.destroy("tgt".into(), false).await; + assert_eq!( + fence_error(&r.unwrap_err()).outcome(), + FenceOutcome::MigrationTargetQuarantined + ); + // A replay answers from the receipt; a new creation of the name is refused. + let replay = store.execute_fence_command(abort, server()).await.unwrap(); + assert_eq!(replay.kind, FenceCommitKind::Replayed); + let r = create(&store, create_request("tgt", 3)).await.unwrap(); + assert!(r.is_err()); + + // After a restart the aborted target is still denied. + store.shutdown().await.unwrap(); + let store = open_store(dir.path()).await; + let fence = controller(&store, "tgt").await; + assert_eq!(fence.gate().state(), FenceState::TargetAborted); + assert_quarantined(&fence); + let (_, conn) = loaded(&store, "tgt").await; + crate::namespace::fence::drain::tests::assert_fenced( + raw(&conn, "create table t (x)").await, + ); + } +} diff --git a/libsql-server/src/namespace/meta_store.rs b/libsql-server/src/namespace/meta_store.rs index ebd1ba2a64..17d23db89c 100644 --- a/libsql-server/src/namespace/meta_store.rs +++ b/libsql-server/src/namespace/meta_store.rs @@ -36,7 +36,7 @@ use super::fence::outcome::{FenceDetail, FenceError, FenceOutcome}; use super::fence::record::{ CommandReceipt, NamespaceFenceRecord, ServerIdentity, ValidationSnapshot, }; -use super::fence::state::OperationClass; +use super::fence::state::{OperationClass, Role}; use super::fence::store::{ self as fence_store, FenceStoreError, MarkerStatus, StoredFence, StoredReceipt, }; @@ -1379,6 +1379,56 @@ impl MetaStore { .map_err(fence_store_error) } + /// Make a migration target that the metastore holds visible in the in-memory config map, + /// which is what makes `exists()` and `lookup()` find it (section 10.1, step 5). The config + /// published is the stored row with the record's own `block_*` values in place of the + /// legacy mirror, as `restore_fences` does at startup. The caller has already installed + /// the target's gate. Returns whether the map changed; `false` also when the namespace is + /// not a target with a stored config. + pub async fn publish_target_config(&self, namespace: NamespaceName) -> Result { + let inner = self.inner.clone(); + tokio::task::spawn_blocking(move || -> std::result::Result { + // The connection lock first, as everywhere else that takes both. + let mut conn = inner.conn.blocking_lock(); + let tx = conn.transaction()?; + let (stored, _) = fence_store::read_fence(&tx, &inner.dbs_path, &namespace)?; + let record = match stored { + StoredFence::Record(r) if r.role == Role::Target => r, + _ => return Ok(false), + }; + let Some(row) = fence_store::read_config_row(&tx, &namespace)? else { + return Ok(false); + }; + drop(tx); + let config = Arc::new(fence_store::with_legacy_blocks(&row, &record.legacy_blocks)); + let mut configs = inner.configs.blocking_lock(); + match configs.get_mut(&namespace) { + Some(sender) + if metadata::DatabaseConfig::from(&*sender.borrow().config) + == metadata::DatabaseConfig::from(&*config) => + { + Ok(false) + } + // An entry that was put in the map by a create or fork of the same name that + // the fence then refused: the durable config replaces it. + Some(sender) => { + sender.send_modify(|c| { + c.version = c.version.wrapping_add(1); + c.config = config; + }); + Ok(true) + } + None => { + let (tx, _) = watch::channel(InnerConfig { version: 0, config }); + configs.insert(namespace, tx); + Ok(true) + } + } + }) + .await? + .map_err(fence_store_error) + } + /// Read a namespace's fence and all of its receipts (`InspectFence`). Never writes. pub async fn inspect_fence(&self, namespace: NamespaceName) -> Result { let inner = self.inner.clone(); diff --git a/libsql-server/src/namespace/store.rs b/libsql-server/src/namespace/store.rs index ca8e7031be..177f2cc471 100644 --- a/libsql-server/src/namespace/store.rs +++ b/libsql-server/src/namespace/store.rs @@ -22,9 +22,14 @@ use crate::stats::Stats; use super::broadcasters::{BroadcasterHandle, BroadcasterRegistry}; use super::configurator::{DynConfigurator, NamespaceConfigurators}; use super::fence::command::{FenceCommand, FenceRequest}; -use super::fence::controller::FenceController; +use super::fence::controller::{FenceController, Transition}; +use super::fence::hooks::HookPoint; +use super::fence::outcome::FenceOutcome; use super::fence::record::ServerIdentity; use super::fence::registry::FenceRegistry; +use super::fence::state::Role; +use super::fence::store::StoredFence; +use super::fence::target::{self, CreateTargetRequest}; use super::meta_store::{FenceCommit, FenceContext, MetaStore, MetaStoreHandle}; use super::schema_lock::SchemaLocksRegistry; use super::{Namespace, ResetCb, ResetOp, ResolveNamespacePathFn, RestoreOption}; @@ -237,6 +242,10 @@ impl NamespaceStore { return Err(Error::NamespaceStoreShutdown); } + // The destination is refused before anything is stored for it when it is being created + // as a migration target or its fence state is unknown. + self.inner.fences.check_available(&to)?; + // check that the source namespace exists if !self.inner.metadata.exists(&from).await { return Err(crate::error::Error::NamespaceDoesntExist(from.to_string())); @@ -450,6 +459,9 @@ impl NamespaceStore { restore_option: RestoreOption, db_config: DatabaseConfig, ) -> crate::Result<()> { + // A name that is being created as a migration target, or whose fence state is unknown, + // is refused before anything is stored for it. + self.inner.fences.check_available(&namespace)?; if let Some(shared_schema_name) = &db_config.shared_schema_name { // we hold a lock for the duration of the namespace creation let _lock = self @@ -550,6 +562,9 @@ impl NamespaceStore { server: ServerIdentity, ) -> crate::Result { let controller = match request.command { + FenceCommand::CreateTargetQuarantined { .. } => { + return self.run_create_target(request, server).await + } FenceCommand::AcquireSourceWriteFence { .. } => { self.with(request.namespace.clone(), |ns| ns.fence().clone()) .await? @@ -565,6 +580,148 @@ impl NamespaceStore { .await } + /// `CreateTargetQuarantined`, atomic with namespace creation (`docs/NAMESPACE_FENCE.md` + /// sections 10.1 and 11): the namespace is quarantined from the first instant it can be + /// observed, and is loaded behind the quarantine gate before this returns `APPLIED`. + /// + /// Fence refusals are [`Error::NamespaceFence`] with their stable outcome code. The work + /// runs on its own task under the namespace's transition lock, so a caller that goes away + /// does not interrupt it; replaying the same request returns the stored result, completes + /// the publication and load of a target whose commit was not acknowledged, and completes a + /// creation interrupted between its marker and its commit. + pub async fn create_target_quarantined( + &self, + request: CreateTargetRequest, + server: ServerIdentity, + ) -> crate::Result { + self.run_create_target(request.into(), server).await + } + + async fn run_create_target( + &self, + request: FenceRequest, + server: ServerIdentity, + ) -> crate::Result { + if self.inner.has_shutdown.load(Ordering::Relaxed) { + return Err(Error::NamespaceStoreShutdown); + } + let controller = self.inner.fences.controller(&request.namespace); + let this = self.clone(); + tokio::spawn(async move { + let mut transition = controller.begin_transition().await; + let ctx = FenceContext::now(server, None); + this.create_target_under(&mut transition, request, ctx) + .await + }) + .await? + } + + /// Section 10.1, steps 1 to 5, under `transition`. + async fn create_target_under( + &self, + transition: &mut Transition, + request: FenceRequest, + ctx: FenceContext, + ) -> crate::Result { + let controller = transition.controller().clone(); + let namespace = request.namespace.clone(); + let key = (request.operation_id, request.command_id); + + // A name with no fence state gets the target-creation gate before the metastore + // transaction writes the marker, so nothing can set the name up, serve it or store a + // config for it from before the commit to the load. A name that already has fence + // state (a replay, a creation interrupted after its marker, or a refusal) already has + // the gate its state implies. A name the server already knows is refused without + // touching its gate; it is checked again once the gate is in place, which closes the + // race with a create or fork that has not stored anything yet. + let fresh = { + let gate = controller.gate(); + matches!(gate.fence, StoredFence::None { .. }) && gate.indeterminate.is_none() + }; + if fresh { + if self.name_in_use(&namespace).await { + return Err(target::name_in_use(&namespace).into()); + } + transition.install_creating_target(key); + let _ = controller.hook(HookPoint::AfterInstallingGate).await; + if self.name_in_use(&namespace).await { + transition.remove_creating_target(); + return Err(target::name_in_use(&namespace).into()); + } + } + + // Steps 2 and 3: the marker, then the rows in one transaction. The commit publishes the + // quarantined record in place of the creation gate. A command proven not to have + // committed removes the creation gate; one whose commit is unknown keeps it, with the + // indeterminate flag, until the same command is replayed. + let commit = match transition.apply(&self.inner.metadata, request, ctx).await { + Ok(commit) => commit, + Err(e) => { + if controller.gate().indeterminate != Some(key) { + transition.remove_creating_target(); + } + return Err(e); + } + }; + let _ = controller.hook(HookPoint::AfterTargetRowsCommitted).await; + + // Steps 4 and 5: the gate is the target's; now make the config visible and load the + // namespace, whose first connection is created behind that gate. A replay does the same, + // which completes a creation whose commit was not acknowledged. + let is_target = commit + .record + .as_ref() + .is_some_and(|r| r.role == Role::Target && r.namespace == namespace); + if is_target { + self.inner + .metadata + .publish_target_config(namespace.clone()) + .await?; + self.clear_empty_entry(&namespace).await; + let loaded = self + .with(namespace.clone(), |ns| ns.fence().clone()) + .await?; + debug_assert!(Arc::ptr_eq(&loaded, &controller)); + } + if commit.receipt.outcome == FenceOutcome::Applied && commit.created_config.is_some() { + tracing::info!( + namespace = %namespace, + operation_id = %key.0, + command_id = %key.1, + "created namespace as a quarantined migration target" + ); + } + Ok(commit) + } + + /// Whether the server already knows `namespace`: its config is in memory, or the namespace + /// cache holds it (a loaded namespace, or a fork in flight, which holds its entry locked). + async fn name_in_use(&self, namespace: &NamespaceName) -> bool { + if self.inner.metadata.exists(namespace).await { + return true; + } + match self.inner.store.get(namespace).await { + Some(entry) => entry.read().await.is_some(), + None => false, + } + } + + /// Drop an empty namespace-cache entry for `namespace`, which a refused fork or a checkpoint + /// of a name that did not exist leaves behind, so that loading the namespace creates it. + async fn clear_empty_entry(&self, namespace: &NamespaceName) { + if let Some(entry) = self.inner.store.get(namespace).await { + if entry.read().await.is_none() { + self.inner.store.invalidate(namespace).await; + } + } + } + + /// The fence controller of `namespace`, creating an `UNFENCED` one if it has none. + #[cfg(test)] + pub(crate) fn fence_controller(&self, namespace: &NamespaceName) -> Arc { + self.inner.fences.controller(namespace) + } + /// The fence controller that admits reads of `namespace` without loading it: `None` when /// the namespace does not exist (and has no fence state). A namespace whose fence state is /// unavailable is refused. From 7f5ba651b16dc27331376425e5db618e9bef66d1 Mon Sep 17 00:00:00 2001 From: River Date: Tue, 29 Sep 2026 18:55:06 +0000 Subject: [PATCH 2/6] libsql-server: import capabilities and target seal drain Add the operation-owned import path into a quarantined migration target and the seal that ends it (docs/NAMESPACE_FENCE.md sections 7, 10.2 and 11). - MigrationCapability (server-issued, fields private) and CapabilityPurpose. The fence controller keeps the live capability set and a count of running import calls; every published transition drops capabilities whose state, owner or revision no longer match. - FenceConnState::with_capability: the WAL admits a capability connection's write transaction only while its capability matches the fence and is live; a validation connection never writes. - NamespaceStore::open_import_session and ImportSession::{with_raw, load_dump}: the only way to write into TARGET_QUARANTINED. The dump loader is split into load_dump_sql so that the loader used for namespaces created from a dump also runs under an import capability. - SealTargetImport closes import admission in memory, persists TARGET_IMPORT_DRAINING (invalidating every import capability), waits on release notifications for running import calls and for any import transaction holding the write slot (force_rollback rolls it back at the deadline), then persists TARGET_VALIDATING. A deadline leaves TARGET_IMPORT_DRAINING durable and closed until the owner's seal resumes it. Co-authored-by: Tomasz Szymczyszyn --- libsql-server/src/connection/legacy.rs | 37 +- libsql-server/src/connection/mod.rs | 5 + .../src/namespace/configurator/helpers.rs | 81 +- .../src/namespace/configurator/mod.rs | 1 + .../src/namespace/fence/capability.rs | 212 +++++ .../src/namespace/fence/controller.rs | 267 +++++- libsql-server/src/namespace/fence/drain.rs | 5 +- libsql-server/src/namespace/fence/import.rs | 838 ++++++++++++++++++ libsql-server/src/namespace/fence/mod.rs | 6 +- libsql-server/src/namespace/store.rs | 69 +- 10 files changed, 1455 insertions(+), 66 deletions(-) create mode 100644 libsql-server/src/namespace/fence/capability.rs create mode 100644 libsql-server/src/namespace/fence/import.rs diff --git a/libsql-server/src/connection/legacy.rs b/libsql-server/src/connection/legacy.rs index efc176cae8..3d441aaded 100644 --- a/libsql-server/src/connection/legacy.rs +++ b/libsql-server/src/connection/legacy.rs @@ -14,6 +14,7 @@ use tokio::time::Duration; use crate::error::Error; use crate::metrics::DESCRIBE_COUNT; use crate::namespace::broadcasters::BroadcasterHandle; +use crate::namespace::fence::capability::MigrationCapability; use crate::namespace::fence::controller::{FenceConnState, FenceController}; use crate::namespace::fence::state::OperationClass; use crate::namespace::meta_store::MetaStoreHandle; @@ -143,6 +144,33 @@ where #[tracing::instrument(skip(self))] pub(super) async fn make_connection(&self) -> Result> { + self.make_connection_with(FenceConnState::new( + self.fence.clone(), + OperationClass::NormalWrite, + )) + .await + } + + /// Open a connection that works under `capability` (an import or validation session, + /// `docs/NAMESPACE_FENCE.md` section 11). It shares the maker's write slot, WAL and + /// replication log with every other connection, and is admitted as the capability's class + /// only while the capability is valid. It is not counted by the connection throttle: it is + /// operation-owned work, and the fence, not the throttle, bounds how much of it runs. + pub(crate) async fn make_capability_connection( + &self, + capability: MigrationCapability, + ) -> Result> { + self.make_connection_with(FenceConnState::with_capability( + self.fence.clone(), + capability, + )) + .await + } + + async fn make_connection_with( + &self, + fence: Arc, + ) -> Result> { LegacyConnection::new( self.db_path.clone(), self.extensions.clone(), @@ -161,7 +189,7 @@ where self.resolve_attach_path.clone(), self.connection_manager.clone(), self.make_wal_manager.clone(), - FenceConnState::new(self.fence.clone(), OperationClass::NormalWrite), + fence, ) .await } @@ -213,6 +241,13 @@ impl LegacyConnection { } } +impl LegacyConnection { + /// The fence state shared by this connection's WAL wrapper and `CoreConnection`. + pub(crate) fn fence_state(&self) -> &Arc { + &self.fence + } +} + impl Clone for LegacyConnection { fn clone(&self) -> Self { Self { diff --git a/libsql-server/src/connection/mod.rs b/libsql-server/src/connection/mod.rs index 167a1a595c..ef8fe0002e 100644 --- a/libsql-server/src/connection/mod.rs +++ b/libsql-server/src/connection/mod.rs @@ -286,6 +286,11 @@ pub struct MakeThrottledConnection { } impl MakeThrottledConnection { + /// The connection maker this one throttles. + pub(crate) fn inner(&self) -> &F { + &self.connection_maker + } + fn new( semaphore: Arc, connection_maker: F, diff --git a/libsql-server/src/namespace/configurator/helpers.rs b/libsql-server/src/namespace/configurator/helpers.rs index a10ef89d6b..6fe8994663 100644 --- a/libsql-server/src/namespace/configurator/helpers.rs +++ b/libsql-server/src/namespace/configurator/helpers.rs @@ -301,6 +301,16 @@ async fn run_periodic_compactions(logger: Arc) -> anyhow::Res } async fn load_dump(dump: S, conn: PrimaryConnection) -> crate::Result<(), LoadDumpError> +where + S: Stream> + Unpin, +{ + let dump_content = read_dump(dump).await?; + tokio::task::spawn_blocking(move || conn.with_raw(|conn| load_dump_sql(&dump_content, conn))) + .await? +} + +/// Read a whole dump into memory. +pub(crate) async fn read_dump(dump: S) -> crate::Result where S: Stream> + Unpin, { @@ -310,13 +320,36 @@ where .read_to_string(&mut dump_content) .await .map_err(|e| LoadDumpError::Internal(format!("Failed to read dump content: {}", e)))?; + Ok(dump_content) +} +/// Parse `dump_content` and run its statements, one at a time, on `conn`. The dump must run +/// inside one transaction that it commits itself; `ATTACH` is refused. This is the loader both +/// for a namespace created from a dump and for an import session into a quarantined migration +/// target, which runs it under its capability (`docs/NAMESPACE_FENCE.md` section 11). +pub(crate) fn load_dump_sql( + dump_content: &str, + conn: &mut rusqlite::Connection, +) -> crate::Result<(), LoadDumpError> { if dump_content.to_lowercase().contains("attach") { return Err(LoadDumpError::InvalidSqlInput( "attach statements are not allowed in dumps".to_string(), )); } + conn.authorizer(Some(|auth: AuthContext<'_>| match auth.action { + AuthAction::Attach { filename: _ } => Authorization::Deny, + _ => Authorization::Allow, + })); + let result = run_dump_statements(dump_content, conn); + conn.authorizer(None::) -> Authorization>); + result +} + +fn run_dump_statements( + dump_content: &str, + conn: &mut rusqlite::Connection, +) -> crate::Result<(), LoadDumpError> { let mut parser = Box::new(Parser::new(dump_content.as_bytes())); let mut skipped_wasm_table = false; let mut n_stmt = 0; @@ -336,37 +369,20 @@ where } } - if n_stmt > 2 && conn.is_autocommit().await.unwrap() { + if n_stmt > 2 && conn.is_autocommit() { return Err(LoadDumpError::NoTxn); } let stmt_sql = cmd.to_string(); - tokio::task::spawn_blocking({ - let conn = conn.clone(); - move || -> crate::Result<(), LoadDumpError> { - conn.with_raw(|conn| { - conn.authorizer(Some(|auth: AuthContext<'_>| match auth.action { - AuthAction::Attach { filename: _ } => Authorization::Deny, - _ => Authorization::Allow, - })); - conn.execute(&stmt_sql, ()) - }) - .map_err(|e| match e { - rusqlite::Error::SqlInputError { - msg, sql, offset, .. - } => LoadDumpError::InvalidSqlInput(format!( - "msg: {}, sql: {}, offset: {}", - msg, sql, offset - )), - e => LoadDumpError::Internal(format!( - "statement: {}, error: {}", - n_stmt, e - )), - })?; - Ok(()) - } - }) - .await??; + conn.execute(&stmt_sql, ()).map_err(|e| match e { + rusqlite::Error::SqlInputError { + msg, sql, offset, .. + } => LoadDumpError::InvalidSqlInput(format!( + "msg: {}, sql: {}, offset: {}", + msg, sql, offset + )), + e => LoadDumpError::Internal(format!("statement: {}, error: {}", n_stmt, e)), + })?; } Ok(None) => break, Err(e) => { @@ -389,15 +405,8 @@ where } } - if !conn.is_autocommit().await.unwrap() { - tokio::task::spawn_blocking({ - let conn = conn.clone(); - move || -> crate::Result<(), LoadDumpError> { - conn.with_raw(|conn| conn.execute("rollback", ()))?; - Ok(()) - } - }) - .await??; + if !conn.is_autocommit() { + conn.execute("rollback", ())?; return Err(LoadDumpError::NoCommit); } diff --git a/libsql-server/src/namespace/configurator/mod.rs b/libsql-server/src/namespace/configurator/mod.rs index 029ab0b3ce..f8d11addd2 100644 --- a/libsql-server/src/namespace/configurator/mod.rs +++ b/libsql-server/src/namespace/configurator/mod.rs @@ -26,6 +26,7 @@ mod primary; mod replica; mod schema; +pub(crate) use helpers::{load_dump_sql, read_dump}; pub use primary::PrimaryConfigurator; pub use replica::ReplicaConfigurator; pub use schema::SchemaConfigurator; diff --git a/libsql-server/src/namespace/fence/capability.rs b/libsql-server/src/namespace/fence/capability.rs new file mode 100644 index 0000000000..46e492f6f0 --- /dev/null +++ b/libsql-server/src/namespace/fence/capability.rs @@ -0,0 +1,212 @@ +//! Migration capabilities (`docs/NAMESPACE_FENCE.md` sections 7.3 and 11). +//! +//! A [`MigrationCapability`] is the only thing that lets operation-owned work through a +//! target's quarantine. The server creates it ([`FenceController::issue_capability`]); its +//! fields are private and it cannot be built from outside this module, so holding one is proof +//! that the fence issued it. It names the namespace, the owning operation, what it is for and +//! the fence revision it was issued at, and it is valid only while the fence is still in the +//! state its purpose needs, at that revision, owned by that operation, and while the controller +//! still lists it as live. Every transition of the target moves the revision, so a capability +//! never outlives the state it was issued in. +//! +//! [`FenceController::issue_capability`]: super::controller::FenceController::issue_capability + +use std::collections::HashMap; +use std::sync::Arc; + +use uuid::Uuid; + +use crate::namespace::NamespaceName; + +use super::controller::FenceController; +use super::outcome::{FenceError, FenceOutcome}; +use super::state::{FenceState, OperationClass}; +use super::store::StoredFence; + +/// What a capability admits. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] +pub enum CapabilityPurpose { + /// Writes of an import session into a `TARGET_QUARANTINED` target. + Import, + /// Read-only validation of a `TARGET_VALIDATING` or `TARGET_WRITE_FENCED` target. + Validate, +} + +impl CapabilityPurpose { + /// The operation class work under a capability of this purpose is admitted as. + pub const fn class(self) -> OperationClass { + match self { + CapabilityPurpose::Import => OperationClass::CapabilityImport, + CapabilityPurpose::Validate => OperationClass::CapabilityValidate, + } + } + + pub const fn as_str(self) -> &'static str { + match self { + CapabilityPurpose::Import => "import", + CapabilityPurpose::Validate => "validate", + } + } + + /// Whether a capability of this purpose may be issued, and stays valid, in `state`. + pub const fn admits(self, state: FenceState) -> bool { + match self { + CapabilityPurpose::Import => matches!(state, FenceState::TargetQuarantined), + CapabilityPurpose::Validate => matches!( + state, + FenceState::TargetValidating | FenceState::TargetWriteFenced + ), + } + } +} + +/// A server-issued grant for operation-owned work on one namespace, valid at one fence +/// revision. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct MigrationCapability { + id: Uuid, + namespace: NamespaceName, + operation_id: Uuid, + purpose: CapabilityPurpose, + fence_revision: u64, +} + +impl MigrationCapability { + /// Only the controller issues capabilities. + pub(super) fn issue( + namespace: NamespaceName, + operation_id: Uuid, + purpose: CapabilityPurpose, + fence_revision: u64, + ) -> Self { + Self { + id: Uuid::new_v4(), + namespace, + operation_id, + purpose, + fence_revision, + } + } + + /// A capability the controller never issued, for tests that prove such a thing is refused. + #[cfg(test)] + pub(crate) fn forged( + namespace: NamespaceName, + operation_id: Uuid, + purpose: CapabilityPurpose, + fence_revision: u64, + ) -> Self { + Self::issue(namespace, operation_id, purpose, fence_revision) + } + + pub fn id(&self) -> Uuid { + self.id + } + + pub fn namespace(&self) -> &NamespaceName { + &self.namespace + } + + pub fn operation_id(&self) -> Uuid { + self.operation_id + } + + pub fn purpose(&self) -> CapabilityPurpose { + self.purpose + } + + pub fn fence_revision(&self) -> u64 { + self.fence_revision + } + + pub fn class(&self) -> OperationClass { + self.purpose.class() + } + + /// Whether `fence` is still the fence this capability was issued against: the state its + /// purpose needs, owned by its operation, at its revision. + pub(crate) fn matches(&self, fence: &StoredFence) -> bool { + self.check(fence).is_ok() + } + + /// [`matches`](Self::matches), with the refusal a holder gets when it does not. + pub(crate) fn check(&self, fence: &StoredFence) -> Result<(), FenceError> { + let Some(record) = fence.record() else { + return Err(stale( + self, + format!("namespace `{}` has no fence record", self.namespace), + )); + }; + if record.operation_id != self.operation_id { + return Err(FenceError::new( + FenceOutcome::FenceOwnedByAnotherOperation, + format!( + "the fence of namespace `{}` is owned by operation {}, not by operation {} \ + that holds this {} capability", + self.namespace, + record.operation_id, + self.operation_id, + self.purpose.as_str() + ), + )); + } + if !self.purpose.admits(record.state) { + return Err(stale( + self, + format!( + "namespace `{}` is in {}, which admits no {} capability", + self.namespace, + record.state, + self.purpose.as_str() + ), + )); + } + if record.revision != self.fence_revision { + return Err(stale( + self, + format!( + "the fence of namespace `{}` is at revision {}, and this {} capability was \ + issued at revision {}", + self.namespace, + record.revision, + self.purpose.as_str(), + self.fence_revision + ), + )); + } + Ok(()) + } +} + +fn stale(cap: &MigrationCapability, why: String) -> FenceError { + FenceError::new( + FenceOutcome::OperationCapabilityRequired, + format!( + "{why}: the {} capability {} is no longer valid", + cap.purpose.as_str(), + cap.id + ), + ) +} + +/// The capabilities a controller has issued and not yet revoked, and the import calls running +/// under them. +#[derive(Debug, Default)] +pub(super) struct CapabilitySet { + pub(super) live: HashMap, + /// Import calls running now ([`ImportWriter`]s). + pub(super) import_writers: usize, +} + +/// One running import call under a capability. The seal waits for every one of them to be +/// dropped (section 10.2). +#[derive(Debug)] +pub struct ImportWriter { + pub(super) controller: Arc, +} + +impl Drop for ImportWriter { + fn drop(&mut self) { + self.controller.end_import_write(); + } +} diff --git a/libsql-server/src/namespace/fence/controller.rs b/libsql-server/src/namespace/fence/controller.rs index 04149a66b4..1eeeffc804 100644 --- a/libsql-server/src/namespace/fence/controller.rs +++ b/libsql-server/src/namespace/fence/controller.rs @@ -24,6 +24,7 @@ use crate::namespace::meta_store::{FenceCommit, FenceContext, MetaStore}; use crate::namespace::NamespaceName; use crate::replication::FrameNo; +use super::capability::{CapabilityPurpose, CapabilitySet, ImportWriter, MigrationCapability}; use super::command::FenceRequest; #[cfg(test)] use super::hooks::FenceTestHooks; @@ -312,6 +313,12 @@ pub struct FenceController { read_leases: Mutex, /// Notified whenever a read lease is released. read_released: Notify, + /// The migration capabilities issued and not revoked, and the import calls running under + /// them (sections 7.2 and 10.2). Lock order: this lock may be taken before borrowing the + /// gate, never while a gate borrow is held. + capabilities: Mutex, + /// Notified whenever an import call ends. + import_released: Notify, #[cfg(test)] hooks: FenceTestHooks, } @@ -337,6 +344,8 @@ impl FenceController { write_drains: Mutex::new(Vec::new()), read_leases: Mutex::new(ReadLeaseSet::default()), read_released: Notify::new(), + capabilities: Mutex::new(CapabilitySet::default()), + import_released: Notify::new(), #[cfg(test)] hooks: FenceTestHooks::default(), }) @@ -484,6 +493,144 @@ impl FenceController { self.read_released.notify_waiters(); } + /// Issue a migration capability for `purpose` to `operation_id` (section 11). The fence + /// must be in a state the purpose admits, owned by `operation_id`, at `expected_revision`, + /// with no command being installed or reconciled. The capability stays valid until the + /// fence moves on (every transition moves the revision) or it is revoked. + pub fn issue_capability( + &self, + purpose: CapabilityPurpose, + operation_id: Uuid, + expected_revision: u64, + ) -> Result { + let mut caps = self.capabilities.lock(); + let gate = self.gate.borrow(); + gate.permits(purpose.class())?; + let Some(record) = gate.fence.record() else { + return Err(FenceError::new( + FenceOutcome::OperationCapabilityRequired, + format!("namespace `{}` has no fence record", self.namespace), + )); + }; + if record.operation_id != operation_id { + return Err(FenceError::new( + FenceOutcome::FenceOwnedByAnotherOperation, + format!( + "the fence of namespace `{}` is owned by operation {}, not by operation \ + {operation_id}", + self.namespace, record.operation_id + ), + )); + } + if record.revision != expected_revision { + return Err(FenceError::new( + FenceOutcome::FenceRevisionMismatch, + format!( + "the fence of namespace `{}` is at revision {}, not at the expected revision \ + {expected_revision}", + self.namespace, record.revision + ), + )); + } + if !purpose.admits(record.state) { + return Err(FenceError::new( + FenceOutcome::OperationCapabilityRequired, + format!( + "namespace `{}` is in {}, which admits no new {} capability", + self.namespace, + record.state, + purpose.as_str() + ), + )); + } + let cap = MigrationCapability::issue( + self.namespace.clone(), + operation_id, + purpose, + record.revision, + ); + drop(gate); + caps.live.insert(cap.id(), cap.clone()); + tracing::debug!( + namespace = %self.namespace, + %operation_id, + capability = %cap.id(), + purpose = purpose.as_str(), + revision = cap.fence_revision(), + "issued migration capability" + ); + Ok(cap) + } + + /// Revoke a capability: nothing is admitted under it any more. + pub fn revoke_capability(&self, id: Uuid) { + self.capabilities.lock().live.remove(&id); + } + + /// Whether `id` was issued by this controller and is neither revoked nor invalidated by a + /// transition. + pub fn capability_is_live(&self, id: Uuid) -> bool { + self.capabilities.lock().live.contains_key(&id) + } + + /// Admit one import call under `cap`, counted until the returned guard is dropped. The + /// capability is checked against the gate under the capability lock, so an import call is + /// either refused by a seal that closed admission before it, or counted by the seal, which + /// reads the count only after closing admission (section 10.2). + pub fn begin_import_write( + self: &Arc, + cap: &MigrationCapability, + ) -> Result { + if cap.namespace() != &self.namespace || cap.purpose() != CapabilityPurpose::Import { + return Err(FenceError::new( + FenceOutcome::OperationCapabilityRequired, + format!( + "a {} capability for namespace `{}` does not admit imports into `{}`", + cap.purpose().as_str(), + cap.namespace(), + self.namespace + ), + )); + } + let mut caps = self.capabilities.lock(); + { + let gate = self.gate.borrow(); + gate.permits(OperationClass::CapabilityImport)?; + cap.check(&gate.fence)?; + } + if !caps.live.contains_key(&cap.id()) { + return Err(revoked(cap)); + } + caps.import_writers += 1; + Ok(ImportWriter { + controller: self.clone(), + }) + } + + pub(super) fn end_import_write(&self) { + { + let mut caps = self.capabilities.lock(); + caps.import_writers = caps.import_writers.saturating_sub(1); + } + self.import_released.notify_waiters(); + } + + /// The import calls running now. + pub fn import_writers(&self) -> usize { + self.capabilities.lock().import_writers + } + + /// The capabilities issued and still live. + pub fn live_capabilities(&self) -> usize { + self.capabilities.lock().live.len() + } + + /// Notified whenever an import call ends. Enable the notification before checking + /// [`import_writers`](Self::import_writers), so an end in between is not missed. + pub(crate) fn import_released(&self) -> &Notify { + &self.import_released + } + /// Take the namespace's transition lock. Every fence command on the namespace runs while /// holding it, from its first check to its response. pub async fn begin_transition(self: &Arc) -> Transition { @@ -558,6 +705,7 @@ impl FenceController { installing: Option, creating_target: Option, ) { + let fence_published = fence.is_some(); let mut generation_changed = false; self.gate.send_modify(|gate| { let fence = fence.unwrap_or_else(|| gate.fence.clone()); @@ -590,6 +738,16 @@ impl FenceController { "published namespace fence gate" ); } + // A published transition invalidates every capability issued against an earlier state, + // owner or revision. Cloned first: the capability lock is never taken under a gate + // borrow. + if fence_published { + let fence = self.gate.borrow().fence.clone(); + self.capabilities + .lock() + .live + .retain(|_, cap| cap.matches(&fence)); + } // After the gate is published, so that every woken writer re-checks against it. if generation_changed { self.write_queues.lock().retain(|wake| wake()); @@ -805,6 +963,20 @@ fn indeterminate(key: CommandKey, why: &str) -> FenceError { .with_detail(FenceDetail::IndeterminateCommit) } +fn revoked(cap: &MigrationCapability) -> FenceError { + FenceError::new( + FenceOutcome::OperationCapabilityRequired, + format!( + "the {} capability {} of operation {} on namespace `{}` was revoked or was never \ + issued by this server", + cap.purpose().as_str(), + cap.id(), + cap.operation_id(), + cap.namespace() + ), + ) +} + fn pending_indeterminate((operation_id, command_id): CommandKey) -> FenceError { FenceError::new( FenceOutcome::FenceCommitIndeterminate, @@ -857,6 +1029,9 @@ impl Drop for ProgramReadLease { pub struct FenceConnState { controller: Arc, class: OperationClass, + /// The migration capability this connection works under, fixed at construction. Only a + /// connection opened for an import or validation session has one. + capability: Option, /// The write generation the current program was admitted under. program_generation: AtomicU64, /// The write generation the current read transaction was opened under. @@ -875,10 +1050,30 @@ pub struct FenceConnState { impl FenceConnState { pub fn new(controller: Arc, class: OperationClass) -> Arc { + Self::build(controller, class, None) + } + + /// The fence state of a connection that works under `capability` (an import or a + /// validation session): it is admitted as the capability's class, and only while the + /// capability is valid. + pub fn with_capability( + controller: Arc, + capability: MigrationCapability, + ) -> Arc { + let class = capability.class(); + Self::build(controller, class, Some(capability)) + } + + fn build( + controller: Arc, + class: OperationClass, + capability: Option, + ) -> Arc { let generation = controller.write_generation(); Arc::new(Self { controller, class, + capability, program_generation: AtomicU64::new(generation), txn_generation: AtomicU64::new(generation), denial: Mutex::new(None), @@ -974,6 +1169,10 @@ impl FenceConnState { self.class } + pub fn capability(&self) -> Option<&MigrationCapability> { + self.capability.as_ref() + } + pub fn program_generation(&self) -> u64 { self.program_generation.load(Ordering::Acquire) } @@ -1000,32 +1199,19 @@ impl FenceConnState { } /// The authoritative write admission (section 8.1, check 2): the live gate permits this - /// connection's class, and the program and its read transaction were both admitted under - /// the gate's current write generation. On refusal the typed outcome is left in the denial - /// slot and returned. + /// connection's class, the program and its read transaction were both admitted under the + /// gate's current write generation, and a capability connection's capability is still the + /// valid one (the fence's state, owner and revision are the ones it was issued at, and it + /// is live). A validation connection never writes. On refusal the typed outcome is left in + /// the denial slot and returned. pub fn admit_write(&self) -> Result<(), FenceError> { - let result = { - let gate = self.controller.gate.borrow(); - gate.permits(self.class).and_then(|()| { - let current = gate.write_generation; - let program = self.program_generation(); - let txn = self.txn_generation(); - if program == current && txn == current { - Ok(()) - } else { - Err(FenceError::new( - FenceOutcome::MigrationWriteFenced, - format!( - "the namespace fence changed after this transaction began \ - (program admitted at generation {program}, transaction opened at \ - generation {txn}, current generation {current}); roll back and \ - begin a new transaction" - ), - ) - .with_detail(FenceDetail::StaleTransaction)) - } - }) - }; + let result = self + .admit_write_under_gate() + .and_then(|()| match &self.capability { + // Outside the gate borrow: the capability lock is never taken under one. + Some(cap) if !self.controller.capability_is_live(cap.id()) => Err(revoked(cap)), + _ => Ok(()), + }); if let Err(e) = &result { tracing::debug!( namespace = %self.controller.namespace, @@ -1037,6 +1223,37 @@ impl FenceConnState { result } + fn admit_write_under_gate(&self) -> Result<(), FenceError> { + let gate = self.controller.gate.borrow(); + gate.permits(self.class)?; + let current = gate.write_generation; + let program = self.program_generation(); + let txn = self.txn_generation(); + if program != current || txn != current { + return Err(FenceError::new( + FenceOutcome::MigrationWriteFenced, + format!( + "the namespace fence changed after this transaction began (program admitted \ + at generation {program}, transaction opened at generation {txn}, current \ + generation {current}); roll back and begin a new transaction" + ), + ) + .with_detail(FenceDetail::StaleTransaction)); + } + match (self.class, &self.capability) { + (OperationClass::CapabilityValidate, _) => Err(FenceError::new( + FenceOutcome::OperationCapabilityRequired, + "a validation capability admits reads only", + )), + (_, Some(cap)) => cap.check(&gate.fence), + (OperationClass::CapabilityImport, None) => Err(FenceError::new( + FenceOutcome::OperationCapabilityRequired, + "an import write needs a migration capability", + )), + _ => Ok(()), + } + } + /// Take the typed outcome of the last refusal at the WAL, if any. pub fn take_denial(&self) -> Option { self.denial.lock().take() diff --git a/libsql-server/src/namespace/fence/drain.rs b/libsql-server/src/namespace/fence/drain.rs index 3e5c8ba5b8..f05924a92c 100644 --- a/libsql-server/src/namespace/fence/drain.rs +++ b/libsql-server/src/namespace/fence/drain.rs @@ -59,6 +59,9 @@ impl FenceController { FenceCommand::SetSourceReadFence { .. } => { super::read::set_source_read_fence(&mut transition, &meta, request, ctx).await } + FenceCommand::SealTargetImport { .. } => { + super::import::seal_target_import(&mut transition, &meta, request, ctx).await + } _ => transition.apply(&meta, request, ctx).await, } }) @@ -234,7 +237,7 @@ async fn drain_writers( /// /// With write admission closed a manager that has been seen without a writer stays without one /// (only checkpoints can take the slot), so the managers are waited for one after the other. -async fn wait_for_writers(sources: &[LiveWriteDrain], deadline: Instant) -> bool { +pub(super) async fn wait_for_writers(sources: &[LiveWriteDrain], deadline: Instant) -> bool { for source in sources { loop { let released = source.manager.released().notified(); diff --git a/libsql-server/src/namespace/fence/import.rs b/libsql-server/src/namespace/fence/import.rs new file mode 100644 index 0000000000..d4d36451c3 --- /dev/null +++ b/libsql-server/src/namespace/fence/import.rs @@ -0,0 +1,838 @@ +//! Import sessions into quarantined migration targets, and the seal that ends them +//! (`docs/NAMESPACE_FENCE.md` sections 10.2 and 11). +//! +//! An [`ImportSession`] is the only way to write into a `TARGET_QUARANTINED` target. It holds a +//! [`MigrationCapability`] issued to the operation that owns the target and a connection whose +//! fence state carries that capability, so the WAL admits its write transactions as +//! `CapabilityImport` for as long as the capability is valid, and refuses every other +//! connection's. Each call into the session counts as an import writer until it returns. +//! +//! `SealTargetImport` ends import for good. It closes import admission in memory, persists +//! `TARGET_IMPORT_DRAINING` (the revision moves, so every issued import capability is +//! invalidated and no new one can be issued), then waits for the running import calls and for +//! any import transaction still holding the write slot to end, and persists +//! `TARGET_VALIDATING`. Like the source write drain, it waits on release notifications and +//! never takes elapsed time as evidence that a writer has finished. + +use std::sync::Arc; +use std::time::Duration; + +use bytes::Bytes; +use futures::Stream; +use tokio::time::Instant; + +use crate::connection::legacy::LegacyConnection; +use crate::connection::Connection as _; +use crate::error::Error; +use crate::namespace::configurator::{load_dump_sql, read_dump}; +use crate::namespace::meta_store::{FenceCommit, FenceContext, MetaStore}; +use crate::namespace::replication_wal::ReplicationWalWrapper; + +use super::capability::MigrationCapability; +use super::command::{DrainPolicy, FenceCommand, FenceRequest, OnDeadline}; +use super::controller::{FenceController, LiveWriteDrain, Transition}; +use super::drain::{now_ms, wait_for_writers, FORCED_ROLLBACK_GRACE}; +use super::hooks::HookPoint; +use super::outcome::{FenceError, FenceOutcome}; +use super::state::FenceState; +use super::transition::DrainCompletion; + +/// An operation's write access to its quarantined target (section 11): a capability and a +/// connection that works under it. Dropping the session revokes the capability and closes the +/// connection, which rolls back a transaction the session left open. +pub struct ImportSession { + capability: MigrationCapability, + controller: Arc, + conn: LegacyConnection, +} + +impl std::fmt::Debug for ImportSession { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("ImportSession") + .field("capability", &self.capability) + .finish_non_exhaustive() + } +} + +impl ImportSession { + pub(crate) fn new( + capability: MigrationCapability, + controller: Arc, + conn: LegacyConnection, + ) -> Self { + Self { + capability, + controller, + conn, + } + } + + pub fn capability(&self) -> &MigrationCapability { + &self.capability + } + + /// Run `f` with the session's raw connection. The call is refused up front when the + /// capability is no longer valid (the target was sealed or aborted, or its fence moved on), + /// and counts as an import writer until `f` returns, so a seal waits for it. Inside `f` the + /// WAL admits a write transaction only while the capability is still valid: a write that + /// the fence refused there is reported as that refusal, whatever `f` made of the + /// `SQLITE_AUTH` it saw. + pub async fn with_raw( + &mut self, + f: impl FnOnce(&mut rusqlite::Connection) -> R + Send + 'static, + ) -> Result { + let writer = self.controller.begin_import_write(&self.capability)?; + let conn = self.conn.clone(); + let joined = tokio::task::spawn_blocking(move || { + let _writer = writer; + let result = conn.with_raw(f); + (result, conn.fence_state().take_denial()) + }) + .await; + match joined { + Ok((_, Some(denial))) => Err(denial), + Ok((result, None)) => Ok(result), + Err(e) if e.is_panic() => std::panic::resume_unwind(e.into_panic()), + Err(e) => Err(FenceError::new( + FenceOutcome::OperationCapabilityRequired, + format!("the import call did not complete: {e}"), + )), + } + } + + /// Load a SQL dump into the target with the server's dump loader, under the capability. + /// The dump must run in one transaction that it commits itself, as for a namespace + /// created from a dump. Fence refusals are [`Error::NamespaceFence`]; dump errors are + /// [`Error::LoadDumpError`]. + pub async fn load_dump(&mut self, dump: S) -> crate::Result<()> + where + S: Stream> + Unpin, + { + let content = read_dump(dump).await?; + self.with_raw(move |conn| load_dump_sql(&content, conn)) + .await??; + Ok(()) + } +} + +impl Drop for ImportSession { + fn drop(&mut self) { + self.controller.revoke_capability(self.capability.id()); + } +} + +/// `SealTargetImport` (section 10.2), under `transition`. +/// +/// Returns the `APPLIED` commit of `TARGET_VALIDATING` once no import call is running and no +/// import transaction holds the write slot, the `DRAINING` commit of `TARGET_IMPORT_DRAINING` +/// when the deadline passes first (import stays closed, and only a replay of the same command +/// resumes the drain), or the stored result of a replay. +pub async fn seal_target_import( + transition: &mut Transition, + meta: &MetaStore, + request: FenceRequest, + mut ctx: FenceContext, +) -> crate::Result { + let controller = transition.controller().clone(); + let key = (request.operation_id, request.command_id); + let policy = match &request.command { + FenceCommand::SealTargetImport { drain_policy } => { + drain_policy.unwrap_or_else(|| meta.fence_default_write_drain()) + } + _ => return transition.apply(meta, request, ctx).await, + }; + + // Close import admission in memory before persisting: the INSTALLING gate refuses new + // import calls and import write transactions, and moving the write generation makes every + // transaction opened before it stale and wakes the queued import writers. Only for a seal + // that can apply (the owner, at the current revision, of a quarantined target), so that a + // refused command does not disturb a running import. + let gate = controller.gate(); + let can_apply = gate.state() == FenceState::TargetQuarantined + && gate.indeterminate.is_none() + && !gate.is_installing() + && gate.operation_id() == Some(request.operation_id) + && gate.revision() == request.expected_revision; + if can_apply { + transition.install_closing_gate(key); + let _ = controller.hook(HookPoint::AfterInstallingGate).await; + } + + // Persist TARGET_IMPORT_DRAINING. Its publication replaces the INSTALLING gate and, the + // revision having moved, drops every issued import capability. + let commit = match transition.apply(meta, request, ctx.clone()).await { + Ok(commit) => commit, + Err(e) => { + transition.remove_installing(); + return Err(e); + } + }; + if commit.receipt.outcome != FenceOutcome::Draining { + return Ok(commit); + } + let drain_key = (commit.receipt.operation_id, commit.receipt.command_id); + + if !drain_import_writers(&controller, policy).await { + return Ok(commit); + } + + ctx.now_ms = now_ms(); + transition + .complete_drain(meta, drain_key, DrainCompletion::TargetImport, ctx) + .await +} + +/// Wait until no import call is running and no connection manager of the target has a writer +/// holding its write slot. At the deadline, `force_rollback` rolls back the import transactions +/// still holding the slot (an idle session's open transaction; a running call ends its own) +/// and waits again for the same deadline, but at least [`FORCED_ROLLBACK_GRACE`]. `false` when +/// the drain could not be proven within the policy. +async fn drain_import_writers(controller: &FenceController, policy: DrainPolicy) -> bool { + let namespace = controller.namespace().clone(); + let deadline_after = Duration::from_millis(policy.deadline_ms); + let mut deadline = Instant::now() + deadline_after; + let mut forced = false; + loop { + if wait_for_import_writers(controller, deadline).await { + // With import closed, a manager seen without a writer under its slot lock stays + // without one. A target with no loaded maker has no connection that could write. + let sources = controller.live_write_drains(); + if sources_have_no_writer(&sources) { + tracing::info!(%namespace, "import drain proven"); + return true; + } + continue; + } + match policy.on_deadline { + OnDeadline::ForceRollback if !forced => { + forced = true; + for source in controller.live_write_drains() { + let manager = source.manager.clone(); + // The rollback takes the connection's lock, which a running import call + // holds; the release it causes is what the drain keeps waiting for. + tokio::task::spawn_blocking(move || { + if let Some(id) = manager.abort_active() { + tracing::info!( + connection = id, + "import drain deadline passed; rolling back the active import \ + transaction" + ); + } + }); + } + deadline = Instant::now() + deadline_after.max(FORCED_ROLLBACK_GRACE); + } + _ => { + tracing::info!( + %namespace, + deadline_ms = policy.deadline_ms, + on_deadline = policy.on_deadline.as_str(), + forced, + import_writers = controller.import_writers(), + "import drain deadline passed with an import writer still active; \ + answering DRAINING" + ); + return false; + } + } + } +} + +/// Wait for the running import calls to end, then for every manager's write slot to be free of +/// writers. `false` when `deadline` passes first. +async fn wait_for_import_writers(controller: &FenceController, deadline: Instant) -> bool { + loop { + let released = controller.import_released().notified(); + tokio::pin!(released); + // Registered before the check, so an end in between is not missed. + released.as_mut().enable(); + if controller.import_writers() == 0 { + break; + } + tokio::select! { + _ = &mut released => {} + _ = tokio::time::sleep_until(deadline) => return false, + } + } + wait_for_writers(&controller.live_write_drains(), deadline).await +} + +fn sources_have_no_writer(sources: &[LiveWriteDrain]) -> bool { + sources + .iter() + .all(|source| source.manager.with_no_writer(|| ()).is_ok()) +} + +/// A refusal of `open_import_session` for a namespace that is not a primary. +pub(crate) fn not_importable(namespace: &crate::namespace::NamespaceName) -> Error { + FenceError::new( + FenceOutcome::FencePreconditionFailed, + format!("namespace `{namespace}` is not a primary database; it cannot be imported into"), + ) + .with_detail(super::outcome::FenceDetail::NotPrimary) + .into() +} + +#[cfg(test)] +mod tests { + use std::sync::Arc; + + use tempfile::{tempdir, TempDir}; + use uuid::Uuid; + + use super::*; + use crate::database::Connection; + use crate::namespace::fence::capability::CapabilityPurpose; + use crate::namespace::fence::drain::tests::{assert_fenced, raw, LONG, PROMPT}; + use crate::namespace::fence::state::OperationClass; + use crate::namespace::fence::target::tests::{create, create_request, server, OP}; + use crate::namespace::meta_store::FenceCommitKind; + use crate::namespace::store::fence_tests::open_store; + use crate::namespace::store::NamespaceStore; + use crate::namespace::{NamespaceName, RestoreOption}; + + const OTHER_OP: Uuid = Uuid::from_u128(0xb); + /// A deadline that has already passed. + const NOW: DrainPolicy = DrainPolicy { + deadline_ms: 0, + on_deadline: OnDeadline::Fail, + }; + + fn tgt() -> NamespaceName { + "tgt".into() + } + + /// A store holding the quarantined target `tgt` (revision 1), and its controller. + async fn target() -> (TempDir, NamespaceStore, Arc) { + let dir = tempdir().unwrap(); + let store = open_store(dir.path()).await; + create(&store, create_request("tgt", 1)) + .await + .unwrap() + .unwrap(); + let fence = store.with(tgt(), |ns| ns.fence().clone()).await.unwrap(); + assert_eq!(fence.gate().state(), FenceState::TargetQuarantined); + (dir, store, fence) + } + + fn seal_request(command_id: u128, expected_revision: u64, policy: DrainPolicy) -> FenceRequest { + FenceRequest { + namespace: tgt(), + operation_id: OP, + command_id: Uuid::from_u128(command_id), + expected_state: FenceState::TargetQuarantined, + expected_revision, + command: FenceCommand::SealTargetImport { + drain_policy: Some(policy), + }, + } + } + + /// Run `request` through the store, on a task of its own. + fn execute( + store: &NamespaceStore, + request: FenceRequest, + ) -> tokio::task::JoinHandle> { + let store = store.clone(); + tokio::spawn(async move { store.execute_fence_command(request, server()).await }) + } + + async fn until_state(fence: &FenceController, state: FenceState) { + let mut gate = fence.subscribe(); + tokio::time::timeout(PROMPT, gate.wait_for(|g| g.state() == state)) + .await + .expect("the fence reaches the state") + .unwrap(); + } + + /// A normal connection to the target (what the admin shell, a SQL request or a dump load + /// outside a capability would use). + async fn plain_conn(store: &NamespaceStore) -> Arc { + let maker = store + .with(tgt(), |ns| ns.db.connection_maker()) + .await + .unwrap(); + Arc::new(maker.create().await.unwrap()) + } + + /// Rows in `t` on the target, read through a raw connection. + async fn count(store: &NamespaceStore) -> i64 { + let conn = plain_conn(store).await; + tokio::task::spawn_blocking(move || { + conn.with_raw(|c| c.query_row("select count(*) from t", (), |r| r.get(0))) + }) + .await + .unwrap() + .unwrap() + } + + fn fence_err(r: crate::Result) -> FenceError { + match r { + Err(Error::NamespaceFence(e)) => e, + other => panic!("expected a fence error, got {other:?}"), + } + } + + /// Write through a connection that carries `capability`, and return the fence's refusal. + async fn refused_with(store: &NamespaceStore, capability: MigrationCapability) -> FenceError { + let conn = store + .capability_connection(&tgt(), capability) + .await + .unwrap(); + let r = conn.with_raw(|c| c.execute_batch("insert into t values (99)")); + assert_fenced(r); + conn.fence_state() + .take_denial() + .expect("the WAL gate records its refusal") + } + + /// Only the operation's own, server-issued capability at the current revision writes into + /// a quarantined target: plain connections (as an admin shell would use), capabilities of + /// another operation or revision, forged, revoked or validation capabilities are all + /// refused at the WAL, and issuing one is refused for another operation or revision. + #[tokio::test(flavor = "multi_thread")] + async fn import_requires_matching_capability() { + let (_dir, store, fence) = target().await; + let mut session = store.open_import_session(tgt(), OP, 1).await.unwrap(); + let cap = session.capability().clone(); + assert_eq!( + ( + cap.namespace(), + cap.operation_id(), + cap.purpose(), + cap.fence_revision() + ), + (&tgt(), OP, CapabilityPurpose::Import, 1) + ); + session + .with_raw(|c| c.execute_batch("create table t (x); insert into t values (1)")) + .await + .unwrap() + .unwrap(); + assert_eq!(fence.import_writers(), 0); + + // A normal connection, with or without raw access. + let conn = plain_conn(&store).await; + assert_fenced(raw(&conn, "insert into t values (2)").await); + + // Issuing: another operation, a stale or future revision. + let e = fence_err(store.open_import_session(tgt(), OTHER_OP, 1).await); + assert_eq!(e.outcome(), FenceOutcome::FenceOwnedByAnotherOperation); + for revision in [0, 2] { + let e = fence_err(store.open_import_session(tgt(), OP, revision).await); + assert_eq!(e.outcome(), FenceOutcome::FenceRevisionMismatch); + } + + // At the WAL: capabilities the server never issued, of this operation at this revision, + // of another operation, at another revision, or for validation. + let forged = + |op, purpose, revision| MigrationCapability::forged(tgt(), op, purpose, revision); + let cases = [ + ( + forged(OP, CapabilityPurpose::Import, 1), + FenceOutcome::OperationCapabilityRequired, + ), + ( + forged(OTHER_OP, CapabilityPurpose::Import, 1), + FenceOutcome::FenceOwnedByAnotherOperation, + ), + ( + forged(OP, CapabilityPurpose::Import, 2), + FenceOutcome::OperationCapabilityRequired, + ), + ( + forged(OP, CapabilityPurpose::Validate, 1), + FenceOutcome::OperationCapabilityRequired, + ), + ]; + for (capability, outcome) in cases { + let e = refused_with(&store, capability.clone()).await; + assert_eq!(e.outcome(), outcome, "{capability:?}: {e}"); + } + + // A capability that was issued, once its session is gone. + let other = store.open_import_session(tgt(), OP, 1).await.unwrap(); + let revoked = other.capability().clone(); + assert!(fence.capability_is_live(revoked.id())); + drop(other); + assert!(!fence.capability_is_live(revoked.id())); + let e = refused_with(&store, revoked).await; + assert_eq!(e.outcome(), FenceOutcome::OperationCapabilityRequired); + + // The live session still writes; nothing else did. + session + .with_raw(|c| c.execute_batch("insert into t values (3)")) + .await + .unwrap() + .unwrap(); + assert_eq!(count(&store).await, 2); + } + + /// The seal closes import at once and waits, on release notifications, for an import call + /// that was already running: its write transaction, which held the slot before the seal, + /// commits, and only then is `TARGET_VALIDATING` persisted. + #[tokio::test(flavor = "multi_thread")] + async fn seal_waits_for_import_writers() { + let (_dir, store, fence) = target().await; + let mut session = store.open_import_session(tgt(), OP, 1).await.unwrap(); + session + .with_raw(|c| c.execute_batch("create table t (x)")) + .await + .unwrap() + .unwrap(); + let mut idle = store.open_import_session(tgt(), OP, 1).await.unwrap(); + + let (entered_tx, entered) = tokio::sync::oneshot::channel(); + let (resume, resume_rx) = std::sync::mpsc::channel::<()>(); + let committed = Arc::new(std::sync::atomic::AtomicBool::new(false)); + let running = tokio::spawn({ + let committed = committed.clone(); + async move { + let r = session + .with_raw(move |c| { + c.execute_batch("begin immediate; insert into t values (1)")?; + entered_tx.send(()).unwrap(); + resume_rx.recv().unwrap(); + c.execute_batch("commit")?; + committed.store(true, std::sync::atomic::Ordering::SeqCst); + Ok::<_, rusqlite::Error>(()) + }) + .await; + (session, r) + } + }); + entered.await.unwrap(); + assert_eq!(fence.import_writers(), 1); + + let sealing = execute(&store, seal_request(10, 1, LONG)); + until_state(&fence, FenceState::TargetImportDraining).await; + assert_eq!(fence.gate().revision(), 2); + + // Import is closed for good: no new capability, and no new call on a live session. + let e = fence_err(store.open_import_session(tgt(), OP, 2).await); + assert_eq!(e.outcome(), FenceOutcome::OperationCapabilityRequired); + let e = idle + .with_raw(|c| c.execute_batch("insert into t values (2)")) + .await + .unwrap_err(); + assert_eq!(e.outcome(), FenceOutcome::OperationCapabilityRequired); + assert!(!sealing.is_finished()); + assert_eq!(fence.gate().state(), FenceState::TargetImportDraining); + + // The running import finishes; its transaction commits, and only then does the seal + // reach the commit of TARGET_VALIDATING. + let completing = fence.hooks().pause_at(HookPoint::BeforeMetastoreCommit); + resume.send(()).unwrap(); + tokio::time::timeout(PROMPT, completing.reached()) + .await + .expect("the seal completes once the import writer is gone"); + assert!( + committed.load(std::sync::atomic::Ordering::SeqCst), + "the seal proceeded while the import transaction was still running" + ); + completing.resume(); + let (mut session, r) = running.await.unwrap(); + r.unwrap().unwrap(); + let commit = sealing.await.unwrap().unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + assert_eq!( + (fence.gate().state(), fence.gate().revision()), + (FenceState::TargetValidating, 3) + ); + // The publication of TARGET_IMPORT_DRAINING dropped both sessions' capabilities. + assert_eq!(fence.live_capabilities(), 0); + assert_eq!(count(&store).await, 1); + let e = session + .with_raw(|c| c.execute_batch("insert into t values (3)")) + .await + .unwrap_err(); + assert_eq!(e.outcome(), FenceOutcome::OperationCapabilityRequired); + } + + /// An import transaction left open by an idle session holds the write slot: the seal does + /// not complete past its deadline (`on_deadline: fail`), `TARGET_IMPORT_DRAINING` stays + /// durable and closed (also across a restart), another operation cannot touch it, and only + /// the owner's seal resumes it: a replay of the same seal completes once the session is + /// gone (its transaction rolled back). + #[tokio::test(flavor = "multi_thread")] + async fn seal_deadline_leaves_import_draining_until_replayed() { + let (dir, store, fence) = target().await; + let mut session = store.open_import_session(tgt(), OP, 1).await.unwrap(); + session + .with_raw(|c| c.execute_batch("create table t (x)")) + .await + .unwrap() + .unwrap(); + session + .with_raw(|c| c.execute_batch("begin immediate; insert into t values (1)")) + .await + .unwrap() + .unwrap(); + assert_eq!(fence.import_writers(), 0); + + let commit = execute(&store, seal_request(10, 1, NOW)) + .await + .unwrap() + .unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Draining); + assert_eq!( + (fence.gate().state(), fence.gate().revision()), + (FenceState::TargetImportDraining, 2) + ); + + // Another operation's seal is refused; a new seal of the owner joins the drain, which + // still cannot complete. + let mut foreign = seal_request(11, 2, NOW); + foreign.operation_id = OTHER_OP; + foreign.expected_state = FenceState::TargetImportDraining; + let e = fence_err(execute(&store, foreign).await.unwrap()); + assert_eq!(e.outcome(), FenceOutcome::FenceOwnedByAnotherOperation); + let mut join = seal_request(12, 2, NOW); + join.expected_state = FenceState::TargetImportDraining; + let joined = execute(&store, join).await.unwrap().unwrap(); + assert_eq!(joined.receipt.outcome, FenceOutcome::Draining); + assert_eq!( + (fence.gate().state(), fence.gate().revision()), + (FenceState::TargetImportDraining, 2) + ); + + // Still draining after a restart, which drops the open transaction. + drop(session); + store.shutdown().await.unwrap(); + let store = open_store(dir.path()).await; + let fence = store.fence_controller(&tgt()); + assert_eq!(fence.gate().state(), FenceState::TargetImportDraining); + assert!(fence.permits(OperationClass::CapabilityImport).is_ok()); + let e = fence_err(store.open_import_session(tgt(), OP, 2).await); + assert_eq!(e.outcome(), FenceOutcome::OperationCapabilityRequired); + + let replay = execute(&store, seal_request(10, 1, NOW)) + .await + .unwrap() + .unwrap(); + assert_eq!(replay.receipt.outcome, FenceOutcome::Applied); + assert_eq!( + (fence.gate().state(), fence.gate().revision()), + (FenceState::TargetValidating, 3) + ); + assert_eq!(count(&store).await, 0); + let replay = execute(&store, seal_request(10, 1, NOW)) + .await + .unwrap() + .unwrap(); + assert_eq!(replay.kind, FenceCommitKind::Replayed); + } + + /// With `on_deadline: force_rollback` the seal rolls back an import transaction that an + /// idle session left holding the write slot, waits for the release, and completes. + #[tokio::test(flavor = "multi_thread")] + async fn seal_force_rollback_ends_open_import_transaction() { + let (_dir, store, fence) = target().await; + let mut session = store.open_import_session(tgt(), OP, 1).await.unwrap(); + session + .with_raw(|c| c.execute_batch("create table t (x)")) + .await + .unwrap() + .unwrap(); + session + .with_raw(|c| c.execute_batch("begin immediate; insert into t values (1)")) + .await + .unwrap() + .unwrap(); + let policy = DrainPolicy { + deadline_ms: 0, + on_deadline: OnDeadline::ForceRollback, + }; + let commit = execute(&store, seal_request(10, 1, policy)) + .await + .unwrap() + .unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + assert_eq!(fence.gate().state(), FenceState::TargetValidating); + drop(session); + assert_eq!(count(&store).await, 0); + } + + /// Once sealed, nothing imports: no capability is issued at the new revision, an older + /// session's calls are refused, and even a capability naming the current revision is + /// refused at the WAL. + #[tokio::test(flavor = "multi_thread")] + async fn sealed_target_rejects_import() { + let (_dir, store, fence) = target().await; + let mut session = store.open_import_session(tgt(), OP, 1).await.unwrap(); + session + .with_raw(|c| c.execute_batch("create table t (x)")) + .await + .unwrap() + .unwrap(); + let commit = execute(&store, seal_request(10, 1, LONG)) + .await + .unwrap() + .unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + assert_eq!( + (fence.gate().state(), fence.gate().revision()), + (FenceState::TargetValidating, 3) + ); + + for revision in [1, 3] { + let e = fence_err(store.open_import_session(tgt(), OP, revision).await); + assert_eq!(e.outcome(), FenceOutcome::OperationCapabilityRequired); + } + let e = session + .with_raw(|c| c.execute_batch("insert into t values (1)")) + .await + .unwrap_err(); + assert_eq!(e.outcome(), FenceOutcome::OperationCapabilityRequired); + let e = refused_with( + &store, + MigrationCapability::forged(tgt(), OP, CapabilityPurpose::Import, 3), + ) + .await; + assert_eq!(e.outcome(), FenceOutcome::OperationCapabilityRequired); + assert_fenced(raw(&plain_conn(&store).await, "insert into t values (1)").await); + assert_eq!(count(&store).await, 0); + } + + /// A synthetic representative schema: tables with keys and a foreign key, indexes, a + /// trigger, a view and an FTS5 table. + const SCHEMA: &str = " + create table users (id integer primary key, email text not null unique, name text); + create table orders ( + id integer primary key, + user_id integer not null references users(id), + total real not null, + note text + ); + create index orders_by_user on orders(user_id, total); + create table audit (id integer primary key autoincrement, what text); + create trigger orders_audit after insert on orders begin + insert into audit (what) values ('order ' || new.id); + end; + create view order_totals as + select u.email, sum(o.total) as total from users u join orders o on o.user_id = u.id + group by u.email; + create virtual table docs using fts5(title, body); + insert into users (email, name) values ('a@example.com', 'A'), ('b@example.com', 'B'); + insert into orders (user_id, total, note) values (1, 10.5, 'first'), (1, 2, null), + (2, 7.25, 'it''s quoted'); + insert into docs (title, body) values ('fence', 'operation owned namespace fence'), + ('import', 'quarantined target import'); + "; + + /// What a target must reproduce of the source: the schema, the rows, and the derived data + /// (the view, the full-text index). + fn contents(c: &rusqlite::Connection) -> Vec { + let mut out = Vec::new(); + let mut q = |sql: &str| { + use rusqlite::types::ValueRef; + let mut stmt = c.prepare(sql).unwrap(); + let n = stmt.column_count(); + let rows = stmt + .query_map((), |r| { + (0..n) + .map(|i| { + r.get_ref(i).map(|v| match v { + ValueRef::Text(t) => String::from_utf8_lossy(t).into_owned(), + other => format!("{other:?}"), + }) + }) + .collect::>>() + }) + .unwrap(); + for row in rows { + out.push(format!("{sql}: {}", row.unwrap().join(", "))); + } + }; + // The loader re-renders every statement it runs, so the stored SQL of an object differs + // from the source's in case and spacing only. + q("select type, name, tbl_name, \ + lower(replace(replace(replace(sql, ' ', ''), char(10), ''), '\"', '')) \ + from sqlite_schema order by type, name"); + q("select * from users order by id"); + q("select * from orders order by id"); + q("select * from audit order by id"); + q("select * from order_totals order by email"); + q("select title from docs where docs match 'quarantined' order by rowid"); + q("select count(*) from docs"); + out + } + + /// The server's own dump loader runs inside an import session: a dump exported from a + /// source with a representative schema loads into the quarantined target, which then holds + /// exactly what the source held, while nothing else could write to it. + #[tokio::test(flavor = "multi_thread")] + async fn import_session_loads_dump_into_quarantined_target() { + let (_dir, store, fence) = target().await; + store + .create("src".into(), RestoreOption::Latest, Default::default()) + .await + .unwrap(); + let src = { + let maker = store + .with("src".into(), |ns| ns.db.connection_maker()) + .await + .unwrap(); + Arc::new(maker.create().await.unwrap()) + }; + let (dump, expected) = { + let src = src.clone(); + tokio::task::spawn_blocking(move || { + src.with_raw(|c| { + c.execute_batch(SCHEMA).unwrap(); + let mut dump = Vec::new(); + crate::connection::dump::exporter::export_dump(c, &mut dump, false).unwrap(); + (dump, contents(c)) + }) + }) + .await + .unwrap() + }; + let text = String::from_utf8(dump.clone()).unwrap(); + assert!(text.contains("CREATE VIRTUAL TABLE"), "{text}"); + + let mut session = store.open_import_session(tgt(), OP, 1).await.unwrap(); + let stream = futures::stream::iter( + dump.chunks(64) + .map(|c| Ok(Bytes::copy_from_slice(c))) + .collect::>(), + ); + session.load_dump(stream).await.unwrap(); + assert_eq!(fence.gate().state(), FenceState::TargetQuarantined); + // Nothing but the session could have written it. + assert_fenced( + raw( + &plain_conn(&store).await, + "insert into audit (what) values ('x')", + ) + .await, + ); + + // A second load fails as the loader always does on a dump that is not in a + // transaction, without the fence being involved. + let r = session + .load_dump(futures::stream::iter(vec![Ok(Bytes::from_static( + b"savepoint a; release a; savepoint b;", + ))])) + .await; + assert!( + matches!( + r, + Err(Error::LoadDumpError(crate::error::LoadDumpError::NoTxn)) + ), + "{r:?}" + ); + drop(session); + + let commit = execute(&store, seal_request(10, 1, LONG)) + .await + .unwrap() + .unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + let conn = plain_conn(&store).await; + let imported = tokio::task::spawn_blocking(move || conn.with_raw(|c| contents(c))) + .await + .unwrap(); + assert_eq!(imported, expected); + } +} diff --git a/libsql-server/src/namespace/fence/mod.rs b/libsql-server/src/namespace/fence/mod.rs index fd8810e0db..a8d80da59c 100644 --- a/libsql-server/src/namespace/fence/mod.rs +++ b/libsql-server/src/namespace/fence/mod.rs @@ -10,8 +10,8 @@ //! ([`store`], driven by `MetaStore::apply_fence_command`), and the in-memory authority built //! on them: the per-namespace [`controller`] with its gate and read leases, the positive write //! [`drain`], the source [`read`] fence and its -//! [`stream`] leases for dump and replication, quarantined migration [`target`]s, the -//! [`registry`] that holds the controllers outside the namespace cache, and the test [`hooks`] +//! [`stream`] leases for dump and replication, quarantined migration [`target`]s with their +//! [`capability`]-scoped [`import`] sessions and seal drain, the [`registry`] that holds the controllers outside the namespace cache, and the test [`hooks`] //! on their paths. // The persistence, controller and protocol layers that consume these types land in the @@ -19,10 +19,12 @@ // the crate. This attribute is removed once they are wired. #![allow(dead_code)] +pub mod capability; pub mod command; pub mod controller; pub mod drain; pub mod hooks; +pub mod import; pub mod outcome; pub mod read; pub mod record; diff --git a/libsql-server/src/namespace/store.rs b/libsql-server/src/namespace/store.rs index 177f2cc471..43414bc738 100644 --- a/libsql-server/src/namespace/store.rs +++ b/libsql-server/src/namespace/store.rs @@ -13,7 +13,7 @@ use tokio_stream::wrappers::BroadcastStream; use crate::auth::Authenticated; use crate::broadcaster::BroadcastMsg; use crate::connection::config::DatabaseConfig; -use crate::database::DatabaseKind; +use crate::database::{Database, DatabaseKind, PrimaryConnectionMaker}; use crate::error::Error; use crate::metrics::NAMESPACE_LOAD_LATENCY; use crate::namespace::{NamespaceBottomlessDbId, NamespaceBottomlessDbIdInit, NamespaceName}; @@ -21,9 +21,11 @@ use crate::stats::Stats; use super::broadcasters::{BroadcasterHandle, BroadcasterRegistry}; use super::configurator::{DynConfigurator, NamespaceConfigurators}; +use super::fence::capability::CapabilityPurpose; use super::fence::command::{FenceCommand, FenceRequest}; use super::fence::controller::{FenceController, Transition}; use super::fence::hooks::HookPoint; +use super::fence::import::{self, ImportSession}; use super::fence::outcome::FenceOutcome; use super::fence::record::ServerIdentity; use super::fence::registry::FenceRegistry; @@ -694,6 +696,71 @@ impl NamespaceStore { Ok(commit) } + /// Issue an import capability to `operation_id` and open a connection that works under it + /// (`docs/NAMESPACE_FENCE.md` section 11): the only way to write into a quarantined target. + /// Valid only while the target is `TARGET_QUARANTINED`, owned by `operation_id`, at + /// `expected_revision`; the target is loaded if it is not. Fence refusals are + /// [`Error::NamespaceFence`] with their stable outcome code (`OPERATION_CAPABILITY_REQUIRED` + /// in any other state, `FENCE_OWNED_BY_ANOTHER_OPERATION`, `FENCE_REVISION_MISMATCH`). + pub async fn open_import_session( + &self, + namespace: NamespaceName, + operation_id: uuid::Uuid, + expected_revision: u64, + ) -> crate::Result { + let (controller, maker) = self.primary_maker(&namespace).await?; + let capability = controller.issue_capability( + CapabilityPurpose::Import, + operation_id, + expected_revision, + )?; + match maker + .inner() + .make_capability_connection(capability.clone()) + .await + { + Ok(conn) => Ok(ImportSession::new(capability, controller, conn)), + Err(e) => { + controller.revoke_capability(capability.id()); + Err(e) + } + } + } + + /// The fence controller and connection maker of the primary `namespace`, loading it. + async fn primary_maker( + &self, + namespace: &NamespaceName, + ) -> crate::Result<(Arc, Arc)> { + let (controller, maker) = self + .with(namespace.clone(), |ns| { + let maker = match &ns.db { + Database::Primary(p) => Some(p.connection_maker()), + _ => None, + }; + (ns.fence().clone(), maker) + }) + .await?; + match maker { + Some(maker) => Ok((controller, maker)), + None => Err(import::not_importable(namespace)), + } + } + + /// A connection to `namespace` that works under `capability`, whether or not this server + /// issued it: for tests that prove the WAL refuses a capability that is not the valid one. + #[cfg(test)] + pub(crate) async fn capability_connection( + &self, + namespace: &NamespaceName, + capability: super::fence::capability::MigrationCapability, + ) -> crate::Result< + crate::connection::legacy::LegacyConnection, + > { + let (_, maker) = self.primary_maker(namespace).await?; + maker.inner().make_capability_connection(capability).await + } + /// Whether the server already knows `namespace`: its config is in memory, or the namespace /// cache holds it (a loaded namespace, or a fork in flight, which holds its entry locked). async fn name_in_use(&self, namespace: &NamespaceName) -> bool { From 7f17de2e9d0ef007d46a714788877c7e16e918d0 Mon Sep 17 00:00:00 2001 From: River Date: Tue, 29 Sep 2026 19:24:41 +0000 Subject: [PATCH 3/6] libsql-server: validate, publish and enable writes on migration targets Co-authored-by: Tomasz Szymczyszyn --- .../src/namespace/fence/controller.rs | 32 + libsql-server/src/namespace/fence/target.rs | 566 +++++++++++++++++- libsql-server/src/namespace/store.rs | 93 ++- 3 files changed, 680 insertions(+), 11 deletions(-) diff --git a/libsql-server/src/namespace/fence/controller.rs b/libsql-server/src/namespace/fence/controller.rs index 1eeeffc804..4dfd45743b 100644 --- a/libsql-server/src/namespace/fence/controller.rs +++ b/libsql-server/src/namespace/fence/controller.rs @@ -573,6 +573,38 @@ impl FenceController { self.capabilities.lock().live.contains_key(&id) } + /// Check that `cap` is the live server-issued capability of `purpose` for this namespace + /// and still matches the published fence. Validation sessions use this before every call; + /// import sessions perform the same checks while also incrementing their writer count. + pub(crate) fn check_capability( + &self, + cap: &MigrationCapability, + purpose: CapabilityPurpose, + ) -> Result<(), FenceError> { + if cap.namespace() != &self.namespace || cap.purpose() != purpose { + return Err(FenceError::new( + FenceOutcome::OperationCapabilityRequired, + format!( + "a {} capability for namespace `{}` does not admit {} work on `{}`", + cap.purpose().as_str(), + cap.namespace(), + purpose.as_str(), + self.namespace + ), + )); + } + let caps = self.capabilities.lock(); + { + let gate = self.gate.borrow(); + gate.permits(purpose.class())?; + cap.check(&gate.fence)?; + } + if !caps.live.contains_key(&cap.id()) { + return Err(revoked(cap)); + } + Ok(()) + } + /// Admit one import call under `cap`, counted until the returned guard is dropped. The /// capability is checked against the gate under the capability lock, so an import call is /// either refused by a seal that closed admission before it, or counted by the seal, which diff --git a/libsql-server/src/namespace/fence/target.rs b/libsql-server/src/namespace/fence/target.rs index 3b69b7efe5..38729dd191 100644 --- a/libsql-server/src/namespace/fence/target.rs +++ b/libsql-server/src/namespace/fence/target.rs @@ -14,10 +14,16 @@ use uuid::Uuid; +use crate::connection::legacy::LegacyConnection; +use crate::connection::Connection as _; +use crate::namespace::replication_wal::ReplicationWalWrapper; use crate::namespace::NamespaceName; +use super::capability::{CapabilityPurpose, MigrationCapability}; use super::command::{FenceCommand, FenceRequest, TargetConfig}; +use super::controller::FenceController; use super::outcome::{FenceDetail, FenceError, FenceOutcome}; +use super::record::ValidationSnapshot; use super::state::FenceState; /// `CreateTargetQuarantined` for `namespace`, by `operation_id`. The expectation is always @@ -45,6 +51,103 @@ impl From for FenceRequest { } } +/// Read-only access used by the owning operation to validate a sealed target. The connection +/// carries a server-issued validation capability and has SQLite's `query_only` mode enabled; +/// every call checks that the capability still matches the target's owner, state and revision. +pub struct ValidationSession { + capability: MigrationCapability, + controller: std::sync::Arc, + conn: LegacyConnection, +} + +impl std::fmt::Debug for ValidationSession { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("ValidationSession") + .field("capability", &self.capability) + .finish_non_exhaustive() + } +} + +impl ValidationSession { + pub(crate) async fn new( + capability: MigrationCapability, + controller: std::sync::Arc, + conn: LegacyConnection, + ) -> crate::Result { + let mut this = Self { + capability, + controller, + conn, + }; + this.with_raw(|conn| conn.pragma_update(None, "query_only", true)) + .await??; + Ok(this) + } + + pub fn capability(&self) -> &MigrationCapability { + &self.capability + } + + /// Run a read-only operation on the capability connection. The capability is checked before + /// the call, and a write refused at the WAL is returned as its typed fence outcome even if + /// the closure swallowed SQLite's `SQLITE_AUTH`. + pub async fn with_raw( + &mut self, + f: impl FnOnce(&mut rusqlite::Connection) -> R + Send + 'static, + ) -> Result { + self.controller + .check_capability(&self.capability, CapabilityPurpose::Validate)?; + let conn = self.conn.clone(); + let joined = tokio::task::spawn_blocking(move || { + let result = conn.with_raw(f); + (result, conn.fence_state().take_denial()) + }) + .await; + match joined { + Ok((_, Some(denial))) => Err(denial), + Ok((result, None)) => Ok(result), + Err(e) if e.is_panic() => std::panic::resume_unwind(e.into_panic()), + Err(e) => Err(FenceError::new( + FenceOutcome::OperationCapabilityRequired, + format!("the validation call did not complete: {e}"), + )), + } + } + + /// What the server records beside `RecordTargetValidation`: the target's current + /// replication-log identity and frame, and SQLite page count. Target writes have already + /// been positively drained, so these observations cannot race a mutation. + pub(crate) async fn snapshot(&mut self) -> crate::Result { + let sources = self.controller.live_write_drains(); + let Some(latest) = sources.last() else { + return Err(FenceError::new( + FenceOutcome::FenceStateUnavailable, + format!( + "validation target `{}` has no live primary replication log", + self.capability.namespace() + ), + ) + .into()); + }; + let log_id = latest.log_id; + let frame_no = (latest.current_frame_no)().unwrap_or(0); + let page_count = self + .with_raw(|conn| conn.query_row("PRAGMA page_count", (), |row| row.get::<_, u64>(0))) + .await??; + Ok(ValidationSnapshot { + log_id, + frame_no, + page_count, + }) + } +} + +impl Drop for ValidationSession { + fn drop(&mut self) { + self.controller.revoke_capability(self.capability.id()); + } +} + /// The refusal of a target name that the server already knows, in memory or in the namespace /// cache, although the metastore may not hold it yet (a create or fork in flight, or one the /// fence refused after it had published its config in memory). @@ -69,9 +172,9 @@ pub(crate) mod tests { use crate::auth::Authenticated; use crate::connection::config::DatabaseConfig; use crate::connection::program::Program; - use crate::connection::{Connection as _, RequestContext}; + use crate::connection::RequestContext; use crate::error::Error; - use crate::namespace::fence::command::FenceCommand; + use crate::namespace::fence::command::{FenceCommand, ValidationResult}; use crate::namespace::fence::controller::FenceController; use crate::namespace::fence::drain::tests::{raw, PROMPT}; use crate::namespace::fence::hooks::{HookAction, HookPoint}; @@ -82,6 +185,7 @@ pub(crate) mod tests { use crate::namespace::store::NamespaceStore; use crate::namespace::RestoreOption; use crate::query_result_builder::test::TestBuilder; + use crate::query_result_builder::QueryResultBuilder as _; use crate::rpc::replication::replication_log::ReplicationLogService; pub(crate) const OP: Uuid = Uuid::from_u128(0xa); @@ -115,6 +219,148 @@ pub(crate) mod tests { tokio::spawn(async move { store.create_target_quarantined(req, server()).await }) } + fn target_command( + command_id: u128, + expected_state: FenceState, + expected_revision: u64, + command: FenceCommand, + ) -> FenceRequest { + FenceRequest { + namespace: "tgt".into(), + operation_id: OP, + command_id: Uuid::from_u128(command_id), + expected_state, + expected_revision, + command, + } + } + + fn execute( + store: &NamespaceStore, + request: FenceRequest, + ) -> tokio::task::JoinHandle> { + let store = store.clone(); + tokio::spawn(async move { store.execute_fence_command(request, server()).await }) + } + + /// A target with a small table, sealed at TARGET_VALIDATING revision 3. + async fn validating_target() -> (TempDir, NamespaceStore, Arc) { + let dir = tempdir().unwrap(); + let store = open_store(dir.path()).await; + create(&store, create_request("tgt", 1)) + .await + .unwrap() + .unwrap(); + let mut session = store + .open_import_session("tgt".into(), OP, 1) + .await + .unwrap(); + session + .with_raw(|conn| { + conn.execute_batch("create table t (x); insert into t values (1), (2)") + }) + .await + .unwrap() + .unwrap(); + drop(session); + let seal = target_command( + 10, + FenceState::TargetQuarantined, + 1, + FenceCommand::SealTargetImport { drain_policy: None }, + ); + let commit = execute(&store, seal).await.unwrap().unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + let fence = controller(&store, "tgt").await; + assert_eq!( + (fence.gate().state(), fence.gate().revision()), + (FenceState::TargetValidating, 3) + ); + (dir, store, fence) + } + + async fn record_validation( + store: &NamespaceStore, + command_id: u128, + expected_revision: u64, + result: ValidationResult, + ) -> FenceCommit { + execute( + store, + target_command( + command_id, + FenceState::TargetValidating, + expected_revision, + FenceCommand::RecordTargetValidation { + result, + summary: format!("validation {result:?}"), + }, + ), + ) + .await + .unwrap() + .unwrap() + } + + /// A target with successful validation, published readable and write-fenced at revision 5. + async fn write_fenced_target() -> (TempDir, NamespaceStore, Arc) { + let (dir, store, fence) = validating_target().await; + record_validation(&store, 20, 3, ValidationResult::Ok).await; + let publish = target_command( + 21, + FenceState::TargetValidating, + 4, + FenceCommand::PublishTargetReadableWriteFenced, + ); + execute(&store, publish).await.unwrap().unwrap(); + assert_eq!( + (fence.gate().state(), fence.gate().revision()), + (FenceState::TargetWriteFenced, 5) + ); + (dir, store, fence) + } + + fn enable_request(command_id: u128) -> FenceRequest { + target_command( + command_id, + FenceState::TargetWriteFenced, + 5, + FenceCommand::EnableTargetWrites, + ) + } + + async fn count_rows(store: &NamespaceStore) -> i64 { + let (_, conn) = loaded(store, "tgt").await; + tokio::task::spawn_blocking(move || { + conn.with_raw(|c| c.query_row("select count(*) from t", (), |row| row.get(0))) + }) + .await + .unwrap() + .unwrap() + } + + /// Run one normal SQL program, including its legacy config checks, and require every step to + /// succeed. Raw access is intentionally not used for restart mirror assertions. + async fn program( + store: &NamespaceStore, + conn: &Arc, + sql: &'static str, + ) { + let ctx = RequestContext::new( + Authenticated::FullAccess, + "tgt".into(), + store.meta_store().clone(), + ); + let steps = conn + .execute_program(Program::seq(&[sql]), ctx, TestBuilder::default(), None) + .await + .unwrap() + .into_ret(); + for (i, step) in steps.iter().enumerate() { + assert!(step.is_ok(), "step {i} failed: {step:?}"); + } + } + fn fence_error(e: &Error) -> &FenceError { match e { Error::NamespaceFence(f) => f, @@ -491,6 +737,322 @@ pub(crate) mod tests { assert!(!fence.gate().is_creating_target()); } + /// A validation session carries the owner's current capability, admits reads through the + /// quarantine, is `query_only`, and is invalidated by the validation receipt's revision. + /// The receipt records the target snapshot observed by the server. + #[tokio::test(flavor = "multi_thread")] + async fn validation_session_is_read_only() { + let (_dir, store, fence) = validating_target().await; + let mut session = store + .open_validation_session("tgt".into(), OP, 3) + .await + .unwrap(); + assert_eq!(session.capability().purpose(), CapabilityPurpose::Validate); + let (query_only, count) = session + .with_raw(|conn| { + let query_only = + conn.query_row("PRAGMA query_only", (), |row| row.get::<_, i64>(0)); + let count = + conn.query_row("select count(*) from t", (), |row| row.get::<_, i64>(0)); + (query_only, count) + }) + .await + .unwrap(); + assert_eq!(query_only.unwrap(), 1); + assert_eq!(count.unwrap(), 2); + match session + .with_raw(|conn| conn.execute_batch("insert into t values (3)")) + .await + .unwrap() + { + Err(rusqlite::Error::SqliteFailure(e, _)) => { + assert_eq!(e.code, rusqlite::ErrorCode::ReadOnly) + } + other => panic!("query_only validation connection accepted a write: {other:?}"), + } + let e = session + .with_raw(|conn| { + conn.pragma_update(None, "query_only", false).unwrap(); + conn.execute_batch("insert into t values (3)") + }) + .await + .unwrap_err(); + assert_eq!(e.outcome(), FenceOutcome::OperationCapabilityRequired); + + let error = store + .open_validation_session("tgt".into(), OTHER_OP, 3) + .await + .unwrap_err(); + assert_eq!( + fence_error(&error).outcome(), + FenceOutcome::FenceOwnedByAnotherOperation + ); + let error = store + .open_validation_session("tgt".into(), OP, 2) + .await + .unwrap_err(); + assert_eq!( + fence_error(&error).outcome(), + FenceOutcome::FenceRevisionMismatch + ); + + let request = target_command( + 20, + FenceState::TargetValidating, + 3, + FenceCommand::RecordTargetValidation { + result: ValidationResult::Ok, + summary: "validation Ok".into(), + }, + ); + let after_commit = fence.hooks().pause_at(HookPoint::AfterMetastoreCommit); + let first = execute(&store, request.clone()); + after_commit.reached().await; + // The metastore has the receipt but the live gate still has revision 3. The concurrent + // replay must not demand another validation snapshot or capability before it waits for + // the first command to publish. + let replay = execute(&store, request); + tokio::task::yield_now().await; + after_commit.resume(); + let commit = first.await.unwrap().unwrap(); + let replay = replay.await.unwrap().unwrap(); + assert_eq!(replay.kind, FenceCommitKind::Replayed); + let validation = commit.record.as_ref().unwrap().validation.as_ref().unwrap(); + let snapshot = validation.snapshot.expect("the server records a snapshot"); + let (log_id, frame_no) = store + .with("tgt".into(), |ns| { + let logger = ns.db.logger().unwrap(); + let frame_no = *logger.new_frame_notifier.borrow(); + (logger.log_id(), frame_no) + }) + .await + .unwrap(); + assert_eq!(snapshot.log_id, log_id); + assert_eq!(snapshot.frame_no, frame_no.unwrap_or(0)); + assert!(snapshot.page_count > 0); + assert_eq!(fence.gate().revision(), 4); + + let e = session + .with_raw(|conn| conn.query_row("select 1", (), |row| row.get::<_, i64>(0))) + .await + .unwrap_err(); + assert_eq!(e.outcome(), FenceOutcome::OperationCapabilityRequired); + let mut current = store + .open_validation_session("tgt".into(), OP, 4) + .await + .unwrap(); + assert_eq!( + current + .with_raw( + |conn| conn.query_row("select count(*) from t", (), |row| row.get::<_, i64>(0)) + ) + .await + .unwrap() + .unwrap(), + 2 + ); + } + + /// Publication cannot make a target readable until the latest durable validation result of + /// the owning operation is successful. + #[tokio::test(flavor = "multi_thread")] + async fn publish_requires_validation_receipt() { + let (_dir, store, fence) = validating_target().await; + let publish = |command_id, revision| { + target_command( + command_id, + FenceState::TargetValidating, + revision, + FenceCommand::PublishTargetReadableWriteFenced, + ) + }; + let e = execute(&store, publish(20, 3)).await.unwrap().unwrap_err(); + assert_eq!( + fence_error(&e).detail(), + Some(FenceDetail::ValidationReceiptRequired) + ); + record_validation(&store, 21, 3, ValidationResult::Failed).await; + let e = execute(&store, publish(22, 4)).await.unwrap().unwrap_err(); + assert_eq!( + fence_error(&e).detail(), + Some(FenceDetail::ValidationReceiptRequired) + ); + assert_eq!( + (fence.gate().state(), fence.gate().revision()), + (FenceState::TargetValidating, 4) + ); + assert!(fence.permits(OperationClass::NormalRead).is_err()); + } + + /// A successful validation makes publication possible once; exact replay returns the stored + /// result and a new command with the same goal answers ALREADY_APPLIED without moving the + /// revision. + #[tokio::test(flavor = "multi_thread")] + async fn publish_is_idempotent() { + let (dir, store, fence) = validating_target().await; + record_validation(&store, 20, 3, ValidationResult::Ok).await; + let publish = target_command( + 21, + FenceState::TargetValidating, + 4, + FenceCommand::PublishTargetReadableWriteFenced, + ); + let commit = execute(&store, publish.clone()).await.unwrap().unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + assert_eq!( + (fence.gate().state(), fence.gate().revision()), + (FenceState::TargetWriteFenced, 5) + ); + assert!(fence.permits(OperationClass::NormalRead).is_ok()); + assert!(fence.permits(OperationClass::NormalWrite).is_err()); + assert_eq!(count_rows(&store).await, 2); + + let replay = execute(&store, publish).await.unwrap().unwrap(); + assert_eq!(replay.kind, FenceCommitKind::Replayed); + assert_eq!(replay.receipt.outcome, FenceOutcome::Applied); + let again = execute( + &store, + target_command( + 22, + FenceState::TargetValidating, + 4, + FenceCommand::PublishTargetReadableWriteFenced, + ), + ) + .await + .unwrap() + .unwrap(); + assert_eq!(again.receipt.outcome, FenceOutcome::AlreadyApplied); + assert_eq!( + (again.receipt.revision_before, again.receipt.revision_after), + (5, 5) + ); + assert_eq!(fence.gate().revision(), 5); + + store.shutdown().await.unwrap(); + let store = open_store(dir.path()).await; + let fence = controller(&store, "tgt").await; + assert_eq!(fence.gate().state(), FenceState::TargetWriteFenced); + let (_, conn) = loaded(&store, "tgt").await; + program(&store, &conn, "select count(*) from t").await; + } + + /// Enabling writes is idempotent but irreversible: it opens normal writes exactly after the + /// commit is published; replay and a new same-goal command are safe, while no target command + /// can close or abort it again. + #[tokio::test(flavor = "multi_thread")] + async fn enable_writes_idempotent_and_irreversible() { + let (_dir, store, fence) = write_fenced_target().await; + let request = enable_request(30); + let commit = execute(&store, request.clone()).await.unwrap().unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + assert_eq!( + (fence.gate().state(), fence.gate().revision()), + (FenceState::TargetWritable, 6) + ); + assert!(fence.permits(OperationClass::NormalWrite).is_ok()); + + let replay = execute(&store, request).await.unwrap().unwrap(); + assert_eq!(replay.kind, FenceCommitKind::Replayed); + assert_eq!(replay.receipt.outcome, FenceOutcome::Applied); + let again = execute(&store, enable_request(31)).await.unwrap().unwrap(); + assert_eq!(again.receipt.outcome, FenceOutcome::AlreadyApplied); + assert_eq!(fence.gate().revision(), 6); + + let abort = target_command( + 32, + FenceState::TargetWritable, + 6, + FenceCommand::AbortQuarantinedTarget, + ); + let e = execute(&store, abort).await.unwrap().unwrap_err(); + assert_eq!( + fence_error(&e).outcome(), + FenceOutcome::InvalidFenceTransition + ); + assert_eq!(fence.gate().state(), FenceState::TargetWritable); + let (_, conn) = loaded(&store, "tgt").await; + raw(&conn, "insert into t values (3)").await.unwrap(); + assert_eq!(count_rows(&store).await, 3); + } + + /// The committed writable state is installed before a restarted server can expose the + /// target, and the legacy config mirror no longer blocks its writes. + #[tokio::test(flavor = "multi_thread")] + async fn enable_writes_survives_restart() { + let (dir, store, _fence) = write_fenced_target().await; + execute(&store, enable_request(30)).await.unwrap().unwrap(); + store.shutdown().await.unwrap(); + + let store = open_store(dir.path()).await; + let fence = controller(&store, "tgt").await; + assert_eq!( + (fence.gate().state(), fence.gate().revision()), + (FenceState::TargetWritable, 6) + ); + let (_, conn) = loaded(&store, "tgt").await; + program(&store, &conn, "insert into t values (3)").await; + assert_eq!(count_rows(&store).await, 3); + } + + /// Losing the response after EnableTargetWrites commits is resolved by inspection and exact + /// replay; the detached command still publishes the writable gate before it releases the + /// transition lock. + #[tokio::test(flavor = "multi_thread")] + async fn enable_writes_response_loss_resolved() { + let (_dir, store, fence) = write_fenced_target().await; + let request = enable_request(30); + let after_commit = fence.hooks().pause_at(HookPoint::AfterMetastoreCommit); + let lost = execute(&store, request.clone()); + after_commit.reached().await; + let before_response = fence.hooks().pause_at(HookPoint::BeforeResponse); + lost.abort(); + assert!(lost.await.unwrap_err().is_cancelled()); + after_commit.resume(); + before_response.reached().await; + + assert_eq!(fence.gate().state(), FenceState::TargetWritable); + let inspected = store + .meta_store() + .inspect_fence("tgt".into()) + .await + .unwrap(); + assert_eq!(inspected.fence.state(), FenceState::TargetWritable); + assert!(inspected.receipts.iter().any(|stored| { + matches!( + &stored.receipt, + Ok(receipt) + if receipt.operation_id == OP + && receipt.command_id == Uuid::from_u128(30) + && receipt.outcome == FenceOutcome::Applied + ) + })); + before_response.resume(); + + let replay = execute(&store, request).await.unwrap().unwrap(); + assert_eq!(replay.kind, FenceCommitKind::Replayed); + assert_eq!(replay.receipt.outcome, FenceOutcome::Applied); + assert_eq!(fence.gate().state(), FenceState::TargetWritable); + } + + /// A read transaction opened before target write authority is published cannot upgrade to + /// a write afterwards; rolling it back and starting a fresh program succeeds. + #[tokio::test(flavor = "multi_thread")] + async fn stale_generation_cannot_write_after_enable_writes() { + let (_dir, store, fence) = write_fenced_target().await; + let (_, conn) = loaded(&store, "tgt").await; + raw(&conn, "begin; select count(*) from t").await.unwrap(); + execute(&store, enable_request(30)).await.unwrap().unwrap(); + assert_eq!(fence.gate().state(), FenceState::TargetWritable); + + crate::namespace::fence::drain::tests::assert_fenced( + raw(&conn, "insert into t values (3)").await, + ); + raw(&conn, "commit").await.unwrap(); + raw(&conn, "insert into t values (4)").await.unwrap(); + assert_eq!(count_rows(&store).await, 3); + } + /// `AbortQuarantinedTarget` finishes the operation and keeps every normal class denied; /// the target is not deletable by the generic lifecycle either. #[tokio::test(flavor = "multi_thread")] diff --git a/libsql-server/src/namespace/store.rs b/libsql-server/src/namespace/store.rs index 43414bc738..889370ad79 100644 --- a/libsql-server/src/namespace/store.rs +++ b/libsql-server/src/namespace/store.rs @@ -29,9 +29,9 @@ use super::fence::import::{self, ImportSession}; use super::fence::outcome::FenceOutcome; use super::fence::record::ServerIdentity; use super::fence::registry::FenceRegistry; -use super::fence::state::Role; +use super::fence::state::{FenceState, Role}; use super::fence::store::StoredFence; -use super::fence::target::{self, CreateTargetRequest}; +use super::fence::target::{self, CreateTargetRequest, ValidationSession}; use super::meta_store::{FenceCommit, FenceContext, MetaStore, MetaStoreHandle}; use super::schema_lock::SchemaLocksRegistry; use super::{Namespace, ResetCb, ResetOp, ResolveNamespacePathFn, RestoreOption}; @@ -573,13 +573,58 @@ impl NamespaceStore { } _ => self.inner.fences.controller(&request.namespace), }; - controller - .execute( - &self.inner.metadata, - request, - FenceContext::now(server, None), - ) - .await + let mut ctx = FenceContext::now(server, None); + // A new validation receipt records what this server observes of the sealed target. Do + // this only when the live gate exactly matches the request: an exact replay after the + // revision or state has advanced must reach the metastore's replay check without first + // trying to issue a now-invalid validation capability. + if matches!( + &request.command, + FenceCommand::RecordTargetValidation { .. } + ) { + let needs_snapshot = { + let gate = controller.gate(); + gate.state() == FenceState::TargetValidating + && gate.operation_id() == Some(request.operation_id) + && gate.revision() == request.expected_revision + }; + if needs_snapshot && !self.validation_command_recorded(&request).await? { + let snapshot = async { + let mut session = self + .open_validation_session( + request.namespace.clone(), + request.operation_id, + request.expected_revision, + ) + .await?; + session.snapshot().await + } + .await; + match snapshot { + Ok(snapshot) => ctx.validation_snapshot = Some(snapshot), + // A concurrent copy of this command can commit between the gate check and + // the capability call. Once its receipt exists, let execute take the + // transition lock and perform the authoritative replay/fingerprint check. + Err(_) if self.validation_command_recorded(&request).await? => {} + Err(e) => return Err(e), + } + } + } + controller.execute(&self.inner.metadata, request, ctx).await + } + + async fn validation_command_recorded(&self, request: &FenceRequest) -> crate::Result { + let operation_id = request.operation_id.to_string(); + let command_id = request.command_id.to_string(); + let inspected = self + .inner + .metadata + .inspect_fence(request.namespace.clone()) + .await?; + Ok(inspected + .receipts + .iter() + .any(|stored| stored.operation_id == operation_id && stored.command_id == command_id)) } /// `CreateTargetQuarantined`, atomic with namespace creation (`docs/NAMESPACE_FENCE.md` @@ -727,6 +772,36 @@ impl NamespaceStore { } } + /// Issue a read-only validation capability to `operation_id` and open a `query_only` + /// connection under it (`docs/NAMESPACE_FENCE.md` sections 10.3 and 11). Valid only while + /// the target is `TARGET_VALIDATING` or `TARGET_WRITE_FENCED`, owned by the operation at + /// `expected_revision`; the target is loaded if it is not. + pub async fn open_validation_session( + &self, + namespace: NamespaceName, + operation_id: uuid::Uuid, + expected_revision: u64, + ) -> crate::Result { + let (controller, maker) = self.primary_maker(&namespace).await?; + let capability = controller.issue_capability( + CapabilityPurpose::Validate, + operation_id, + expected_revision, + )?; + let conn = match maker + .inner() + .make_capability_connection(capability.clone()) + .await + { + Ok(conn) => conn, + Err(e) => { + controller.revoke_capability(capability.id()); + return Err(e); + } + }; + ValidationSession::new(capability, controller, conn).await + } + /// The fence controller and connection maker of the primary `namespace`, loading it. async fn primary_maker( &self, From c5f3f7ef1efc0bdbb185a6a5b52d6d41bcd4b0b6 Mon Sep 17 00:00:00 2001 From: River Date: Wed, 30 Sep 2026 17:24:26 +0000 Subject: [PATCH 4/6] libsql-server: fix fence replay and resume edge cases Keep stale DRAINING receipts historical once their exact drain state and revision have been superseded, and preserve resumed attribution when a live drain completes. This prevents old commands from cancelling or completing work in newer states while keeping replay responses and audit classification accurate. Co-authored-by: Tomasz Szymczyszyn --- libsql-server/src/namespace/fence/drain.rs | 58 +++++++++++-- libsql-server/src/namespace/fence/import.rs | 15 +++- libsql-server/src/namespace/fence/read.rs | 15 +++- libsql-server/src/namespace/fence/tests.rs | 2 +- .../src/namespace/fence/transition.rs | 84 +++++++++++++++++-- libsql-server/src/namespace/meta_store.rs | 17 ++-- 6 files changed, 163 insertions(+), 28 deletions(-) diff --git a/libsql-server/src/namespace/fence/drain.rs b/libsql-server/src/namespace/fence/drain.rs index f05924a92c..4c93334b10 100644 --- a/libsql-server/src/namespace/fence/drain.rs +++ b/libsql-server/src/namespace/fence/drain.rs @@ -14,7 +14,7 @@ use std::time::Duration; use tokio::time::Instant; -use crate::namespace::meta_store::{FenceCommit, FenceContext, MetaStore}; +use crate::namespace::meta_store::{FenceCommit, FenceCommitKind, FenceContext, MetaStore}; use super::command::{DrainPolicy, FenceCommand, FenceRequest, OnDeadline}; use super::controller::{FenceController, LiveWriteDrain, Transition}; @@ -114,10 +114,13 @@ pub async fn acquire_source_write_fence( return Err(e); } }; - if commit.receipt.outcome != FenceOutcome::Draining { - // A replay of a finished acquisition, or ALREADY_APPLIED. + if commit.kind == FenceCommitKind::Replayed || commit.receipt.outcome != FenceOutcome::Draining + { + // A replay whose drain was superseded, a replay of a finished acquisition, or + // ALREADY_APPLIED. return Ok(commit); } + let resumed = commit.kind == FenceCommitKind::Resumed; let drain_key = (commit.receipt.operation_id, commit.receipt.command_id); // Steps 5 and 6. @@ -138,14 +141,18 @@ pub async fn acquire_source_write_fence( ); } ctx.now_ms = now_ms(); - transition + let mut completed = transition .complete_drain( meta, drain_key, DrainCompletion::SourceWrites { boundary }, ctx, ) - .await + .await?; + if resumed { + completed.kind = FenceCommitKind::Resumed; + } + Ok(completed) } /// Wait until no connection manager of the namespace has a writer holding its write slot, and @@ -685,6 +692,7 @@ pub(crate) mod tests { raw(&holder, "commit").await.unwrap(); let committed = s.frame_no(); let done = s.execute(s.acquire(OP, 1, policy)).await.unwrap().unwrap(); + assert_eq!(done.kind, FenceCommitKind::Resumed); assert_eq!(done.receipt.outcome, FenceOutcome::Applied); assert_eq!(done.receipt.command_id, Uuid::from_u128(1)); assert_eq!( @@ -703,6 +711,46 @@ pub(crate) mod tests { assert_eq!(boundary(&again), boundary(&done)); } + /// A drain that was explicitly rolled back by `ReleaseSourceWriteFence` is historical: an + /// exact replay returns its stored `DRAINING` receipt without trying to complete that drain + /// against the newer `RELEASED` record. + #[tokio::test(flavor = "multi_thread")] + async fn replay_of_released_drain_does_not_resume_it() { + let s = Source::new().await; + let holder = s.conn().await; + raw(&holder, "begin immediate; insert into t values (1);") + .await + .unwrap(); + let policy = DrainPolicy { + deadline_ms: 0, + on_deadline: OnDeadline::Fail, + }; + let acquire = s.acquire(OP, 1, policy); + let first = s.execute(acquire.clone()).await.unwrap().unwrap(); + assert_eq!(first.receipt.outcome, FenceOutcome::Draining); + + let release = FenceRequest { + namespace: "ns".into(), + operation_id: OP, + command_id: Uuid::from_u128(2), + expected_state: FenceState::SourceDraining, + expected_revision: s.fence.gate().revision(), + command: FenceCommand::ReleaseSourceWriteFence, + }; + let released = s.execute(release).await.unwrap().unwrap(); + assert_eq!(released.receipt.outcome, FenceOutcome::Applied); + assert_eq!(s.fence.gate().state(), FenceState::Released); + raw(&holder, "commit").await.unwrap(); + + let replay = s.execute(acquire).await.unwrap().unwrap(); + assert_eq!(replay.kind, FenceCommitKind::Replayed); + assert_eq!(replay.receipt.outcome, FenceOutcome::Draining); + assert_eq!(s.fence.gate().state(), FenceState::Released); + raw(&s.conn().await, "insert into t values (2)") + .await + .unwrap(); + } + /// Releasing the write fence commits, publishes a new write generation and only then /// answers: new programs write again, and a transaction that began under the fence cannot. #[tokio::test(flavor = "multi_thread")] diff --git a/libsql-server/src/namespace/fence/import.rs b/libsql-server/src/namespace/fence/import.rs index d4d36451c3..54fea99606 100644 --- a/libsql-server/src/namespace/fence/import.rs +++ b/libsql-server/src/namespace/fence/import.rs @@ -25,7 +25,7 @@ use crate::connection::legacy::LegacyConnection; use crate::connection::Connection as _; use crate::error::Error; use crate::namespace::configurator::{load_dump_sql, read_dump}; -use crate::namespace::meta_store::{FenceCommit, FenceContext, MetaStore}; +use crate::namespace::meta_store::{FenceCommit, FenceCommitKind, FenceContext, MetaStore}; use crate::namespace::replication_wal::ReplicationWalWrapper; use super::capability::MigrationCapability; @@ -167,9 +167,11 @@ pub async fn seal_target_import( return Err(e); } }; - if commit.receipt.outcome != FenceOutcome::Draining { + if commit.kind == FenceCommitKind::Replayed || commit.receipt.outcome != FenceOutcome::Draining + { return Ok(commit); } + let resumed = commit.kind == FenceCommitKind::Resumed; let drain_key = (commit.receipt.operation_id, commit.receipt.command_id); if !drain_import_writers(&controller, policy).await { @@ -177,9 +179,13 @@ pub async fn seal_target_import( } ctx.now_ms = now_ms(); - transition + let mut completed = transition .complete_drain(meta, drain_key, DrainCompletion::TargetImport, ctx) - .await + .await?; + if resumed { + completed.kind = FenceCommitKind::Resumed; + } + Ok(completed) } /// Wait until no import call is running and no connection manager of the target has a writer @@ -610,6 +616,7 @@ mod tests { .await .unwrap() .unwrap(); + assert_eq!(replay.kind, FenceCommitKind::Resumed); assert_eq!(replay.receipt.outcome, FenceOutcome::Applied); assert_eq!( (fence.gate().state(), fence.gate().revision()), diff --git a/libsql-server/src/namespace/fence/read.rs b/libsql-server/src/namespace/fence/read.rs index 50feb0b3b1..95d7beb877 100644 --- a/libsql-server/src/namespace/fence/read.rs +++ b/libsql-server/src/namespace/fence/read.rs @@ -13,7 +13,7 @@ use std::time::Duration; use tokio::time::Instant; -use crate::namespace::meta_store::{FenceCommit, FenceContext, MetaStore}; +use crate::namespace::meta_store::{FenceCommit, FenceCommitKind, FenceContext, MetaStore}; use super::command::{DrainPolicy, FenceCommand, FenceRequest}; use super::controller::{FenceController, Transition}; @@ -67,9 +67,11 @@ pub async fn set_source_read_fence( return Err(e); } }; - if commit.receipt.outcome != FenceOutcome::Draining { + if commit.kind == FenceCommitKind::Replayed || commit.receipt.outcome != FenceOutcome::Draining + { return Ok(commit); } + let resumed = commit.kind == FenceCommitKind::Resumed; let drain_key = (commit.receipt.operation_id, commit.receipt.command_id); // Step 4. @@ -79,9 +81,13 @@ pub async fn set_source_read_fence( // Step 5. ctx.now_ms = now_ms(); - transition + let mut completed = transition .complete_drain(meta, drain_key, DrainCompletion::SourceReads, ctx) - .await + .await?; + if resumed { + completed.kind = FenceCommitKind::Resumed; + } + Ok(completed) } /// Wait until every read lease of the namespace is released. At the deadline the leases still @@ -420,6 +426,7 @@ pub(crate) mod tests { // The program was cancelled by the fence; it reports the fence, not its rows. read_fenced(&running.await.unwrap().unwrap_err()); let replayed = s.execute(request).await.unwrap(); + assert_eq!(replayed.as_ref().unwrap().kind, FenceCommitKind::Resumed); assert_eq!(fence_outcome(&replayed), FenceOutcome::Applied); assert_eq!(s.fence.gate().state(), FenceState::SourceReadFenced); } diff --git a/libsql-server/src/namespace/fence/tests.rs b/libsql-server/src/namespace/fence/tests.rs index bb41fbeecd..20a91e353a 100644 --- a/libsql-server/src/namespace/fence/tests.rs +++ b/libsql-server/src/namespace/fence/tests.rs @@ -582,7 +582,7 @@ fn restart_at(case: &Boundary) { // The drain that was requested resumes and completes at once: recovery // discarded any uncommitted work. The boundary is on the live, rebuilt log. let commit = replay.unwrap(); - assert_eq!(commit.kind, FenceCommitKind::Committed, "{name}"); + assert_eq!(commit.kind, FenceCommitKind::Resumed, "{name}"); assert_eq!(commit.receipt.outcome, FenceOutcome::Applied, "{name}"); assert_eq!(commit.receipt.command_id, request.command_id); assert_eq!(commit.receipt.revision_after, 2, "{name}"); diff --git a/libsql-server/src/namespace/fence/transition.rs b/libsql-server/src/namespace/fence/transition.rs index 9c664b320c..1a4b5fceb7 100644 --- a/libsql-server/src/namespace/fence/transition.rs +++ b/libsql-server/src/namespace/fence/transition.rs @@ -10,8 +10,9 @@ //! Checks run in this order, and the order is part of the contract: //! //! 1. **Replay.** A stored receipt with the same fingerprint is answered from the receipt -//! (`Replay`, or `Resume` for a drain still in progress), whatever has happened to the -//! record since. A stored receipt with a different fingerprint is `FENCE_COMMAND_CONFLICT`. +//! (`Resume` only while the record is still in that command's draining state and revision, +//! otherwise `Replay`), whatever has happened to the record since. A stored receipt with a +//! different fingerprint is `FENCE_COMMAND_CONFLICT`. //! 2. **Unavailable state.** A record the server cannot establish refuses everything with //! `FENCE_STATE_UNAVAILABLE`, except the two commands that can reconcile it: a replay of the //! `CreateTargetQuarantined` that left the marker, and an adoption after a metastore @@ -260,10 +261,22 @@ pub fn apply( ), )); } - return Ok(if existing.is_final() { - Decision::Replay(existing.clone()) - } else { + // A DRAINING receipt resumes only while its exact drain is still the durable state. + // Release, clear-read and abort may supersede one, and a source may begin another read + // drain later under the same operation and state; the record revision distinguishes that + // later cycle. Replaying an older command must return its stored answer without running + // it against the newer record (and potentially cancelling newly admitted work). + let still_draining = matches!( + current, + CurrentFence::Record(record) + if record.operation_id == existing.operation_id + && drain_state(existing.command) == Some(record.state) + && existing.revision_after == record.revision + ); + return Ok(if !existing.is_final() && still_draining { Decision::Resume(existing.clone()) + } else { + Decision::Replay(existing.clone()) }); } @@ -646,9 +659,12 @@ pub fn complete_drain( completion: DrainCompletion, env: &ApplyEnv, ) -> Result<(NamespaceFenceRecord, CommandReceipt), FenceError> { - if receipt.outcome != FenceOutcome::Draining || receipt.operation_id != record.operation_id { + if receipt.outcome != FenceOutcome::Draining + || receipt.operation_id != record.operation_id + || receipt.revision_after != record.revision + { return Err(invalid( - "there is no drain of the owning operation to complete", + "there is no matching drain of the owning operation to complete", )); } @@ -1274,6 +1290,57 @@ mod tests { assert_eq!(receipt.revision_after, 2); } + #[test] + fn superseded_draining_receipt_is_replayed_without_resuming() { + let mut source = Harness::source(); + let acquire = source.request(OP, command(CommandKind::AcquireSourceWriteFence)); + source.run_request(&acquire, &env()).unwrap(); + source + .run(OP, CommandKind::ReleaseSourceWriteFence) + .unwrap(); + let replay = source.decide(&acquire, &env()).unwrap(); + assert!(matches!( + replay, + Decision::Replay(ref receipt) if receipt.outcome == O::Draining + )); + assert_eq!(source.state(), S::Released); + + let mut source = Harness::in_state(S::SourceWriteFenced); + let read_fence = source.request(OP, command(CommandKind::SetSourceReadFence)); + source.run_request(&read_fence, &env()).unwrap(); + source.run(OP, CommandKind::ClearSourceReadFence).unwrap(); + let replay = source.decide(&read_fence, &env()).unwrap(); + assert!(matches!( + replay, + Decision::Replay(ref receipt) if receipt.outcome == O::Draining + )); + assert_eq!(source.state(), S::SourceWriteFenced); + + // Starting another read drain under the same operation and state does not make the + // earlier cycle live again: only receipts at the current drain revision may resume it. + let later_read_fence = source.request(OP, command(CommandKind::SetSourceReadFence)); + source.run_request(&later_read_fence, &env()).unwrap(); + assert!(matches!( + source.decide(&read_fence, &env()).unwrap(), + Decision::Replay(ref receipt) if receipt.outcome == O::Draining + )); + assert!(matches!( + source.decide(&later_read_fence, &env()).unwrap(), + Decision::Resume(ref receipt) if receipt.outcome == O::Draining + )); + + let mut target = Harness::in_state(S::TargetQuarantined); + let seal = target.request(OP, command(CommandKind::SealTargetImport)); + target.run_request(&seal, &env()).unwrap(); + target.run(OP, CommandKind::AbortQuarantinedTarget).unwrap(); + let replay = target.decide(&seal, &env()).unwrap(); + assert!(matches!( + replay, + Decision::Replay(ref receipt) if receipt.outcome == O::Draining + )); + assert_eq!(target.state(), S::TargetAborted); + } + #[test] fn command_id_reuse_with_different_fingerprint_conflicts() { let mut h = Harness::source(); @@ -1515,6 +1582,9 @@ mod tests { let mut other = receipt.clone(); other.operation_id = OTHER_OP; assert!(complete_drain(&record, &other, boundary, &env()).is_err()); + let mut stale = receipt.clone(); + stale.revision_after -= 1; + assert!(complete_drain(&record, &stale, boundary, &env()).is_err()); // Not draining any more. let h = Harness::in_state(S::SourceWriteFenced); diff --git a/libsql-server/src/namespace/meta_store.rs b/libsql-server/src/namespace/meta_store.rs index 17d23db89c..7c7fc5c9c1 100644 --- a/libsql-server/src/namespace/meta_store.rs +++ b/libsql-server/src/namespace/meta_store.rs @@ -884,14 +884,17 @@ fn apply_fence_command( let decision = transition::apply(stored.as_current(), existing.as_ref(), request, &env)?; let (record, receipt) = match decision { - Decision::Replay(receipt) | Decision::Resume(receipt) => { - let kind = if receipt.is_final() { - FenceCommitKind::Replayed - } else { - FenceCommitKind::Resumed - }; + Decision::Replay(receipt) => { + return Ok(FenceCommit { + kind: FenceCommitKind::Replayed, + receipt, + record: stored.record().cloned(), + created_config: None, + }); + } + Decision::Resume(receipt) => { return Ok(FenceCommit { - kind, + kind: FenceCommitKind::Resumed, receipt, record: stored.record().cloned(), created_config: None, From 3c8d41d0d2be87cf4e1486d9f3c97a5fa40a8be6 Mon Sep 17 00:00:00 2001 From: River Date: Wed, 30 Sep 2026 11:36:57 +0000 Subject: [PATCH 5/6] libsql-server: test legacy fence mirror and read/target crash boundaries Walk a source and two targets through every stored fence state and check the protection an older binary gets (docs/NAMESPACE_FENCE.md section 13.2): the config row's block_* fields hold the fence's mirror and nothing else in the row changes, a config write through the metastore is refused and changes nothing, and an older binary's delete of the config row fails on the fence row's foreign key. Release and write enable put the namespace's own values back. The test found that a restart overwrote the in-memory config of a released source or a writable target with the block_* values saved when the fence was acquired, although config writes after the operation had stored the namespace's own values in the row since: a namespace blocked after its release could come back unblocked. Loading the metastore, the target config publication and adoption now take the namespace's own config through fence_store::own_config, which uses the row as it is once the record no longer mirrors the fence. Crash the server at each persistence boundary of SetSourceReadFence, SealTargetImport and EnableTargetWrites, and while a read or seal drain waits for a reader or an import call, then restart it on the same directory: the prior or the committed state is recovered with no in-memory gate, reader or import call left, reads and import stay closed once their draining state committed, target writes open only if TARGET_WRITABLE committed, the marker is repaired, and a replay of the same command completes an interrupted drain at once. Co-authored-by: Tomasz Szymczyszyn --- libsql-server/src/namespace/fence/record.rs | 14 +- libsql-server/src/namespace/fence/store.rs | 12 + libsql-server/src/namespace/fence/target.rs | 4 +- libsql-server/src/namespace/fence/tests.rs | 1122 ++++++++++++++++++- libsql-server/src/namespace/meta_store.rs | 16 +- 5 files changed, 1152 insertions(+), 16 deletions(-) diff --git a/libsql-server/src/namespace/fence/record.rs b/libsql-server/src/namespace/fence/record.rs index 247b481ad2..a8baa25e60 100644 --- a/libsql-server/src/namespace/fence/record.rs +++ b/libsql-server/src/namespace/fence/record.rs @@ -118,12 +118,24 @@ impl NamespaceFenceRecord { self.state.read_admission() } + /// Whether the stored config's `block_*` fields hold the fence's mirror (section 13.2) + /// rather than the namespace's own values. They stop holding it when the operation releases + /// the namespace or enables target writes: that transition puts the saved values back, and + /// from then on config writes store the namespace's own values in the row again, so the row + /// is authoritative and `legacy_blocks` may be out of date. + pub fn mirrors_legacy_blocks(&self) -> bool { + !matches!( + self.state, + FenceState::Released | FenceState::TargetWritable + ) + } + /// Values of the legacy `block_*` configuration fields while this record is in force: the /// fence state mirrored for an older binary, or the pre-fence values once the operation /// has released the namespace. pub fn legacy_mirror(&self) -> LegacyBlocks { match self.state { - FenceState::Released | FenceState::TargetWritable => self.legacy_blocks.clone(), + _ if !self.mirrors_legacy_blocks() => self.legacy_blocks.clone(), state => LegacyBlocks { block_reads: !state.read_admission().is_open(), block_writes: !state.write_admission().is_open(), diff --git a/libsql-server/src/namespace/fence/store.rs b/libsql-server/src/namespace/fence/store.rs index bb7a92a20f..fc7e60e062 100644 --- a/libsql-server/src/namespace/fence/store.rs +++ b/libsql-server/src/namespace/fence/store.rs @@ -675,6 +675,18 @@ pub fn with_legacy_blocks(config: &DatabaseConfig, blocks: &LegacyBlocks) -> Dat } } +/// The namespace's own config, given its stored config row `stored` and its fence record: the +/// row with the record's saved `block_*` values in place of the mirror while the record +/// mirrors them, and the row itself once the operation has finished (section 13.2), since +/// config writes after a release or a write enable store the namespace's own values. +pub fn own_config(stored: &DatabaseConfig, record: &NamespaceFenceRecord) -> DatabaseConfig { + if record.mirrors_legacy_blocks() { + with_legacy_blocks(stored, &record.legacy_blocks) + } else { + stored.clone() + } +} + /// The `block_*` fields of `config`. pub fn legacy_blocks_of(config: &DatabaseConfig) -> LegacyBlocks { LegacyBlocks { diff --git a/libsql-server/src/namespace/fence/target.rs b/libsql-server/src/namespace/fence/target.rs index 38729dd191..44006e40bd 100644 --- a/libsql-server/src/namespace/fence/target.rs +++ b/libsql-server/src/namespace/fence/target.rs @@ -219,7 +219,7 @@ pub(crate) mod tests { tokio::spawn(async move { store.create_target_quarantined(req, server()).await }) } - fn target_command( + pub(crate) fn target_command( command_id: u128, expected_state: FenceState, expected_revision: u64, @@ -320,7 +320,7 @@ pub(crate) mod tests { (dir, store, fence) } - fn enable_request(command_id: u128) -> FenceRequest { + pub(crate) fn enable_request(command_id: u128) -> FenceRequest { target_command( command_id, FenceState::TargetWriteFenced, diff --git a/libsql-server/src/namespace/fence/tests.rs b/libsql-server/src/namespace/fence/tests.rs index 20a91e353a..6d297a0231 100644 --- a/libsql-server/src/namespace/fence/tests.rs +++ b/libsql-server/src/namespace/fence/tests.rs @@ -114,16 +114,24 @@ impl Server { /// The namespace's controller, loading the namespace if it is not loaded. async fn fence(&self) -> Arc { + self.fence_of("ns").await + } + + async fn fence_of(&self, ns: &'static str) -> Arc { self.store - .with("ns".into(), |ns| ns.fence().clone()) + .with(ns.into(), |ns| ns.fence().clone()) .await .unwrap() } async fn conn(&self) -> Arc { + self.conn_to("ns").await + } + + async fn conn_to(&self, ns: &'static str) -> Arc { let maker = self .store - .with("ns".into(), |ns| ns.db.connection_maker()) + .with(ns.into(), |ns| ns.db.connection_maker()) .await .unwrap(); Arc::new(maker.create().await.unwrap()) @@ -157,15 +165,23 @@ impl Server { } async fn inspect(&self) -> FenceInspection { + self.inspect_of("ns").await + } + + async fn inspect_of(&self, ns: &'static str) -> FenceInspection { self.store .meta_store() - .inspect_fence("ns".into()) + .inspect_fence(ns.into()) .await .unwrap() } async fn count(&self) -> i64 { - let conn = self.conn().await; + self.count_in("ns").await + } + + async fn count_in(&self, ns: &'static str) -> i64 { + let conn = self.conn_to(ns).await; tokio::task::spawn_blocking(move || { conn.with_raw(|c| c.query_row("select count(*) from t", (), |r| r.get(0))) }) @@ -176,7 +192,11 @@ impl Server { /// Whether a new connection may begin a write transaction. Writes nothing. async fn writes_admitted(&self) -> bool { - let conn = self.conn().await; + self.writes_admitted_in("ns").await + } + + async fn writes_admitted_in(&self, ns: &'static str) -> bool { + let conn = self.conn_to(ns).await; match raw(&conn, "begin immediate; rollback;").await { Ok(()) => true, Err(rusqlite::Error::SqliteFailure(e, _)) @@ -254,7 +274,11 @@ fn boundary(commit: &FenceCommit) -> FrozenBoundary { /// The marker file's bytes, if there is one. fn read_marker_bytes(dbs: &Path) -> Option> { - match std::fs::read(fence_store::marker_path(dbs, &"ns".into())) { + read_marker_bytes_of(dbs, "ns") +} + +fn read_marker_bytes_of(dbs: &Path, ns: &'static str) -> Option> { + match std::fs::read(fence_store::marker_path(dbs, &ns.into())) { Ok(bytes) => Some(bytes), Err(e) if e.kind() == std::io::ErrorKind::NotFound => None, Err(e) => panic!("{e}"), @@ -263,7 +287,11 @@ fn read_marker_bytes(dbs: &Path) -> Option> { /// Put the marker file back to `bytes` (`None`: no marker). fn restore_marker_bytes(dbs: &Path, bytes: Option<&[u8]>) { - let path = fence_store::marker_path(dbs, &"ns".into()); + restore_marker_bytes_of(dbs, "ns", bytes) +} + +fn restore_marker_bytes_of(dbs: &Path, ns: &'static str, bytes: Option<&[u8]>) { + let path = fence_store::marker_path(dbs, &ns.into()); match bytes { Some(bytes) => std::fs::write(path, bytes).unwrap(), None => std::fs::remove_file(path).unwrap(), @@ -896,3 +924,1083 @@ fn acquire_response_loss_resolved_by_replay_and_inspect() { server.crash(); } } + +// --------------------------------------------------------------------------------------------- +// Read fence, seal and write enable: restart at each persistence boundary (section 8.5) + +mod read_and_target_boundaries { + use super::*; + use crate::namespace::fence::command::ValidationResult; + use crate::namespace::fence::controller::LeaseKind; + use crate::namespace::fence::target::tests::{create_request, enable_request, target_command}; + + #[derive(Debug, Clone, Copy, PartialEq, Eq)] + enum Later { + /// `SetSourceReadFence` on the write-fenced source `ns`. + ReadFence, + /// `SealTargetImport` on the quarantined target `tgt`. + Seal, + /// `EnableTargetWrites` on the write-fenced target `tgt`. + Enable, + } + + impl Later { + fn namespace(self) -> &'static str { + match self { + Later::ReadFence => "ns", + Later::Seal | Later::Enable => "tgt", + } + } + + /// State and revision before the command. + fn before(self) -> (FenceState, u64) { + match self { + Later::ReadFence => (FenceState::SourceWriteFenced, 2), + Later::Seal => (FenceState::TargetQuarantined, 1), + Later::Enable => (FenceState::TargetWriteFenced, 5), + } + } + + /// State and revision while the command drains. + fn draining(self) -> (FenceState, u64) { + match self { + Later::ReadFence => (FenceState::SourceReadDraining, 3), + Later::Seal => (FenceState::TargetImportDraining, 2), + Later::Enable => unreachable!("EnableTargetWrites does not drain"), + } + } + + /// State and revision once the command has applied. + fn applied(self) -> (FenceState, u64) { + match self { + Later::ReadFence => (FenceState::SourceReadFenced, 4), + Later::Seal => (FenceState::TargetValidating, 3), + Later::Enable => (FenceState::TargetWritable, 6), + } + } + + fn state_after(self, durable: Durable) -> (FenceState, u64) { + match durable { + Durable::Nothing => self.before(), + Durable::Draining => self.draining(), + Durable::Final => self.applied(), + } + } + + fn request(self) -> FenceRequest { + match self { + Later::ReadFence => FenceRequest { + namespace: "ns".into(), + operation_id: OP, + command_id: Uuid::from_u128(2), + expected_state: FenceState::SourceWriteFenced, + expected_revision: 2, + command: FenceCommand::SetSourceReadFence { + drain_policy: Some(LONG), + }, + }, + Later::Seal => target_command( + 10, + FenceState::TargetQuarantined, + 1, + FenceCommand::SealTargetImport { + drain_policy: Some(LONG), + }, + ), + Later::Enable => enable_request(30), + } + } + } + + /// Where the command is when the process dies. + #[derive(Debug, Clone, Copy)] + enum Park { + /// At a point of the command's first (or only) commit. + First(HookPoint), + /// At a point of the drain's completion, the command's second commit. + Second(HookPoint), + /// In the drain, waiting for a reader (read fence) or an import call (seal) that was + /// already running when the command started. + WaitingForHolder, + } + + #[derive(Debug, Clone, Copy)] + struct LaterBoundary { + name: &'static str, + command: Later, + park: Park, + /// The marker file is put back to what it held before the commit, as a crash between + /// the metastore commit and the marker write leaves it. + marker_lags: bool, + durable: Durable, + } + + const fn case( + name: &'static str, + command: Later, + park: Park, + marker_lags: bool, + durable: Durable, + ) -> LaterBoundary { + LaterBoundary { + name, + command, + park, + marker_lags, + durable, + } + } + + use Durable::{Draining, Final, Nothing}; + use HookPoint::{ + AfterClosingReads, AfterInstallingGate, AfterMetastoreCommit, BeforeGatePublish, + BeforeMetastoreCommit, BeforeResponse, + }; + use Later::{Enable, ReadFence, Seal}; + use Park::{First, Second, WaitingForHolder}; + + const LATER_BOUNDARIES: &[LaterBoundary] = &[ + case( + "read/after-closing-reads", + ReadFence, + First(AfterClosingReads), + false, + Nothing, + ), + case( + "read/before-draining-commit", + ReadFence, + First(BeforeMetastoreCommit), + false, + Nothing, + ), + case( + "read/after-draining-commit", + ReadFence, + First(AfterMetastoreCommit), + false, + Draining, + ), + case( + "read/after-draining-commit/marker-lags", + ReadFence, + First(AfterMetastoreCommit), + true, + Draining, + ), + case( + "read/before-draining-publish", + ReadFence, + First(BeforeGatePublish), + false, + Draining, + ), + case( + "read/draining-published", + ReadFence, + First(BeforeResponse), + false, + Draining, + ), + case( + "read/waiting-for-reader", + ReadFence, + WaitingForHolder, + false, + Draining, + ), + case( + "read/before-fenced-commit", + ReadFence, + Second(BeforeMetastoreCommit), + false, + Draining, + ), + case( + "read/after-fenced-commit", + ReadFence, + Second(AfterMetastoreCommit), + false, + Final, + ), + case( + "read/after-fenced-commit/marker-lags", + ReadFence, + Second(AfterMetastoreCommit), + true, + Final, + ), + case( + "read/before-fenced-publish", + ReadFence, + Second(BeforeGatePublish), + false, + Final, + ), + case( + "read/before-fenced-response", + ReadFence, + Second(BeforeResponse), + false, + Final, + ), + case( + "seal/after-installing-gate", + Seal, + First(AfterInstallingGate), + false, + Nothing, + ), + case( + "seal/before-draining-commit", + Seal, + First(BeforeMetastoreCommit), + false, + Nothing, + ), + case( + "seal/after-draining-commit", + Seal, + First(AfterMetastoreCommit), + false, + Draining, + ), + case( + "seal/after-draining-commit/marker-lags", + Seal, + First(AfterMetastoreCommit), + true, + Draining, + ), + case( + "seal/before-draining-publish", + Seal, + First(BeforeGatePublish), + false, + Draining, + ), + case( + "seal/draining-published", + Seal, + First(BeforeResponse), + false, + Draining, + ), + case( + "seal/waiting-for-import-call", + Seal, + WaitingForHolder, + false, + Draining, + ), + case( + "seal/before-validating-commit", + Seal, + Second(BeforeMetastoreCommit), + false, + Draining, + ), + case( + "seal/after-validating-commit", + Seal, + Second(AfterMetastoreCommit), + false, + Final, + ), + case( + "seal/after-validating-commit/marker-lags", + Seal, + Second(AfterMetastoreCommit), + true, + Final, + ), + case( + "seal/before-validating-publish", + Seal, + Second(BeforeGatePublish), + false, + Final, + ), + case( + "seal/before-validating-response", + Seal, + Second(BeforeResponse), + false, + Final, + ), + case( + "enable/before-commit", + Enable, + First(BeforeMetastoreCommit), + false, + Nothing, + ), + case( + "enable/after-commit", + Enable, + First(AfterMetastoreCommit), + false, + Final, + ), + case( + "enable/after-commit/marker-lags", + Enable, + First(AfterMetastoreCommit), + true, + Final, + ), + case( + "enable/before-publish", + Enable, + First(BeforeGatePublish), + false, + Final, + ), + case( + "enable/before-response", + Enable, + First(BeforeResponse), + false, + Final, + ), + ]; + + /// Create the quarantined target `tgt` (revision 1) and import a table `t` of [`ROWS`] + /// rows into it. + fn create_target(server: &Server) { + server.run(async { + let commit = server + .store + .create_target_quarantined(create_request("tgt", 1), server_identity()) + .await + .unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + let mut session = server + .store + .open_import_session("tgt".into(), OP, 1) + .await + .unwrap(); + session + .with_raw(|c| { + c.execute_batch("create table t (x)")?; + for _ in 0..ROWS { + c.execute("insert into t values (1)", ())?; + } + Ok::<_, rusqlite::Error>(()) + }) + .await + .unwrap() + .unwrap(); + }) + } + + /// Seal, validate and publish `tgt`: `TARGET_WRITE_FENCED` at revision 5. + fn write_fence_target(server: &Server) { + server.run(async { + let steps = [ + target_command( + 10, + FenceState::TargetQuarantined, + 1, + FenceCommand::SealTargetImport { + drain_policy: Some(LONG), + }, + ), + target_command( + 20, + FenceState::TargetValidating, + 3, + FenceCommand::RecordTargetValidation { + result: ValidationResult::Ok, + summary: "rows match".into(), + }, + ), + target_command( + 21, + FenceState::TargetValidating, + 4, + FenceCommand::PublishTargetReadableWriteFenced, + ), + ]; + for step in steps { + let commit = server.execute(step).await.unwrap().unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + } + }) + } + + fn prepare(server: &Server, command: Later) { + match command { + Later::ReadFence => { + server.create_source(); + server.fence_source(); + } + Later::Seal => create_target(server), + Later::Enable => { + create_target(server); + write_fence_target(server); + } + } + } + + /// What the drain of [`Park::WaitingForHolder`] waits for: a read lease, as a running SQL + /// program holds one, or an admitted import call, as a running `ImportSession::with_raw` + /// holds one. Neither holds a SQLite lock, so it can outlive the crash without disturbing + /// the next lifetime's recovery. Only dropped. + type Holder = Box; + + async fn hold(server: &Server, fence: &Arc, command: Later) -> Holder { + match command { + Later::ReadFence => Box::new( + fence + .acquire_read_lease(OperationClass::NormalRead, LeaseKind::Sql, || {}) + .unwrap(), + ), + Later::Seal => { + let session = server + .store + .open_import_session("tgt".into(), OP, 1) + .await + .unwrap(); + let call = fence.begin_import_write(session.capability()).unwrap(); + drop(session); + assert_eq!(fence.import_writers(), 1); + Box::new(call) + } + Later::Enable => unreachable!("EnableTargetWrites does not drain"), + } + } + + /// Admission of new work in the gate's current state: writes only on a writable target, + /// normal reads wherever the state admits them. + async fn assert_admission( + server: &Server, + fence: &Arc, + ns: &'static str, + name: &str, + ) { + let state = fence.gate().state(); + let writes = state == FenceState::TargetWritable; + let reads = matches!( + state, + FenceState::SourceWriteFenced + | FenceState::TargetWriteFenced + | FenceState::TargetWritable + ); + assert_eq!( + server.writes_admitted_in(ns).await, + writes, + "{name}: write admission in {state}" + ); + let lease = fence.acquire_read_lease(OperationClass::NormalRead, LeaseKind::Sql, || {}); + assert_eq!(lease.is_ok(), reads, "{name}: read admission in {state}"); + } + + /// Kill the process at every point where `SetSourceReadFence`, `SealTargetImport` and + /// `EnableTargetWrites` persist, publish, answer or wait, and restart it on the same + /// directory. As for the source write fence (`restart_at_each_boundary`), the restarted + /// server recovers exactly the state before the command or the state it committed and + /// installs that gate before serving the namespace: reads stay closed once + /// `SOURCE_READ_DRAINING` committed, import stays closed once `TARGET_IMPORT_DRAINING` + /// committed, and target writes open only if `TARGET_WRITABLE` committed. No reader or + /// import call survives a restart, so a replay of the same command completes an + /// interrupted drain at once; a replay of a finished one returns its stored result. + #[test] + fn restart_at_each_read_and_target_boundary() { + for case in LATER_BOUNDARIES { + restart_later_at(case); + } + } + + fn restart_later_at(case: &LaterBoundary) { + let name = case.name; + let ns = case.command.namespace(); + let dir = tempdir().unwrap(); + let dbs = dir.path().join("dbs"); + let request = case.command.request(); + let before = case.command.before(); + + // First lifetime: run the command until it reaches the boundary, then crash. + let server = Server::boot(dir.path()); + prepare(&server, case.command); + let holder = server.run(async { + let fence = server.fence_of(ns).await; + assert_eq!( + (fence.gate().state(), fence.gate().revision()), + before, + "{name}" + ); + let hooks = fence.hooks(); + let mut marker_before = read_marker_bytes_of(&dbs, ns); + let mut holder = None; + let paused = match case.park { + Park::First(point) => { + let paused = hooks.pause_at(point); + let task = server.execute(request.clone()); + reached(&paused, name, point).await; + drop(task); + Some(paused) + } + Park::Second(point) => { + // The first commit's response point comes after its publication and + // before the drain, which has nothing to wait for. + let first = hooks.pause_at(HookPoint::BeforeResponse); + let task = server.execute(request.clone()); + reached(&first, name, HookPoint::BeforeResponse).await; + marker_before = read_marker_bytes_of(&dbs, ns); + let paused = hooks.pause_at(point); + first.resume(); + reached(&paused, name, point).await; + drop(task); + Some(paused) + } + Park::WaitingForHolder => { + holder = Some(hold(&server, &fence, case.command).await); + let (draining, _) = case.command.draining(); + let mut gate = fence.subscribe(); + let task = server.execute(request.clone()); + tokio::time::timeout(PROMPT, gate.wait_for(|g| g.state() == draining)) + .await + .unwrap_or_else(|_| panic!("{name}: the command never started draining")) + .unwrap(); + assert!(!task.is_finished(), "{name}: the drain did not wait"); + drop(task); + None + } + }; + if case.marker_lags { + restore_marker_bytes_of(&dbs, ns, marker_before.as_deref()); + } + // What the metastore holds at the moment of the crash. + let inspected = server.inspect_of(ns).await; + assert_eq!( + (inspected.fence.state(), inspected.fence.revision()), + case.command.state_after(case.durable), + "{name}: durable at the crash" + ); + drop(paused); + holder + }); + server.crash(); + // The reader or import call belonged to the dead process. + drop(holder); + + // Second lifetime. + let server = Server::boot(dir.path()); + server.run(async { + let recovered = case.command.state_after(case.durable); + let fence = server.fence_of(ns).await; + let gate = fence.gate(); + assert_eq!( + (gate.state(), gate.revision()), + recovered, + "{name}: recovered state" + ); + assert!( + gate.indeterminate.is_none() + && !gate.is_installing() + && gate.closing_reads.is_none(), + "{name}: an in-memory gate survived the restart" + ); + assert_eq!(fence.read_lease_counts().total(), 0, "{name}"); + assert_eq!(fence.import_writers(), 0, "{name}"); + assert_admission(&server, &fence, ns, name).await; + // Committed data survived, nothing else was written. + assert_eq!(server.count_in(ns).await, ROWS, "{name}"); + // The marker was repaired if it had fallen behind. + let marker = fence_store::read_marker(&dbs, &ns.into()).unwrap(); + assert_eq!( + marker.and_then(|m| m.ok()).map(|m| m.record.revision), + gate.fence.record().map(|r| r.revision), + "{name}: marker" + ); + if case.command == Later::Seal { + // Import resumes only if the seal never committed. + let session = server.store.open_import_session(ns.into(), OP, 1).await; + match case.durable { + Durable::Nothing => drop(session.unwrap()), + Durable::Draining | Durable::Final => { + assert!( + matches!(session, Err(Error::NamespaceFence(_))), + "{name}: import reopened after the seal committed" + ); + } + } + } + + let replay = server.execute(request.clone()).await.unwrap().unwrap(); + let kind = match case.durable { + Durable::Nothing => FenceCommitKind::Committed, + Durable::Draining => FenceCommitKind::Resumed, + Durable::Final => FenceCommitKind::Replayed, + }; + assert_eq!(replay.kind, kind, "{name}"); + assert_eq!(replay.receipt.outcome, FenceOutcome::Applied, "{name}"); + assert_eq!(replay.receipt.command_id, request.command_id, "{name}"); + assert_eq!(replay.receipt.revision_before, before.1, "{name}"); + assert_eq!( + (replay.receipt.state_after, replay.receipt.revision_after), + case.command.applied(), + "{name}" + ); + + // Settled: the gate is the durable state, and a further replay answers the same. + let gate = fence.gate(); + let durable = server.inspect_of(ns).await; + assert_eq!(gate.fence, durable.fence, "{name}"); + assert_eq!( + (gate.state(), gate.revision()), + case.command.applied(), + "{name}" + ); + assert_admission(&server, &fence, ns, name).await; + assert_eq!(server.count_in(ns).await, ROWS, "{name}"); + let again = server.execute(request.clone()).await.unwrap().unwrap(); + assert_eq!(again.kind, FenceCommitKind::Replayed, "{name}"); + assert_eq!(again.receipt, replay.receipt, "{name}"); + }); + server.crash(); + } +} + +// --------------------------------------------------------------------------------------------- +// Protection against an older binary (section 13.2) + +mod legacy_mirror { + use super::*; + use crate::namespace::fence::command::ValidationResult; + use crate::namespace::fence::target::tests::{create_request, enable_request, target_command}; + use crate::namespace::meta_store::{metastore_connection_maker, MetaStoreConnection}; + use libsql_replication::rpc::metadata; + + /// A metastore connection set up the way an older binary sets up its own: foreign keys on, + /// and no knowledge of the fence tables. + async fn older_binary(dir: &Path) -> MetaStoreConnection { + let (maker, _) = metastore_connection_maker(None, dir).await.unwrap(); + let conn = maker().unwrap(); + conn.execute("PRAGMA foreign_keys=ON", ()).unwrap(); + conn + } + + fn encoded(config: &DatabaseConfig) -> metadata::DatabaseConfig { + metadata::DatabaseConfig::from(config) + } + + fn stored(meta: &rusqlite::Connection, ns: &'static str) -> DatabaseConfig { + fence_store::read_config_row(meta, &ns.into()) + .unwrap() + .expect("the namespace has a config row") + } + + /// The `block_*` values section 13.2 says a fenced namespace's stored config holds in + /// `state`, derived from the permission matrix (section 3.3) rather than from the record. + fn mirror(state: FenceState) -> (bool, bool, Option) { + let (block_reads, block_writes) = match state { + FenceState::SourceDraining + | FenceState::SourceWriteFenced + | FenceState::TargetWriteFenced => (false, true), + FenceState::SourceReadDraining + | FenceState::SourceReadFenced + | FenceState::TargetQuarantined + | FenceState::TargetImportDraining + | FenceState::TargetValidating + | FenceState::TargetAborted => (true, true), + other => unreachable!("{other} does not mirror the fence"), + }; + let reason = format!("namespace fence: {state} (operation {OP})"); + (block_reads, block_writes, Some(reason)) + } + + /// In `state`, the stored config of `ns` is the namespace's own config `own` with the fence + /// mirrored into its `block_*` fields (or with its own values once the operation has + /// finished); the in-memory config is `own`; a config write through the metastore is + /// refused and changes nothing while the fence denies lifecycle work; and an older + /// binary's delete of the config row fails on the fence row's foreign key. + async fn check( + server: &Server, + meta: &rusqlite::Connection, + ns: &'static str, + own: &DatabaseConfig, + state: FenceState, + ) { + let fence = server.fence_of(ns).await; + assert_eq!(fence.gate().state(), state, "{ns}"); + let row = stored(meta, ns); + let blocks = (row.block_reads, row.block_writes, row.block_reason.clone()); + let finished = matches!( + state, + FenceState::Unfenced | FenceState::Released | FenceState::TargetWritable + ); + if finished { + assert_eq!( + blocks, + (own.block_reads, own.block_writes, own.block_reason.clone()), + "{ns} in {state}: the namespace's own block_* values" + ); + } else { + assert_eq!(blocks, mirror(state), "{ns} in {state}: the legacy mirror"); + } + // Only the block_* fields carry the mirror. + assert_eq!( + encoded(&fence_store::with_legacy_blocks( + &row, + &fence_store::legacy_blocks_of(own) + )), + encoded(own), + "{ns} in {state}" + ); + let handle = server + .store + .meta_store() + .lookup(&ns.into()) + .await + .unwrap() + .expect("the namespace has a config"); + assert_eq!( + encoded(&handle.get()), + encoded(own), + "{ns} in {state}: in memory" + ); + + if !finished { + let overwrite = DatabaseConfig { + block_reads: false, + block_writes: false, + block_reason: None, + max_db_pages: own.max_db_pages + 1, + ..own.clone() + }; + match handle.store(overwrite).await { + Err(Error::NamespaceFence(_)) => (), + other => panic!("{ns} in {state}: config write not refused: {other:?}"), + } + assert_eq!(encoded(&stored(meta, ns)), encoded(&row), "{ns} in {state}"); + assert_eq!(encoded(&handle.get()), encoded(own), "{ns} in {state}"); + } + + if state != FenceState::Unfenced { + // SQLite enforces `ON DELETE RESTRICT` with an action trigger, so the refusal is + // `SQLITE_CONSTRAINT_TRIGGER` carrying the foreign key message. + match meta.execute("DELETE FROM namespace_configs WHERE namespace = ?1", [ns]) { + Err(rusqlite::Error::SqliteFailure(e, message)) => { + assert_eq!( + e.code, + ErrorCode::ConstraintViolation, + "{ns} in {state}: {e}" + ); + assert_eq!( + message.as_deref(), + Some("FOREIGN KEY constraint failed"), + "{ns} in {state}" + ); + } + other => { + panic!("{ns} in {state}: an older binary's delete was not refused: {other:?}") + } + } + assert_eq!(encoded(&stored(meta, ns)), encoded(&row), "{ns} in {state}"); + } + } + + fn applied(result: Result, tokio::task::JoinError>) -> FenceCommit { + let commit = result.unwrap().unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + commit + } + + fn source_command( + command_id: u128, + expected_state: FenceState, + expected_revision: u64, + command: FenceCommand, + ) -> FenceRequest { + FenceRequest { + namespace: "ns".into(), + operation_id: OP, + command_id: Uuid::from_u128(command_id), + expected_state, + expected_revision, + command, + } + } + + /// Store `config` through the metastore, as `POST /v1/namespaces/:ns/config` does. + async fn store_config(server: &Server, ns: &'static str, config: &DatabaseConfig) { + server + .store + .meta_store() + .lookup(&ns.into()) + .await + .unwrap() + .unwrap() + .store(config.clone()) + .await + .unwrap(); + } + + /// Walk a source and two targets through every stored state: in each, the config row + /// carries the legacy mirror of section 13.2 (reads blocked where the state denies reads, + /// writes blocked where it denies writes, and a reason naming the state and the + /// operation), nothing else in the row changes, a config write cannot overwrite it, and the + /// foreign key refuses an older binary's delete. Release and write enable put the + /// namespace's own values back, after which config writes follow the existing policy; the + /// foreign key stays with the fence row. A restart keeps the rows and gives the in-memory + /// config the namespace's own values. + #[test] + fn legacy_mirror_and_fk_guard() { + let dir = tempdir().unwrap(); + let server = Server::boot(dir.path()); + server.create_source(); + let owns = server.run(async { + let meta = older_binary(dir.path()).await; + let mut own = DatabaseConfig { + block_reason: Some("pre-fence note".into()), + max_db_pages: 1234, + ..(*server + .store + .meta_store() + .lookup(&"ns".into()) + .await + .unwrap() + .unwrap() + .get()) + .clone() + }; + store_config(&server, "ns", &own).await; + check(&server, &meta, "ns", &own, FenceState::Unfenced).await; + + // SOURCE_DRAINING, parked after its commit and before the boundary is captured. + let fence = server.fence().await; + let (log_id, _) = server.log().await; + let paused = fence.hooks().pause_at(HookPoint::BeforeBoundaryCapture); + let task = server.execute(acquire(log_id, 1, LONG)); + reached(&paused, "acquire", HookPoint::BeforeBoundaryCapture).await; + check(&server, &meta, "ns", &own, FenceState::SourceDraining).await; + paused.resume(); + applied(task.await); + check(&server, &meta, "ns", &own, FenceState::SourceWriteFenced).await; + + // SOURCE_READ_DRAINING, parked after its publication and before the drain. + let paused = fence.hooks().pause_at(HookPoint::BeforeResponse); + let task = server.execute(source_command( + 2, + FenceState::SourceWriteFenced, + 2, + FenceCommand::SetSourceReadFence { + drain_policy: Some(LONG), + }, + )); + reached(&paused, "read fence", HookPoint::BeforeResponse).await; + check(&server, &meta, "ns", &own, FenceState::SourceReadDraining).await; + paused.resume(); + applied(task.await); + check(&server, &meta, "ns", &own, FenceState::SourceReadFenced).await; + + applied( + server + .execute(source_command( + 3, + FenceState::SourceReadFenced, + 4, + FenceCommand::ClearSourceReadFence, + )) + .await, + ); + check(&server, &meta, "ns", &own, FenceState::SourceWriteFenced).await; + applied(server.execute(release(4, 5)).await); + check(&server, &meta, "ns", &own, FenceState::Released).await; + // Released: config writes follow the existing policy and are stored as written. + own = DatabaseConfig { + block_writes: true, + block_reason: Some("after the operation".into()), + max_db_pages: 4321, + ..own + }; + store_config(&server, "ns", &own).await; + check(&server, &meta, "ns", &own, FenceState::Released).await; + + // A target, created with the default block_* values. + applied(Ok(server + .store + .create_target_quarantined(create_request("tgt", 1), server_identity()) + .await)); + let mut own_target = (*server + .store + .meta_store() + .lookup(&"tgt".into()) + .await + .unwrap() + .unwrap() + .get()) + .clone(); + assert!( + !own_target.block_reads + && !own_target.block_writes + && own_target.block_reason.is_none() + ); + check( + &server, + &meta, + "tgt", + &own_target, + FenceState::TargetQuarantined, + ) + .await; + + // TARGET_IMPORT_DRAINING, parked after its publication and before the drain. + let target = server.fence_of("tgt").await; + let paused = target.hooks().pause_at(HookPoint::BeforeResponse); + let task = server.execute(target_command( + 10, + FenceState::TargetQuarantined, + 1, + FenceCommand::SealTargetImport { + drain_policy: Some(LONG), + }, + )); + reached(&paused, "seal", HookPoint::BeforeResponse).await; + check( + &server, + &meta, + "tgt", + &own_target, + FenceState::TargetImportDraining, + ) + .await; + paused.resume(); + applied(task.await); + check( + &server, + &meta, + "tgt", + &own_target, + FenceState::TargetValidating, + ) + .await; + + applied( + server + .execute(target_command( + 20, + FenceState::TargetValidating, + 3, + FenceCommand::RecordTargetValidation { + result: ValidationResult::Ok, + summary: "rows match".into(), + }, + )) + .await, + ); + check( + &server, + &meta, + "tgt", + &own_target, + FenceState::TargetValidating, + ) + .await; + applied( + server + .execute(target_command( + 21, + FenceState::TargetValidating, + 4, + FenceCommand::PublishTargetReadableWriteFenced, + )) + .await, + ); + check( + &server, + &meta, + "tgt", + &own_target, + FenceState::TargetWriteFenced, + ) + .await; + applied(server.execute(enable_request(30)).await); + check( + &server, + &meta, + "tgt", + &own_target, + FenceState::TargetWritable, + ) + .await; + own_target = DatabaseConfig { + max_db_pages: 777, + ..own_target + }; + store_config(&server, "tgt", &own_target).await; + check( + &server, + &meta, + "tgt", + &own_target, + FenceState::TargetWritable, + ) + .await; + + // An aborted target keeps everything blocked. + applied(Ok(server + .store + .create_target_quarantined(create_request("tgt2", 40), server_identity()) + .await)); + let own_aborted = (*server + .store + .meta_store() + .lookup(&"tgt2".into()) + .await + .unwrap() + .unwrap() + .get()) + .clone(); + applied( + server + .execute(FenceRequest { + namespace: "tgt2".into(), + operation_id: OP, + command_id: Uuid::from_u128(41), + expected_state: FenceState::TargetQuarantined, + expected_revision: 1, + command: FenceCommand::AbortQuarantinedTarget, + }) + .await, + ); + check( + &server, + &meta, + "tgt2", + &own_aborted, + FenceState::TargetAborted, + ) + .await; + (own, own_target, own_aborted) + }); + server.crash(); + + // The rows are kept across a restart, and the in-memory config is the namespace's own. + let (own, own_target, own_aborted) = owns; + let server = Server::boot(dir.path()); + server.run(async { + let meta = older_binary(dir.path()).await; + check(&server, &meta, "ns", &own, FenceState::Released).await; + check( + &server, + &meta, + "tgt", + &own_target, + FenceState::TargetWritable, + ) + .await; + check( + &server, + &meta, + "tgt2", + &own_aborted, + FenceState::TargetAborted, + ) + .await; + }); + server.crash(); + } +} diff --git a/libsql-server/src/namespace/meta_store.rs b/libsql-server/src/namespace/meta_store.rs index 7c7fc5c9c1..5f67045640 100644 --- a/libsql-server/src/namespace/meta_store.rs +++ b/libsql-server/src/namespace/meta_store.rs @@ -466,8 +466,11 @@ impl MetaStoreInner { /// Load every namespace's fence after the configs (section 5.6). The stored config row of /// a fenced namespace carries the legacy mirror of the fence in its `block_*` fields - /// (section 13.2); the in-memory config is the namespace's own configuration, so those - /// fields are put back to the values the record saved. A marker that fell behind its + /// (section 13.2) while the record is in force; the in-memory config is the namespace's own + /// configuration, so those fields are put back to the values the record saved + /// ([`fence_store::own_config`]). Once the operation has released the namespace or enabled + /// target writes, the row holds the namespace's own values (including any config written + /// since) and is used as it is. A marker that fell behind its /// record is rewritten. A namespace whose fence cannot be established is logged and keeps /// its stored config, mirror included. fn restore_fences(&mut self) -> Result<()> { @@ -501,7 +504,7 @@ impl MetaStoreInner { } let sender = self.configs.get_mut().get_mut(&ns).expect("listed above"); let config = sender.borrow().config.clone(); - let config = fence_store::with_legacy_blocks(&config, &record.legacy_blocks); + let config = fence_store::own_config(&config, record); sender.send_modify(|c| c.config = Arc::new(config)); } StoredFence::Unavailable { @@ -1384,8 +1387,9 @@ impl MetaStore { /// Make a migration target that the metastore holds visible in the in-memory config map, /// which is what makes `exists()` and `lookup()` find it (section 10.1, step 5). The config - /// published is the stored row with the record's own `block_*` values in place of the - /// legacy mirror, as `restore_fences` does at startup. The caller has already installed + /// published is the namespace's own config ([`fence_store::own_config`]): the stored row + /// with the record's saved `block_*` values in place of the legacy mirror, or the row + /// itself once target writes are enabled, as `restore_fences` does at startup. The caller has already installed /// the target's gate. Returns whether the map changed; `false` also when the namespace is /// not a target with a stored config. pub async fn publish_target_config(&self, namespace: NamespaceName) -> Result { @@ -1403,7 +1407,7 @@ impl MetaStore { return Ok(false); }; drop(tx); - let config = Arc::new(fence_store::with_legacy_blocks(&row, &record.legacy_blocks)); + let config = Arc::new(fence_store::own_config(&row, &record)); let mut configs = inner.configs.blocking_lock(); match configs.get_mut(&namespace) { Some(sender) From e3b4ab46c10740b7c9cfe1874fc6e18e259f60b1 Mon Sep 17 00:00:00 2001 From: River Date: Wed, 30 Sep 2026 12:34:59 +0000 Subject: [PATCH 6/6] libsql-server: complete namespace fence acceptance tests Add a test that the admin shell can neither read nor write a quarantined migration target, through its own entry point and with a raw write that skips its read admission, while the operation's import session keeps working. Map every acceptance requirement in docs/NAMESPACE_FENCE.md section 17 to the tests that cover it, correct the names of the corrupt-state tests, list the transition and CAS tests that cover ownership, revision and replay outcomes, and note the replica connection path in the code-path coverage table. Co-authored-by: Tomasz Szymczyszyn --- libsql-server/src/admin_shell.rs | 91 ++++++++++++++++++++++++++++++ libsql-server/src/namespace/mod.rs | 3 + 2 files changed, 94 insertions(+) diff --git a/libsql-server/src/admin_shell.rs b/libsql-server/src/admin_shell.rs index 84f11e7fe8..83f6a40e27 100644 --- a/libsql-server/src/admin_shell.rs +++ b/libsql-server/src/admin_shell.rs @@ -343,4 +343,95 @@ mod fence_tests { ); assert_eq!(s.fence.read_lease_counts().total(), 0); } + + /// Section 17 row 12: a quarantined target is written only through its operation's import + /// capability. The admin shell, which reaches the namespace with admin authority and runs raw + /// SQL, can neither read nor write it: every query is refused by the read admission with the + /// quarantine code, and a raw write that skipped the admission is still refused at the WAL. + /// The import session keeps working, and nothing the shell sent changed the data. + #[tokio::test(flavor = "multi_thread")] + async fn admin_shell_cannot_write_quarantined() { + use crate::namespace::fence::target::tests::{create, create_request, OP as TARGET_OP}; + use crate::namespace::open_test_store as open_store; + + let dir = tempfile::tempdir().unwrap(); + let store = open_store(dir.path()).await; + create(&store, create_request("tgt", 1)) + .await + .unwrap() + .unwrap(); + let mut session = store + .open_import_session("tgt".into(), TARGET_OP, 1) + .await + .unwrap(); + session + .with_raw(|c| c.execute_batch("create table t (x); insert into t values (1)")) + .await + .unwrap() + .unwrap(); + + let shell = AdminShell::new(store.clone()); + let sql = [ + "insert into t values (2)", + "delete from t", + "create table u (y)", + "pragma user_version = 7", + "begin immediate", + "select count(*) from t", + ]; + let queries = tokio_stream::iter(sql.map(|q| Ok(rpc::Query { query: q.into() }))); + let responses: Vec<_> = shell + .with_namespace(Bytes::from_static(b"tgt"), queries) + .await + .unwrap() + .collect() + .await; + assert_eq!(responses.len(), sql.len()); + for (q, resp) in sql.iter().zip(&responses) { + let resp = resp.as_ref().unwrap(); + assert!( + error(resp).starts_with("MIGRATION_TARGET_QUARANTINED"), + "{q}: {}", + error(resp) + ); + } + + // Without the shell's read admission, the raw write is refused by the WAL gate: the + // connection holds no capability. + let (fence, maker) = store + .with("tgt".into(), |ns| { + (ns.fence().clone(), ns.db.connection_maker()) + }) + .await + .unwrap(); + let conn = maker.create().await.unwrap(); + for q in ["insert into t values (3)", "create table v (z)"] { + let resp = conn.with_raw(|c| run_one(c, q.into())).unwrap(); + assert!(error(&resp).contains("authoriz"), "{q}: {}", error(&resp)); + } + assert_eq!(fence.read_lease_counts().total(), 0); + + // The capability still writes, and it sees only its own rows. + let rows: i64 = session + .with_raw(|c| { + c.execute("insert into t values (4)", ())?; + c.query_row("select count(*) from t", (), |r| r.get(0)) + }) + .await + .unwrap() + .unwrap(); + assert_eq!(rows, 2); + let tables: i64 = session + .with_raw(|c| { + c.query_row( + "select count(*) from sqlite_schema where type = 'table'", + (), + |r| r.get(0), + ) + }) + .await + .unwrap() + .unwrap(); + assert_eq!(tables, 1); + } } diff --git a/libsql-server/src/namespace/mod.rs b/libsql-server/src/namespace/mod.rs index f75dbd700f..69d93368f3 100644 --- a/libsql-server/src/namespace/mod.rs +++ b/libsql-server/src/namespace/mod.rs @@ -28,6 +28,9 @@ pub mod replication_wal; mod schema_lock; mod store; +#[cfg(test)] +pub(crate) use store::fence_tests::open_store as open_test_store; + pub type ResetCb = Box; /// Resolves a namespace that a program ATTACHes: its directory, and its fence controller, which /// admits the attachment as a read of that namespace (`docs/NAMESPACE_FENCE.md` section 9).