diff --git a/libsql-server/src/admin_shell.rs b/libsql-server/src/admin_shell.rs index 84f11e7fe8..83f6a40e27 100644 --- a/libsql-server/src/admin_shell.rs +++ b/libsql-server/src/admin_shell.rs @@ -343,4 +343,95 @@ mod fence_tests { ); assert_eq!(s.fence.read_lease_counts().total(), 0); } + + /// Section 17 row 12: a quarantined target is written only through its operation's import + /// capability. The admin shell, which reaches the namespace with admin authority and runs raw + /// SQL, can neither read nor write it: every query is refused by the read admission with the + /// quarantine code, and a raw write that skipped the admission is still refused at the WAL. + /// The import session keeps working, and nothing the shell sent changed the data. + #[tokio::test(flavor = "multi_thread")] + async fn admin_shell_cannot_write_quarantined() { + use crate::namespace::fence::target::tests::{create, create_request, OP as TARGET_OP}; + use crate::namespace::open_test_store as open_store; + + let dir = tempfile::tempdir().unwrap(); + let store = open_store(dir.path()).await; + create(&store, create_request("tgt", 1)) + .await + .unwrap() + .unwrap(); + let mut session = store + .open_import_session("tgt".into(), TARGET_OP, 1) + .await + .unwrap(); + session + .with_raw(|c| c.execute_batch("create table t (x); insert into t values (1)")) + .await + .unwrap() + .unwrap(); + + let shell = AdminShell::new(store.clone()); + let sql = [ + "insert into t values (2)", + "delete from t", + "create table u (y)", + "pragma user_version = 7", + "begin immediate", + "select count(*) from t", + ]; + let queries = tokio_stream::iter(sql.map(|q| Ok(rpc::Query { query: q.into() }))); + let responses: Vec<_> = shell + .with_namespace(Bytes::from_static(b"tgt"), queries) + .await + .unwrap() + .collect() + .await; + assert_eq!(responses.len(), sql.len()); + for (q, resp) in sql.iter().zip(&responses) { + let resp = resp.as_ref().unwrap(); + assert!( + error(resp).starts_with("MIGRATION_TARGET_QUARANTINED"), + "{q}: {}", + error(resp) + ); + } + + // Without the shell's read admission, the raw write is refused by the WAL gate: the + // connection holds no capability. + let (fence, maker) = store + .with("tgt".into(), |ns| { + (ns.fence().clone(), ns.db.connection_maker()) + }) + .await + .unwrap(); + let conn = maker.create().await.unwrap(); + for q in ["insert into t values (3)", "create table v (z)"] { + let resp = conn.with_raw(|c| run_one(c, q.into())).unwrap(); + assert!(error(&resp).contains("authoriz"), "{q}: {}", error(&resp)); + } + assert_eq!(fence.read_lease_counts().total(), 0); + + // The capability still writes, and it sees only its own rows. + let rows: i64 = session + .with_raw(|c| { + c.execute("insert into t values (4)", ())?; + c.query_row("select count(*) from t", (), |r| r.get(0)) + }) + .await + .unwrap() + .unwrap(); + assert_eq!(rows, 2); + let tables: i64 = session + .with_raw(|c| { + c.query_row( + "select count(*) from sqlite_schema where type = 'table'", + (), + |r| r.get(0), + ) + }) + .await + .unwrap() + .unwrap(); + assert_eq!(tables, 1); + } } diff --git a/libsql-server/src/connection/legacy.rs b/libsql-server/src/connection/legacy.rs index efc176cae8..3d441aaded 100644 --- a/libsql-server/src/connection/legacy.rs +++ b/libsql-server/src/connection/legacy.rs @@ -14,6 +14,7 @@ use tokio::time::Duration; use crate::error::Error; use crate::metrics::DESCRIBE_COUNT; use crate::namespace::broadcasters::BroadcasterHandle; +use crate::namespace::fence::capability::MigrationCapability; use crate::namespace::fence::controller::{FenceConnState, FenceController}; use crate::namespace::fence::state::OperationClass; use crate::namespace::meta_store::MetaStoreHandle; @@ -143,6 +144,33 @@ where #[tracing::instrument(skip(self))] pub(super) async fn make_connection(&self) -> Result> { + self.make_connection_with(FenceConnState::new( + self.fence.clone(), + OperationClass::NormalWrite, + )) + .await + } + + /// Open a connection that works under `capability` (an import or validation session, + /// `docs/NAMESPACE_FENCE.md` section 11). It shares the maker's write slot, WAL and + /// replication log with every other connection, and is admitted as the capability's class + /// only while the capability is valid. It is not counted by the connection throttle: it is + /// operation-owned work, and the fence, not the throttle, bounds how much of it runs. + pub(crate) async fn make_capability_connection( + &self, + capability: MigrationCapability, + ) -> Result> { + self.make_connection_with(FenceConnState::with_capability( + self.fence.clone(), + capability, + )) + .await + } + + async fn make_connection_with( + &self, + fence: Arc, + ) -> Result> { LegacyConnection::new( self.db_path.clone(), self.extensions.clone(), @@ -161,7 +189,7 @@ where self.resolve_attach_path.clone(), self.connection_manager.clone(), self.make_wal_manager.clone(), - FenceConnState::new(self.fence.clone(), OperationClass::NormalWrite), + fence, ) .await } @@ -213,6 +241,13 @@ impl LegacyConnection { } } +impl LegacyConnection { + /// The fence state shared by this connection's WAL wrapper and `CoreConnection`. + pub(crate) fn fence_state(&self) -> &Arc { + &self.fence + } +} + impl Clone for LegacyConnection { fn clone(&self) -> Self { Self { diff --git a/libsql-server/src/connection/mod.rs b/libsql-server/src/connection/mod.rs index 167a1a595c..ef8fe0002e 100644 --- a/libsql-server/src/connection/mod.rs +++ b/libsql-server/src/connection/mod.rs @@ -286,6 +286,11 @@ pub struct MakeThrottledConnection { } impl MakeThrottledConnection { + /// The connection maker this one throttles. + pub(crate) fn inner(&self) -> &F { + &self.connection_maker + } + fn new( semaphore: Arc, connection_maker: F, diff --git a/libsql-server/src/namespace/configurator/helpers.rs b/libsql-server/src/namespace/configurator/helpers.rs index a10ef89d6b..6fe8994663 100644 --- a/libsql-server/src/namespace/configurator/helpers.rs +++ b/libsql-server/src/namespace/configurator/helpers.rs @@ -301,6 +301,16 @@ async fn run_periodic_compactions(logger: Arc) -> anyhow::Res } async fn load_dump(dump: S, conn: PrimaryConnection) -> crate::Result<(), LoadDumpError> +where + S: Stream> + Unpin, +{ + let dump_content = read_dump(dump).await?; + tokio::task::spawn_blocking(move || conn.with_raw(|conn| load_dump_sql(&dump_content, conn))) + .await? +} + +/// Read a whole dump into memory. +pub(crate) async fn read_dump(dump: S) -> crate::Result where S: Stream> + Unpin, { @@ -310,13 +320,36 @@ where .read_to_string(&mut dump_content) .await .map_err(|e| LoadDumpError::Internal(format!("Failed to read dump content: {}", e)))?; + Ok(dump_content) +} +/// Parse `dump_content` and run its statements, one at a time, on `conn`. The dump must run +/// inside one transaction that it commits itself; `ATTACH` is refused. This is the loader both +/// for a namespace created from a dump and for an import session into a quarantined migration +/// target, which runs it under its capability (`docs/NAMESPACE_FENCE.md` section 11). +pub(crate) fn load_dump_sql( + dump_content: &str, + conn: &mut rusqlite::Connection, +) -> crate::Result<(), LoadDumpError> { if dump_content.to_lowercase().contains("attach") { return Err(LoadDumpError::InvalidSqlInput( "attach statements are not allowed in dumps".to_string(), )); } + conn.authorizer(Some(|auth: AuthContext<'_>| match auth.action { + AuthAction::Attach { filename: _ } => Authorization::Deny, + _ => Authorization::Allow, + })); + let result = run_dump_statements(dump_content, conn); + conn.authorizer(None::) -> Authorization>); + result +} + +fn run_dump_statements( + dump_content: &str, + conn: &mut rusqlite::Connection, +) -> crate::Result<(), LoadDumpError> { let mut parser = Box::new(Parser::new(dump_content.as_bytes())); let mut skipped_wasm_table = false; let mut n_stmt = 0; @@ -336,37 +369,20 @@ where } } - if n_stmt > 2 && conn.is_autocommit().await.unwrap() { + if n_stmt > 2 && conn.is_autocommit() { return Err(LoadDumpError::NoTxn); } let stmt_sql = cmd.to_string(); - tokio::task::spawn_blocking({ - let conn = conn.clone(); - move || -> crate::Result<(), LoadDumpError> { - conn.with_raw(|conn| { - conn.authorizer(Some(|auth: AuthContext<'_>| match auth.action { - AuthAction::Attach { filename: _ } => Authorization::Deny, - _ => Authorization::Allow, - })); - conn.execute(&stmt_sql, ()) - }) - .map_err(|e| match e { - rusqlite::Error::SqlInputError { - msg, sql, offset, .. - } => LoadDumpError::InvalidSqlInput(format!( - "msg: {}, sql: {}, offset: {}", - msg, sql, offset - )), - e => LoadDumpError::Internal(format!( - "statement: {}, error: {}", - n_stmt, e - )), - })?; - Ok(()) - } - }) - .await??; + conn.execute(&stmt_sql, ()).map_err(|e| match e { + rusqlite::Error::SqlInputError { + msg, sql, offset, .. + } => LoadDumpError::InvalidSqlInput(format!( + "msg: {}, sql: {}, offset: {}", + msg, sql, offset + )), + e => LoadDumpError::Internal(format!("statement: {}, error: {}", n_stmt, e)), + })?; } Ok(None) => break, Err(e) => { @@ -389,15 +405,8 @@ where } } - if !conn.is_autocommit().await.unwrap() { - tokio::task::spawn_blocking({ - let conn = conn.clone(); - move || -> crate::Result<(), LoadDumpError> { - conn.with_raw(|conn| conn.execute("rollback", ()))?; - Ok(()) - } - }) - .await??; + if !conn.is_autocommit() { + conn.execute("rollback", ())?; return Err(LoadDumpError::NoCommit); } diff --git a/libsql-server/src/namespace/configurator/mod.rs b/libsql-server/src/namespace/configurator/mod.rs index 029ab0b3ce..f8d11addd2 100644 --- a/libsql-server/src/namespace/configurator/mod.rs +++ b/libsql-server/src/namespace/configurator/mod.rs @@ -26,6 +26,7 @@ mod primary; mod replica; mod schema; +pub(crate) use helpers::{load_dump_sql, read_dump}; pub use primary::PrimaryConfigurator; pub use replica::ReplicaConfigurator; pub use schema::SchemaConfigurator; diff --git a/libsql-server/src/namespace/fence/capability.rs b/libsql-server/src/namespace/fence/capability.rs new file mode 100644 index 0000000000..46e492f6f0 --- /dev/null +++ b/libsql-server/src/namespace/fence/capability.rs @@ -0,0 +1,212 @@ +//! Migration capabilities (`docs/NAMESPACE_FENCE.md` sections 7.3 and 11). +//! +//! A [`MigrationCapability`] is the only thing that lets operation-owned work through a +//! target's quarantine. The server creates it ([`FenceController::issue_capability`]); its +//! fields are private and it cannot be built from outside this module, so holding one is proof +//! that the fence issued it. It names the namespace, the owning operation, what it is for and +//! the fence revision it was issued at, and it is valid only while the fence is still in the +//! state its purpose needs, at that revision, owned by that operation, and while the controller +//! still lists it as live. Every transition of the target moves the revision, so a capability +//! never outlives the state it was issued in. +//! +//! [`FenceController::issue_capability`]: super::controller::FenceController::issue_capability + +use std::collections::HashMap; +use std::sync::Arc; + +use uuid::Uuid; + +use crate::namespace::NamespaceName; + +use super::controller::FenceController; +use super::outcome::{FenceError, FenceOutcome}; +use super::state::{FenceState, OperationClass}; +use super::store::StoredFence; + +/// What a capability admits. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] +pub enum CapabilityPurpose { + /// Writes of an import session into a `TARGET_QUARANTINED` target. + Import, + /// Read-only validation of a `TARGET_VALIDATING` or `TARGET_WRITE_FENCED` target. + Validate, +} + +impl CapabilityPurpose { + /// The operation class work under a capability of this purpose is admitted as. + pub const fn class(self) -> OperationClass { + match self { + CapabilityPurpose::Import => OperationClass::CapabilityImport, + CapabilityPurpose::Validate => OperationClass::CapabilityValidate, + } + } + + pub const fn as_str(self) -> &'static str { + match self { + CapabilityPurpose::Import => "import", + CapabilityPurpose::Validate => "validate", + } + } + + /// Whether a capability of this purpose may be issued, and stays valid, in `state`. + pub const fn admits(self, state: FenceState) -> bool { + match self { + CapabilityPurpose::Import => matches!(state, FenceState::TargetQuarantined), + CapabilityPurpose::Validate => matches!( + state, + FenceState::TargetValidating | FenceState::TargetWriteFenced + ), + } + } +} + +/// A server-issued grant for operation-owned work on one namespace, valid at one fence +/// revision. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct MigrationCapability { + id: Uuid, + namespace: NamespaceName, + operation_id: Uuid, + purpose: CapabilityPurpose, + fence_revision: u64, +} + +impl MigrationCapability { + /// Only the controller issues capabilities. + pub(super) fn issue( + namespace: NamespaceName, + operation_id: Uuid, + purpose: CapabilityPurpose, + fence_revision: u64, + ) -> Self { + Self { + id: Uuid::new_v4(), + namespace, + operation_id, + purpose, + fence_revision, + } + } + + /// A capability the controller never issued, for tests that prove such a thing is refused. + #[cfg(test)] + pub(crate) fn forged( + namespace: NamespaceName, + operation_id: Uuid, + purpose: CapabilityPurpose, + fence_revision: u64, + ) -> Self { + Self::issue(namespace, operation_id, purpose, fence_revision) + } + + pub fn id(&self) -> Uuid { + self.id + } + + pub fn namespace(&self) -> &NamespaceName { + &self.namespace + } + + pub fn operation_id(&self) -> Uuid { + self.operation_id + } + + pub fn purpose(&self) -> CapabilityPurpose { + self.purpose + } + + pub fn fence_revision(&self) -> u64 { + self.fence_revision + } + + pub fn class(&self) -> OperationClass { + self.purpose.class() + } + + /// Whether `fence` is still the fence this capability was issued against: the state its + /// purpose needs, owned by its operation, at its revision. + pub(crate) fn matches(&self, fence: &StoredFence) -> bool { + self.check(fence).is_ok() + } + + /// [`matches`](Self::matches), with the refusal a holder gets when it does not. + pub(crate) fn check(&self, fence: &StoredFence) -> Result<(), FenceError> { + let Some(record) = fence.record() else { + return Err(stale( + self, + format!("namespace `{}` has no fence record", self.namespace), + )); + }; + if record.operation_id != self.operation_id { + return Err(FenceError::new( + FenceOutcome::FenceOwnedByAnotherOperation, + format!( + "the fence of namespace `{}` is owned by operation {}, not by operation {} \ + that holds this {} capability", + self.namespace, + record.operation_id, + self.operation_id, + self.purpose.as_str() + ), + )); + } + if !self.purpose.admits(record.state) { + return Err(stale( + self, + format!( + "namespace `{}` is in {}, which admits no {} capability", + self.namespace, + record.state, + self.purpose.as_str() + ), + )); + } + if record.revision != self.fence_revision { + return Err(stale( + self, + format!( + "the fence of namespace `{}` is at revision {}, and this {} capability was \ + issued at revision {}", + self.namespace, + record.revision, + self.purpose.as_str(), + self.fence_revision + ), + )); + } + Ok(()) + } +} + +fn stale(cap: &MigrationCapability, why: String) -> FenceError { + FenceError::new( + FenceOutcome::OperationCapabilityRequired, + format!( + "{why}: the {} capability {} is no longer valid", + cap.purpose.as_str(), + cap.id + ), + ) +} + +/// The capabilities a controller has issued and not yet revoked, and the import calls running +/// under them. +#[derive(Debug, Default)] +pub(super) struct CapabilitySet { + pub(super) live: HashMap, + /// Import calls running now ([`ImportWriter`]s). + pub(super) import_writers: usize, +} + +/// One running import call under a capability. The seal waits for every one of them to be +/// dropped (section 10.2). +#[derive(Debug)] +pub struct ImportWriter { + pub(super) controller: Arc, +} + +impl Drop for ImportWriter { + fn drop(&mut self) { + self.controller.end_import_write(); + } +} diff --git a/libsql-server/src/namespace/fence/controller.rs b/libsql-server/src/namespace/fence/controller.rs index ff946554c6..4dfd45743b 100644 --- a/libsql-server/src/namespace/fence/controller.rs +++ b/libsql-server/src/namespace/fence/controller.rs @@ -24,6 +24,7 @@ use crate::namespace::meta_store::{FenceCommit, FenceContext, MetaStore}; use crate::namespace::NamespaceName; use crate::replication::FrameNo; +use super::capability::{CapabilityPurpose, CapabilitySet, ImportWriter, MigrationCapability}; use super::command::FenceRequest; #[cfg(test)] use super::hooks::FenceTestHooks; @@ -58,6 +59,11 @@ pub struct GateSnapshot { /// allows. Never persisted, and cleared by every publication of a commit. It does not move /// the write generation: write admission is already closed wherever a read fence can be set. pub closing_reads: Option, + /// The in-memory gate of a `CreateTargetQuarantined` that is being persisted (section + /// 10.1): the name is becoming a quarantined target, so everything but maintenance and + /// observability is refused with `MIGRATION_TARGET_QUARANTINED`, and the namespace is not + /// set up. Never persisted; replaced by the record the command's commit publishes. + pub creating_target: Option, } impl GateSnapshot { @@ -68,6 +74,7 @@ impl GateSnapshot { indeterminate: None, installing: None, closing_reads: None, + creating_target: None, } } @@ -104,6 +111,20 @@ impl GateSnapshot { .with_detail(FenceDetail::IndeterminateCommit)); } } + if let Some((operation_id, command_id)) = self.creating_target { + if !matches!( + class, + OperationClass::Maintenance | OperationClass::Observability + ) { + return Err(FenceError::new( + FenceOutcome::MigrationTargetQuarantined, + format!( + "{class:?} is not permitted: fence command {command_id} of operation \ + {operation_id} is creating this namespace as a migration target" + ), + )); + } + } self.fence.permits(class)?; if let Some((operation_id, command_id)) = self.installing { if matches!( @@ -141,6 +162,11 @@ impl GateSnapshot { self.installing.is_some() } + /// Whether the namespace is being created as a quarantined target. + pub fn is_creating_target(&self) -> bool { + self.creating_target.is_some() + } + /// Normal write admission. pub fn write(&self) -> Admission { Admission::from( @@ -287,6 +313,12 @@ pub struct FenceController { read_leases: Mutex, /// Notified whenever a read lease is released. read_released: Notify, + /// The migration capabilities issued and not revoked, and the import calls running under + /// them (sections 7.2 and 10.2). Lock order: this lock may be taken before borrowing the + /// gate, never while a gate borrow is held. + capabilities: Mutex, + /// Notified whenever an import call ends. + import_released: Notify, #[cfg(test)] hooks: FenceTestHooks, } @@ -312,6 +344,8 @@ impl FenceController { write_drains: Mutex::new(Vec::new()), read_leases: Mutex::new(ReadLeaseSet::default()), read_released: Notify::new(), + capabilities: Mutex::new(CapabilitySet::default()), + import_released: Notify::new(), #[cfg(test)] hooks: FenceTestHooks::default(), }) @@ -459,6 +493,176 @@ impl FenceController { self.read_released.notify_waiters(); } + /// Issue a migration capability for `purpose` to `operation_id` (section 11). The fence + /// must be in a state the purpose admits, owned by `operation_id`, at `expected_revision`, + /// with no command being installed or reconciled. The capability stays valid until the + /// fence moves on (every transition moves the revision) or it is revoked. + pub fn issue_capability( + &self, + purpose: CapabilityPurpose, + operation_id: Uuid, + expected_revision: u64, + ) -> Result { + let mut caps = self.capabilities.lock(); + let gate = self.gate.borrow(); + gate.permits(purpose.class())?; + let Some(record) = gate.fence.record() else { + return Err(FenceError::new( + FenceOutcome::OperationCapabilityRequired, + format!("namespace `{}` has no fence record", self.namespace), + )); + }; + if record.operation_id != operation_id { + return Err(FenceError::new( + FenceOutcome::FenceOwnedByAnotherOperation, + format!( + "the fence of namespace `{}` is owned by operation {}, not by operation \ + {operation_id}", + self.namespace, record.operation_id + ), + )); + } + if record.revision != expected_revision { + return Err(FenceError::new( + FenceOutcome::FenceRevisionMismatch, + format!( + "the fence of namespace `{}` is at revision {}, not at the expected revision \ + {expected_revision}", + self.namespace, record.revision + ), + )); + } + if !purpose.admits(record.state) { + return Err(FenceError::new( + FenceOutcome::OperationCapabilityRequired, + format!( + "namespace `{}` is in {}, which admits no new {} capability", + self.namespace, + record.state, + purpose.as_str() + ), + )); + } + let cap = MigrationCapability::issue( + self.namespace.clone(), + operation_id, + purpose, + record.revision, + ); + drop(gate); + caps.live.insert(cap.id(), cap.clone()); + tracing::debug!( + namespace = %self.namespace, + %operation_id, + capability = %cap.id(), + purpose = purpose.as_str(), + revision = cap.fence_revision(), + "issued migration capability" + ); + Ok(cap) + } + + /// Revoke a capability: nothing is admitted under it any more. + pub fn revoke_capability(&self, id: Uuid) { + self.capabilities.lock().live.remove(&id); + } + + /// Whether `id` was issued by this controller and is neither revoked nor invalidated by a + /// transition. + pub fn capability_is_live(&self, id: Uuid) -> bool { + self.capabilities.lock().live.contains_key(&id) + } + + /// Check that `cap` is the live server-issued capability of `purpose` for this namespace + /// and still matches the published fence. Validation sessions use this before every call; + /// import sessions perform the same checks while also incrementing their writer count. + pub(crate) fn check_capability( + &self, + cap: &MigrationCapability, + purpose: CapabilityPurpose, + ) -> Result<(), FenceError> { + if cap.namespace() != &self.namespace || cap.purpose() != purpose { + return Err(FenceError::new( + FenceOutcome::OperationCapabilityRequired, + format!( + "a {} capability for namespace `{}` does not admit {} work on `{}`", + cap.purpose().as_str(), + cap.namespace(), + purpose.as_str(), + self.namespace + ), + )); + } + let caps = self.capabilities.lock(); + { + let gate = self.gate.borrow(); + gate.permits(purpose.class())?; + cap.check(&gate.fence)?; + } + if !caps.live.contains_key(&cap.id()) { + return Err(revoked(cap)); + } + Ok(()) + } + + /// Admit one import call under `cap`, counted until the returned guard is dropped. The + /// capability is checked against the gate under the capability lock, so an import call is + /// either refused by a seal that closed admission before it, or counted by the seal, which + /// reads the count only after closing admission (section 10.2). + pub fn begin_import_write( + self: &Arc, + cap: &MigrationCapability, + ) -> Result { + if cap.namespace() != &self.namespace || cap.purpose() != CapabilityPurpose::Import { + return Err(FenceError::new( + FenceOutcome::OperationCapabilityRequired, + format!( + "a {} capability for namespace `{}` does not admit imports into `{}`", + cap.purpose().as_str(), + cap.namespace(), + self.namespace + ), + )); + } + let mut caps = self.capabilities.lock(); + { + let gate = self.gate.borrow(); + gate.permits(OperationClass::CapabilityImport)?; + cap.check(&gate.fence)?; + } + if !caps.live.contains_key(&cap.id()) { + return Err(revoked(cap)); + } + caps.import_writers += 1; + Ok(ImportWriter { + controller: self.clone(), + }) + } + + pub(super) fn end_import_write(&self) { + { + let mut caps = self.capabilities.lock(); + caps.import_writers = caps.import_writers.saturating_sub(1); + } + self.import_released.notify_waiters(); + } + + /// The import calls running now. + pub fn import_writers(&self) -> usize { + self.capabilities.lock().import_writers + } + + /// The capabilities issued and still live. + pub fn live_capabilities(&self) -> usize { + self.capabilities.lock().live.len() + } + + /// Notified whenever an import call ends. Enable the notification before checking + /// [`import_writers`](Self::import_writers), so an end in between is not missed. + pub(crate) fn import_released(&self) -> &Notify { + &self.import_released + } + /// Take the namespace's transition lock. Every fence command on the namespace runs while /// holding it, from its first check to its response. pub async fn begin_transition(self: &Arc) -> Transition { @@ -507,15 +711,33 @@ impl FenceController { } /// Publish a new gate. `fence: None` keeps the published fence. The write generation moves - /// whenever the state, the owning operation, the indeterminate flag or the installing gate - /// changes. Every publication removes the read-closing gate: the commit that follows it - /// either persists the read fence or proves that nothing changed. + /// whenever the state, the owning operation, the indeterminate flag, the installing gate or + /// the target-creation gate changes. Every publication removes the read-closing gate: the + /// commit that follows it either persists the read fence or proves that nothing changed. + /// A publication with a fence also removes the target-creation gate, which the committed + /// record replaces; one without keeps it. fn publish( &self, fence: Option, indeterminate: Option, installing: Option, ) { + let creating_target = if fence.is_some() { + None + } else { + self.gate.borrow().creating_target + }; + self.publish_gate(fence, indeterminate, installing, creating_target); + } + + fn publish_gate( + &self, + fence: Option, + indeterminate: Option, + installing: Option, + creating_target: Option, + ) { + let fence_published = fence.is_some(); let mut generation_changed = false; self.gate.send_modify(|gate| { let fence = fence.unwrap_or_else(|| gate.fence.clone()); @@ -523,10 +745,12 @@ impl FenceController { || fence.record().map(|r| r.operation_id) != gate.fence.record().map(|r| r.operation_id) || indeterminate != gate.indeterminate - || installing != gate.installing; + || installing != gate.installing + || creating_target != gate.creating_target; gate.fence = fence; gate.indeterminate = indeterminate; gate.installing = installing; + gate.creating_target = creating_target; gate.closing_reads = None; if changed { gate.write_generation += 1; @@ -542,9 +766,20 @@ impl FenceController { write_generation = gate.write_generation, indeterminate = gate.indeterminate.is_some(), installing = gate.installing.is_some(), + creating_target = gate.creating_target.is_some(), "published namespace fence gate" ); } + // A published transition invalidates every capability issued against an earlier state, + // owner or revision. Cloned first: the capability lock is never taken under a gate + // borrow. + if fence_published { + let fence = self.gate.borrow().fence.clone(); + self.capabilities + .lock() + .live + .retain(|_, cap| cap.matches(&fence)); + } // After the gate is published, so that every woken writer re-checks against it. if generation_changed { self.write_queues.lock().retain(|wake| wake()); @@ -586,6 +821,33 @@ impl Transition { } } + /// Publish the in-memory target-creation gate for `CreateTargetQuarantined` `key` (section + /// 10.1): until the command's commit publishes the quarantined record, every class but + /// maintenance and observability is refused and the namespace is not set up. It is removed + /// with [`remove_creating_target`](Self::remove_creating_target) when the command is proven + /// not to have committed, and kept (with the indeterminate flag) when its outcome is + /// unknown. + pub fn install_creating_target(&mut self, key: CommandKey) { + let (indeterminate, installing) = { + let gate = self.controller.gate.borrow(); + (gate.indeterminate, gate.installing) + }; + self.controller + .publish_gate(None, indeterminate, installing, Some(key)); + } + + /// Remove the target-creation gate of a command that was proven not to have committed. + pub fn remove_creating_target(&mut self) { + let (indeterminate, installing, creating) = { + let gate = self.controller.gate.borrow(); + (gate.indeterminate, gate.installing, gate.creating_target) + }; + if creating.is_some() { + self.controller + .publish_gate(None, indeterminate, installing, None); + } + } + /// Publish the in-memory read-closing gate for `SetSourceReadFence` `key` (section 9, /// step 2): new SQL programs, dumps, replication calls and ATTACHes of the namespace are /// refused with `MIGRATION_READ_FENCED`. A read lease is only ever taken after checking the @@ -733,6 +995,20 @@ fn indeterminate(key: CommandKey, why: &str) -> FenceError { .with_detail(FenceDetail::IndeterminateCommit) } +fn revoked(cap: &MigrationCapability) -> FenceError { + FenceError::new( + FenceOutcome::OperationCapabilityRequired, + format!( + "the {} capability {} of operation {} on namespace `{}` was revoked or was never \ + issued by this server", + cap.purpose().as_str(), + cap.id(), + cap.operation_id(), + cap.namespace() + ), + ) +} + fn pending_indeterminate((operation_id, command_id): CommandKey) -> FenceError { FenceError::new( FenceOutcome::FenceCommitIndeterminate, @@ -785,6 +1061,9 @@ impl Drop for ProgramReadLease { pub struct FenceConnState { controller: Arc, class: OperationClass, + /// The migration capability this connection works under, fixed at construction. Only a + /// connection opened for an import or validation session has one. + capability: Option, /// The write generation the current program was admitted under. program_generation: AtomicU64, /// The write generation the current read transaction was opened under. @@ -803,10 +1082,30 @@ pub struct FenceConnState { impl FenceConnState { pub fn new(controller: Arc, class: OperationClass) -> Arc { + Self::build(controller, class, None) + } + + /// The fence state of a connection that works under `capability` (an import or a + /// validation session): it is admitted as the capability's class, and only while the + /// capability is valid. + pub fn with_capability( + controller: Arc, + capability: MigrationCapability, + ) -> Arc { + let class = capability.class(); + Self::build(controller, class, Some(capability)) + } + + fn build( + controller: Arc, + class: OperationClass, + capability: Option, + ) -> Arc { let generation = controller.write_generation(); Arc::new(Self { controller, class, + capability, program_generation: AtomicU64::new(generation), txn_generation: AtomicU64::new(generation), denial: Mutex::new(None), @@ -902,6 +1201,10 @@ impl FenceConnState { self.class } + pub fn capability(&self) -> Option<&MigrationCapability> { + self.capability.as_ref() + } + pub fn program_generation(&self) -> u64 { self.program_generation.load(Ordering::Acquire) } @@ -928,32 +1231,19 @@ impl FenceConnState { } /// The authoritative write admission (section 8.1, check 2): the live gate permits this - /// connection's class, and the program and its read transaction were both admitted under - /// the gate's current write generation. On refusal the typed outcome is left in the denial - /// slot and returned. + /// connection's class, the program and its read transaction were both admitted under the + /// gate's current write generation, and a capability connection's capability is still the + /// valid one (the fence's state, owner and revision are the ones it was issued at, and it + /// is live). A validation connection never writes. On refusal the typed outcome is left in + /// the denial slot and returned. pub fn admit_write(&self) -> Result<(), FenceError> { - let result = { - let gate = self.controller.gate.borrow(); - gate.permits(self.class).and_then(|()| { - let current = gate.write_generation; - let program = self.program_generation(); - let txn = self.txn_generation(); - if program == current && txn == current { - Ok(()) - } else { - Err(FenceError::new( - FenceOutcome::MigrationWriteFenced, - format!( - "the namespace fence changed after this transaction began \ - (program admitted at generation {program}, transaction opened at \ - generation {txn}, current generation {current}); roll back and \ - begin a new transaction" - ), - ) - .with_detail(FenceDetail::StaleTransaction)) - } - }) - }; + let result = self + .admit_write_under_gate() + .and_then(|()| match &self.capability { + // Outside the gate borrow: the capability lock is never taken under one. + Some(cap) if !self.controller.capability_is_live(cap.id()) => Err(revoked(cap)), + _ => Ok(()), + }); if let Err(e) = &result { tracing::debug!( namespace = %self.controller.namespace, @@ -965,6 +1255,37 @@ impl FenceConnState { result } + fn admit_write_under_gate(&self) -> Result<(), FenceError> { + let gate = self.controller.gate.borrow(); + gate.permits(self.class)?; + let current = gate.write_generation; + let program = self.program_generation(); + let txn = self.txn_generation(); + if program != current || txn != current { + return Err(FenceError::new( + FenceOutcome::MigrationWriteFenced, + format!( + "the namespace fence changed after this transaction began (program admitted \ + at generation {program}, transaction opened at generation {txn}, current \ + generation {current}); roll back and begin a new transaction" + ), + ) + .with_detail(FenceDetail::StaleTransaction)); + } + match (self.class, &self.capability) { + (OperationClass::CapabilityValidate, _) => Err(FenceError::new( + FenceOutcome::OperationCapabilityRequired, + "a validation capability admits reads only", + )), + (_, Some(cap)) => cap.check(&gate.fence), + (OperationClass::CapabilityImport, None) => Err(FenceError::new( + FenceOutcome::OperationCapabilityRequired, + "an import write needs a migration capability", + )), + _ => Ok(()), + } + } + /// Take the typed outcome of the last refusal at the WAL, if any. pub fn take_denial(&self) -> Option { self.denial.lock().take() diff --git a/libsql-server/src/namespace/fence/drain.rs b/libsql-server/src/namespace/fence/drain.rs index 3e5c8ba5b8..4c93334b10 100644 --- a/libsql-server/src/namespace/fence/drain.rs +++ b/libsql-server/src/namespace/fence/drain.rs @@ -14,7 +14,7 @@ use std::time::Duration; use tokio::time::Instant; -use crate::namespace::meta_store::{FenceCommit, FenceContext, MetaStore}; +use crate::namespace::meta_store::{FenceCommit, FenceCommitKind, FenceContext, MetaStore}; use super::command::{DrainPolicy, FenceCommand, FenceRequest, OnDeadline}; use super::controller::{FenceController, LiveWriteDrain, Transition}; @@ -59,6 +59,9 @@ impl FenceController { FenceCommand::SetSourceReadFence { .. } => { super::read::set_source_read_fence(&mut transition, &meta, request, ctx).await } + FenceCommand::SealTargetImport { .. } => { + super::import::seal_target_import(&mut transition, &meta, request, ctx).await + } _ => transition.apply(&meta, request, ctx).await, } }) @@ -111,10 +114,13 @@ pub async fn acquire_source_write_fence( return Err(e); } }; - if commit.receipt.outcome != FenceOutcome::Draining { - // A replay of a finished acquisition, or ALREADY_APPLIED. + if commit.kind == FenceCommitKind::Replayed || commit.receipt.outcome != FenceOutcome::Draining + { + // A replay whose drain was superseded, a replay of a finished acquisition, or + // ALREADY_APPLIED. return Ok(commit); } + let resumed = commit.kind == FenceCommitKind::Resumed; let drain_key = (commit.receipt.operation_id, commit.receipt.command_id); // Steps 5 and 6. @@ -135,14 +141,18 @@ pub async fn acquire_source_write_fence( ); } ctx.now_ms = now_ms(); - transition + let mut completed = transition .complete_drain( meta, drain_key, DrainCompletion::SourceWrites { boundary }, ctx, ) - .await + .await?; + if resumed { + completed.kind = FenceCommitKind::Resumed; + } + Ok(completed) } /// Wait until no connection manager of the namespace has a writer holding its write slot, and @@ -234,7 +244,7 @@ async fn drain_writers( /// /// With write admission closed a manager that has been seen without a writer stays without one /// (only checkpoints can take the slot), so the managers are waited for one after the other. -async fn wait_for_writers(sources: &[LiveWriteDrain], deadline: Instant) -> bool { +pub(super) async fn wait_for_writers(sources: &[LiveWriteDrain], deadline: Instant) -> bool { for source in sources { loop { let released = source.manager.released().notified(); @@ -682,6 +692,7 @@ pub(crate) mod tests { raw(&holder, "commit").await.unwrap(); let committed = s.frame_no(); let done = s.execute(s.acquire(OP, 1, policy)).await.unwrap().unwrap(); + assert_eq!(done.kind, FenceCommitKind::Resumed); assert_eq!(done.receipt.outcome, FenceOutcome::Applied); assert_eq!(done.receipt.command_id, Uuid::from_u128(1)); assert_eq!( @@ -700,6 +711,46 @@ pub(crate) mod tests { assert_eq!(boundary(&again), boundary(&done)); } + /// A drain that was explicitly rolled back by `ReleaseSourceWriteFence` is historical: an + /// exact replay returns its stored `DRAINING` receipt without trying to complete that drain + /// against the newer `RELEASED` record. + #[tokio::test(flavor = "multi_thread")] + async fn replay_of_released_drain_does_not_resume_it() { + let s = Source::new().await; + let holder = s.conn().await; + raw(&holder, "begin immediate; insert into t values (1);") + .await + .unwrap(); + let policy = DrainPolicy { + deadline_ms: 0, + on_deadline: OnDeadline::Fail, + }; + let acquire = s.acquire(OP, 1, policy); + let first = s.execute(acquire.clone()).await.unwrap().unwrap(); + assert_eq!(first.receipt.outcome, FenceOutcome::Draining); + + let release = FenceRequest { + namespace: "ns".into(), + operation_id: OP, + command_id: Uuid::from_u128(2), + expected_state: FenceState::SourceDraining, + expected_revision: s.fence.gate().revision(), + command: FenceCommand::ReleaseSourceWriteFence, + }; + let released = s.execute(release).await.unwrap().unwrap(); + assert_eq!(released.receipt.outcome, FenceOutcome::Applied); + assert_eq!(s.fence.gate().state(), FenceState::Released); + raw(&holder, "commit").await.unwrap(); + + let replay = s.execute(acquire).await.unwrap().unwrap(); + assert_eq!(replay.kind, FenceCommitKind::Replayed); + assert_eq!(replay.receipt.outcome, FenceOutcome::Draining); + assert_eq!(s.fence.gate().state(), FenceState::Released); + raw(&s.conn().await, "insert into t values (2)") + .await + .unwrap(); + } + /// Releasing the write fence commits, publishes a new write generation and only then /// answers: new programs write again, and a transaction that began under the fence cannot. #[tokio::test(flavor = "multi_thread")] diff --git a/libsql-server/src/namespace/fence/import.rs b/libsql-server/src/namespace/fence/import.rs new file mode 100644 index 0000000000..54fea99606 --- /dev/null +++ b/libsql-server/src/namespace/fence/import.rs @@ -0,0 +1,845 @@ +//! Import sessions into quarantined migration targets, and the seal that ends them +//! (`docs/NAMESPACE_FENCE.md` sections 10.2 and 11). +//! +//! An [`ImportSession`] is the only way to write into a `TARGET_QUARANTINED` target. It holds a +//! [`MigrationCapability`] issued to the operation that owns the target and a connection whose +//! fence state carries that capability, so the WAL admits its write transactions as +//! `CapabilityImport` for as long as the capability is valid, and refuses every other +//! connection's. Each call into the session counts as an import writer until it returns. +//! +//! `SealTargetImport` ends import for good. It closes import admission in memory, persists +//! `TARGET_IMPORT_DRAINING` (the revision moves, so every issued import capability is +//! invalidated and no new one can be issued), then waits for the running import calls and for +//! any import transaction still holding the write slot to end, and persists +//! `TARGET_VALIDATING`. Like the source write drain, it waits on release notifications and +//! never takes elapsed time as evidence that a writer has finished. + +use std::sync::Arc; +use std::time::Duration; + +use bytes::Bytes; +use futures::Stream; +use tokio::time::Instant; + +use crate::connection::legacy::LegacyConnection; +use crate::connection::Connection as _; +use crate::error::Error; +use crate::namespace::configurator::{load_dump_sql, read_dump}; +use crate::namespace::meta_store::{FenceCommit, FenceCommitKind, FenceContext, MetaStore}; +use crate::namespace::replication_wal::ReplicationWalWrapper; + +use super::capability::MigrationCapability; +use super::command::{DrainPolicy, FenceCommand, FenceRequest, OnDeadline}; +use super::controller::{FenceController, LiveWriteDrain, Transition}; +use super::drain::{now_ms, wait_for_writers, FORCED_ROLLBACK_GRACE}; +use super::hooks::HookPoint; +use super::outcome::{FenceError, FenceOutcome}; +use super::state::FenceState; +use super::transition::DrainCompletion; + +/// An operation's write access to its quarantined target (section 11): a capability and a +/// connection that works under it. Dropping the session revokes the capability and closes the +/// connection, which rolls back a transaction the session left open. +pub struct ImportSession { + capability: MigrationCapability, + controller: Arc, + conn: LegacyConnection, +} + +impl std::fmt::Debug for ImportSession { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("ImportSession") + .field("capability", &self.capability) + .finish_non_exhaustive() + } +} + +impl ImportSession { + pub(crate) fn new( + capability: MigrationCapability, + controller: Arc, + conn: LegacyConnection, + ) -> Self { + Self { + capability, + controller, + conn, + } + } + + pub fn capability(&self) -> &MigrationCapability { + &self.capability + } + + /// Run `f` with the session's raw connection. The call is refused up front when the + /// capability is no longer valid (the target was sealed or aborted, or its fence moved on), + /// and counts as an import writer until `f` returns, so a seal waits for it. Inside `f` the + /// WAL admits a write transaction only while the capability is still valid: a write that + /// the fence refused there is reported as that refusal, whatever `f` made of the + /// `SQLITE_AUTH` it saw. + pub async fn with_raw( + &mut self, + f: impl FnOnce(&mut rusqlite::Connection) -> R + Send + 'static, + ) -> Result { + let writer = self.controller.begin_import_write(&self.capability)?; + let conn = self.conn.clone(); + let joined = tokio::task::spawn_blocking(move || { + let _writer = writer; + let result = conn.with_raw(f); + (result, conn.fence_state().take_denial()) + }) + .await; + match joined { + Ok((_, Some(denial))) => Err(denial), + Ok((result, None)) => Ok(result), + Err(e) if e.is_panic() => std::panic::resume_unwind(e.into_panic()), + Err(e) => Err(FenceError::new( + FenceOutcome::OperationCapabilityRequired, + format!("the import call did not complete: {e}"), + )), + } + } + + /// Load a SQL dump into the target with the server's dump loader, under the capability. + /// The dump must run in one transaction that it commits itself, as for a namespace + /// created from a dump. Fence refusals are [`Error::NamespaceFence`]; dump errors are + /// [`Error::LoadDumpError`]. + pub async fn load_dump(&mut self, dump: S) -> crate::Result<()> + where + S: Stream> + Unpin, + { + let content = read_dump(dump).await?; + self.with_raw(move |conn| load_dump_sql(&content, conn)) + .await??; + Ok(()) + } +} + +impl Drop for ImportSession { + fn drop(&mut self) { + self.controller.revoke_capability(self.capability.id()); + } +} + +/// `SealTargetImport` (section 10.2), under `transition`. +/// +/// Returns the `APPLIED` commit of `TARGET_VALIDATING` once no import call is running and no +/// import transaction holds the write slot, the `DRAINING` commit of `TARGET_IMPORT_DRAINING` +/// when the deadline passes first (import stays closed, and only a replay of the same command +/// resumes the drain), or the stored result of a replay. +pub async fn seal_target_import( + transition: &mut Transition, + meta: &MetaStore, + request: FenceRequest, + mut ctx: FenceContext, +) -> crate::Result { + let controller = transition.controller().clone(); + let key = (request.operation_id, request.command_id); + let policy = match &request.command { + FenceCommand::SealTargetImport { drain_policy } => { + drain_policy.unwrap_or_else(|| meta.fence_default_write_drain()) + } + _ => return transition.apply(meta, request, ctx).await, + }; + + // Close import admission in memory before persisting: the INSTALLING gate refuses new + // import calls and import write transactions, and moving the write generation makes every + // transaction opened before it stale and wakes the queued import writers. Only for a seal + // that can apply (the owner, at the current revision, of a quarantined target), so that a + // refused command does not disturb a running import. + let gate = controller.gate(); + let can_apply = gate.state() == FenceState::TargetQuarantined + && gate.indeterminate.is_none() + && !gate.is_installing() + && gate.operation_id() == Some(request.operation_id) + && gate.revision() == request.expected_revision; + if can_apply { + transition.install_closing_gate(key); + let _ = controller.hook(HookPoint::AfterInstallingGate).await; + } + + // Persist TARGET_IMPORT_DRAINING. Its publication replaces the INSTALLING gate and, the + // revision having moved, drops every issued import capability. + let commit = match transition.apply(meta, request, ctx.clone()).await { + Ok(commit) => commit, + Err(e) => { + transition.remove_installing(); + return Err(e); + } + }; + if commit.kind == FenceCommitKind::Replayed || commit.receipt.outcome != FenceOutcome::Draining + { + return Ok(commit); + } + let resumed = commit.kind == FenceCommitKind::Resumed; + let drain_key = (commit.receipt.operation_id, commit.receipt.command_id); + + if !drain_import_writers(&controller, policy).await { + return Ok(commit); + } + + ctx.now_ms = now_ms(); + let mut completed = transition + .complete_drain(meta, drain_key, DrainCompletion::TargetImport, ctx) + .await?; + if resumed { + completed.kind = FenceCommitKind::Resumed; + } + Ok(completed) +} + +/// Wait until no import call is running and no connection manager of the target has a writer +/// holding its write slot. At the deadline, `force_rollback` rolls back the import transactions +/// still holding the slot (an idle session's open transaction; a running call ends its own) +/// and waits again for the same deadline, but at least [`FORCED_ROLLBACK_GRACE`]. `false` when +/// the drain could not be proven within the policy. +async fn drain_import_writers(controller: &FenceController, policy: DrainPolicy) -> bool { + let namespace = controller.namespace().clone(); + let deadline_after = Duration::from_millis(policy.deadline_ms); + let mut deadline = Instant::now() + deadline_after; + let mut forced = false; + loop { + if wait_for_import_writers(controller, deadline).await { + // With import closed, a manager seen without a writer under its slot lock stays + // without one. A target with no loaded maker has no connection that could write. + let sources = controller.live_write_drains(); + if sources_have_no_writer(&sources) { + tracing::info!(%namespace, "import drain proven"); + return true; + } + continue; + } + match policy.on_deadline { + OnDeadline::ForceRollback if !forced => { + forced = true; + for source in controller.live_write_drains() { + let manager = source.manager.clone(); + // The rollback takes the connection's lock, which a running import call + // holds; the release it causes is what the drain keeps waiting for. + tokio::task::spawn_blocking(move || { + if let Some(id) = manager.abort_active() { + tracing::info!( + connection = id, + "import drain deadline passed; rolling back the active import \ + transaction" + ); + } + }); + } + deadline = Instant::now() + deadline_after.max(FORCED_ROLLBACK_GRACE); + } + _ => { + tracing::info!( + %namespace, + deadline_ms = policy.deadline_ms, + on_deadline = policy.on_deadline.as_str(), + forced, + import_writers = controller.import_writers(), + "import drain deadline passed with an import writer still active; \ + answering DRAINING" + ); + return false; + } + } + } +} + +/// Wait for the running import calls to end, then for every manager's write slot to be free of +/// writers. `false` when `deadline` passes first. +async fn wait_for_import_writers(controller: &FenceController, deadline: Instant) -> bool { + loop { + let released = controller.import_released().notified(); + tokio::pin!(released); + // Registered before the check, so an end in between is not missed. + released.as_mut().enable(); + if controller.import_writers() == 0 { + break; + } + tokio::select! { + _ = &mut released => {} + _ = tokio::time::sleep_until(deadline) => return false, + } + } + wait_for_writers(&controller.live_write_drains(), deadline).await +} + +fn sources_have_no_writer(sources: &[LiveWriteDrain]) -> bool { + sources + .iter() + .all(|source| source.manager.with_no_writer(|| ()).is_ok()) +} + +/// A refusal of `open_import_session` for a namespace that is not a primary. +pub(crate) fn not_importable(namespace: &crate::namespace::NamespaceName) -> Error { + FenceError::new( + FenceOutcome::FencePreconditionFailed, + format!("namespace `{namespace}` is not a primary database; it cannot be imported into"), + ) + .with_detail(super::outcome::FenceDetail::NotPrimary) + .into() +} + +#[cfg(test)] +mod tests { + use std::sync::Arc; + + use tempfile::{tempdir, TempDir}; + use uuid::Uuid; + + use super::*; + use crate::database::Connection; + use crate::namespace::fence::capability::CapabilityPurpose; + use crate::namespace::fence::drain::tests::{assert_fenced, raw, LONG, PROMPT}; + use crate::namespace::fence::state::OperationClass; + use crate::namespace::fence::target::tests::{create, create_request, server, OP}; + use crate::namespace::meta_store::FenceCommitKind; + use crate::namespace::store::fence_tests::open_store; + use crate::namespace::store::NamespaceStore; + use crate::namespace::{NamespaceName, RestoreOption}; + + const OTHER_OP: Uuid = Uuid::from_u128(0xb); + /// A deadline that has already passed. + const NOW: DrainPolicy = DrainPolicy { + deadline_ms: 0, + on_deadline: OnDeadline::Fail, + }; + + fn tgt() -> NamespaceName { + "tgt".into() + } + + /// A store holding the quarantined target `tgt` (revision 1), and its controller. + async fn target() -> (TempDir, NamespaceStore, Arc) { + let dir = tempdir().unwrap(); + let store = open_store(dir.path()).await; + create(&store, create_request("tgt", 1)) + .await + .unwrap() + .unwrap(); + let fence = store.with(tgt(), |ns| ns.fence().clone()).await.unwrap(); + assert_eq!(fence.gate().state(), FenceState::TargetQuarantined); + (dir, store, fence) + } + + fn seal_request(command_id: u128, expected_revision: u64, policy: DrainPolicy) -> FenceRequest { + FenceRequest { + namespace: tgt(), + operation_id: OP, + command_id: Uuid::from_u128(command_id), + expected_state: FenceState::TargetQuarantined, + expected_revision, + command: FenceCommand::SealTargetImport { + drain_policy: Some(policy), + }, + } + } + + /// Run `request` through the store, on a task of its own. + fn execute( + store: &NamespaceStore, + request: FenceRequest, + ) -> tokio::task::JoinHandle> { + let store = store.clone(); + tokio::spawn(async move { store.execute_fence_command(request, server()).await }) + } + + async fn until_state(fence: &FenceController, state: FenceState) { + let mut gate = fence.subscribe(); + tokio::time::timeout(PROMPT, gate.wait_for(|g| g.state() == state)) + .await + .expect("the fence reaches the state") + .unwrap(); + } + + /// A normal connection to the target (what the admin shell, a SQL request or a dump load + /// outside a capability would use). + async fn plain_conn(store: &NamespaceStore) -> Arc { + let maker = store + .with(tgt(), |ns| ns.db.connection_maker()) + .await + .unwrap(); + Arc::new(maker.create().await.unwrap()) + } + + /// Rows in `t` on the target, read through a raw connection. + async fn count(store: &NamespaceStore) -> i64 { + let conn = plain_conn(store).await; + tokio::task::spawn_blocking(move || { + conn.with_raw(|c| c.query_row("select count(*) from t", (), |r| r.get(0))) + }) + .await + .unwrap() + .unwrap() + } + + fn fence_err(r: crate::Result) -> FenceError { + match r { + Err(Error::NamespaceFence(e)) => e, + other => panic!("expected a fence error, got {other:?}"), + } + } + + /// Write through a connection that carries `capability`, and return the fence's refusal. + async fn refused_with(store: &NamespaceStore, capability: MigrationCapability) -> FenceError { + let conn = store + .capability_connection(&tgt(), capability) + .await + .unwrap(); + let r = conn.with_raw(|c| c.execute_batch("insert into t values (99)")); + assert_fenced(r); + conn.fence_state() + .take_denial() + .expect("the WAL gate records its refusal") + } + + /// Only the operation's own, server-issued capability at the current revision writes into + /// a quarantined target: plain connections (as an admin shell would use), capabilities of + /// another operation or revision, forged, revoked or validation capabilities are all + /// refused at the WAL, and issuing one is refused for another operation or revision. + #[tokio::test(flavor = "multi_thread")] + async fn import_requires_matching_capability() { + let (_dir, store, fence) = target().await; + let mut session = store.open_import_session(tgt(), OP, 1).await.unwrap(); + let cap = session.capability().clone(); + assert_eq!( + ( + cap.namespace(), + cap.operation_id(), + cap.purpose(), + cap.fence_revision() + ), + (&tgt(), OP, CapabilityPurpose::Import, 1) + ); + session + .with_raw(|c| c.execute_batch("create table t (x); insert into t values (1)")) + .await + .unwrap() + .unwrap(); + assert_eq!(fence.import_writers(), 0); + + // A normal connection, with or without raw access. + let conn = plain_conn(&store).await; + assert_fenced(raw(&conn, "insert into t values (2)").await); + + // Issuing: another operation, a stale or future revision. + let e = fence_err(store.open_import_session(tgt(), OTHER_OP, 1).await); + assert_eq!(e.outcome(), FenceOutcome::FenceOwnedByAnotherOperation); + for revision in [0, 2] { + let e = fence_err(store.open_import_session(tgt(), OP, revision).await); + assert_eq!(e.outcome(), FenceOutcome::FenceRevisionMismatch); + } + + // At the WAL: capabilities the server never issued, of this operation at this revision, + // of another operation, at another revision, or for validation. + let forged = + |op, purpose, revision| MigrationCapability::forged(tgt(), op, purpose, revision); + let cases = [ + ( + forged(OP, CapabilityPurpose::Import, 1), + FenceOutcome::OperationCapabilityRequired, + ), + ( + forged(OTHER_OP, CapabilityPurpose::Import, 1), + FenceOutcome::FenceOwnedByAnotherOperation, + ), + ( + forged(OP, CapabilityPurpose::Import, 2), + FenceOutcome::OperationCapabilityRequired, + ), + ( + forged(OP, CapabilityPurpose::Validate, 1), + FenceOutcome::OperationCapabilityRequired, + ), + ]; + for (capability, outcome) in cases { + let e = refused_with(&store, capability.clone()).await; + assert_eq!(e.outcome(), outcome, "{capability:?}: {e}"); + } + + // A capability that was issued, once its session is gone. + let other = store.open_import_session(tgt(), OP, 1).await.unwrap(); + let revoked = other.capability().clone(); + assert!(fence.capability_is_live(revoked.id())); + drop(other); + assert!(!fence.capability_is_live(revoked.id())); + let e = refused_with(&store, revoked).await; + assert_eq!(e.outcome(), FenceOutcome::OperationCapabilityRequired); + + // The live session still writes; nothing else did. + session + .with_raw(|c| c.execute_batch("insert into t values (3)")) + .await + .unwrap() + .unwrap(); + assert_eq!(count(&store).await, 2); + } + + /// The seal closes import at once and waits, on release notifications, for an import call + /// that was already running: its write transaction, which held the slot before the seal, + /// commits, and only then is `TARGET_VALIDATING` persisted. + #[tokio::test(flavor = "multi_thread")] + async fn seal_waits_for_import_writers() { + let (_dir, store, fence) = target().await; + let mut session = store.open_import_session(tgt(), OP, 1).await.unwrap(); + session + .with_raw(|c| c.execute_batch("create table t (x)")) + .await + .unwrap() + .unwrap(); + let mut idle = store.open_import_session(tgt(), OP, 1).await.unwrap(); + + let (entered_tx, entered) = tokio::sync::oneshot::channel(); + let (resume, resume_rx) = std::sync::mpsc::channel::<()>(); + let committed = Arc::new(std::sync::atomic::AtomicBool::new(false)); + let running = tokio::spawn({ + let committed = committed.clone(); + async move { + let r = session + .with_raw(move |c| { + c.execute_batch("begin immediate; insert into t values (1)")?; + entered_tx.send(()).unwrap(); + resume_rx.recv().unwrap(); + c.execute_batch("commit")?; + committed.store(true, std::sync::atomic::Ordering::SeqCst); + Ok::<_, rusqlite::Error>(()) + }) + .await; + (session, r) + } + }); + entered.await.unwrap(); + assert_eq!(fence.import_writers(), 1); + + let sealing = execute(&store, seal_request(10, 1, LONG)); + until_state(&fence, FenceState::TargetImportDraining).await; + assert_eq!(fence.gate().revision(), 2); + + // Import is closed for good: no new capability, and no new call on a live session. + let e = fence_err(store.open_import_session(tgt(), OP, 2).await); + assert_eq!(e.outcome(), FenceOutcome::OperationCapabilityRequired); + let e = idle + .with_raw(|c| c.execute_batch("insert into t values (2)")) + .await + .unwrap_err(); + assert_eq!(e.outcome(), FenceOutcome::OperationCapabilityRequired); + assert!(!sealing.is_finished()); + assert_eq!(fence.gate().state(), FenceState::TargetImportDraining); + + // The running import finishes; its transaction commits, and only then does the seal + // reach the commit of TARGET_VALIDATING. + let completing = fence.hooks().pause_at(HookPoint::BeforeMetastoreCommit); + resume.send(()).unwrap(); + tokio::time::timeout(PROMPT, completing.reached()) + .await + .expect("the seal completes once the import writer is gone"); + assert!( + committed.load(std::sync::atomic::Ordering::SeqCst), + "the seal proceeded while the import transaction was still running" + ); + completing.resume(); + let (mut session, r) = running.await.unwrap(); + r.unwrap().unwrap(); + let commit = sealing.await.unwrap().unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + assert_eq!( + (fence.gate().state(), fence.gate().revision()), + (FenceState::TargetValidating, 3) + ); + // The publication of TARGET_IMPORT_DRAINING dropped both sessions' capabilities. + assert_eq!(fence.live_capabilities(), 0); + assert_eq!(count(&store).await, 1); + let e = session + .with_raw(|c| c.execute_batch("insert into t values (3)")) + .await + .unwrap_err(); + assert_eq!(e.outcome(), FenceOutcome::OperationCapabilityRequired); + } + + /// An import transaction left open by an idle session holds the write slot: the seal does + /// not complete past its deadline (`on_deadline: fail`), `TARGET_IMPORT_DRAINING` stays + /// durable and closed (also across a restart), another operation cannot touch it, and only + /// the owner's seal resumes it: a replay of the same seal completes once the session is + /// gone (its transaction rolled back). + #[tokio::test(flavor = "multi_thread")] + async fn seal_deadline_leaves_import_draining_until_replayed() { + let (dir, store, fence) = target().await; + let mut session = store.open_import_session(tgt(), OP, 1).await.unwrap(); + session + .with_raw(|c| c.execute_batch("create table t (x)")) + .await + .unwrap() + .unwrap(); + session + .with_raw(|c| c.execute_batch("begin immediate; insert into t values (1)")) + .await + .unwrap() + .unwrap(); + assert_eq!(fence.import_writers(), 0); + + let commit = execute(&store, seal_request(10, 1, NOW)) + .await + .unwrap() + .unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Draining); + assert_eq!( + (fence.gate().state(), fence.gate().revision()), + (FenceState::TargetImportDraining, 2) + ); + + // Another operation's seal is refused; a new seal of the owner joins the drain, which + // still cannot complete. + let mut foreign = seal_request(11, 2, NOW); + foreign.operation_id = OTHER_OP; + foreign.expected_state = FenceState::TargetImportDraining; + let e = fence_err(execute(&store, foreign).await.unwrap()); + assert_eq!(e.outcome(), FenceOutcome::FenceOwnedByAnotherOperation); + let mut join = seal_request(12, 2, NOW); + join.expected_state = FenceState::TargetImportDraining; + let joined = execute(&store, join).await.unwrap().unwrap(); + assert_eq!(joined.receipt.outcome, FenceOutcome::Draining); + assert_eq!( + (fence.gate().state(), fence.gate().revision()), + (FenceState::TargetImportDraining, 2) + ); + + // Still draining after a restart, which drops the open transaction. + drop(session); + store.shutdown().await.unwrap(); + let store = open_store(dir.path()).await; + let fence = store.fence_controller(&tgt()); + assert_eq!(fence.gate().state(), FenceState::TargetImportDraining); + assert!(fence.permits(OperationClass::CapabilityImport).is_ok()); + let e = fence_err(store.open_import_session(tgt(), OP, 2).await); + assert_eq!(e.outcome(), FenceOutcome::OperationCapabilityRequired); + + let replay = execute(&store, seal_request(10, 1, NOW)) + .await + .unwrap() + .unwrap(); + assert_eq!(replay.kind, FenceCommitKind::Resumed); + assert_eq!(replay.receipt.outcome, FenceOutcome::Applied); + assert_eq!( + (fence.gate().state(), fence.gate().revision()), + (FenceState::TargetValidating, 3) + ); + assert_eq!(count(&store).await, 0); + let replay = execute(&store, seal_request(10, 1, NOW)) + .await + .unwrap() + .unwrap(); + assert_eq!(replay.kind, FenceCommitKind::Replayed); + } + + /// With `on_deadline: force_rollback` the seal rolls back an import transaction that an + /// idle session left holding the write slot, waits for the release, and completes. + #[tokio::test(flavor = "multi_thread")] + async fn seal_force_rollback_ends_open_import_transaction() { + let (_dir, store, fence) = target().await; + let mut session = store.open_import_session(tgt(), OP, 1).await.unwrap(); + session + .with_raw(|c| c.execute_batch("create table t (x)")) + .await + .unwrap() + .unwrap(); + session + .with_raw(|c| c.execute_batch("begin immediate; insert into t values (1)")) + .await + .unwrap() + .unwrap(); + let policy = DrainPolicy { + deadline_ms: 0, + on_deadline: OnDeadline::ForceRollback, + }; + let commit = execute(&store, seal_request(10, 1, policy)) + .await + .unwrap() + .unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + assert_eq!(fence.gate().state(), FenceState::TargetValidating); + drop(session); + assert_eq!(count(&store).await, 0); + } + + /// Once sealed, nothing imports: no capability is issued at the new revision, an older + /// session's calls are refused, and even a capability naming the current revision is + /// refused at the WAL. + #[tokio::test(flavor = "multi_thread")] + async fn sealed_target_rejects_import() { + let (_dir, store, fence) = target().await; + let mut session = store.open_import_session(tgt(), OP, 1).await.unwrap(); + session + .with_raw(|c| c.execute_batch("create table t (x)")) + .await + .unwrap() + .unwrap(); + let commit = execute(&store, seal_request(10, 1, LONG)) + .await + .unwrap() + .unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + assert_eq!( + (fence.gate().state(), fence.gate().revision()), + (FenceState::TargetValidating, 3) + ); + + for revision in [1, 3] { + let e = fence_err(store.open_import_session(tgt(), OP, revision).await); + assert_eq!(e.outcome(), FenceOutcome::OperationCapabilityRequired); + } + let e = session + .with_raw(|c| c.execute_batch("insert into t values (1)")) + .await + .unwrap_err(); + assert_eq!(e.outcome(), FenceOutcome::OperationCapabilityRequired); + let e = refused_with( + &store, + MigrationCapability::forged(tgt(), OP, CapabilityPurpose::Import, 3), + ) + .await; + assert_eq!(e.outcome(), FenceOutcome::OperationCapabilityRequired); + assert_fenced(raw(&plain_conn(&store).await, "insert into t values (1)").await); + assert_eq!(count(&store).await, 0); + } + + /// A synthetic representative schema: tables with keys and a foreign key, indexes, a + /// trigger, a view and an FTS5 table. + const SCHEMA: &str = " + create table users (id integer primary key, email text not null unique, name text); + create table orders ( + id integer primary key, + user_id integer not null references users(id), + total real not null, + note text + ); + create index orders_by_user on orders(user_id, total); + create table audit (id integer primary key autoincrement, what text); + create trigger orders_audit after insert on orders begin + insert into audit (what) values ('order ' || new.id); + end; + create view order_totals as + select u.email, sum(o.total) as total from users u join orders o on o.user_id = u.id + group by u.email; + create virtual table docs using fts5(title, body); + insert into users (email, name) values ('a@example.com', 'A'), ('b@example.com', 'B'); + insert into orders (user_id, total, note) values (1, 10.5, 'first'), (1, 2, null), + (2, 7.25, 'it''s quoted'); + insert into docs (title, body) values ('fence', 'operation owned namespace fence'), + ('import', 'quarantined target import'); + "; + + /// What a target must reproduce of the source: the schema, the rows, and the derived data + /// (the view, the full-text index). + fn contents(c: &rusqlite::Connection) -> Vec { + let mut out = Vec::new(); + let mut q = |sql: &str| { + use rusqlite::types::ValueRef; + let mut stmt = c.prepare(sql).unwrap(); + let n = stmt.column_count(); + let rows = stmt + .query_map((), |r| { + (0..n) + .map(|i| { + r.get_ref(i).map(|v| match v { + ValueRef::Text(t) => String::from_utf8_lossy(t).into_owned(), + other => format!("{other:?}"), + }) + }) + .collect::>>() + }) + .unwrap(); + for row in rows { + out.push(format!("{sql}: {}", row.unwrap().join(", "))); + } + }; + // The loader re-renders every statement it runs, so the stored SQL of an object differs + // from the source's in case and spacing only. + q("select type, name, tbl_name, \ + lower(replace(replace(replace(sql, ' ', ''), char(10), ''), '\"', '')) \ + from sqlite_schema order by type, name"); + q("select * from users order by id"); + q("select * from orders order by id"); + q("select * from audit order by id"); + q("select * from order_totals order by email"); + q("select title from docs where docs match 'quarantined' order by rowid"); + q("select count(*) from docs"); + out + } + + /// The server's own dump loader runs inside an import session: a dump exported from a + /// source with a representative schema loads into the quarantined target, which then holds + /// exactly what the source held, while nothing else could write to it. + #[tokio::test(flavor = "multi_thread")] + async fn import_session_loads_dump_into_quarantined_target() { + let (_dir, store, fence) = target().await; + store + .create("src".into(), RestoreOption::Latest, Default::default()) + .await + .unwrap(); + let src = { + let maker = store + .with("src".into(), |ns| ns.db.connection_maker()) + .await + .unwrap(); + Arc::new(maker.create().await.unwrap()) + }; + let (dump, expected) = { + let src = src.clone(); + tokio::task::spawn_blocking(move || { + src.with_raw(|c| { + c.execute_batch(SCHEMA).unwrap(); + let mut dump = Vec::new(); + crate::connection::dump::exporter::export_dump(c, &mut dump, false).unwrap(); + (dump, contents(c)) + }) + }) + .await + .unwrap() + }; + let text = String::from_utf8(dump.clone()).unwrap(); + assert!(text.contains("CREATE VIRTUAL TABLE"), "{text}"); + + let mut session = store.open_import_session(tgt(), OP, 1).await.unwrap(); + let stream = futures::stream::iter( + dump.chunks(64) + .map(|c| Ok(Bytes::copy_from_slice(c))) + .collect::>(), + ); + session.load_dump(stream).await.unwrap(); + assert_eq!(fence.gate().state(), FenceState::TargetQuarantined); + // Nothing but the session could have written it. + assert_fenced( + raw( + &plain_conn(&store).await, + "insert into audit (what) values ('x')", + ) + .await, + ); + + // A second load fails as the loader always does on a dump that is not in a + // transaction, without the fence being involved. + let r = session + .load_dump(futures::stream::iter(vec![Ok(Bytes::from_static( + b"savepoint a; release a; savepoint b;", + ))])) + .await; + assert!( + matches!( + r, + Err(Error::LoadDumpError(crate::error::LoadDumpError::NoTxn)) + ), + "{r:?}" + ); + drop(session); + + let commit = execute(&store, seal_request(10, 1, LONG)) + .await + .unwrap() + .unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + let conn = plain_conn(&store).await; + let imported = tokio::task::spawn_blocking(move || conn.with_raw(|c| contents(c))) + .await + .unwrap(); + assert_eq!(imported, expected); + } +} diff --git a/libsql-server/src/namespace/fence/mod.rs b/libsql-server/src/namespace/fence/mod.rs index 74f227cf23..a8d80da59c 100644 --- a/libsql-server/src/namespace/fence/mod.rs +++ b/libsql-server/src/namespace/fence/mod.rs @@ -10,18 +10,21 @@ //! ([`store`], driven by `MetaStore::apply_fence_command`), and the in-memory authority built //! on them: the per-namespace [`controller`] with its gate and read leases, the positive write //! [`drain`], the source [`read`] fence and its -//! [`stream`] leases for dump and replication, the [`registry`] that holds the controllers outside the -//! namespace cache, and the test [`hooks`] on their paths. +//! [`stream`] leases for dump and replication, quarantined migration [`target`]s with their +//! [`capability`]-scoped [`import`] sessions and seal drain, the [`registry`] that holds the controllers outside the namespace cache, and the test [`hooks`] +//! on their paths. // The persistence, controller and protocol layers that consume these types land in the // following commits of this series; until then most of the module is unused by the rest of // the crate. This attribute is removed once they are wired. #![allow(dead_code)] +pub mod capability; pub mod command; pub mod controller; pub mod drain; pub mod hooks; +pub mod import; pub mod outcome; pub mod read; pub mod record; @@ -29,6 +32,7 @@ pub mod registry; pub mod state; pub mod store; pub mod stream; +pub mod target; pub mod transition; #[cfg(test)] diff --git a/libsql-server/src/namespace/fence/read.rs b/libsql-server/src/namespace/fence/read.rs index 50feb0b3b1..95d7beb877 100644 --- a/libsql-server/src/namespace/fence/read.rs +++ b/libsql-server/src/namespace/fence/read.rs @@ -13,7 +13,7 @@ use std::time::Duration; use tokio::time::Instant; -use crate::namespace::meta_store::{FenceCommit, FenceContext, MetaStore}; +use crate::namespace::meta_store::{FenceCommit, FenceCommitKind, FenceContext, MetaStore}; use super::command::{DrainPolicy, FenceCommand, FenceRequest}; use super::controller::{FenceController, Transition}; @@ -67,9 +67,11 @@ pub async fn set_source_read_fence( return Err(e); } }; - if commit.receipt.outcome != FenceOutcome::Draining { + if commit.kind == FenceCommitKind::Replayed || commit.receipt.outcome != FenceOutcome::Draining + { return Ok(commit); } + let resumed = commit.kind == FenceCommitKind::Resumed; let drain_key = (commit.receipt.operation_id, commit.receipt.command_id); // Step 4. @@ -79,9 +81,13 @@ pub async fn set_source_read_fence( // Step 5. ctx.now_ms = now_ms(); - transition + let mut completed = transition .complete_drain(meta, drain_key, DrainCompletion::SourceReads, ctx) - .await + .await?; + if resumed { + completed.kind = FenceCommitKind::Resumed; + } + Ok(completed) } /// Wait until every read lease of the namespace is released. At the deadline the leases still @@ -420,6 +426,7 @@ pub(crate) mod tests { // The program was cancelled by the fence; it reports the fence, not its rows. read_fenced(&running.await.unwrap().unwrap_err()); let replayed = s.execute(request).await.unwrap(); + assert_eq!(replayed.as_ref().unwrap().kind, FenceCommitKind::Resumed); assert_eq!(fence_outcome(&replayed), FenceOutcome::Applied); assert_eq!(s.fence.gate().state(), FenceState::SourceReadFenced); } diff --git a/libsql-server/src/namespace/fence/record.rs b/libsql-server/src/namespace/fence/record.rs index 247b481ad2..a8baa25e60 100644 --- a/libsql-server/src/namespace/fence/record.rs +++ b/libsql-server/src/namespace/fence/record.rs @@ -118,12 +118,24 @@ impl NamespaceFenceRecord { self.state.read_admission() } + /// Whether the stored config's `block_*` fields hold the fence's mirror (section 13.2) + /// rather than the namespace's own values. They stop holding it when the operation releases + /// the namespace or enables target writes: that transition puts the saved values back, and + /// from then on config writes store the namespace's own values in the row again, so the row + /// is authoritative and `legacy_blocks` may be out of date. + pub fn mirrors_legacy_blocks(&self) -> bool { + !matches!( + self.state, + FenceState::Released | FenceState::TargetWritable + ) + } + /// Values of the legacy `block_*` configuration fields while this record is in force: the /// fence state mirrored for an older binary, or the pre-fence values once the operation /// has released the namespace. pub fn legacy_mirror(&self) -> LegacyBlocks { match self.state { - FenceState::Released | FenceState::TargetWritable => self.legacy_blocks.clone(), + _ if !self.mirrors_legacy_blocks() => self.legacy_blocks.clone(), state => LegacyBlocks { block_reads: !state.read_admission().is_open(), block_writes: !state.write_admission().is_open(), diff --git a/libsql-server/src/namespace/fence/registry.rs b/libsql-server/src/namespace/fence/registry.rs index 610189677b..7c03102cd1 100644 --- a/libsql-server/src/namespace/fence/registry.rs +++ b/libsql-server/src/namespace/fence/registry.rs @@ -66,13 +66,13 @@ impl FenceRegistry { self.controllers.lock().remove(namespace) } - /// Refuse a namespace whose fence state is `UNKNOWN_UNAVAILABLE`, before any work is done - /// to serve it. + /// Refuse a namespace whose fence state is `UNKNOWN_UNAVAILABLE`, or that is being created + /// as a quarantined target, before any work is done to serve it. pub fn check_available(&self, namespace: &NamespaceName) -> Result<(), FenceError> { match self.get(namespace) { Some(controller) => { let gate = controller.gate(); - if gate.is_unavailable() { + if gate.is_unavailable() || gate.is_creating_target() { gate.permits(OperationClass::NormalRead) } else { Ok(()) diff --git a/libsql-server/src/namespace/fence/store.rs b/libsql-server/src/namespace/fence/store.rs index bb7a92a20f..fc7e60e062 100644 --- a/libsql-server/src/namespace/fence/store.rs +++ b/libsql-server/src/namespace/fence/store.rs @@ -675,6 +675,18 @@ pub fn with_legacy_blocks(config: &DatabaseConfig, blocks: &LegacyBlocks) -> Dat } } +/// The namespace's own config, given its stored config row `stored` and its fence record: the +/// row with the record's saved `block_*` values in place of the mirror while the record +/// mirrors them, and the row itself once the operation has finished (section 13.2), since +/// config writes after a release or a write enable store the namespace's own values. +pub fn own_config(stored: &DatabaseConfig, record: &NamespaceFenceRecord) -> DatabaseConfig { + if record.mirrors_legacy_blocks() { + with_legacy_blocks(stored, &record.legacy_blocks) + } else { + stored.clone() + } +} + /// The `block_*` fields of `config`. pub fn legacy_blocks_of(config: &DatabaseConfig) -> LegacyBlocks { LegacyBlocks { diff --git a/libsql-server/src/namespace/fence/target.rs b/libsql-server/src/namespace/fence/target.rs new file mode 100644 index 0000000000..44006e40bd --- /dev/null +++ b/libsql-server/src/namespace/fence/target.rs @@ -0,0 +1,1108 @@ +//! Migration targets (`docs/NAMESPACE_FENCE.md` sections 10 and 11). +//! +//! A target namespace is created by the operation that will fill it, already in +//! `TARGET_QUARANTINED`, and it is quarantined from the first instant anything else could +//! observe it: the in-memory target-creation gate is installed before the metastore transaction +//! that writes its marker, config row, record and receipt; the committed record replaces that +//! gate; and only then is the config put where `exists()` and `lookup()` find it and the +//! namespace loaded, so its first connection maker is created behind the quarantine gate. +//! +//! [`CreateTargetRequest`] is the typed entry point that the admin route and bulk import both +//! use; `NamespaceStore::create_target_quarantined` runs it. `AbortQuarantinedTarget` needs +//! nothing of its own: it is an ordinary transition, and `TARGET_ABORTED` denies every +//! normal class exactly as the quarantine does. + +use uuid::Uuid; + +use crate::connection::legacy::LegacyConnection; +use crate::connection::Connection as _; +use crate::namespace::replication_wal::ReplicationWalWrapper; +use crate::namespace::NamespaceName; + +use super::capability::{CapabilityPurpose, MigrationCapability}; +use super::command::{FenceCommand, FenceRequest, TargetConfig}; +use super::controller::FenceController; +use super::outcome::{FenceDetail, FenceError, FenceOutcome}; +use super::record::ValidationSnapshot; +use super::state::FenceState; + +/// `CreateTargetQuarantined` for `namespace`, by `operation_id`. The expectation is always +/// `ABSENT` at revision 0, so it is not part of the request. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct CreateTargetRequest { + pub namespace: NamespaceName, + pub operation_id: Uuid, + /// Idempotency key: replaying the same command returns its stored result, and completes a + /// creation that was interrupted between its marker and its commit. + pub command_id: Uuid, + pub config: TargetConfig, +} + +impl From for FenceRequest { + fn from(req: CreateTargetRequest) -> Self { + FenceRequest { + namespace: req.namespace, + operation_id: req.operation_id, + command_id: req.command_id, + expected_state: FenceState::Absent, + expected_revision: 0, + command: FenceCommand::CreateTargetQuarantined { config: req.config }, + } + } +} + +/// Read-only access used by the owning operation to validate a sealed target. The connection +/// carries a server-issued validation capability and has SQLite's `query_only` mode enabled; +/// every call checks that the capability still matches the target's owner, state and revision. +pub struct ValidationSession { + capability: MigrationCapability, + controller: std::sync::Arc, + conn: LegacyConnection, +} + +impl std::fmt::Debug for ValidationSession { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("ValidationSession") + .field("capability", &self.capability) + .finish_non_exhaustive() + } +} + +impl ValidationSession { + pub(crate) async fn new( + capability: MigrationCapability, + controller: std::sync::Arc, + conn: LegacyConnection, + ) -> crate::Result { + let mut this = Self { + capability, + controller, + conn, + }; + this.with_raw(|conn| conn.pragma_update(None, "query_only", true)) + .await??; + Ok(this) + } + + pub fn capability(&self) -> &MigrationCapability { + &self.capability + } + + /// Run a read-only operation on the capability connection. The capability is checked before + /// the call, and a write refused at the WAL is returned as its typed fence outcome even if + /// the closure swallowed SQLite's `SQLITE_AUTH`. + pub async fn with_raw( + &mut self, + f: impl FnOnce(&mut rusqlite::Connection) -> R + Send + 'static, + ) -> Result { + self.controller + .check_capability(&self.capability, CapabilityPurpose::Validate)?; + let conn = self.conn.clone(); + let joined = tokio::task::spawn_blocking(move || { + let result = conn.with_raw(f); + (result, conn.fence_state().take_denial()) + }) + .await; + match joined { + Ok((_, Some(denial))) => Err(denial), + Ok((result, None)) => Ok(result), + Err(e) if e.is_panic() => std::panic::resume_unwind(e.into_panic()), + Err(e) => Err(FenceError::new( + FenceOutcome::OperationCapabilityRequired, + format!("the validation call did not complete: {e}"), + )), + } + } + + /// What the server records beside `RecordTargetValidation`: the target's current + /// replication-log identity and frame, and SQLite page count. Target writes have already + /// been positively drained, so these observations cannot race a mutation. + pub(crate) async fn snapshot(&mut self) -> crate::Result { + let sources = self.controller.live_write_drains(); + let Some(latest) = sources.last() else { + return Err(FenceError::new( + FenceOutcome::FenceStateUnavailable, + format!( + "validation target `{}` has no live primary replication log", + self.capability.namespace() + ), + ) + .into()); + }; + let log_id = latest.log_id; + let frame_no = (latest.current_frame_no)().unwrap_or(0); + let page_count = self + .with_raw(|conn| conn.query_row("PRAGMA page_count", (), |row| row.get::<_, u64>(0))) + .await??; + Ok(ValidationSnapshot { + log_id, + frame_no, + page_count, + }) + } +} + +impl Drop for ValidationSession { + fn drop(&mut self) { + self.controller.revoke_capability(self.capability.id()); + } +} + +/// The refusal of a target name that the server already knows, in memory or in the namespace +/// cache, although the metastore may not hold it yet (a create or fork in flight, or one the +/// fence refused after it had published its config in memory). +pub(crate) fn name_in_use(namespace: &NamespaceName) -> FenceError { + FenceError::new( + FenceOutcome::FencePreconditionFailed, + format!("namespace `{namespace}` already exists on this server"), + ) + .with_detail(FenceDetail::NamespaceExists) +} + +#[cfg(test)] +pub(crate) mod tests { + use std::sync::Arc; + + use libsql_replication::rpc::replication::replication_log_server::ReplicationLog; + use libsql_replication::rpc::replication::{HelloRequest, NAMESPACE_METADATA_KEY}; + use tempfile::{tempdir, TempDir}; + use tonic::metadata::BinaryMetadataValue; + + use super::*; + use crate::auth::Authenticated; + use crate::connection::config::DatabaseConfig; + use crate::connection::program::Program; + use crate::connection::RequestContext; + use crate::error::Error; + use crate::namespace::fence::command::{FenceCommand, ValidationResult}; + use crate::namespace::fence::controller::FenceController; + use crate::namespace::fence::drain::tests::{raw, PROMPT}; + use crate::namespace::fence::hooks::{HookAction, HookPoint}; + use crate::namespace::fence::record::ServerIdentity; + use crate::namespace::fence::state::OperationClass; + use crate::namespace::meta_store::{metastore_connection_maker, FenceCommit, FenceCommitKind}; + use crate::namespace::store::fence_tests::open_store; + use crate::namespace::store::NamespaceStore; + use crate::namespace::RestoreOption; + use crate::query_result_builder::test::TestBuilder; + use crate::query_result_builder::QueryResultBuilder as _; + use crate::rpc::replication::replication_log::ReplicationLogService; + + pub(crate) const OP: Uuid = Uuid::from_u128(0xa); + const OTHER_OP: Uuid = Uuid::from_u128(0xb); + + pub(crate) fn server() -> ServerIdentity { + ServerIdentity { + build: "test".into(), + instance_id: Uuid::from_u128(0x99), + } + } + + pub(crate) fn create_request(ns: &'static str, command_id: u128) -> CreateTargetRequest { + CreateTargetRequest { + namespace: ns.into(), + operation_id: OP, + command_id: Uuid::from_u128(command_id), + config: TargetConfig { + max_db_size: Some(4096 * 1000), + ..Default::default() + }, + } + } + + /// Run `CreateTargetQuarantined` through the store on a task of its own. + pub(crate) fn create( + store: &NamespaceStore, + req: CreateTargetRequest, + ) -> tokio::task::JoinHandle> { + let store = store.clone(); + tokio::spawn(async move { store.create_target_quarantined(req, server()).await }) + } + + pub(crate) fn target_command( + command_id: u128, + expected_state: FenceState, + expected_revision: u64, + command: FenceCommand, + ) -> FenceRequest { + FenceRequest { + namespace: "tgt".into(), + operation_id: OP, + command_id: Uuid::from_u128(command_id), + expected_state, + expected_revision, + command, + } + } + + fn execute( + store: &NamespaceStore, + request: FenceRequest, + ) -> tokio::task::JoinHandle> { + let store = store.clone(); + tokio::spawn(async move { store.execute_fence_command(request, server()).await }) + } + + /// A target with a small table, sealed at TARGET_VALIDATING revision 3. + async fn validating_target() -> (TempDir, NamespaceStore, Arc) { + let dir = tempdir().unwrap(); + let store = open_store(dir.path()).await; + create(&store, create_request("tgt", 1)) + .await + .unwrap() + .unwrap(); + let mut session = store + .open_import_session("tgt".into(), OP, 1) + .await + .unwrap(); + session + .with_raw(|conn| { + conn.execute_batch("create table t (x); insert into t values (1), (2)") + }) + .await + .unwrap() + .unwrap(); + drop(session); + let seal = target_command( + 10, + FenceState::TargetQuarantined, + 1, + FenceCommand::SealTargetImport { drain_policy: None }, + ); + let commit = execute(&store, seal).await.unwrap().unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + let fence = controller(&store, "tgt").await; + assert_eq!( + (fence.gate().state(), fence.gate().revision()), + (FenceState::TargetValidating, 3) + ); + (dir, store, fence) + } + + async fn record_validation( + store: &NamespaceStore, + command_id: u128, + expected_revision: u64, + result: ValidationResult, + ) -> FenceCommit { + execute( + store, + target_command( + command_id, + FenceState::TargetValidating, + expected_revision, + FenceCommand::RecordTargetValidation { + result, + summary: format!("validation {result:?}"), + }, + ), + ) + .await + .unwrap() + .unwrap() + } + + /// A target with successful validation, published readable and write-fenced at revision 5. + async fn write_fenced_target() -> (TempDir, NamespaceStore, Arc) { + let (dir, store, fence) = validating_target().await; + record_validation(&store, 20, 3, ValidationResult::Ok).await; + let publish = target_command( + 21, + FenceState::TargetValidating, + 4, + FenceCommand::PublishTargetReadableWriteFenced, + ); + execute(&store, publish).await.unwrap().unwrap(); + assert_eq!( + (fence.gate().state(), fence.gate().revision()), + (FenceState::TargetWriteFenced, 5) + ); + (dir, store, fence) + } + + pub(crate) fn enable_request(command_id: u128) -> FenceRequest { + target_command( + command_id, + FenceState::TargetWriteFenced, + 5, + FenceCommand::EnableTargetWrites, + ) + } + + async fn count_rows(store: &NamespaceStore) -> i64 { + let (_, conn) = loaded(store, "tgt").await; + tokio::task::spawn_blocking(move || { + conn.with_raw(|c| c.query_row("select count(*) from t", (), |row| row.get(0))) + }) + .await + .unwrap() + .unwrap() + } + + /// Run one normal SQL program, including its legacy config checks, and require every step to + /// succeed. Raw access is intentionally not used for restart mirror assertions. + async fn program( + store: &NamespaceStore, + conn: &Arc, + sql: &'static str, + ) { + let ctx = RequestContext::new( + Authenticated::FullAccess, + "tgt".into(), + store.meta_store().clone(), + ); + let steps = conn + .execute_program(Program::seq(&[sql]), ctx, TestBuilder::default(), None) + .await + .unwrap() + .into_ret(); + for (i, step) in steps.iter().enumerate() { + assert!(step.is_ok(), "step {i} failed: {step:?}"); + } + } + + fn fence_error(e: &Error) -> &FenceError { + match e { + Error::NamespaceFence(f) => f, + other => panic!("expected a fence error, got {other:?}"), + } + } + + /// Denied by the fence, or not there at all: never served. + fn assert_not_served(what: &str, r: &crate::Result) { + match r { + Err(Error::NamespaceDoesntExist(_)) => (), + Err(Error::NamespaceFence(_)) => (), + other => panic!("{what}: expected a denial, got {other:?}"), + } + } + + fn assert_quarantined(fence: &FenceController) { + for class in [ + OperationClass::NormalRead, + OperationClass::NormalWrite, + OperationClass::Stream, + OperationClass::Lifecycle, + OperationClass::Vacuum, + ] { + let e = fence.permits(class).unwrap_err(); + assert_eq!( + e.outcome(), + FenceOutcome::MigrationTargetQuarantined, + "{class:?}" + ); + } + } + + async fn controller(store: &NamespaceStore, ns: &'static str) -> Arc { + store + .fence_gate(&ns.into()) + .await + .unwrap() + .expect("the target has a controller") + } + + /// The loaded target's controller, and a connection to it. + async fn loaded( + store: &NamespaceStore, + ns: &'static str, + ) -> (Arc, Arc) { + let (fence, maker) = store + .with(ns.into(), |ns| { + (ns.fence().clone(), ns.db.connection_maker()) + }) + .await + .unwrap(); + (fence, Arc::new(maker.create().await.unwrap())) + } + + async fn replication_hello(store: &NamespaceStore, ns: &'static str) -> tonic::Status { + let service = ReplicationLogService::new(store.clone(), None, None, false, false, true); + let mut req = tonic::Request::new(HelloRequest { + handshake_version: Some(1), + }); + req.metadata_mut().insert_bin( + NAMESPACE_METADATA_KEY, + BinaryMetadataValue::from_bytes(ns.as_bytes()), + ); + service.hello(req).await.unwrap_err() + } + + /// Every way of reaching `ns` other than the fence commands: none of them is served. + async fn attempt_everything(store: &NamespaceStore, ns: &'static str) { + let r = store + .with(ns.into(), |ns| ns.db.connection_maker()) + .await + .map(|_| ()); + assert_not_served("SQL connection", &r); + let r = store.stats(ns.into()).await.map(|_| ()); + assert_not_served("stats", &r); + // The dump route and the replication service reach the namespace the same way. + let hello = replication_hello(store, ns).await; + assert_ne!(hello.code(), tonic::Code::Ok); + assert_ne!(hello.code(), tonic::Code::Unavailable, "{hello:?}"); + let r = store + .create(ns.into(), RestoreOption::Latest, DatabaseConfig::default()) + .await; + assert_not_served("create", &r); + let r = store.destroy(ns.into(), false).await; + assert!(r.is_err(), "delete: {r:?}"); + let r = store + .fork("src".into(), ns.into(), DatabaseConfig::default(), None) + .await; + assert!(r.is_err(), "fork: {r:?}"); + } + + async fn store_with_source(dir: &TempDir) -> NamespaceStore { + let store = open_store(dir.path()).await; + store + .create("src".into(), RestoreOption::Latest, Default::default()) + .await + .unwrap(); + store + } + + /// Parked after its rows are committed and its quarantine gate published, and before its + /// config is published and the namespace loaded, the target is never observable: SQL, + /// dump/replication, create, delete and fork of the name are all denied or find nothing. + /// Afterwards it is loaded behind the quarantine gate. + #[tokio::test(flavor = "multi_thread")] + async fn create_race_never_observable() { + let dir = tempdir().unwrap(); + let store = store_with_source(&dir).await; + let fence = store.fence_controller(&"tgt".into()); + let paused = fence.hooks().pause_at(HookPoint::AfterTargetRowsCommitted); + let creating = create(&store, create_request("tgt", 1)); + paused.reached().await; + + assert_eq!(fence.gate().state(), FenceState::TargetQuarantined); + assert!(!store.exists(&"tgt".into()).await); + attempt_everything(&store, "tgt").await; + + paused.resume(); + let commit = creating.await.unwrap().unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + assert!(store.exists(&"tgt".into()).await); + + // Loaded behind the gate it was created with. + let (loaded_fence, conn) = loaded(&store, "tgt").await; + assert!(Arc::ptr_eq(&fence, &loaded_fence)); + assert_quarantined(&fence); + let ctx = RequestContext::new( + Authenticated::FullAccess, + "tgt".into(), + store.meta_store().clone(), + ); + let e = conn + .execute_program( + Program::seq(&["select 1"]), + ctx, + TestBuilder::default(), + None, + ) + .await + .map(|_| ()) + .unwrap_err(); + assert_eq!( + fence_error(&e).outcome(), + FenceOutcome::MigrationTargetQuarantined + ); + crate::namespace::fence::drain::tests::assert_fenced( + raw(&conn, "create table t (x)").await, + ); + // The in-memory config is the logical one; the stored row carries the legacy mirror. + let config = store.config_store("tgt".into()).await.unwrap().get(); + assert_eq!(config.max_db_pages, 1000); + assert!(!config.block_reads && !config.block_writes); + // Everything but the commands is still refused once it is loaded. + attempt_everything_loaded(&store, "tgt").await; + } + + /// Once loaded, the lifecycle paths are refused by the fence. + async fn attempt_everything_loaded(store: &NamespaceStore, ns: &'static str) { + let r = store + .create(ns.into(), RestoreOption::Latest, DatabaseConfig::default()) + .await; + assert!(r.is_err(), "create: {r:?}"); + let r = store.destroy(ns.into(), false).await; + assert_eq!( + fence_error(&r.unwrap_err()).outcome(), + FenceOutcome::MigrationTargetQuarantined + ); + let r = store + .fork("src".into(), ns.into(), DatabaseConfig::default(), None) + .await; + assert!(r.is_err(), "fork: {r:?}"); + let hello = replication_hello(store, ns).await; + assert_eq!( + FenceError::outcome_from_grpc_status(&hello), + Some(FenceOutcome::MigrationTargetQuarantined), + "{hello:?}" + ); + assert!(store.exists(&ns.into()).await); + } + + /// Before its commit, while the target-creation gate is in place, the name is refused + /// before any setup work. + #[tokio::test(flavor = "multi_thread")] + async fn creating_gate_refuses_before_commit() { + let dir = tempdir().unwrap(); + let store = store_with_source(&dir).await; + let fence = store.fence_controller(&"tgt".into()); + let paused = fence.hooks().pause_at(HookPoint::BeforeMetastoreCommit); + let creating = create(&store, create_request("tgt", 1)); + paused.reached().await; + + assert!(fence.gate().is_creating_target()); + assert_quarantined(&fence); + attempt_everything(&store, "tgt").await; + // No database was set up under the name. + assert!(!dir.path().join("dbs").join("tgt").join("data").exists()); + + paused.resume(); + creating.await.unwrap().unwrap(); + assert!(!fence.gate().is_creating_target()); + assert_eq!(fence.gate().state(), FenceState::TargetQuarantined); + } + + /// A creation interrupted between its marker and its commit leaves the name + /// `UNKNOWN_UNAVAILABLE` after a restart; only the same command completes it, and the + /// completed target is loaded quarantined. + #[tokio::test(flavor = "multi_thread")] + async fn create_replay_completes_interrupted_creation() { + let dir = tempdir().unwrap(); + { + let store = store_with_source(&dir).await; + create(&store, create_request("tgt", 1)) + .await + .unwrap() + .unwrap(); + store.shutdown().await.unwrap(); + } + // What a crash after the marker and before the commit leaves: the marker alone. + { + let (maker, _) = metastore_connection_maker(None, dir.path()).await.unwrap(); + let conn = maker().unwrap(); + for sql in [ + "DELETE FROM namespace_fence_receipts WHERE namespace = 'tgt'", + "DELETE FROM namespace_fences WHERE namespace = 'tgt'", + "DELETE FROM namespace_configs WHERE namespace = 'tgt'", + ] { + conn.execute(sql, ()).unwrap(); + } + } + std::fs::remove_file(dir.path().join("dbs").join("tgt").join("data")).ok(); + + let store = open_store(dir.path()).await; + let fence = store.fence_controller(&"tgt".into()); + assert!(fence.gate().is_unavailable()); + let r = store.with("tgt".into(), |_| ()).await; + assert_eq!( + fence_error(&r.unwrap_err()).detail(), + Some(FenceDetail::IncompleteTargetCreation) + ); + // Another command does not complete it. + let r = create(&store, create_request("tgt", 2)).await.unwrap(); + assert_eq!( + fence_error(&r.unwrap_err()).outcome(), + FenceOutcome::FenceStateUnavailable + ); + + let commit = create(&store, create_request("tgt", 1)) + .await + .unwrap() + .unwrap(); + assert_eq!(commit.kind, FenceCommitKind::Committed); + assert_eq!(fence.gate().state(), FenceState::TargetQuarantined); + let (loaded_fence, _) = loaded(&store, "tgt").await; + assert!(Arc::ptr_eq(&fence, &loaded_fence)); + assert_quarantined(&fence); + let config = store.config_store("tgt".into()).await.unwrap().get(); + assert!(!config.block_reads && !config.block_writes); + } + + /// The caller going away does not stop a creation: it is published and loaded, and a replay + /// returns the stored result. + #[tokio::test(flavor = "multi_thread")] + async fn create_completes_when_the_caller_goes_away() { + let dir = tempdir().unwrap(); + let store = store_with_source(&dir).await; + let fence = store.fence_controller(&"tgt".into()); + let paused = fence.hooks().pause_at(HookPoint::AfterTargetRowsCommitted); + let creating = create(&store, create_request("tgt", 1)); + paused.reached().await; + creating.abort(); + let _ = creating.await; + paused.resume(); + + // The creation carries on without its caller; the replay waits for it on the + // transition lock and then answers from the receipt. + let replay = tokio::time::timeout(PROMPT, create(&store, create_request("tgt", 1))) + .await + .unwrap() + .unwrap() + .unwrap(); + assert_eq!(replay.kind, FenceCommitKind::Replayed); + assert_eq!(replay.receipt.outcome, FenceOutcome::Applied); + assert!(store.exists(&"tgt".into()).await); + let (loaded_fence, _) = loaded(&store, "tgt").await; + assert!(Arc::ptr_eq(&fence, &loaded_fence)); + } + + /// A commit whose acknowledgement is lost keeps the name closed; the replay reconciles it + /// from the durable rows and publishes and loads the target. + #[tokio::test(flavor = "multi_thread")] + async fn indeterminate_create_is_completed_by_replay() { + let dir = tempdir().unwrap(); + let store = store_with_source(&dir).await; + let fence = store.fence_controller(&"tgt".into()); + fence + .hooks() + .arm(HookPoint::AfterMetastoreCommit, HookAction::Indeterminate); + let r = create(&store, create_request("tgt", 1)).await.unwrap(); + assert_eq!( + fence_error(&r.unwrap_err()).outcome(), + FenceOutcome::FenceCommitIndeterminate + ); + // Still refused before any setup, and not published. + assert!(fence.gate().is_creating_target()); + assert!(!store.exists(&"tgt".into()).await); + assert_not_served("SQL", &store.with("tgt".into(), |_| ()).await); + + let replay = create(&store, create_request("tgt", 1)) + .await + .unwrap() + .unwrap(); + assert_eq!(replay.kind, FenceCommitKind::Replayed); + assert!(!fence.gate().is_creating_target()); + assert!(fence.gate().indeterminate.is_none()); + assert_eq!(fence.gate().state(), FenceState::TargetQuarantined); + assert!(store.exists(&"tgt".into()).await); + loaded(&store, "tgt").await; + assert_quarantined(&fence); + } + + /// A name the server already has is refused, and its traffic is not disturbed; a name that + /// is refused keeps no creation gate. + #[tokio::test(flavor = "multi_thread")] + async fn create_rejects_existing_name() { + let dir = tempdir().unwrap(); + let store = store_with_source(&dir).await; + let src_conn = { + let maker = store + .with("src".into(), |ns| ns.db.connection_maker()) + .await + .unwrap(); + Arc::new(maker.create().await.unwrap()) + }; + raw(&src_conn, "create table t (x)").await.unwrap(); + let generation = store.fence_controller(&"src".into()).write_generation(); + + let r = create(&store, create_request("src", 1)).await.unwrap(); + let e = r.unwrap_err(); + assert_eq!( + fence_error(&e).outcome(), + FenceOutcome::FencePreconditionFailed + ); + assert_eq!(fence_error(&e).detail(), Some(FenceDetail::NamespaceExists)); + // The source's gate never moved, and it still takes writes. + let src_fence = store.fence_controller(&"src".into()); + assert_eq!(src_fence.write_generation(), generation); + assert!(!src_fence.gate().is_creating_target()); + raw(&src_conn, "insert into t values (1)").await.unwrap(); + + // A name that exists in the metastore but is not loaded is refused the same way. + store + .create("cold".into(), RestoreOption::Latest, Default::default()) + .await + .unwrap(); + let r = create(&store, create_request("cold", 2)).await.unwrap(); + assert_eq!( + fence_error(&r.unwrap_err()).detail(), + Some(FenceDetail::NamespaceExists) + ); + + // A second target of the same name by another operation is refused, and the first is + // untouched. + create(&store, create_request("tgt", 3)) + .await + .unwrap() + .unwrap(); + let mut other = create_request("tgt", 4); + other.operation_id = OTHER_OP; + let r = create(&store, other).await.unwrap(); + assert!(r.is_err()); + let fence = store.fence_controller(&"tgt".into()); + assert_eq!(fence.gate().operation_id(), Some(OP)); + assert!(!fence.gate().is_creating_target()); + } + + /// A validation session carries the owner's current capability, admits reads through the + /// quarantine, is `query_only`, and is invalidated by the validation receipt's revision. + /// The receipt records the target snapshot observed by the server. + #[tokio::test(flavor = "multi_thread")] + async fn validation_session_is_read_only() { + let (_dir, store, fence) = validating_target().await; + let mut session = store + .open_validation_session("tgt".into(), OP, 3) + .await + .unwrap(); + assert_eq!(session.capability().purpose(), CapabilityPurpose::Validate); + let (query_only, count) = session + .with_raw(|conn| { + let query_only = + conn.query_row("PRAGMA query_only", (), |row| row.get::<_, i64>(0)); + let count = + conn.query_row("select count(*) from t", (), |row| row.get::<_, i64>(0)); + (query_only, count) + }) + .await + .unwrap(); + assert_eq!(query_only.unwrap(), 1); + assert_eq!(count.unwrap(), 2); + match session + .with_raw(|conn| conn.execute_batch("insert into t values (3)")) + .await + .unwrap() + { + Err(rusqlite::Error::SqliteFailure(e, _)) => { + assert_eq!(e.code, rusqlite::ErrorCode::ReadOnly) + } + other => panic!("query_only validation connection accepted a write: {other:?}"), + } + let e = session + .with_raw(|conn| { + conn.pragma_update(None, "query_only", false).unwrap(); + conn.execute_batch("insert into t values (3)") + }) + .await + .unwrap_err(); + assert_eq!(e.outcome(), FenceOutcome::OperationCapabilityRequired); + + let error = store + .open_validation_session("tgt".into(), OTHER_OP, 3) + .await + .unwrap_err(); + assert_eq!( + fence_error(&error).outcome(), + FenceOutcome::FenceOwnedByAnotherOperation + ); + let error = store + .open_validation_session("tgt".into(), OP, 2) + .await + .unwrap_err(); + assert_eq!( + fence_error(&error).outcome(), + FenceOutcome::FenceRevisionMismatch + ); + + let request = target_command( + 20, + FenceState::TargetValidating, + 3, + FenceCommand::RecordTargetValidation { + result: ValidationResult::Ok, + summary: "validation Ok".into(), + }, + ); + let after_commit = fence.hooks().pause_at(HookPoint::AfterMetastoreCommit); + let first = execute(&store, request.clone()); + after_commit.reached().await; + // The metastore has the receipt but the live gate still has revision 3. The concurrent + // replay must not demand another validation snapshot or capability before it waits for + // the first command to publish. + let replay = execute(&store, request); + tokio::task::yield_now().await; + after_commit.resume(); + let commit = first.await.unwrap().unwrap(); + let replay = replay.await.unwrap().unwrap(); + assert_eq!(replay.kind, FenceCommitKind::Replayed); + let validation = commit.record.as_ref().unwrap().validation.as_ref().unwrap(); + let snapshot = validation.snapshot.expect("the server records a snapshot"); + let (log_id, frame_no) = store + .with("tgt".into(), |ns| { + let logger = ns.db.logger().unwrap(); + let frame_no = *logger.new_frame_notifier.borrow(); + (logger.log_id(), frame_no) + }) + .await + .unwrap(); + assert_eq!(snapshot.log_id, log_id); + assert_eq!(snapshot.frame_no, frame_no.unwrap_or(0)); + assert!(snapshot.page_count > 0); + assert_eq!(fence.gate().revision(), 4); + + let e = session + .with_raw(|conn| conn.query_row("select 1", (), |row| row.get::<_, i64>(0))) + .await + .unwrap_err(); + assert_eq!(e.outcome(), FenceOutcome::OperationCapabilityRequired); + let mut current = store + .open_validation_session("tgt".into(), OP, 4) + .await + .unwrap(); + assert_eq!( + current + .with_raw( + |conn| conn.query_row("select count(*) from t", (), |row| row.get::<_, i64>(0)) + ) + .await + .unwrap() + .unwrap(), + 2 + ); + } + + /// Publication cannot make a target readable until the latest durable validation result of + /// the owning operation is successful. + #[tokio::test(flavor = "multi_thread")] + async fn publish_requires_validation_receipt() { + let (_dir, store, fence) = validating_target().await; + let publish = |command_id, revision| { + target_command( + command_id, + FenceState::TargetValidating, + revision, + FenceCommand::PublishTargetReadableWriteFenced, + ) + }; + let e = execute(&store, publish(20, 3)).await.unwrap().unwrap_err(); + assert_eq!( + fence_error(&e).detail(), + Some(FenceDetail::ValidationReceiptRequired) + ); + record_validation(&store, 21, 3, ValidationResult::Failed).await; + let e = execute(&store, publish(22, 4)).await.unwrap().unwrap_err(); + assert_eq!( + fence_error(&e).detail(), + Some(FenceDetail::ValidationReceiptRequired) + ); + assert_eq!( + (fence.gate().state(), fence.gate().revision()), + (FenceState::TargetValidating, 4) + ); + assert!(fence.permits(OperationClass::NormalRead).is_err()); + } + + /// A successful validation makes publication possible once; exact replay returns the stored + /// result and a new command with the same goal answers ALREADY_APPLIED without moving the + /// revision. + #[tokio::test(flavor = "multi_thread")] + async fn publish_is_idempotent() { + let (dir, store, fence) = validating_target().await; + record_validation(&store, 20, 3, ValidationResult::Ok).await; + let publish = target_command( + 21, + FenceState::TargetValidating, + 4, + FenceCommand::PublishTargetReadableWriteFenced, + ); + let commit = execute(&store, publish.clone()).await.unwrap().unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + assert_eq!( + (fence.gate().state(), fence.gate().revision()), + (FenceState::TargetWriteFenced, 5) + ); + assert!(fence.permits(OperationClass::NormalRead).is_ok()); + assert!(fence.permits(OperationClass::NormalWrite).is_err()); + assert_eq!(count_rows(&store).await, 2); + + let replay = execute(&store, publish).await.unwrap().unwrap(); + assert_eq!(replay.kind, FenceCommitKind::Replayed); + assert_eq!(replay.receipt.outcome, FenceOutcome::Applied); + let again = execute( + &store, + target_command( + 22, + FenceState::TargetValidating, + 4, + FenceCommand::PublishTargetReadableWriteFenced, + ), + ) + .await + .unwrap() + .unwrap(); + assert_eq!(again.receipt.outcome, FenceOutcome::AlreadyApplied); + assert_eq!( + (again.receipt.revision_before, again.receipt.revision_after), + (5, 5) + ); + assert_eq!(fence.gate().revision(), 5); + + store.shutdown().await.unwrap(); + let store = open_store(dir.path()).await; + let fence = controller(&store, "tgt").await; + assert_eq!(fence.gate().state(), FenceState::TargetWriteFenced); + let (_, conn) = loaded(&store, "tgt").await; + program(&store, &conn, "select count(*) from t").await; + } + + /// Enabling writes is idempotent but irreversible: it opens normal writes exactly after the + /// commit is published; replay and a new same-goal command are safe, while no target command + /// can close or abort it again. + #[tokio::test(flavor = "multi_thread")] + async fn enable_writes_idempotent_and_irreversible() { + let (_dir, store, fence) = write_fenced_target().await; + let request = enable_request(30); + let commit = execute(&store, request.clone()).await.unwrap().unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + assert_eq!( + (fence.gate().state(), fence.gate().revision()), + (FenceState::TargetWritable, 6) + ); + assert!(fence.permits(OperationClass::NormalWrite).is_ok()); + + let replay = execute(&store, request).await.unwrap().unwrap(); + assert_eq!(replay.kind, FenceCommitKind::Replayed); + assert_eq!(replay.receipt.outcome, FenceOutcome::Applied); + let again = execute(&store, enable_request(31)).await.unwrap().unwrap(); + assert_eq!(again.receipt.outcome, FenceOutcome::AlreadyApplied); + assert_eq!(fence.gate().revision(), 6); + + let abort = target_command( + 32, + FenceState::TargetWritable, + 6, + FenceCommand::AbortQuarantinedTarget, + ); + let e = execute(&store, abort).await.unwrap().unwrap_err(); + assert_eq!( + fence_error(&e).outcome(), + FenceOutcome::InvalidFenceTransition + ); + assert_eq!(fence.gate().state(), FenceState::TargetWritable); + let (_, conn) = loaded(&store, "tgt").await; + raw(&conn, "insert into t values (3)").await.unwrap(); + assert_eq!(count_rows(&store).await, 3); + } + + /// The committed writable state is installed before a restarted server can expose the + /// target, and the legacy config mirror no longer blocks its writes. + #[tokio::test(flavor = "multi_thread")] + async fn enable_writes_survives_restart() { + let (dir, store, _fence) = write_fenced_target().await; + execute(&store, enable_request(30)).await.unwrap().unwrap(); + store.shutdown().await.unwrap(); + + let store = open_store(dir.path()).await; + let fence = controller(&store, "tgt").await; + assert_eq!( + (fence.gate().state(), fence.gate().revision()), + (FenceState::TargetWritable, 6) + ); + let (_, conn) = loaded(&store, "tgt").await; + program(&store, &conn, "insert into t values (3)").await; + assert_eq!(count_rows(&store).await, 3); + } + + /// Losing the response after EnableTargetWrites commits is resolved by inspection and exact + /// replay; the detached command still publishes the writable gate before it releases the + /// transition lock. + #[tokio::test(flavor = "multi_thread")] + async fn enable_writes_response_loss_resolved() { + let (_dir, store, fence) = write_fenced_target().await; + let request = enable_request(30); + let after_commit = fence.hooks().pause_at(HookPoint::AfterMetastoreCommit); + let lost = execute(&store, request.clone()); + after_commit.reached().await; + let before_response = fence.hooks().pause_at(HookPoint::BeforeResponse); + lost.abort(); + assert!(lost.await.unwrap_err().is_cancelled()); + after_commit.resume(); + before_response.reached().await; + + assert_eq!(fence.gate().state(), FenceState::TargetWritable); + let inspected = store + .meta_store() + .inspect_fence("tgt".into()) + .await + .unwrap(); + assert_eq!(inspected.fence.state(), FenceState::TargetWritable); + assert!(inspected.receipts.iter().any(|stored| { + matches!( + &stored.receipt, + Ok(receipt) + if receipt.operation_id == OP + && receipt.command_id == Uuid::from_u128(30) + && receipt.outcome == FenceOutcome::Applied + ) + })); + before_response.resume(); + + let replay = execute(&store, request).await.unwrap().unwrap(); + assert_eq!(replay.kind, FenceCommitKind::Replayed); + assert_eq!(replay.receipt.outcome, FenceOutcome::Applied); + assert_eq!(fence.gate().state(), FenceState::TargetWritable); + } + + /// A read transaction opened before target write authority is published cannot upgrade to + /// a write afterwards; rolling it back and starting a fresh program succeeds. + #[tokio::test(flavor = "multi_thread")] + async fn stale_generation_cannot_write_after_enable_writes() { + let (_dir, store, fence) = write_fenced_target().await; + let (_, conn) = loaded(&store, "tgt").await; + raw(&conn, "begin; select count(*) from t").await.unwrap(); + execute(&store, enable_request(30)).await.unwrap().unwrap(); + assert_eq!(fence.gate().state(), FenceState::TargetWritable); + + crate::namespace::fence::drain::tests::assert_fenced( + raw(&conn, "insert into t values (3)").await, + ); + raw(&conn, "commit").await.unwrap(); + raw(&conn, "insert into t values (4)").await.unwrap(); + assert_eq!(count_rows(&store).await, 3); + } + + /// `AbortQuarantinedTarget` finishes the operation and keeps every normal class denied; + /// the target is not deletable by the generic lifecycle either. + #[tokio::test(flavor = "multi_thread")] + async fn abort_keeps_traffic_denied() { + let dir = tempdir().unwrap(); + let store = store_with_source(&dir).await; + create(&store, create_request("tgt", 1)) + .await + .unwrap() + .unwrap(); + let abort = FenceRequest { + namespace: "tgt".into(), + operation_id: OP, + command_id: Uuid::from_u128(2), + expected_state: FenceState::TargetQuarantined, + expected_revision: 1, + command: FenceCommand::AbortQuarantinedTarget, + }; + let commit = store + .execute_fence_command(abort.clone(), server()) + .await + .unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + let fence = controller(&store, "tgt").await; + assert_eq!(fence.gate().state(), FenceState::TargetAborted); + assert_quarantined(&fence); + let (_, conn) = loaded(&store, "tgt").await; + crate::namespace::fence::drain::tests::assert_fenced( + raw(&conn, "create table t (x)").await, + ); + let r = store.destroy("tgt".into(), false).await; + assert_eq!( + fence_error(&r.unwrap_err()).outcome(), + FenceOutcome::MigrationTargetQuarantined + ); + // A replay answers from the receipt; a new creation of the name is refused. + let replay = store.execute_fence_command(abort, server()).await.unwrap(); + assert_eq!(replay.kind, FenceCommitKind::Replayed); + let r = create(&store, create_request("tgt", 3)).await.unwrap(); + assert!(r.is_err()); + + // After a restart the aborted target is still denied. + store.shutdown().await.unwrap(); + let store = open_store(dir.path()).await; + let fence = controller(&store, "tgt").await; + assert_eq!(fence.gate().state(), FenceState::TargetAborted); + assert_quarantined(&fence); + let (_, conn) = loaded(&store, "tgt").await; + crate::namespace::fence::drain::tests::assert_fenced( + raw(&conn, "create table t (x)").await, + ); + } +} diff --git a/libsql-server/src/namespace/fence/tests.rs b/libsql-server/src/namespace/fence/tests.rs index bb41fbeecd..6d297a0231 100644 --- a/libsql-server/src/namespace/fence/tests.rs +++ b/libsql-server/src/namespace/fence/tests.rs @@ -114,16 +114,24 @@ impl Server { /// The namespace's controller, loading the namespace if it is not loaded. async fn fence(&self) -> Arc { + self.fence_of("ns").await + } + + async fn fence_of(&self, ns: &'static str) -> Arc { self.store - .with("ns".into(), |ns| ns.fence().clone()) + .with(ns.into(), |ns| ns.fence().clone()) .await .unwrap() } async fn conn(&self) -> Arc { + self.conn_to("ns").await + } + + async fn conn_to(&self, ns: &'static str) -> Arc { let maker = self .store - .with("ns".into(), |ns| ns.db.connection_maker()) + .with(ns.into(), |ns| ns.db.connection_maker()) .await .unwrap(); Arc::new(maker.create().await.unwrap()) @@ -157,15 +165,23 @@ impl Server { } async fn inspect(&self) -> FenceInspection { + self.inspect_of("ns").await + } + + async fn inspect_of(&self, ns: &'static str) -> FenceInspection { self.store .meta_store() - .inspect_fence("ns".into()) + .inspect_fence(ns.into()) .await .unwrap() } async fn count(&self) -> i64 { - let conn = self.conn().await; + self.count_in("ns").await + } + + async fn count_in(&self, ns: &'static str) -> i64 { + let conn = self.conn_to(ns).await; tokio::task::spawn_blocking(move || { conn.with_raw(|c| c.query_row("select count(*) from t", (), |r| r.get(0))) }) @@ -176,7 +192,11 @@ impl Server { /// Whether a new connection may begin a write transaction. Writes nothing. async fn writes_admitted(&self) -> bool { - let conn = self.conn().await; + self.writes_admitted_in("ns").await + } + + async fn writes_admitted_in(&self, ns: &'static str) -> bool { + let conn = self.conn_to(ns).await; match raw(&conn, "begin immediate; rollback;").await { Ok(()) => true, Err(rusqlite::Error::SqliteFailure(e, _)) @@ -254,7 +274,11 @@ fn boundary(commit: &FenceCommit) -> FrozenBoundary { /// The marker file's bytes, if there is one. fn read_marker_bytes(dbs: &Path) -> Option> { - match std::fs::read(fence_store::marker_path(dbs, &"ns".into())) { + read_marker_bytes_of(dbs, "ns") +} + +fn read_marker_bytes_of(dbs: &Path, ns: &'static str) -> Option> { + match std::fs::read(fence_store::marker_path(dbs, &ns.into())) { Ok(bytes) => Some(bytes), Err(e) if e.kind() == std::io::ErrorKind::NotFound => None, Err(e) => panic!("{e}"), @@ -263,7 +287,11 @@ fn read_marker_bytes(dbs: &Path) -> Option> { /// Put the marker file back to `bytes` (`None`: no marker). fn restore_marker_bytes(dbs: &Path, bytes: Option<&[u8]>) { - let path = fence_store::marker_path(dbs, &"ns".into()); + restore_marker_bytes_of(dbs, "ns", bytes) +} + +fn restore_marker_bytes_of(dbs: &Path, ns: &'static str, bytes: Option<&[u8]>) { + let path = fence_store::marker_path(dbs, &ns.into()); match bytes { Some(bytes) => std::fs::write(path, bytes).unwrap(), None => std::fs::remove_file(path).unwrap(), @@ -582,7 +610,7 @@ fn restart_at(case: &Boundary) { // The drain that was requested resumes and completes at once: recovery // discarded any uncommitted work. The boundary is on the live, rebuilt log. let commit = replay.unwrap(); - assert_eq!(commit.kind, FenceCommitKind::Committed, "{name}"); + assert_eq!(commit.kind, FenceCommitKind::Resumed, "{name}"); assert_eq!(commit.receipt.outcome, FenceOutcome::Applied, "{name}"); assert_eq!(commit.receipt.command_id, request.command_id); assert_eq!(commit.receipt.revision_after, 2, "{name}"); @@ -896,3 +924,1083 @@ fn acquire_response_loss_resolved_by_replay_and_inspect() { server.crash(); } } + +// --------------------------------------------------------------------------------------------- +// Read fence, seal and write enable: restart at each persistence boundary (section 8.5) + +mod read_and_target_boundaries { + use super::*; + use crate::namespace::fence::command::ValidationResult; + use crate::namespace::fence::controller::LeaseKind; + use crate::namespace::fence::target::tests::{create_request, enable_request, target_command}; + + #[derive(Debug, Clone, Copy, PartialEq, Eq)] + enum Later { + /// `SetSourceReadFence` on the write-fenced source `ns`. + ReadFence, + /// `SealTargetImport` on the quarantined target `tgt`. + Seal, + /// `EnableTargetWrites` on the write-fenced target `tgt`. + Enable, + } + + impl Later { + fn namespace(self) -> &'static str { + match self { + Later::ReadFence => "ns", + Later::Seal | Later::Enable => "tgt", + } + } + + /// State and revision before the command. + fn before(self) -> (FenceState, u64) { + match self { + Later::ReadFence => (FenceState::SourceWriteFenced, 2), + Later::Seal => (FenceState::TargetQuarantined, 1), + Later::Enable => (FenceState::TargetWriteFenced, 5), + } + } + + /// State and revision while the command drains. + fn draining(self) -> (FenceState, u64) { + match self { + Later::ReadFence => (FenceState::SourceReadDraining, 3), + Later::Seal => (FenceState::TargetImportDraining, 2), + Later::Enable => unreachable!("EnableTargetWrites does not drain"), + } + } + + /// State and revision once the command has applied. + fn applied(self) -> (FenceState, u64) { + match self { + Later::ReadFence => (FenceState::SourceReadFenced, 4), + Later::Seal => (FenceState::TargetValidating, 3), + Later::Enable => (FenceState::TargetWritable, 6), + } + } + + fn state_after(self, durable: Durable) -> (FenceState, u64) { + match durable { + Durable::Nothing => self.before(), + Durable::Draining => self.draining(), + Durable::Final => self.applied(), + } + } + + fn request(self) -> FenceRequest { + match self { + Later::ReadFence => FenceRequest { + namespace: "ns".into(), + operation_id: OP, + command_id: Uuid::from_u128(2), + expected_state: FenceState::SourceWriteFenced, + expected_revision: 2, + command: FenceCommand::SetSourceReadFence { + drain_policy: Some(LONG), + }, + }, + Later::Seal => target_command( + 10, + FenceState::TargetQuarantined, + 1, + FenceCommand::SealTargetImport { + drain_policy: Some(LONG), + }, + ), + Later::Enable => enable_request(30), + } + } + } + + /// Where the command is when the process dies. + #[derive(Debug, Clone, Copy)] + enum Park { + /// At a point of the command's first (or only) commit. + First(HookPoint), + /// At a point of the drain's completion, the command's second commit. + Second(HookPoint), + /// In the drain, waiting for a reader (read fence) or an import call (seal) that was + /// already running when the command started. + WaitingForHolder, + } + + #[derive(Debug, Clone, Copy)] + struct LaterBoundary { + name: &'static str, + command: Later, + park: Park, + /// The marker file is put back to what it held before the commit, as a crash between + /// the metastore commit and the marker write leaves it. + marker_lags: bool, + durable: Durable, + } + + const fn case( + name: &'static str, + command: Later, + park: Park, + marker_lags: bool, + durable: Durable, + ) -> LaterBoundary { + LaterBoundary { + name, + command, + park, + marker_lags, + durable, + } + } + + use Durable::{Draining, Final, Nothing}; + use HookPoint::{ + AfterClosingReads, AfterInstallingGate, AfterMetastoreCommit, BeforeGatePublish, + BeforeMetastoreCommit, BeforeResponse, + }; + use Later::{Enable, ReadFence, Seal}; + use Park::{First, Second, WaitingForHolder}; + + const LATER_BOUNDARIES: &[LaterBoundary] = &[ + case( + "read/after-closing-reads", + ReadFence, + First(AfterClosingReads), + false, + Nothing, + ), + case( + "read/before-draining-commit", + ReadFence, + First(BeforeMetastoreCommit), + false, + Nothing, + ), + case( + "read/after-draining-commit", + ReadFence, + First(AfterMetastoreCommit), + false, + Draining, + ), + case( + "read/after-draining-commit/marker-lags", + ReadFence, + First(AfterMetastoreCommit), + true, + Draining, + ), + case( + "read/before-draining-publish", + ReadFence, + First(BeforeGatePublish), + false, + Draining, + ), + case( + "read/draining-published", + ReadFence, + First(BeforeResponse), + false, + Draining, + ), + case( + "read/waiting-for-reader", + ReadFence, + WaitingForHolder, + false, + Draining, + ), + case( + "read/before-fenced-commit", + ReadFence, + Second(BeforeMetastoreCommit), + false, + Draining, + ), + case( + "read/after-fenced-commit", + ReadFence, + Second(AfterMetastoreCommit), + false, + Final, + ), + case( + "read/after-fenced-commit/marker-lags", + ReadFence, + Second(AfterMetastoreCommit), + true, + Final, + ), + case( + "read/before-fenced-publish", + ReadFence, + Second(BeforeGatePublish), + false, + Final, + ), + case( + "read/before-fenced-response", + ReadFence, + Second(BeforeResponse), + false, + Final, + ), + case( + "seal/after-installing-gate", + Seal, + First(AfterInstallingGate), + false, + Nothing, + ), + case( + "seal/before-draining-commit", + Seal, + First(BeforeMetastoreCommit), + false, + Nothing, + ), + case( + "seal/after-draining-commit", + Seal, + First(AfterMetastoreCommit), + false, + Draining, + ), + case( + "seal/after-draining-commit/marker-lags", + Seal, + First(AfterMetastoreCommit), + true, + Draining, + ), + case( + "seal/before-draining-publish", + Seal, + First(BeforeGatePublish), + false, + Draining, + ), + case( + "seal/draining-published", + Seal, + First(BeforeResponse), + false, + Draining, + ), + case( + "seal/waiting-for-import-call", + Seal, + WaitingForHolder, + false, + Draining, + ), + case( + "seal/before-validating-commit", + Seal, + Second(BeforeMetastoreCommit), + false, + Draining, + ), + case( + "seal/after-validating-commit", + Seal, + Second(AfterMetastoreCommit), + false, + Final, + ), + case( + "seal/after-validating-commit/marker-lags", + Seal, + Second(AfterMetastoreCommit), + true, + Final, + ), + case( + "seal/before-validating-publish", + Seal, + Second(BeforeGatePublish), + false, + Final, + ), + case( + "seal/before-validating-response", + Seal, + Second(BeforeResponse), + false, + Final, + ), + case( + "enable/before-commit", + Enable, + First(BeforeMetastoreCommit), + false, + Nothing, + ), + case( + "enable/after-commit", + Enable, + First(AfterMetastoreCommit), + false, + Final, + ), + case( + "enable/after-commit/marker-lags", + Enable, + First(AfterMetastoreCommit), + true, + Final, + ), + case( + "enable/before-publish", + Enable, + First(BeforeGatePublish), + false, + Final, + ), + case( + "enable/before-response", + Enable, + First(BeforeResponse), + false, + Final, + ), + ]; + + /// Create the quarantined target `tgt` (revision 1) and import a table `t` of [`ROWS`] + /// rows into it. + fn create_target(server: &Server) { + server.run(async { + let commit = server + .store + .create_target_quarantined(create_request("tgt", 1), server_identity()) + .await + .unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + let mut session = server + .store + .open_import_session("tgt".into(), OP, 1) + .await + .unwrap(); + session + .with_raw(|c| { + c.execute_batch("create table t (x)")?; + for _ in 0..ROWS { + c.execute("insert into t values (1)", ())?; + } + Ok::<_, rusqlite::Error>(()) + }) + .await + .unwrap() + .unwrap(); + }) + } + + /// Seal, validate and publish `tgt`: `TARGET_WRITE_FENCED` at revision 5. + fn write_fence_target(server: &Server) { + server.run(async { + let steps = [ + target_command( + 10, + FenceState::TargetQuarantined, + 1, + FenceCommand::SealTargetImport { + drain_policy: Some(LONG), + }, + ), + target_command( + 20, + FenceState::TargetValidating, + 3, + FenceCommand::RecordTargetValidation { + result: ValidationResult::Ok, + summary: "rows match".into(), + }, + ), + target_command( + 21, + FenceState::TargetValidating, + 4, + FenceCommand::PublishTargetReadableWriteFenced, + ), + ]; + for step in steps { + let commit = server.execute(step).await.unwrap().unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + } + }) + } + + fn prepare(server: &Server, command: Later) { + match command { + Later::ReadFence => { + server.create_source(); + server.fence_source(); + } + Later::Seal => create_target(server), + Later::Enable => { + create_target(server); + write_fence_target(server); + } + } + } + + /// What the drain of [`Park::WaitingForHolder`] waits for: a read lease, as a running SQL + /// program holds one, or an admitted import call, as a running `ImportSession::with_raw` + /// holds one. Neither holds a SQLite lock, so it can outlive the crash without disturbing + /// the next lifetime's recovery. Only dropped. + type Holder = Box; + + async fn hold(server: &Server, fence: &Arc, command: Later) -> Holder { + match command { + Later::ReadFence => Box::new( + fence + .acquire_read_lease(OperationClass::NormalRead, LeaseKind::Sql, || {}) + .unwrap(), + ), + Later::Seal => { + let session = server + .store + .open_import_session("tgt".into(), OP, 1) + .await + .unwrap(); + let call = fence.begin_import_write(session.capability()).unwrap(); + drop(session); + assert_eq!(fence.import_writers(), 1); + Box::new(call) + } + Later::Enable => unreachable!("EnableTargetWrites does not drain"), + } + } + + /// Admission of new work in the gate's current state: writes only on a writable target, + /// normal reads wherever the state admits them. + async fn assert_admission( + server: &Server, + fence: &Arc, + ns: &'static str, + name: &str, + ) { + let state = fence.gate().state(); + let writes = state == FenceState::TargetWritable; + let reads = matches!( + state, + FenceState::SourceWriteFenced + | FenceState::TargetWriteFenced + | FenceState::TargetWritable + ); + assert_eq!( + server.writes_admitted_in(ns).await, + writes, + "{name}: write admission in {state}" + ); + let lease = fence.acquire_read_lease(OperationClass::NormalRead, LeaseKind::Sql, || {}); + assert_eq!(lease.is_ok(), reads, "{name}: read admission in {state}"); + } + + /// Kill the process at every point where `SetSourceReadFence`, `SealTargetImport` and + /// `EnableTargetWrites` persist, publish, answer or wait, and restart it on the same + /// directory. As for the source write fence (`restart_at_each_boundary`), the restarted + /// server recovers exactly the state before the command or the state it committed and + /// installs that gate before serving the namespace: reads stay closed once + /// `SOURCE_READ_DRAINING` committed, import stays closed once `TARGET_IMPORT_DRAINING` + /// committed, and target writes open only if `TARGET_WRITABLE` committed. No reader or + /// import call survives a restart, so a replay of the same command completes an + /// interrupted drain at once; a replay of a finished one returns its stored result. + #[test] + fn restart_at_each_read_and_target_boundary() { + for case in LATER_BOUNDARIES { + restart_later_at(case); + } + } + + fn restart_later_at(case: &LaterBoundary) { + let name = case.name; + let ns = case.command.namespace(); + let dir = tempdir().unwrap(); + let dbs = dir.path().join("dbs"); + let request = case.command.request(); + let before = case.command.before(); + + // First lifetime: run the command until it reaches the boundary, then crash. + let server = Server::boot(dir.path()); + prepare(&server, case.command); + let holder = server.run(async { + let fence = server.fence_of(ns).await; + assert_eq!( + (fence.gate().state(), fence.gate().revision()), + before, + "{name}" + ); + let hooks = fence.hooks(); + let mut marker_before = read_marker_bytes_of(&dbs, ns); + let mut holder = None; + let paused = match case.park { + Park::First(point) => { + let paused = hooks.pause_at(point); + let task = server.execute(request.clone()); + reached(&paused, name, point).await; + drop(task); + Some(paused) + } + Park::Second(point) => { + // The first commit's response point comes after its publication and + // before the drain, which has nothing to wait for. + let first = hooks.pause_at(HookPoint::BeforeResponse); + let task = server.execute(request.clone()); + reached(&first, name, HookPoint::BeforeResponse).await; + marker_before = read_marker_bytes_of(&dbs, ns); + let paused = hooks.pause_at(point); + first.resume(); + reached(&paused, name, point).await; + drop(task); + Some(paused) + } + Park::WaitingForHolder => { + holder = Some(hold(&server, &fence, case.command).await); + let (draining, _) = case.command.draining(); + let mut gate = fence.subscribe(); + let task = server.execute(request.clone()); + tokio::time::timeout(PROMPT, gate.wait_for(|g| g.state() == draining)) + .await + .unwrap_or_else(|_| panic!("{name}: the command never started draining")) + .unwrap(); + assert!(!task.is_finished(), "{name}: the drain did not wait"); + drop(task); + None + } + }; + if case.marker_lags { + restore_marker_bytes_of(&dbs, ns, marker_before.as_deref()); + } + // What the metastore holds at the moment of the crash. + let inspected = server.inspect_of(ns).await; + assert_eq!( + (inspected.fence.state(), inspected.fence.revision()), + case.command.state_after(case.durable), + "{name}: durable at the crash" + ); + drop(paused); + holder + }); + server.crash(); + // The reader or import call belonged to the dead process. + drop(holder); + + // Second lifetime. + let server = Server::boot(dir.path()); + server.run(async { + let recovered = case.command.state_after(case.durable); + let fence = server.fence_of(ns).await; + let gate = fence.gate(); + assert_eq!( + (gate.state(), gate.revision()), + recovered, + "{name}: recovered state" + ); + assert!( + gate.indeterminate.is_none() + && !gate.is_installing() + && gate.closing_reads.is_none(), + "{name}: an in-memory gate survived the restart" + ); + assert_eq!(fence.read_lease_counts().total(), 0, "{name}"); + assert_eq!(fence.import_writers(), 0, "{name}"); + assert_admission(&server, &fence, ns, name).await; + // Committed data survived, nothing else was written. + assert_eq!(server.count_in(ns).await, ROWS, "{name}"); + // The marker was repaired if it had fallen behind. + let marker = fence_store::read_marker(&dbs, &ns.into()).unwrap(); + assert_eq!( + marker.and_then(|m| m.ok()).map(|m| m.record.revision), + gate.fence.record().map(|r| r.revision), + "{name}: marker" + ); + if case.command == Later::Seal { + // Import resumes only if the seal never committed. + let session = server.store.open_import_session(ns.into(), OP, 1).await; + match case.durable { + Durable::Nothing => drop(session.unwrap()), + Durable::Draining | Durable::Final => { + assert!( + matches!(session, Err(Error::NamespaceFence(_))), + "{name}: import reopened after the seal committed" + ); + } + } + } + + let replay = server.execute(request.clone()).await.unwrap().unwrap(); + let kind = match case.durable { + Durable::Nothing => FenceCommitKind::Committed, + Durable::Draining => FenceCommitKind::Resumed, + Durable::Final => FenceCommitKind::Replayed, + }; + assert_eq!(replay.kind, kind, "{name}"); + assert_eq!(replay.receipt.outcome, FenceOutcome::Applied, "{name}"); + assert_eq!(replay.receipt.command_id, request.command_id, "{name}"); + assert_eq!(replay.receipt.revision_before, before.1, "{name}"); + assert_eq!( + (replay.receipt.state_after, replay.receipt.revision_after), + case.command.applied(), + "{name}" + ); + + // Settled: the gate is the durable state, and a further replay answers the same. + let gate = fence.gate(); + let durable = server.inspect_of(ns).await; + assert_eq!(gate.fence, durable.fence, "{name}"); + assert_eq!( + (gate.state(), gate.revision()), + case.command.applied(), + "{name}" + ); + assert_admission(&server, &fence, ns, name).await; + assert_eq!(server.count_in(ns).await, ROWS, "{name}"); + let again = server.execute(request.clone()).await.unwrap().unwrap(); + assert_eq!(again.kind, FenceCommitKind::Replayed, "{name}"); + assert_eq!(again.receipt, replay.receipt, "{name}"); + }); + server.crash(); + } +} + +// --------------------------------------------------------------------------------------------- +// Protection against an older binary (section 13.2) + +mod legacy_mirror { + use super::*; + use crate::namespace::fence::command::ValidationResult; + use crate::namespace::fence::target::tests::{create_request, enable_request, target_command}; + use crate::namespace::meta_store::{metastore_connection_maker, MetaStoreConnection}; + use libsql_replication::rpc::metadata; + + /// A metastore connection set up the way an older binary sets up its own: foreign keys on, + /// and no knowledge of the fence tables. + async fn older_binary(dir: &Path) -> MetaStoreConnection { + let (maker, _) = metastore_connection_maker(None, dir).await.unwrap(); + let conn = maker().unwrap(); + conn.execute("PRAGMA foreign_keys=ON", ()).unwrap(); + conn + } + + fn encoded(config: &DatabaseConfig) -> metadata::DatabaseConfig { + metadata::DatabaseConfig::from(config) + } + + fn stored(meta: &rusqlite::Connection, ns: &'static str) -> DatabaseConfig { + fence_store::read_config_row(meta, &ns.into()) + .unwrap() + .expect("the namespace has a config row") + } + + /// The `block_*` values section 13.2 says a fenced namespace's stored config holds in + /// `state`, derived from the permission matrix (section 3.3) rather than from the record. + fn mirror(state: FenceState) -> (bool, bool, Option) { + let (block_reads, block_writes) = match state { + FenceState::SourceDraining + | FenceState::SourceWriteFenced + | FenceState::TargetWriteFenced => (false, true), + FenceState::SourceReadDraining + | FenceState::SourceReadFenced + | FenceState::TargetQuarantined + | FenceState::TargetImportDraining + | FenceState::TargetValidating + | FenceState::TargetAborted => (true, true), + other => unreachable!("{other} does not mirror the fence"), + }; + let reason = format!("namespace fence: {state} (operation {OP})"); + (block_reads, block_writes, Some(reason)) + } + + /// In `state`, the stored config of `ns` is the namespace's own config `own` with the fence + /// mirrored into its `block_*` fields (or with its own values once the operation has + /// finished); the in-memory config is `own`; a config write through the metastore is + /// refused and changes nothing while the fence denies lifecycle work; and an older + /// binary's delete of the config row fails on the fence row's foreign key. + async fn check( + server: &Server, + meta: &rusqlite::Connection, + ns: &'static str, + own: &DatabaseConfig, + state: FenceState, + ) { + let fence = server.fence_of(ns).await; + assert_eq!(fence.gate().state(), state, "{ns}"); + let row = stored(meta, ns); + let blocks = (row.block_reads, row.block_writes, row.block_reason.clone()); + let finished = matches!( + state, + FenceState::Unfenced | FenceState::Released | FenceState::TargetWritable + ); + if finished { + assert_eq!( + blocks, + (own.block_reads, own.block_writes, own.block_reason.clone()), + "{ns} in {state}: the namespace's own block_* values" + ); + } else { + assert_eq!(blocks, mirror(state), "{ns} in {state}: the legacy mirror"); + } + // Only the block_* fields carry the mirror. + assert_eq!( + encoded(&fence_store::with_legacy_blocks( + &row, + &fence_store::legacy_blocks_of(own) + )), + encoded(own), + "{ns} in {state}" + ); + let handle = server + .store + .meta_store() + .lookup(&ns.into()) + .await + .unwrap() + .expect("the namespace has a config"); + assert_eq!( + encoded(&handle.get()), + encoded(own), + "{ns} in {state}: in memory" + ); + + if !finished { + let overwrite = DatabaseConfig { + block_reads: false, + block_writes: false, + block_reason: None, + max_db_pages: own.max_db_pages + 1, + ..own.clone() + }; + match handle.store(overwrite).await { + Err(Error::NamespaceFence(_)) => (), + other => panic!("{ns} in {state}: config write not refused: {other:?}"), + } + assert_eq!(encoded(&stored(meta, ns)), encoded(&row), "{ns} in {state}"); + assert_eq!(encoded(&handle.get()), encoded(own), "{ns} in {state}"); + } + + if state != FenceState::Unfenced { + // SQLite enforces `ON DELETE RESTRICT` with an action trigger, so the refusal is + // `SQLITE_CONSTRAINT_TRIGGER` carrying the foreign key message. + match meta.execute("DELETE FROM namespace_configs WHERE namespace = ?1", [ns]) { + Err(rusqlite::Error::SqliteFailure(e, message)) => { + assert_eq!( + e.code, + ErrorCode::ConstraintViolation, + "{ns} in {state}: {e}" + ); + assert_eq!( + message.as_deref(), + Some("FOREIGN KEY constraint failed"), + "{ns} in {state}" + ); + } + other => { + panic!("{ns} in {state}: an older binary's delete was not refused: {other:?}") + } + } + assert_eq!(encoded(&stored(meta, ns)), encoded(&row), "{ns} in {state}"); + } + } + + fn applied(result: Result, tokio::task::JoinError>) -> FenceCommit { + let commit = result.unwrap().unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + commit + } + + fn source_command( + command_id: u128, + expected_state: FenceState, + expected_revision: u64, + command: FenceCommand, + ) -> FenceRequest { + FenceRequest { + namespace: "ns".into(), + operation_id: OP, + command_id: Uuid::from_u128(command_id), + expected_state, + expected_revision, + command, + } + } + + /// Store `config` through the metastore, as `POST /v1/namespaces/:ns/config` does. + async fn store_config(server: &Server, ns: &'static str, config: &DatabaseConfig) { + server + .store + .meta_store() + .lookup(&ns.into()) + .await + .unwrap() + .unwrap() + .store(config.clone()) + .await + .unwrap(); + } + + /// Walk a source and two targets through every stored state: in each, the config row + /// carries the legacy mirror of section 13.2 (reads blocked where the state denies reads, + /// writes blocked where it denies writes, and a reason naming the state and the + /// operation), nothing else in the row changes, a config write cannot overwrite it, and the + /// foreign key refuses an older binary's delete. Release and write enable put the + /// namespace's own values back, after which config writes follow the existing policy; the + /// foreign key stays with the fence row. A restart keeps the rows and gives the in-memory + /// config the namespace's own values. + #[test] + fn legacy_mirror_and_fk_guard() { + let dir = tempdir().unwrap(); + let server = Server::boot(dir.path()); + server.create_source(); + let owns = server.run(async { + let meta = older_binary(dir.path()).await; + let mut own = DatabaseConfig { + block_reason: Some("pre-fence note".into()), + max_db_pages: 1234, + ..(*server + .store + .meta_store() + .lookup(&"ns".into()) + .await + .unwrap() + .unwrap() + .get()) + .clone() + }; + store_config(&server, "ns", &own).await; + check(&server, &meta, "ns", &own, FenceState::Unfenced).await; + + // SOURCE_DRAINING, parked after its commit and before the boundary is captured. + let fence = server.fence().await; + let (log_id, _) = server.log().await; + let paused = fence.hooks().pause_at(HookPoint::BeforeBoundaryCapture); + let task = server.execute(acquire(log_id, 1, LONG)); + reached(&paused, "acquire", HookPoint::BeforeBoundaryCapture).await; + check(&server, &meta, "ns", &own, FenceState::SourceDraining).await; + paused.resume(); + applied(task.await); + check(&server, &meta, "ns", &own, FenceState::SourceWriteFenced).await; + + // SOURCE_READ_DRAINING, parked after its publication and before the drain. + let paused = fence.hooks().pause_at(HookPoint::BeforeResponse); + let task = server.execute(source_command( + 2, + FenceState::SourceWriteFenced, + 2, + FenceCommand::SetSourceReadFence { + drain_policy: Some(LONG), + }, + )); + reached(&paused, "read fence", HookPoint::BeforeResponse).await; + check(&server, &meta, "ns", &own, FenceState::SourceReadDraining).await; + paused.resume(); + applied(task.await); + check(&server, &meta, "ns", &own, FenceState::SourceReadFenced).await; + + applied( + server + .execute(source_command( + 3, + FenceState::SourceReadFenced, + 4, + FenceCommand::ClearSourceReadFence, + )) + .await, + ); + check(&server, &meta, "ns", &own, FenceState::SourceWriteFenced).await; + applied(server.execute(release(4, 5)).await); + check(&server, &meta, "ns", &own, FenceState::Released).await; + // Released: config writes follow the existing policy and are stored as written. + own = DatabaseConfig { + block_writes: true, + block_reason: Some("after the operation".into()), + max_db_pages: 4321, + ..own + }; + store_config(&server, "ns", &own).await; + check(&server, &meta, "ns", &own, FenceState::Released).await; + + // A target, created with the default block_* values. + applied(Ok(server + .store + .create_target_quarantined(create_request("tgt", 1), server_identity()) + .await)); + let mut own_target = (*server + .store + .meta_store() + .lookup(&"tgt".into()) + .await + .unwrap() + .unwrap() + .get()) + .clone(); + assert!( + !own_target.block_reads + && !own_target.block_writes + && own_target.block_reason.is_none() + ); + check( + &server, + &meta, + "tgt", + &own_target, + FenceState::TargetQuarantined, + ) + .await; + + // TARGET_IMPORT_DRAINING, parked after its publication and before the drain. + let target = server.fence_of("tgt").await; + let paused = target.hooks().pause_at(HookPoint::BeforeResponse); + let task = server.execute(target_command( + 10, + FenceState::TargetQuarantined, + 1, + FenceCommand::SealTargetImport { + drain_policy: Some(LONG), + }, + )); + reached(&paused, "seal", HookPoint::BeforeResponse).await; + check( + &server, + &meta, + "tgt", + &own_target, + FenceState::TargetImportDraining, + ) + .await; + paused.resume(); + applied(task.await); + check( + &server, + &meta, + "tgt", + &own_target, + FenceState::TargetValidating, + ) + .await; + + applied( + server + .execute(target_command( + 20, + FenceState::TargetValidating, + 3, + FenceCommand::RecordTargetValidation { + result: ValidationResult::Ok, + summary: "rows match".into(), + }, + )) + .await, + ); + check( + &server, + &meta, + "tgt", + &own_target, + FenceState::TargetValidating, + ) + .await; + applied( + server + .execute(target_command( + 21, + FenceState::TargetValidating, + 4, + FenceCommand::PublishTargetReadableWriteFenced, + )) + .await, + ); + check( + &server, + &meta, + "tgt", + &own_target, + FenceState::TargetWriteFenced, + ) + .await; + applied(server.execute(enable_request(30)).await); + check( + &server, + &meta, + "tgt", + &own_target, + FenceState::TargetWritable, + ) + .await; + own_target = DatabaseConfig { + max_db_pages: 777, + ..own_target + }; + store_config(&server, "tgt", &own_target).await; + check( + &server, + &meta, + "tgt", + &own_target, + FenceState::TargetWritable, + ) + .await; + + // An aborted target keeps everything blocked. + applied(Ok(server + .store + .create_target_quarantined(create_request("tgt2", 40), server_identity()) + .await)); + let own_aborted = (*server + .store + .meta_store() + .lookup(&"tgt2".into()) + .await + .unwrap() + .unwrap() + .get()) + .clone(); + applied( + server + .execute(FenceRequest { + namespace: "tgt2".into(), + operation_id: OP, + command_id: Uuid::from_u128(41), + expected_state: FenceState::TargetQuarantined, + expected_revision: 1, + command: FenceCommand::AbortQuarantinedTarget, + }) + .await, + ); + check( + &server, + &meta, + "tgt2", + &own_aborted, + FenceState::TargetAborted, + ) + .await; + (own, own_target, own_aborted) + }); + server.crash(); + + // The rows are kept across a restart, and the in-memory config is the namespace's own. + let (own, own_target, own_aborted) = owns; + let server = Server::boot(dir.path()); + server.run(async { + let meta = older_binary(dir.path()).await; + check(&server, &meta, "ns", &own, FenceState::Released).await; + check( + &server, + &meta, + "tgt", + &own_target, + FenceState::TargetWritable, + ) + .await; + check( + &server, + &meta, + "tgt2", + &own_aborted, + FenceState::TargetAborted, + ) + .await; + }); + server.crash(); + } +} diff --git a/libsql-server/src/namespace/fence/transition.rs b/libsql-server/src/namespace/fence/transition.rs index 9c664b320c..1a4b5fceb7 100644 --- a/libsql-server/src/namespace/fence/transition.rs +++ b/libsql-server/src/namespace/fence/transition.rs @@ -10,8 +10,9 @@ //! Checks run in this order, and the order is part of the contract: //! //! 1. **Replay.** A stored receipt with the same fingerprint is answered from the receipt -//! (`Replay`, or `Resume` for a drain still in progress), whatever has happened to the -//! record since. A stored receipt with a different fingerprint is `FENCE_COMMAND_CONFLICT`. +//! (`Resume` only while the record is still in that command's draining state and revision, +//! otherwise `Replay`), whatever has happened to the record since. A stored receipt with a +//! different fingerprint is `FENCE_COMMAND_CONFLICT`. //! 2. **Unavailable state.** A record the server cannot establish refuses everything with //! `FENCE_STATE_UNAVAILABLE`, except the two commands that can reconcile it: a replay of the //! `CreateTargetQuarantined` that left the marker, and an adoption after a metastore @@ -260,10 +261,22 @@ pub fn apply( ), )); } - return Ok(if existing.is_final() { - Decision::Replay(existing.clone()) - } else { + // A DRAINING receipt resumes only while its exact drain is still the durable state. + // Release, clear-read and abort may supersede one, and a source may begin another read + // drain later under the same operation and state; the record revision distinguishes that + // later cycle. Replaying an older command must return its stored answer without running + // it against the newer record (and potentially cancelling newly admitted work). + let still_draining = matches!( + current, + CurrentFence::Record(record) + if record.operation_id == existing.operation_id + && drain_state(existing.command) == Some(record.state) + && existing.revision_after == record.revision + ); + return Ok(if !existing.is_final() && still_draining { Decision::Resume(existing.clone()) + } else { + Decision::Replay(existing.clone()) }); } @@ -646,9 +659,12 @@ pub fn complete_drain( completion: DrainCompletion, env: &ApplyEnv, ) -> Result<(NamespaceFenceRecord, CommandReceipt), FenceError> { - if receipt.outcome != FenceOutcome::Draining || receipt.operation_id != record.operation_id { + if receipt.outcome != FenceOutcome::Draining + || receipt.operation_id != record.operation_id + || receipt.revision_after != record.revision + { return Err(invalid( - "there is no drain of the owning operation to complete", + "there is no matching drain of the owning operation to complete", )); } @@ -1274,6 +1290,57 @@ mod tests { assert_eq!(receipt.revision_after, 2); } + #[test] + fn superseded_draining_receipt_is_replayed_without_resuming() { + let mut source = Harness::source(); + let acquire = source.request(OP, command(CommandKind::AcquireSourceWriteFence)); + source.run_request(&acquire, &env()).unwrap(); + source + .run(OP, CommandKind::ReleaseSourceWriteFence) + .unwrap(); + let replay = source.decide(&acquire, &env()).unwrap(); + assert!(matches!( + replay, + Decision::Replay(ref receipt) if receipt.outcome == O::Draining + )); + assert_eq!(source.state(), S::Released); + + let mut source = Harness::in_state(S::SourceWriteFenced); + let read_fence = source.request(OP, command(CommandKind::SetSourceReadFence)); + source.run_request(&read_fence, &env()).unwrap(); + source.run(OP, CommandKind::ClearSourceReadFence).unwrap(); + let replay = source.decide(&read_fence, &env()).unwrap(); + assert!(matches!( + replay, + Decision::Replay(ref receipt) if receipt.outcome == O::Draining + )); + assert_eq!(source.state(), S::SourceWriteFenced); + + // Starting another read drain under the same operation and state does not make the + // earlier cycle live again: only receipts at the current drain revision may resume it. + let later_read_fence = source.request(OP, command(CommandKind::SetSourceReadFence)); + source.run_request(&later_read_fence, &env()).unwrap(); + assert!(matches!( + source.decide(&read_fence, &env()).unwrap(), + Decision::Replay(ref receipt) if receipt.outcome == O::Draining + )); + assert!(matches!( + source.decide(&later_read_fence, &env()).unwrap(), + Decision::Resume(ref receipt) if receipt.outcome == O::Draining + )); + + let mut target = Harness::in_state(S::TargetQuarantined); + let seal = target.request(OP, command(CommandKind::SealTargetImport)); + target.run_request(&seal, &env()).unwrap(); + target.run(OP, CommandKind::AbortQuarantinedTarget).unwrap(); + let replay = target.decide(&seal, &env()).unwrap(); + assert!(matches!( + replay, + Decision::Replay(ref receipt) if receipt.outcome == O::Draining + )); + assert_eq!(target.state(), S::TargetAborted); + } + #[test] fn command_id_reuse_with_different_fingerprint_conflicts() { let mut h = Harness::source(); @@ -1515,6 +1582,9 @@ mod tests { let mut other = receipt.clone(); other.operation_id = OTHER_OP; assert!(complete_drain(&record, &other, boundary, &env()).is_err()); + let mut stale = receipt.clone(); + stale.revision_after -= 1; + assert!(complete_drain(&record, &stale, boundary, &env()).is_err()); // Not draining any more. let h = Harness::in_state(S::SourceWriteFenced); diff --git a/libsql-server/src/namespace/meta_store.rs b/libsql-server/src/namespace/meta_store.rs index ebd1ba2a64..5f67045640 100644 --- a/libsql-server/src/namespace/meta_store.rs +++ b/libsql-server/src/namespace/meta_store.rs @@ -36,7 +36,7 @@ use super::fence::outcome::{FenceDetail, FenceError, FenceOutcome}; use super::fence::record::{ CommandReceipt, NamespaceFenceRecord, ServerIdentity, ValidationSnapshot, }; -use super::fence::state::OperationClass; +use super::fence::state::{OperationClass, Role}; use super::fence::store::{ self as fence_store, FenceStoreError, MarkerStatus, StoredFence, StoredReceipt, }; @@ -466,8 +466,11 @@ impl MetaStoreInner { /// Load every namespace's fence after the configs (section 5.6). The stored config row of /// a fenced namespace carries the legacy mirror of the fence in its `block_*` fields - /// (section 13.2); the in-memory config is the namespace's own configuration, so those - /// fields are put back to the values the record saved. A marker that fell behind its + /// (section 13.2) while the record is in force; the in-memory config is the namespace's own + /// configuration, so those fields are put back to the values the record saved + /// ([`fence_store::own_config`]). Once the operation has released the namespace or enabled + /// target writes, the row holds the namespace's own values (including any config written + /// since) and is used as it is. A marker that fell behind its /// record is rewritten. A namespace whose fence cannot be established is logged and keeps /// its stored config, mirror included. fn restore_fences(&mut self) -> Result<()> { @@ -501,7 +504,7 @@ impl MetaStoreInner { } let sender = self.configs.get_mut().get_mut(&ns).expect("listed above"); let config = sender.borrow().config.clone(); - let config = fence_store::with_legacy_blocks(&config, &record.legacy_blocks); + let config = fence_store::own_config(&config, record); sender.send_modify(|c| c.config = Arc::new(config)); } StoredFence::Unavailable { @@ -884,14 +887,17 @@ fn apply_fence_command( let decision = transition::apply(stored.as_current(), existing.as_ref(), request, &env)?; let (record, receipt) = match decision { - Decision::Replay(receipt) | Decision::Resume(receipt) => { - let kind = if receipt.is_final() { - FenceCommitKind::Replayed - } else { - FenceCommitKind::Resumed - }; + Decision::Replay(receipt) => { return Ok(FenceCommit { - kind, + kind: FenceCommitKind::Replayed, + receipt, + record: stored.record().cloned(), + created_config: None, + }); + } + Decision::Resume(receipt) => { + return Ok(FenceCommit { + kind: FenceCommitKind::Resumed, receipt, record: stored.record().cloned(), created_config: None, @@ -1379,6 +1385,57 @@ impl MetaStore { .map_err(fence_store_error) } + /// Make a migration target that the metastore holds visible in the in-memory config map, + /// which is what makes `exists()` and `lookup()` find it (section 10.1, step 5). The config + /// published is the namespace's own config ([`fence_store::own_config`]): the stored row + /// with the record's saved `block_*` values in place of the legacy mirror, or the row + /// itself once target writes are enabled, as `restore_fences` does at startup. The caller has already installed + /// the target's gate. Returns whether the map changed; `false` also when the namespace is + /// not a target with a stored config. + pub async fn publish_target_config(&self, namespace: NamespaceName) -> Result { + let inner = self.inner.clone(); + tokio::task::spawn_blocking(move || -> std::result::Result { + // The connection lock first, as everywhere else that takes both. + let mut conn = inner.conn.blocking_lock(); + let tx = conn.transaction()?; + let (stored, _) = fence_store::read_fence(&tx, &inner.dbs_path, &namespace)?; + let record = match stored { + StoredFence::Record(r) if r.role == Role::Target => r, + _ => return Ok(false), + }; + let Some(row) = fence_store::read_config_row(&tx, &namespace)? else { + return Ok(false); + }; + drop(tx); + let config = Arc::new(fence_store::own_config(&row, &record)); + let mut configs = inner.configs.blocking_lock(); + match configs.get_mut(&namespace) { + Some(sender) + if metadata::DatabaseConfig::from(&*sender.borrow().config) + == metadata::DatabaseConfig::from(&*config) => + { + Ok(false) + } + // An entry that was put in the map by a create or fork of the same name that + // the fence then refused: the durable config replaces it. + Some(sender) => { + sender.send_modify(|c| { + c.version = c.version.wrapping_add(1); + c.config = config; + }); + Ok(true) + } + None => { + let (tx, _) = watch::channel(InnerConfig { version: 0, config }); + configs.insert(namespace, tx); + Ok(true) + } + } + }) + .await? + .map_err(fence_store_error) + } + /// Read a namespace's fence and all of its receipts (`InspectFence`). Never writes. pub async fn inspect_fence(&self, namespace: NamespaceName) -> Result { let inner = self.inner.clone(); diff --git a/libsql-server/src/namespace/mod.rs b/libsql-server/src/namespace/mod.rs index f75dbd700f..69d93368f3 100644 --- a/libsql-server/src/namespace/mod.rs +++ b/libsql-server/src/namespace/mod.rs @@ -28,6 +28,9 @@ pub mod replication_wal; mod schema_lock; mod store; +#[cfg(test)] +pub(crate) use store::fence_tests::open_store as open_test_store; + pub type ResetCb = Box; /// Resolves a namespace that a program ATTACHes: its directory, and its fence controller, which /// admits the attachment as a read of that namespace (`docs/NAMESPACE_FENCE.md` section 9). diff --git a/libsql-server/src/namespace/store.rs b/libsql-server/src/namespace/store.rs index ca8e7031be..889370ad79 100644 --- a/libsql-server/src/namespace/store.rs +++ b/libsql-server/src/namespace/store.rs @@ -13,7 +13,7 @@ use tokio_stream::wrappers::BroadcastStream; use crate::auth::Authenticated; use crate::broadcaster::BroadcastMsg; use crate::connection::config::DatabaseConfig; -use crate::database::DatabaseKind; +use crate::database::{Database, DatabaseKind, PrimaryConnectionMaker}; use crate::error::Error; use crate::metrics::NAMESPACE_LOAD_LATENCY; use crate::namespace::{NamespaceBottomlessDbId, NamespaceBottomlessDbIdInit, NamespaceName}; @@ -21,10 +21,17 @@ use crate::stats::Stats; use super::broadcasters::{BroadcasterHandle, BroadcasterRegistry}; use super::configurator::{DynConfigurator, NamespaceConfigurators}; +use super::fence::capability::CapabilityPurpose; use super::fence::command::{FenceCommand, FenceRequest}; -use super::fence::controller::FenceController; +use super::fence::controller::{FenceController, Transition}; +use super::fence::hooks::HookPoint; +use super::fence::import::{self, ImportSession}; +use super::fence::outcome::FenceOutcome; use super::fence::record::ServerIdentity; use super::fence::registry::FenceRegistry; +use super::fence::state::{FenceState, Role}; +use super::fence::store::StoredFence; +use super::fence::target::{self, CreateTargetRequest, ValidationSession}; use super::meta_store::{FenceCommit, FenceContext, MetaStore, MetaStoreHandle}; use super::schema_lock::SchemaLocksRegistry; use super::{Namespace, ResetCb, ResetOp, ResolveNamespacePathFn, RestoreOption}; @@ -237,6 +244,10 @@ impl NamespaceStore { return Err(Error::NamespaceStoreShutdown); } + // The destination is refused before anything is stored for it when it is being created + // as a migration target or its fence state is unknown. + self.inner.fences.check_available(&to)?; + // check that the source namespace exists if !self.inner.metadata.exists(&from).await { return Err(crate::error::Error::NamespaceDoesntExist(from.to_string())); @@ -450,6 +461,9 @@ impl NamespaceStore { restore_option: RestoreOption, db_config: DatabaseConfig, ) -> crate::Result<()> { + // A name that is being created as a migration target, or whose fence state is unknown, + // is refused before anything is stored for it. + self.inner.fences.check_available(&namespace)?; if let Some(shared_schema_name) = &db_config.shared_schema_name { // we hold a lock for the duration of the namespace creation let _lock = self @@ -550,19 +564,304 @@ impl NamespaceStore { server: ServerIdentity, ) -> crate::Result { let controller = match request.command { + FenceCommand::CreateTargetQuarantined { .. } => { + return self.run_create_target(request, server).await + } FenceCommand::AcquireSourceWriteFence { .. } => { self.with(request.namespace.clone(), |ns| ns.fence().clone()) .await? } _ => self.inner.fences.controller(&request.namespace), }; - controller - .execute( - &self.inner.metadata, - request, - FenceContext::now(server, None), - ) + let mut ctx = FenceContext::now(server, None); + // A new validation receipt records what this server observes of the sealed target. Do + // this only when the live gate exactly matches the request: an exact replay after the + // revision or state has advanced must reach the metastore's replay check without first + // trying to issue a now-invalid validation capability. + if matches!( + &request.command, + FenceCommand::RecordTargetValidation { .. } + ) { + let needs_snapshot = { + let gate = controller.gate(); + gate.state() == FenceState::TargetValidating + && gate.operation_id() == Some(request.operation_id) + && gate.revision() == request.expected_revision + }; + if needs_snapshot && !self.validation_command_recorded(&request).await? { + let snapshot = async { + let mut session = self + .open_validation_session( + request.namespace.clone(), + request.operation_id, + request.expected_revision, + ) + .await?; + session.snapshot().await + } + .await; + match snapshot { + Ok(snapshot) => ctx.validation_snapshot = Some(snapshot), + // A concurrent copy of this command can commit between the gate check and + // the capability call. Once its receipt exists, let execute take the + // transition lock and perform the authoritative replay/fingerprint check. + Err(_) if self.validation_command_recorded(&request).await? => {} + Err(e) => return Err(e), + } + } + } + controller.execute(&self.inner.metadata, request, ctx).await + } + + async fn validation_command_recorded(&self, request: &FenceRequest) -> crate::Result { + let operation_id = request.operation_id.to_string(); + let command_id = request.command_id.to_string(); + let inspected = self + .inner + .metadata + .inspect_fence(request.namespace.clone()) + .await?; + Ok(inspected + .receipts + .iter() + .any(|stored| stored.operation_id == operation_id && stored.command_id == command_id)) + } + + /// `CreateTargetQuarantined`, atomic with namespace creation (`docs/NAMESPACE_FENCE.md` + /// sections 10.1 and 11): the namespace is quarantined from the first instant it can be + /// observed, and is loaded behind the quarantine gate before this returns `APPLIED`. + /// + /// Fence refusals are [`Error::NamespaceFence`] with their stable outcome code. The work + /// runs on its own task under the namespace's transition lock, so a caller that goes away + /// does not interrupt it; replaying the same request returns the stored result, completes + /// the publication and load of a target whose commit was not acknowledged, and completes a + /// creation interrupted between its marker and its commit. + pub async fn create_target_quarantined( + &self, + request: CreateTargetRequest, + server: ServerIdentity, + ) -> crate::Result { + self.run_create_target(request.into(), server).await + } + + async fn run_create_target( + &self, + request: FenceRequest, + server: ServerIdentity, + ) -> crate::Result { + if self.inner.has_shutdown.load(Ordering::Relaxed) { + return Err(Error::NamespaceStoreShutdown); + } + let controller = self.inner.fences.controller(&request.namespace); + let this = self.clone(); + tokio::spawn(async move { + let mut transition = controller.begin_transition().await; + let ctx = FenceContext::now(server, None); + this.create_target_under(&mut transition, request, ctx) + .await + }) + .await? + } + + /// Section 10.1, steps 1 to 5, under `transition`. + async fn create_target_under( + &self, + transition: &mut Transition, + request: FenceRequest, + ctx: FenceContext, + ) -> crate::Result { + let controller = transition.controller().clone(); + let namespace = request.namespace.clone(); + let key = (request.operation_id, request.command_id); + + // A name with no fence state gets the target-creation gate before the metastore + // transaction writes the marker, so nothing can set the name up, serve it or store a + // config for it from before the commit to the load. A name that already has fence + // state (a replay, a creation interrupted after its marker, or a refusal) already has + // the gate its state implies. A name the server already knows is refused without + // touching its gate; it is checked again once the gate is in place, which closes the + // race with a create or fork that has not stored anything yet. + let fresh = { + let gate = controller.gate(); + matches!(gate.fence, StoredFence::None { .. }) && gate.indeterminate.is_none() + }; + if fresh { + if self.name_in_use(&namespace).await { + return Err(target::name_in_use(&namespace).into()); + } + transition.install_creating_target(key); + let _ = controller.hook(HookPoint::AfterInstallingGate).await; + if self.name_in_use(&namespace).await { + transition.remove_creating_target(); + return Err(target::name_in_use(&namespace).into()); + } + } + + // Steps 2 and 3: the marker, then the rows in one transaction. The commit publishes the + // quarantined record in place of the creation gate. A command proven not to have + // committed removes the creation gate; one whose commit is unknown keeps it, with the + // indeterminate flag, until the same command is replayed. + let commit = match transition.apply(&self.inner.metadata, request, ctx).await { + Ok(commit) => commit, + Err(e) => { + if controller.gate().indeterminate != Some(key) { + transition.remove_creating_target(); + } + return Err(e); + } + }; + let _ = controller.hook(HookPoint::AfterTargetRowsCommitted).await; + + // Steps 4 and 5: the gate is the target's; now make the config visible and load the + // namespace, whose first connection is created behind that gate. A replay does the same, + // which completes a creation whose commit was not acknowledged. + let is_target = commit + .record + .as_ref() + .is_some_and(|r| r.role == Role::Target && r.namespace == namespace); + if is_target { + self.inner + .metadata + .publish_target_config(namespace.clone()) + .await?; + self.clear_empty_entry(&namespace).await; + let loaded = self + .with(namespace.clone(), |ns| ns.fence().clone()) + .await?; + debug_assert!(Arc::ptr_eq(&loaded, &controller)); + } + if commit.receipt.outcome == FenceOutcome::Applied && commit.created_config.is_some() { + tracing::info!( + namespace = %namespace, + operation_id = %key.0, + command_id = %key.1, + "created namespace as a quarantined migration target" + ); + } + Ok(commit) + } + + /// Issue an import capability to `operation_id` and open a connection that works under it + /// (`docs/NAMESPACE_FENCE.md` section 11): the only way to write into a quarantined target. + /// Valid only while the target is `TARGET_QUARANTINED`, owned by `operation_id`, at + /// `expected_revision`; the target is loaded if it is not. Fence refusals are + /// [`Error::NamespaceFence`] with their stable outcome code (`OPERATION_CAPABILITY_REQUIRED` + /// in any other state, `FENCE_OWNED_BY_ANOTHER_OPERATION`, `FENCE_REVISION_MISMATCH`). + pub async fn open_import_session( + &self, + namespace: NamespaceName, + operation_id: uuid::Uuid, + expected_revision: u64, + ) -> crate::Result { + let (controller, maker) = self.primary_maker(&namespace).await?; + let capability = controller.issue_capability( + CapabilityPurpose::Import, + operation_id, + expected_revision, + )?; + match maker + .inner() + .make_capability_connection(capability.clone()) .await + { + Ok(conn) => Ok(ImportSession::new(capability, controller, conn)), + Err(e) => { + controller.revoke_capability(capability.id()); + Err(e) + } + } + } + + /// Issue a read-only validation capability to `operation_id` and open a `query_only` + /// connection under it (`docs/NAMESPACE_FENCE.md` sections 10.3 and 11). Valid only while + /// the target is `TARGET_VALIDATING` or `TARGET_WRITE_FENCED`, owned by the operation at + /// `expected_revision`; the target is loaded if it is not. + pub async fn open_validation_session( + &self, + namespace: NamespaceName, + operation_id: uuid::Uuid, + expected_revision: u64, + ) -> crate::Result { + let (controller, maker) = self.primary_maker(&namespace).await?; + let capability = controller.issue_capability( + CapabilityPurpose::Validate, + operation_id, + expected_revision, + )?; + let conn = match maker + .inner() + .make_capability_connection(capability.clone()) + .await + { + Ok(conn) => conn, + Err(e) => { + controller.revoke_capability(capability.id()); + return Err(e); + } + }; + ValidationSession::new(capability, controller, conn).await + } + + /// The fence controller and connection maker of the primary `namespace`, loading it. + async fn primary_maker( + &self, + namespace: &NamespaceName, + ) -> crate::Result<(Arc, Arc)> { + let (controller, maker) = self + .with(namespace.clone(), |ns| { + let maker = match &ns.db { + Database::Primary(p) => Some(p.connection_maker()), + _ => None, + }; + (ns.fence().clone(), maker) + }) + .await?; + match maker { + Some(maker) => Ok((controller, maker)), + None => Err(import::not_importable(namespace)), + } + } + + /// A connection to `namespace` that works under `capability`, whether or not this server + /// issued it: for tests that prove the WAL refuses a capability that is not the valid one. + #[cfg(test)] + pub(crate) async fn capability_connection( + &self, + namespace: &NamespaceName, + capability: super::fence::capability::MigrationCapability, + ) -> crate::Result< + crate::connection::legacy::LegacyConnection, + > { + let (_, maker) = self.primary_maker(namespace).await?; + maker.inner().make_capability_connection(capability).await + } + + /// Whether the server already knows `namespace`: its config is in memory, or the namespace + /// cache holds it (a loaded namespace, or a fork in flight, which holds its entry locked). + async fn name_in_use(&self, namespace: &NamespaceName) -> bool { + if self.inner.metadata.exists(namespace).await { + return true; + } + match self.inner.store.get(namespace).await { + Some(entry) => entry.read().await.is_some(), + None => false, + } + } + + /// Drop an empty namespace-cache entry for `namespace`, which a refused fork or a checkpoint + /// of a name that did not exist leaves behind, so that loading the namespace creates it. + async fn clear_empty_entry(&self, namespace: &NamespaceName) { + if let Some(entry) = self.inner.store.get(namespace).await { + if entry.read().await.is_none() { + self.inner.store.invalidate(namespace).await; + } + } + } + + /// The fence controller of `namespace`, creating an `UNFENCED` one if it has none. + #[cfg(test)] + pub(crate) fn fence_controller(&self, namespace: &NamespaceName) -> Arc { + self.inner.fences.controller(namespace) } /// The fence controller that admits reads of `namespace` without loading it: `None` when