From dc79909875ac9111ca9d412f38e2647cb08674ac Mon Sep 17 00:00:00 2001 From: River Date: Tue, 29 Sep 2026 15:24:54 +0000 Subject: [PATCH 1/5] libsql-server: add fence registry and controller, install before first connection Add the in-memory authority for namespace fences: - `FenceRegistry`, held by `NamespaceStore` outside the namespace cache and seeded from `MetaStore::load_fences()` before anything is served, so an evicted and reloaded namespace gets the controller it had. Namespaces without fence state get an UNFENCED controller on first load; deleting a namespace drops its controller. - `FenceController`: a per-namespace transition lock and a `watch` gate (`GateSnapshot`: the durable fence, a write generation and an indeterminate flag). Commands commit in the metastore, are published to the gate and only then answered, on their own task so a lost response does not lose the publication. An error before COMMIT leaves the gate unchanged; a failed COMMIT closes the gate and refuses other commands with FENCE_COMMIT_INDETERMINATE until the same command is replayed. - `FenceConnState`, bound to the controller for every connection a `MakeLegacyConnection` opens, starting with its held connection, and shared with the connection's WAL wrapper. The checks that use it land in the next commit. - `cfg(test)` `FenceTestHooks` with the named hook points of the design. `NamespaceStore::with` and `make_namespace` refuse a namespace whose fence state is unavailable before any setup. Co-authored-by: Tomasz Szymczyszyn --- .../src/connection/connection_core.rs | 6 + .../src/connection/connection_manager.rs | 11 +- libsql-server/src/connection/legacy.rs | 23 +- .../src/namespace/configurator/helpers.rs | 3 + .../src/namespace/configurator/mod.rs | 2 + .../src/namespace/configurator/primary.rs | 6 + .../src/namespace/configurator/replica.rs | 5 + .../src/namespace/configurator/schema.rs | 4 + .../src/namespace/fence/controller.rs | 865 ++++++++++++++++++ libsql-server/src/namespace/fence/hooks.rs | 141 +++ libsql-server/src/namespace/fence/mod.rs | 17 +- libsql-server/src/namespace/fence/registry.rs | 221 +++++ libsql-server/src/namespace/meta_store.rs | 15 +- libsql-server/src/namespace/mod.rs | 12 + libsql-server/src/namespace/store.rs | 229 +++++ 15 files changed, 1548 insertions(+), 12 deletions(-) create mode 100644 libsql-server/src/namespace/fence/controller.rs create mode 100644 libsql-server/src/namespace/fence/hooks.rs create mode 100644 libsql-server/src/namespace/fence/registry.rs diff --git a/libsql-server/src/connection/connection_core.rs b/libsql-server/src/connection/connection_core.rs index 17a6f52961..9025914c62 100644 --- a/libsql-server/src/connection/connection_core.rs +++ b/libsql-server/src/connection/connection_core.rs @@ -394,6 +394,7 @@ mod test { use crate::auth::Authenticated; use crate::connection::legacy::MakeLegacyConnection; use crate::connection::{Connection as _, RequestContext, TXN_TIMEOUT}; + use crate::namespace::fence::controller::FenceController; use crate::namespace::meta_store::{metastore_connection_maker, MetaStore}; use crate::namespace::NamespaceName; use crate::query_result_builder::test::{test_driver, TestBuilder}; @@ -454,6 +455,7 @@ mod test { Default::default(), Arc::new(|_| unreachable!()), Arc::new(|| Sqlite3WalManager::default()), + FenceController::unfenced(Default::default()), ) .await .unwrap(); @@ -500,6 +502,7 @@ mod test { Default::default(), Arc::new(|_| unreachable!()), Arc::new(|| Sqlite3WalManager::default()), + FenceController::unfenced(Default::default()), ) .await .unwrap(); @@ -551,6 +554,7 @@ mod test { Default::default(), Arc::new(|_| unreachable!()), Arc::new(|| Sqlite3WalManager::default()), + FenceController::unfenced(Default::default()), ) .await .unwrap(); @@ -634,6 +638,7 @@ mod test { Default::default(), Arc::new(|_| unreachable!()), Arc::new(|| Sqlite3WalManager::default()), + FenceController::unfenced(Default::default()), ) .await .unwrap(); @@ -727,6 +732,7 @@ mod test { Default::default(), Arc::new(|_| unreachable!()), Arc::new(|| Sqlite3WalManager::default()), + FenceController::unfenced(Default::default()), ) .await .unwrap(); diff --git a/libsql-server/src/connection/connection_manager.rs b/libsql-server/src/connection/connection_manager.rs index 9fa22fbb8e..4baaa0ddc0 100644 --- a/libsql-server/src/connection/connection_manager.rs +++ b/libsql-server/src/connection/connection_manager.rs @@ -14,6 +14,7 @@ use rusqlite::ErrorCode; use super::connection_core::CoreConnection; use super::TXN_TIMEOUT; +use crate::namespace::fence::controller::FenceConnState; pub type ConnId = u64; pub type InnerWalManager = Sqlite3WalManager; @@ -117,12 +118,18 @@ impl Default for ConnectionManagerInner { pub struct ManagedConnectionWalWrapper { id: ConnId, manager: ConnectionManager, + /// The connection's fence state, which `begin_write_txn` checks against the namespace's + /// gate (`docs/NAMESPACE_FENCE.md` section 8.1). + // Installed here so that no connection exists without it; the check itself lands in the + // next commit of this series. + #[allow(dead_code)] + fence: Arc, } impl ManagedConnectionWalWrapper { - pub(crate) fn new(manager: ConnectionManager) -> Self { + pub(crate) fn new(manager: ConnectionManager, fence: Arc) -> Self { let id = manager.inner.next_conn_id.fetch_add(1, Ordering::SeqCst); - Self { id, manager } + Self { id, manager, fence } } pub fn id(&self) -> ConnId { diff --git a/libsql-server/src/connection/legacy.rs b/libsql-server/src/connection/legacy.rs index ae5addd70d..29676d237a 100644 --- a/libsql-server/src/connection/legacy.rs +++ b/libsql-server/src/connection/legacy.rs @@ -14,6 +14,8 @@ use tokio::time::Duration; use crate::error::Error; use crate::metrics::DESCRIBE_COUNT; use crate::namespace::broadcasters::BroadcasterHandle; +use crate::namespace::fence::controller::{FenceConnState, FenceController}; +use crate::namespace::fence::state::OperationClass; use crate::namespace::meta_store::MetaStoreHandle; use crate::namespace::ResolveNamespacePathFn; use crate::query_result_builder::{QueryBuilderConfig, QueryResultBuilder}; @@ -47,6 +49,9 @@ pub struct MakeLegacyConnection { block_writes: Arc, resolve_attach_path: ResolveNamespacePathFn, make_wal_manager: Arc InnerWalManager + Sync + Send + 'static>, + /// The namespace's fence controller. Every connection this maker opens, starting with the + /// held `_db` connection, carries a fence state bound to it. + fence: Arc, } impl MakeLegacyConnection @@ -69,6 +74,7 @@ where block_writes: Arc, resolve_attach_path: ResolveNamespacePathFn, make_wal_manager: Arc InnerWalManager + Sync + Send + 'static>, + fence: Arc, ) -> Result { let txn_timeout = config_store.get().txn_timeout.unwrap_or(TXN_TIMEOUT); @@ -89,6 +95,7 @@ where resolve_attach_path, connection_manager: ConnectionManager::new(txn_timeout), make_wal_manager, + fence, }; let db = this.try_create_db().await?; @@ -146,6 +153,7 @@ where self.resolve_attach_path.clone(), self.connection_manager.clone(), self.make_wal_manager.clone(), + FenceConnState::new(self.fence.clone(), OperationClass::NormalWrite), ) .await } @@ -165,6 +173,10 @@ where pub struct LegacyConnection { pub(super) inner: Arc>>>, + /// Shared with the connection's WAL wrapper. + // Read by the WAL gate and the program admission check in the next commit of this series. + #[allow(dead_code)] + pub(super) fence: Arc, } #[cfg(test)] @@ -185,6 +197,10 @@ impl LegacyConnection { Arc::new(|_| unreachable!()), ConnectionManager::new(TXN_TIMEOUT), Arc::new(|| Sqlite3WalManager::default()), + FenceConnState::new( + FenceController::unfenced(Default::default()), + OperationClass::NormalWrite, + ), ) .await .unwrap() @@ -195,6 +211,7 @@ impl Clone for LegacyConnection { fn clone(&self) -> Self { Self { inner: self.inner.clone(), + fence: self.fence.clone(), } } } @@ -321,11 +338,13 @@ where resolve_attach_path: ResolveNamespacePathFn, connection_manager: ConnectionManager, make_wal: Arc InnerWalManager + Sync + Send + 'static>, + fence: Arc, ) -> crate::Result { let (conn, id) = tokio::task::spawn_blocking({ let connection_manager = connection_manager.clone(); + let fence = fence.clone(); move || -> crate::Result<_> { - let manager = ManagedConnectionWalWrapper::new(connection_manager); + let manager = ManagedConnectionWalWrapper::new(connection_manager, fence); let id = manager.id(); let wal = make_wal().wrap(manager).wrap(wal_wrapper); @@ -366,7 +385,7 @@ where connection_manager.register_connection(&inner, id); - Ok(Self { inner }) + Ok(Self { inner, fence }) } pub async fn execute( diff --git a/libsql-server/src/namespace/configurator/helpers.rs b/libsql-server/src/namespace/configurator/helpers.rs index 599320783d..1f2524cade 100644 --- a/libsql-server/src/namespace/configurator/helpers.rs +++ b/libsql-server/src/namespace/configurator/helpers.rs @@ -24,6 +24,7 @@ use crate::connection::{Connection as _, MakeConnection, MakeThrottledConnection use crate::database::{PrimaryConnection, PrimaryConnectionMaker}; use crate::error::LoadDumpError; use crate::namespace::broadcasters::BroadcasterHandle; +use crate::namespace::fence::controller::FenceController; use crate::namespace::meta_store::MetaStoreHandle; use crate::namespace::replication_wal::{make_replication_wal_wrapper, ReplicationWalWrapper}; use crate::namespace::{ @@ -50,6 +51,7 @@ pub(super) async fn make_primary_connection_maker( broadcaster: BroadcasterHandle, make_wal_manager: Arc InnerWalManager + Sync + Send + 'static>, encryption_config: Option, + fence: Arc, ) -> crate::Result<( Arc, ReplicationWalWrapper, @@ -174,6 +176,7 @@ pub(super) async fn make_primary_connection_maker( block_writes, resolve_attach_path, make_wal_manager.clone(), + fence, ) .await? .throttled( diff --git a/libsql-server/src/namespace/configurator/mod.rs b/libsql-server/src/namespace/configurator/mod.rs index 517b21ca5a..029ab0b3ce 100644 --- a/libsql-server/src/namespace/configurator/mod.rs +++ b/libsql-server/src/namespace/configurator/mod.rs @@ -13,6 +13,7 @@ use crate::replication::script_backup_manager::ScriptBackupManager; use crate::StatsSender; use super::broadcasters::BroadcasterHandle; +use super::fence::controller::FenceController; use super::meta_store::MetaStoreHandle; use super::{ Namespace, NamespaceBottomlessDbIdInit, NamespaceName, NamespaceStore, ResetCb, @@ -119,6 +120,7 @@ pub trait ConfigureNamespace { resolve_attach_path: ResolveNamespacePathFn, store: NamespaceStore, broadcaster: BroadcasterHandle, + fence: Arc, ) -> Pin> + Send + 'a>>; fn cleanup<'a>( diff --git a/libsql-server/src/namespace/configurator/primary.rs b/libsql-server/src/namespace/configurator/primary.rs index f68405fad6..7b63671ed1 100644 --- a/libsql-server/src/namespace/configurator/primary.rs +++ b/libsql-server/src/namespace/configurator/primary.rs @@ -13,6 +13,7 @@ use crate::connection::{Connection as _, MakeConnection}; use crate::database::{Database, PrimaryDatabase}; use crate::namespace::broadcasters::BroadcasterHandle; use crate::namespace::configurator::helpers::make_primary_connection_maker; +use crate::namespace::fence::controller::FenceController; use crate::namespace::meta_store::MetaStoreHandle; use crate::namespace::{ Namespace, NamespaceBottomlessDbIdInit, NamespaceName, NamespaceStore, ResetCb, @@ -53,6 +54,7 @@ impl PrimaryConfigurator { db_path: Arc, broadcaster: BroadcasterHandle, encryption_config: Option, + fence: Arc, ) -> crate::Result { let mut join_set = JoinSet::new(); @@ -72,6 +74,7 @@ impl PrimaryConfigurator { broadcaster, self.make_wal_manager.clone(), encryption_config, + fence.clone(), ) .await?; @@ -112,6 +115,7 @@ impl PrimaryConfigurator { stats, db_config_store: meta_store_handle, path: db_path.into(), + fence, }) } } @@ -126,6 +130,7 @@ impl ConfigureNamespace for PrimaryConfigurator { resolve_attach_path: ResolveNamespacePathFn, _store: NamespaceStore, broadcaster: BroadcasterHandle, + fence: Arc, ) -> Pin> + Send + 'a>> { Box::pin(async move { let db_path: Arc = self.base.base_path.join("dbs").join(name.as_str()).into(); @@ -140,6 +145,7 @@ impl ConfigureNamespace for PrimaryConfigurator { db_path.clone(), broadcaster, self.base.encryption_config.clone(), + fence, ) .await { diff --git a/libsql-server/src/namespace/configurator/replica.rs b/libsql-server/src/namespace/configurator/replica.rs index b1a108af73..adea0fd406 100644 --- a/libsql-server/src/namespace/configurator/replica.rs +++ b/libsql-server/src/namespace/configurator/replica.rs @@ -18,6 +18,7 @@ use crate::connection::MakeConnection; use crate::database::{Database, ReplicaDatabase}; use crate::namespace::broadcasters::BroadcasterHandle; use crate::namespace::configurator::helpers::{make_stats, run_storage_monitor}; +use crate::namespace::fence::controller::FenceController; use crate::namespace::meta_store::MetaStoreHandle; use crate::namespace::{Namespace, NamespaceBottomlessDbIdInit, RestoreOption}; use crate::namespace::{NamespaceName, NamespaceStore, ResetCb, ResetOp, ResolveNamespacePathFn}; @@ -60,6 +61,7 @@ impl ConfigureNamespace for ReplicaConfigurator { resolve_attach_path: ResolveNamespacePathFn, store: NamespaceStore, broadcaster: BroadcasterHandle, + fence: Arc, ) -> Pin> + Send + 'a>> { Box::pin(async move { tracing::debug!("creating replica namespace"); @@ -104,6 +106,7 @@ impl ConfigureNamespace for ReplicaConfigurator { resolve_attach_path, store, broadcaster, + fence, ) .await; } @@ -220,6 +223,7 @@ impl ConfigureNamespace for ReplicaConfigurator { Arc::new(AtomicBool::new(false)), // this is always false for write proxy resolve_attach_path, self.make_wal_manager.clone(), + fence.clone(), ) .await?; @@ -274,6 +278,7 @@ impl ConfigureNamespace for ReplicaConfigurator { stats, db_config_store: meta_store_handle, path: db_path.into(), + fence, }) }) } diff --git a/libsql-server/src/namespace/configurator/schema.rs b/libsql-server/src/namespace/configurator/schema.rs index 275fd71e93..411ec91271 100644 --- a/libsql-server/src/namespace/configurator/schema.rs +++ b/libsql-server/src/namespace/configurator/schema.rs @@ -7,6 +7,7 @@ use crate::connection::config::DatabaseConfig; use crate::connection::connection_manager::InnerWalManager; use crate::database::{Database, SchemaDatabase}; use crate::namespace::broadcasters::BroadcasterHandle; +use crate::namespace::fence::controller::FenceController; use crate::namespace::meta_store::MetaStoreHandle; use crate::namespace::{ Namespace, NamespaceName, NamespaceStore, ResetCb, ResolveNamespacePathFn, RestoreOption, @@ -49,6 +50,7 @@ impl ConfigureNamespace for SchemaConfigurator { resolve_attach_path: ResolveNamespacePathFn, _store: NamespaceStore, broadcaster: BroadcasterHandle, + fence: Arc, ) -> std::pin::Pin> + Send + 'a>> { Box::pin(async move { let mut join_set = JoinSet::new(); @@ -69,6 +71,7 @@ impl ConfigureNamespace for SchemaConfigurator { broadcaster, self.make_wal_manager.clone(), self.base.encryption_config.clone(), + fence.clone(), ) .await?; @@ -90,6 +93,7 @@ impl ConfigureNamespace for SchemaConfigurator { stats, db_config_store: db_config.clone(), path: db_path.into(), + fence, }) }) } diff --git a/libsql-server/src/namespace/fence/controller.rs b/libsql-server/src/namespace/fence/controller.rs new file mode 100644 index 0000000000..265e04e40f --- /dev/null +++ b/libsql-server/src/namespace/fence/controller.rs @@ -0,0 +1,865 @@ +//! The per-namespace fence controller (`docs/NAMESPACE_FENCE.md` sections 7 and 8.4). +//! +//! A [`FenceController`] is the in-memory authority for one namespace's fence. It owns the +//! transition lock that serialises fence commands on the namespace, and the gate that every +//! admission path reads: a `watch` of [`GateSnapshot`]. The gate changes only after the +//! metastore has committed (commit → publish → respond), except that a commit whose outcome is +//! unknown closes it until the same command is replayed. +//! +//! Controllers live in the [`FenceRegistry`](super::registry::FenceRegistry), not in the +//! namespace cache, so evicting and reloading a namespace hands the reloaded namespace the +//! same controller. + +use std::sync::atomic::{AtomicU64, Ordering}; +use std::sync::Arc; + +use parking_lot::Mutex; +use tokio::sync::{watch, OwnedMutexGuard}; +use uuid::Uuid; + +use crate::error::Error; +use crate::namespace::meta_store::{FenceCommit, FenceContext, MetaStore}; +use crate::namespace::NamespaceName; + +use super::command::FenceRequest; +#[cfg(test)] +use super::hooks::FenceTestHooks; +use super::hooks::{HookOutcome, HookPoint}; +use super::outcome::{FenceDetail, FenceError, FenceOutcome}; +use super::state::{Admission, FenceState, OperationClass}; +use super::store::StoredFence; +use super::transition::DrainCompletion; + +/// `(operation_id, command_id)` of a fence command. +pub type CommandKey = (Uuid, Uuid); + +/// What every admission path reads: the fence as last published, the write-admission +/// generation, and whether a commit is indeterminate. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct GateSnapshot { + /// The durable fence as of the last publication (or as loaded at startup). + pub fence: StoredFence, + /// Incremented on every published change of the fence state, of its owning operation, or + /// of the indeterminate flag. A write transaction may only start when the generation its + /// program and its read transaction were admitted under equals this one (section 8.1). + pub write_generation: u64, + /// A command whose commit outcome is unknown. While set, every class except maintenance + /// and observability is denied, and every other command is refused. + pub indeterminate: Option, +} + +impl GateSnapshot { + fn new(fence: StoredFence) -> Self { + Self { + fence, + write_generation: 0, + indeterminate: None, + } + } + + pub fn state(&self) -> FenceState { + self.fence.state() + } + + pub fn revision(&self) -> u64 { + self.fence.revision() + } + + pub fn operation_id(&self) -> Option { + self.fence.record().map(|r| r.operation_id) + } + + pub fn is_unavailable(&self) -> bool { + matches!(self.fence, StoredFence::Unavailable { .. }) + } + + /// The gate's decision for work of `class`. + pub fn permits(&self, class: OperationClass) -> Result<(), FenceError> { + if let Some((operation_id, command_id)) = self.indeterminate { + if !matches!( + class, + OperationClass::Maintenance | OperationClass::Observability + ) { + return Err(FenceError::new( + FenceOutcome::FenceStateUnavailable, + format!( + "the outcome of fence command {command_id} of operation {operation_id} \ + is not known yet" + ), + ) + .with_detail(FenceDetail::IndeterminateCommit)); + } + } + self.fence.permits(class) + } + + /// Normal write admission. + pub fn write(&self) -> Admission { + Admission::from( + self.permits(OperationClass::NormalWrite) + .map_err(|e| e.outcome()), + ) + } + + /// Normal read admission. + pub fn read(&self) -> Admission { + Admission::from( + self.permits(OperationClass::NormalRead) + .map_err(|e| e.outcome()), + ) + } +} + +/// The fence controller of one namespace. +pub struct FenceController { + namespace: NamespaceName, + transition_lock: Arc>, + gate: watch::Sender, + #[cfg(test)] + hooks: FenceTestHooks, +} + +impl std::fmt::Debug for FenceController { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("FenceController") + .field("namespace", &self.namespace) + .field("gate", &*self.gate.borrow()) + .finish_non_exhaustive() + } +} + +impl FenceController { + /// A controller whose gate starts from `fence`, as established by the metastore. + pub fn new(namespace: NamespaceName, fence: StoredFence) -> Arc { + let (gate, _) = watch::channel(GateSnapshot::new(fence)); + Arc::new(Self { + namespace, + transition_lock: Default::default(), + gate, + #[cfg(test)] + hooks: FenceTestHooks::default(), + }) + } + + /// A controller for a namespace without fence state. + pub fn unfenced(namespace: NamespaceName) -> Arc { + Self::new( + namespace, + StoredFence::None { + namespace_exists: true, + }, + ) + } + + pub fn namespace(&self) -> &NamespaceName { + &self.namespace + } + + /// A copy of the current gate. + pub fn gate(&self) -> GateSnapshot { + self.gate.borrow().clone() + } + + /// A receiver that observes every publication. + pub fn subscribe(&self) -> watch::Receiver { + self.gate.subscribe() + } + + pub fn write_generation(&self) -> u64 { + self.gate.borrow().write_generation + } + + /// The live gate's decision for work of `class`. + pub fn permits(&self, class: OperationClass) -> Result<(), FenceError> { + self.gate.borrow().permits(class) + } + + /// Take the namespace's transition lock. Every fence command on the namespace runs while + /// holding it, from its first check to its response. + pub async fn begin_transition(self: &Arc) -> Transition { + let guard = self.transition_lock.clone().lock_owned().await; + Transition { + controller: self.clone(), + _guard: guard, + } + } + + /// Run one fence command to completion: take the transition lock, commit it in the + /// metastore, publish the result to the gate and return it. + /// + /// The work runs on its own task, so a caller that goes away (a lost response) does not + /// stop the publication of a command that committed. + pub async fn apply_command( + self: &Arc, + meta: &MetaStore, + request: FenceRequest, + ctx: FenceContext, + ) -> crate::Result { + let this = self.clone(); + let meta = meta.clone(); + tokio::spawn(async move { + let mut transition = this.begin_transition().await; + transition.apply(&meta, request, ctx).await + }) + .await? + } + + #[cfg(test)] + pub fn hooks(&self) -> &FenceTestHooks { + &self.hooks + } + + /// Reach a test hook point. Outside the library's own test build this does nothing. + #[cfg(test)] + pub(crate) async fn hook(&self, point: HookPoint) -> HookOutcome { + self.hooks.hit(point).await + } + + #[cfg(not(test))] + #[inline(always)] + pub(crate) async fn hook(&self, _point: HookPoint) -> HookOutcome { + HookOutcome::Continue + } + + /// Publish a new gate. `fence: None` keeps the published fence. The write generation moves + /// whenever the state, the owning operation or the indeterminate flag changes. + fn publish(&self, fence: Option, indeterminate: Option) { + self.gate.send_modify(|gate| { + let fence = fence.unwrap_or_else(|| gate.fence.clone()); + let changed = fence.state() != gate.fence.state() + || fence.record().map(|r| r.operation_id) + != gate.fence.record().map(|r| r.operation_id) + || indeterminate != gate.indeterminate; + gate.fence = fence; + gate.indeterminate = indeterminate; + if changed { + gate.write_generation += 1; + } + }); + let gate = self.gate.borrow(); + tracing::debug!( + namespace = %self.namespace, + state = %gate.state(), + revision = gate.revision(), + write_generation = gate.write_generation, + indeterminate = gate.indeterminate.is_some(), + "published namespace fence gate" + ); + } +} + +/// A fence command in progress on one namespace. Holds the namespace's transition lock until +/// dropped. +pub struct Transition { + controller: Arc, + _guard: OwnedMutexGuard<()>, +} + +impl Transition { + pub fn controller(&self) -> &Arc { + &self.controller + } + + /// Commit `request` in the metastore and publish the result. + pub async fn apply( + &mut self, + meta: &MetaStore, + request: FenceRequest, + ctx: FenceContext, + ) -> crate::Result { + debug_assert_eq!(&request.namespace, self.controller.namespace()); + let key = (request.operation_id, request.command_id); + self.commit( + key, + async move { meta.apply_fence_command(request, ctx).await }, + ) + .await + } + + /// Complete the drain that the `DRAINING` receipt `key` started, once the caller has + /// proven `completion`, and publish the result. + pub async fn complete_drain( + &mut self, + meta: &MetaStore, + key: CommandKey, + completion: DrainCompletion, + ctx: FenceContext, + ) -> crate::Result { + let namespace = self.controller.namespace().clone(); + self.commit(key, async move { + meta.complete_fence_drain(namespace, key.0, key.1, completion, ctx) + .await + }) + .await + } + + /// The commit → publish → respond sequence of section 8.4. + /// + /// - An error that proves nothing was committed leaves the gate exactly as it was. + /// - A commit whose outcome is unknown closes the gate (every class but maintenance and + /// observability) and marks `key` indeterminate: other commands get + /// `FENCE_COMMIT_INDETERMINATE` until `key` is replayed, and the replay, which the + /// metastore answers from the durable row, reopens it to whatever is durable. + /// - A commit is published before it is answered. + async fn commit(&mut self, key: CommandKey, run: F) -> crate::Result + where + F: std::future::Future>, + { + let controller = self.controller.clone(); + if let Some(pending) = controller.gate.borrow().indeterminate { + if pending != key { + return Err(pending_indeterminate(pending).into()); + } + } + + if let HookOutcome::Fail(e) = controller.hook(HookPoint::BeforeMetastoreCommit).await { + return Err(e.into()); + } + + let result = match run.await { + Ok(commit) => match controller.hook(HookPoint::AfterMetastoreCommit).await { + HookOutcome::Continue => Ok(commit), + HookOutcome::Indeterminate | HookOutcome::Fail(_) => Err(indeterminate( + key, + "the commit was not acknowledged (test hook)", + )), + }, + Err(e) if is_indeterminate(&e) => Err(indeterminate(key, &e.to_string())), + Err(e) => return Err(e), + }; + + match result { + Ok(commit) => { + let _ = controller.hook(HookPoint::BeforeGatePublish).await; + controller.publish(commit.record.clone().map(StoredFence::Record), None); + let _ = controller.hook(HookPoint::BeforeResponse).await; + Ok(commit) + } + Err(e) => { + tracing::error!( + namespace = %controller.namespace, + operation_id = %key.0, + command_id = %key.1, + "fence commit outcome unknown; the namespace stays closed until the command \ + is replayed: {e}" + ); + controller.publish(None, Some(key)); + Err(e.into()) + } + } + } +} + +/// Whether a metastore error leaves the commit's outcome unknown: the commit itself failed, or +/// the task running it died. +fn is_indeterminate(e: &Error) -> bool { + match e { + Error::NamespaceFence(f) => f.outcome() == FenceOutcome::FenceCommitIndeterminate, + Error::RuntimeTaskJoinError(_) => true, + _ => false, + } +} + +fn indeterminate(key: CommandKey, why: &str) -> FenceError { + FenceError::new( + FenceOutcome::FenceCommitIndeterminate, + format!( + "whether fence command {} of operation {} was committed is unknown ({why}); replay \ + the same command to reconcile", + key.1, key.0 + ), + ) + .with_detail(FenceDetail::IndeterminateCommit) +} + +fn pending_indeterminate((operation_id, command_id): CommandKey) -> FenceError { + FenceError::new( + FenceOutcome::FenceCommitIndeterminate, + format!( + "fence command {command_id} of operation {operation_id} has an unknown outcome; \ + only a replay of that command is accepted until it is reconciled" + ), + ) + .with_detail(FenceDetail::IndeterminateCommit) +} + +/// The fence state of one connection, shared by its WAL wrapper and its `CoreConnection` +/// (section 7.4). The WAL gate reads and writes it; this commit only installs it. +#[derive(Debug)] +pub struct FenceConnState { + controller: Arc, + class: OperationClass, + /// The write generation the current program was admitted under. + program_generation: AtomicU64, + /// The write generation the current read transaction was opened under. + txn_generation: AtomicU64, + /// The typed outcome of the last refusal at the WAL. + denial: Mutex>, +} + +impl FenceConnState { + pub fn new(controller: Arc, class: OperationClass) -> Arc { + let generation = controller.write_generation(); + Arc::new(Self { + controller, + class, + program_generation: AtomicU64::new(generation), + txn_generation: AtomicU64::new(generation), + denial: Mutex::new(None), + }) + } + + pub fn controller(&self) -> &Arc { + &self.controller + } + + pub fn class(&self) -> OperationClass { + self.class + } + + pub fn program_generation(&self) -> u64 { + self.program_generation.load(Ordering::Acquire) + } + + pub fn txn_generation(&self) -> u64 { + self.txn_generation.load(Ordering::Acquire) + } + + pub fn take_denial(&self) -> Option { + self.denial.lock().take() + } +} + +#[cfg(test)] +mod tests { + use std::path::Path; + + use tempfile::tempdir; + + use super::*; + use crate::config::MetaStoreConfig; + use crate::connection::config::DatabaseConfig; + use crate::database::DatabaseKind; + use crate::namespace::fence::command::FenceCommand; + use crate::namespace::fence::record::{FrozenBoundary, ServerIdentity}; + use crate::namespace::meta_store::{metastore_connection_maker, FenceCommitKind}; + + const LOG: Uuid = Uuid::from_u128(0x10); + const OP: Uuid = Uuid::from_u128(0xa); + const OTHER_OP: Uuid = Uuid::from_u128(0xb); + + pub(crate) async fn open_metastore(dir: &Path) -> MetaStore { + let (maker, manager) = metastore_connection_maker(None, dir).await.unwrap(); + let conn = maker().unwrap(); + MetaStore::new( + MetaStoreConfig { + namespace_fence: true, + ..Default::default() + }, + dir, + conn, + manager, + DatabaseKind::Primary, + ) + .await + .unwrap() + } + + pub(crate) async fn create_namespace(meta: &MetaStore, ns: &'static str) { + meta.handle(ns.into()) + .await + .unwrap() + .store(DatabaseConfig::default()) + .await + .unwrap(); + } + + fn ctx() -> FenceContext { + FenceContext::now( + ServerIdentity { + build: "test".into(), + instance_id: Uuid::from_u128(0x99), + }, + Some(LOG), + ) + } + + fn acquire(ns: &'static str, op: Uuid, command_id: u128) -> FenceRequest { + FenceRequest { + namespace: ns.into(), + operation_id: op, + command_id: Uuid::from_u128(command_id), + expected_state: FenceState::Unfenced, + expected_revision: 0, + command: FenceCommand::AcquireSourceWriteFence { + expected_log_id: LOG, + drain_policy: None, + }, + } + } + + fn release(ns: &'static str, op: Uuid, command_id: u128, revision: u64) -> FenceRequest { + FenceRequest { + namespace: ns.into(), + operation_id: op, + command_id: Uuid::from_u128(command_id), + expected_state: FenceState::SourceWriteFenced, + expected_revision: revision, + command: FenceCommand::ReleaseSourceWriteFence, + } + } + + fn outcome(r: &crate::Result) -> FenceOutcome { + match r { + Ok(c) => c.receipt.outcome, + Err(Error::NamespaceFence(e)) => e.outcome(), + Err(e) => panic!("unexpected error: {e}"), + } + } + + /// Acquire and complete the drain directly, as the write drain will: the controller ends in + /// `SOURCE_WRITE_FENCED`. + async fn fence_source(meta: &MetaStore, controller: &Arc, op: Uuid) { + let commit = controller + .apply_command(meta, acquire("ns", op, 1), ctx()) + .await + .unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Draining); + let mut t = controller.begin_transition().await; + let commit = t + .complete_drain( + meta, + (op, Uuid::from_u128(1)), + DrainCompletion::SourceWrites { + boundary: FrozenBoundary { + log_id: LOG, + frame_no: 0, + }, + }, + ctx(), + ) + .await + .unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + } + + #[tokio::test] + async fn committed_command_publishes_new_revision_and_generation() { + let tmp = tempdir().unwrap(); + let meta = open_metastore(tmp.path()).await; + create_namespace(&meta, "ns").await; + let controller = FenceController::unfenced("ns".into()); + let mut rx = controller.subscribe(); + assert!(controller.permits(OperationClass::NormalWrite).is_ok()); + assert_eq!(controller.write_generation(), 0); + + let commit = controller + .apply_command(&meta, acquire("ns", OP, 1), ctx()) + .await + .unwrap(); + assert_eq!(commit.kind, FenceCommitKind::Committed); + assert!(rx.has_changed().unwrap()); + let gate = rx.borrow_and_update().clone(); + assert_eq!(gate.state(), FenceState::SourceDraining); + assert_eq!(gate.revision(), 1); + assert_eq!(gate.operation_id(), Some(OP)); + assert_eq!(gate.write_generation, 1); + assert_eq!( + controller + .permits(OperationClass::NormalWrite) + .unwrap_err() + .outcome(), + FenceOutcome::MigrationWriteFenced + ); + assert!(controller.permits(OperationClass::NormalRead).is_ok()); + assert!(controller.permits(OperationClass::Maintenance).is_ok()); + + // The published gate is the durable one. + let inspected = meta.inspect_fence("ns".into()).await.unwrap(); + assert_eq!(inspected.fence, gate.fence); + } + + #[tokio::test] + async fn write_generation_bumps_on_every_write_admission_change() { + let tmp = tempdir().unwrap(); + let meta = open_metastore(tmp.path()).await; + create_namespace(&meta, "ns").await; + let controller = FenceController::unfenced("ns".into()); + + let mut generations = vec![controller.write_generation()]; + fence_source(&meta, &controller, OP).await; + // UNFENCED -> SOURCE_DRAINING -> SOURCE_WRITE_FENCED: two changes of state. + generations.push(controller.write_generation()); + let revision = controller.gate().revision(); + controller + .apply_command(&meta, release("ns", OP, 2, revision), ctx()) + .await + .unwrap(); + assert_eq!(controller.gate().state(), FenceState::Released); + assert!(controller.permits(OperationClass::NormalWrite).is_ok()); + generations.push(controller.write_generation()); + assert_eq!(generations, vec![0, 2, 3]); + + // A replay publishes the same durable state and does not move the generation. + let before = controller.gate(); + let replay = controller + .apply_command(&meta, release("ns", OP, 2, revision), ctx()) + .await + .unwrap(); + assert_eq!(replay.kind, FenceCommitKind::Replayed); + assert_eq!(controller.gate(), before); + } + + #[tokio::test] + async fn failed_before_commit_leaves_gate_unchanged() { + let tmp = tempdir().unwrap(); + let meta = open_metastore(tmp.path()).await; + create_namespace(&meta, "ns").await; + let controller = FenceController::unfenced("ns".into()); + let before = controller.gate(); + + // A refusal by the transition function. + let mut wrong = acquire("ns", OP, 1); + wrong.expected_revision = 7; + let r = controller.apply_command(&meta, wrong, ctx()).await; + assert_eq!(outcome(&r), FenceOutcome::FenceRevisionMismatch); + assert_eq!(controller.gate(), before); + + // An injected failure before the metastore transaction. + controller.hooks().fail_at( + HookPoint::BeforeMetastoreCommit, + FenceError::new(FenceOutcome::FencePreconditionFailed, "injected"), + ); + let r = controller + .apply_command(&meta, acquire("ns", OP, 1), ctx()) + .await; + assert_eq!(outcome(&r), FenceOutcome::FencePreconditionFailed); + assert_eq!(controller.gate(), before); + + // The metastore is busy (another connection holds its write lock): nothing is written + // and the gate does not move. + let (maker, _) = metastore_connection_maker(None, tmp.path()).await.unwrap(); + let mut other = maker().unwrap(); + let lock = other + .transaction_with_behavior(rusqlite::TransactionBehavior::Immediate) + .unwrap(); + let r = controller + .apply_command(&meta, acquire("ns", OP, 1), ctx()) + .await; + assert!( + matches!(r, Err(Error::RusqliteError(_))), + "expected a busy metastore, got {r:?}" + ); + assert_eq!(controller.gate(), before); + drop(lock); + let inspected = meta.inspect_fence("ns".into()).await.unwrap(); + assert!(matches!(inspected.fence, StoredFence::None { .. })); + assert!(inspected.receipts.is_empty()); + } + + #[tokio::test] + async fn indeterminate_commit_keeps_writes_closed_until_replayed() { + let tmp = tempdir().unwrap(); + let meta = open_metastore(tmp.path()).await; + create_namespace(&meta, "ns").await; + let controller = FenceController::unfenced("ns".into()); + fence_source(&meta, &controller, OP).await; + let fenced = controller.gate(); + let revision = fenced.revision(); + + // Release commits, but its acknowledgement is lost. + controller.hooks().arm( + HookPoint::AfterMetastoreCommit, + super::super::hooks::HookAction::Indeterminate, + ); + let r = controller + .apply_command(&meta, release("ns", OP, 2, revision), ctx()) + .await; + assert_eq!(outcome(&r), FenceOutcome::FenceCommitIndeterminate); + let gate = controller.gate(); + assert_eq!(gate.indeterminate, Some((OP, Uuid::from_u128(2)))); + assert!(gate.write_generation > fenced.write_generation); + // The durable state says released, but nothing is admitted until it is reconciled. + for class in [ + OperationClass::NormalWrite, + OperationClass::NormalRead, + OperationClass::Stream, + OperationClass::Lifecycle, + ] { + let e = controller.permits(class).unwrap_err(); + assert_eq!(e.outcome(), FenceOutcome::FenceStateUnavailable); + assert_eq!(e.detail(), Some(FenceDetail::IndeterminateCommit)); + } + assert!(controller.permits(OperationClass::Maintenance).is_ok()); + + // Any other command is refused with the indeterminate code. + let r = controller + .apply_command(&meta, release("ns", OTHER_OP, 3, revision), ctx()) + .await; + assert_eq!(outcome(&r), FenceOutcome::FenceCommitIndeterminate); + + // The replay reconciles from the durable row and publishes it. + let r = controller + .apply_command(&meta, release("ns", OP, 2, revision), ctx()) + .await + .unwrap(); + assert_eq!(r.kind, FenceCommitKind::Replayed); + let gate = controller.gate(); + assert_eq!(gate.indeterminate, None); + assert_eq!(gate.state(), FenceState::Released); + assert!(controller.permits(OperationClass::NormalWrite).is_ok()); + } + + #[tokio::test] + async fn indeterminate_commit_that_did_not_apply_is_retried_by_replay() { + let tmp = tempdir().unwrap(); + let meta = open_metastore(tmp.path()).await; + create_namespace(&meta, "ns").await; + let controller = FenceController::unfenced("ns".into()); + + // Simulate an indeterminate outcome for a command that never reached the metastore: + // the replay applies it. + controller.publish(None, Some((OP, Uuid::from_u128(1)))); + assert!(controller.permits(OperationClass::NormalWrite).is_err()); + let r = controller + .apply_command(&meta, acquire("ns", OP, 1), ctx()) + .await + .unwrap(); + assert_eq!(r.kind, FenceCommitKind::Committed); + let gate = controller.gate(); + assert_eq!(gate.indeterminate, None); + assert_eq!(gate.state(), FenceState::SourceDraining); + } + + #[tokio::test] + async fn publication_happens_before_the_response() { + let tmp = tempdir().unwrap(); + let meta = open_metastore(tmp.path()).await; + create_namespace(&meta, "ns").await; + let controller = FenceController::unfenced("ns".into()); + let before_publish = controller.hooks().pause_at(HookPoint::BeforeGatePublish); + + let task = tokio::spawn({ + let controller = controller.clone(); + let meta = meta.clone(); + async move { + controller + .apply_command(&meta, acquire("ns", OP, 1), ctx()) + .await + } + }); + before_publish.reached().await; + // Committed, not yet published: the gate still shows the old state. + assert_eq!(controller.gate().state(), FenceState::Unfenced); + assert!(matches!( + meta.inspect_fence("ns".into()).await.unwrap().fence, + StoredFence::Record(_) + )); + let before_response = controller.hooks().pause_at(HookPoint::BeforeResponse); + before_publish.resume(); + before_response.reached().await; + assert_eq!(controller.gate().state(), FenceState::SourceDraining); + assert!(!task.is_finished()); + before_response.resume(); + assert_eq!(outcome(&task.await.unwrap()), FenceOutcome::Draining); + } + + #[tokio::test] + async fn committed_command_is_published_when_the_caller_goes_away() { + let tmp = tempdir().unwrap(); + let meta = open_metastore(tmp.path()).await; + create_namespace(&meta, "ns").await; + let controller = FenceController::unfenced("ns".into()); + let after_commit = controller.hooks().pause_at(HookPoint::AfterMetastoreCommit); + + let caller = tokio::spawn({ + let controller = controller.clone(); + let meta = meta.clone(); + async move { + controller + .apply_command(&meta, acquire("ns", OP, 1), ctx()) + .await + } + }); + after_commit.reached().await; + // The response is lost: the caller is cancelled after the commit. + caller.abort(); + assert!(caller.await.unwrap_err().is_cancelled()); + let published = controller.hooks().pause_at(HookPoint::BeforeResponse); + after_commit.resume(); + published.reached().await; + assert_eq!(controller.gate().state(), FenceState::SourceDraining); + published.resume(); + + // The transition lock is free again and a replay answers from the receipt. + let replay = controller + .apply_command(&meta, acquire("ns", OP, 1), ctx()) + .await + .unwrap(); + assert_eq!(replay.receipt.outcome, FenceOutcome::Draining); + } + + #[tokio::test] + async fn transition_lock_serialises_commands() { + let tmp = tempdir().unwrap(); + let meta = open_metastore(tmp.path()).await; + create_namespace(&meta, "ns").await; + let controller = FenceController::unfenced("ns".into()); + + let held = controller.begin_transition().await; + let second = tokio::spawn({ + let controller = controller.clone(); + let meta = meta.clone(); + async move { + controller + .apply_command(&meta, acquire("ns", OP, 1), ctx()) + .await + } + }); + tokio::task::yield_now().await; + assert!(!second.is_finished()); + assert!(controller.transition_lock.try_lock().is_err()); + drop(held); + assert_eq!(outcome(&second.await.unwrap()), FenceOutcome::Draining); + } + + #[test] + fn unavailable_gate_denies_every_class_but_maintenance() { + let controller = FenceController::new( + "ns".into(), + StoredFence::Unavailable { + detail: FenceDetail::CorruptRecord, + reason: "test".into(), + marker: None, + }, + ); + let gate = controller.gate(); + assert!(gate.is_unavailable()); + for class in OperationClass::ALL { + let r = gate.permits(class); + match class { + OperationClass::Maintenance | OperationClass::Observability => { + assert!(r.is_ok()) + } + _ => { + let e = r.unwrap_err(); + assert_eq!(e.outcome(), FenceOutcome::FenceStateUnavailable); + assert_eq!(e.detail(), Some(FenceDetail::CorruptRecord)); + } + } + } + } + + #[test] + fn conn_state_starts_at_the_current_generation() { + let controller = FenceController::unfenced("ns".into()); + controller.publish(None, Some((OP, OP))); + controller.publish(None, None); + let state = FenceConnState::new(controller.clone(), OperationClass::NormalWrite); + assert_eq!(state.program_generation(), 2); + assert_eq!(state.txn_generation(), 2); + assert!(Arc::ptr_eq(state.controller(), &controller)); + assert_eq!(state.class(), OperationClass::NormalWrite); + assert!(state.take_denial().is_none()); + } +} diff --git a/libsql-server/src/namespace/fence/hooks.rs b/libsql-server/src/namespace/fence/hooks.rs new file mode 100644 index 0000000000..90ef0d1685 --- /dev/null +++ b/libsql-server/src/namespace/fence/hooks.rs @@ -0,0 +1,141 @@ +//! Test hooks for fence race tests (`docs/NAMESPACE_FENCE.md` section 16). +//! +//! The crate has no failpoint library. Instead, a [`FenceController`] built by the library's +//! own test build carries a `FenceTestHooks`: named points on the transition path where a +//! test can park the task until it releases it, or inject a failure. Race tests are written +//! against these points and never against elapsed time. Outside `cfg(test)` only the point +//! names exist, and reaching a point does nothing. +//! +//! [`FenceController`]: super::controller::FenceController + +#[cfg(test)] +use std::collections::HashMap; +#[cfg(test)] +use std::sync::Arc; + +#[cfg(test)] +use parking_lot::Mutex; +#[cfg(test)] +use tokio::sync::Notify; + +use super::outcome::FenceError; + +/// A named point on a fence transition or a gated path. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] +pub enum HookPoint { + /// The in-memory `INSTALLING` gate of a closing transition has been published. + AfterInstallingGate, + /// Under the transition lock, immediately before the metastore transaction runs. + BeforeMetastoreCommit, + /// The metastore transaction returned a committed result. + AfterMetastoreCommit, + /// The committed result is about to be published to the gate. + BeforeGatePublish, + /// In `begin_write_txn`, after the gate check admitted the transaction. + InBeginWriteTxnAfterCheck, + /// The connection manager released the write slot. + AfterManagerRelease, + /// A drain is about to read the frozen boundary. + BeforeBoundaryCapture, + /// The rows of a quarantined target were committed; the target is not published yet. + AfterTargetRowsCommitted, + /// The gate is published; the answer is about to be returned. + BeforeResponse, +} + +#[cfg(test)] +/// What happens when a task reaches an armed point. Every action fires once: reaching the +/// point disarms it. +#[derive(Debug, Clone)] +pub enum HookAction { + /// Signal `reached`, then wait until `resume` is notified. + Pause { + reached: Arc, + resume: Arc, + }, + /// Fail at this point with `error`, as if the step had failed before it took effect. + Fail(FenceError), + /// At `AfterMetastoreCommit`: report the commit as indeterminate even though it happened, + /// which is what a lost commit acknowledgement looks like to the controller. + Indeterminate, +} + +/// What the task that reached a point has to do next. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum HookOutcome { + Continue, + Fail(FenceError), + Indeterminate, +} + +#[cfg(test)] +/// The armed points of one controller. +#[derive(Debug, Default)] +pub struct FenceTestHooks { + armed: Mutex>, +} + +#[cfg(test)] +/// The handles a test uses to follow a paused task. +#[derive(Debug, Clone)] +pub struct Paused { + pub reached: Arc, + pub resume: Arc, +} + +#[cfg(test)] +impl Paused { + /// Wait until the task reaches the point. + pub async fn reached(&self) { + self.reached.notified().await + } + + /// Let the task continue. + pub fn resume(&self) { + self.resume.notify_one() + } +} + +#[cfg(test)] +impl FenceTestHooks { + pub fn arm(&self, point: HookPoint, action: HookAction) { + self.armed.lock().insert(point, action); + } + + /// Arm `point` to pause, returning the handles to wait for it and to release it. + pub fn pause_at(&self, point: HookPoint) -> Paused { + let paused = Paused { + reached: Arc::new(Notify::new()), + resume: Arc::new(Notify::new()), + }; + self.arm( + point, + HookAction::Pause { + reached: paused.reached.clone(), + resume: paused.resume.clone(), + }, + ); + paused + } + + pub fn fail_at(&self, point: HookPoint, error: FenceError) { + self.arm(point, HookAction::Fail(error)); + } + + /// Called by the code under test when it reaches `point`. + pub async fn hit(&self, point: HookPoint) -> HookOutcome { + let action = self.armed.lock().remove(&point); + match action { + None => HookOutcome::Continue, + Some(HookAction::Pause { reached, resume }) => { + // `notify_one` stores a permit, so a test that starts waiting after the task + // got here still sees it. + reached.notify_one(); + resume.notified().await; + HookOutcome::Continue + } + Some(HookAction::Fail(e)) => HookOutcome::Fail(e), + Some(HookAction::Indeterminate) => HookOutcome::Indeterminate, + } + } +} diff --git a/libsql-server/src/namespace/fence/mod.rs b/libsql-server/src/namespace/fence/mod.rs index 672b0565e8..d5ac933e2e 100644 --- a/libsql-server/src/namespace/fence/mod.rs +++ b/libsql-server/src/namespace/fence/mod.rs @@ -2,12 +2,14 @@ //! example, moving a database between servers) uses as the data-plane authority boundary for //! one namespace. //! -//! `docs/NAMESPACE_FENCE.md` is the contract and the design. This module holds the parts with -//! no I/O: the states and permission matrix ([`state`]), the stable outcome codes and their -//! protocol mappings ([`outcome`]), commands and their canonical fingerprint ([`command`]), -//! records, receipts and markers with their strict durable encoding ([`record`]), the pure -//! transition function ([`transition`]), and the metastore tables, compare-and-swap and marker -//! file that persist them ([`store`], driven by `MetaStore::apply_fence_command`). +//! `docs/NAMESPACE_FENCE.md` is the contract and the design. This module holds the states and +//! permission matrix ([`state`]), the stable outcome codes and their protocol mappings +//! ([`outcome`]), commands and their canonical fingerprint ([`command`]), records, receipts and +//! markers with their strict durable encoding ([`record`]), the pure transition function +//! ([`transition`]), the metastore tables, compare-and-swap and marker file that persist them +//! ([`store`], driven by `MetaStore::apply_fence_command`), and the in-memory authority built +//! on them: the per-namespace [`controller`] with its gate, the [`registry`] that holds the +//! controllers outside the namespace cache, and the test [`hooks`] on their paths. // The persistence, controller and protocol layers that consume these types land in the // following commits of this series; until then most of the module is unused by the rest of @@ -15,8 +17,11 @@ #![allow(dead_code)] pub mod command; +pub mod controller; +pub mod hooks; pub mod outcome; pub mod record; +pub mod registry; pub mod state; pub mod store; pub mod transition; diff --git a/libsql-server/src/namespace/fence/registry.rs b/libsql-server/src/namespace/fence/registry.rs new file mode 100644 index 0000000000..610189677b --- /dev/null +++ b/libsql-server/src/namespace/fence/registry.rs @@ -0,0 +1,221 @@ +//! The fence registry (`docs/NAMESPACE_FENCE.md` section 7.1). +//! +//! One [`FenceController`] per namespace, held outside the namespace cache so that eviction and +//! lazy reload hand a reloaded namespace the controller it had, with the same gate, revision +//! and write generation. The registry is seeded from the metastore before the namespace store +//! serves anything, and namespaces without fence state get an `UNFENCED` controller on first +//! use. + +use std::collections::HashMap; +use std::sync::Arc; + +use parking_lot::Mutex; + +use crate::namespace::NamespaceName; + +use super::controller::FenceController; +use super::outcome::FenceError; +use super::state::OperationClass; +use super::store::StoredFence; + +#[derive(Debug, Default)] +pub struct FenceRegistry { + controllers: Mutex>>, +} + +impl FenceRegistry { + /// A registry holding a controller for every namespace with fence state, as returned by + /// `MetaStore::load_fences` (which includes the namespaces startup could not recover). + pub fn seeded(fences: impl IntoIterator) -> Self { + let controllers = fences + .into_iter() + .map(|(ns, fence)| { + if let StoredFence::Unavailable { detail, reason, .. } = &fence { + tracing::error!( + namespace = %ns, + %detail, + "namespace fence state is unavailable; the namespace is not served: {reason}" + ); + } + let controller = FenceController::new(ns.clone(), fence); + (ns, controller) + }) + .collect(); + Self { + controllers: Mutex::new(controllers), + } + } + + /// The controller of `namespace`, creating an `UNFENCED` one if it has none yet. + pub fn controller(&self, namespace: &NamespaceName) -> Arc { + self.controllers + .lock() + .entry(namespace.clone()) + .or_insert_with(|| FenceController::unfenced(namespace.clone())) + .clone() + } + + /// The controller of `namespace`, if it has one. + pub fn get(&self, namespace: &NamespaceName) -> Option> { + self.controllers.lock().get(namespace).cloned() + } + + /// Forget `namespace`'s controller. Only for a namespace that was deleted together with its + /// fence state. + pub fn remove(&self, namespace: &NamespaceName) -> Option> { + self.controllers.lock().remove(namespace) + } + + /// Refuse a namespace whose fence state is `UNKNOWN_UNAVAILABLE`, before any work is done + /// to serve it. + pub fn check_available(&self, namespace: &NamespaceName) -> Result<(), FenceError> { + match self.get(namespace) { + Some(controller) => { + let gate = controller.gate(); + if gate.is_unavailable() { + gate.permits(OperationClass::NormalRead) + } else { + Ok(()) + } + } + None => Ok(()), + } + } + + pub fn len(&self) -> usize { + self.controllers.lock().len() + } +} + +#[cfg(test)] +mod tests { + use tempfile::tempdir; + use uuid::Uuid; + + use super::*; + use crate::config::MetaStoreConfig; + use crate::connection::config::DatabaseConfig; + use crate::database::DatabaseKind; + use crate::namespace::fence::command::{FenceCommand, FenceRequest}; + use crate::namespace::fence::outcome::{FenceDetail, FenceOutcome}; + use crate::namespace::fence::record::ServerIdentity; + use crate::namespace::fence::state::FenceState; + use crate::namespace::fence::store as fence_store; + use crate::namespace::meta_store::{metastore_connection_maker, FenceContext, MetaStore}; + + const LOG: Uuid = Uuid::from_u128(0x10); + const OP: Uuid = Uuid::from_u128(0xa); + + async fn open(dir: &std::path::Path) -> MetaStore { + let (maker, manager) = metastore_connection_maker(None, dir).await.unwrap(); + let conn = maker().unwrap(); + MetaStore::new( + MetaStoreConfig { + namespace_fence: true, + ..Default::default() + }, + dir, + conn, + manager, + DatabaseKind::Primary, + ) + .await + .unwrap() + } + + fn ctx() -> FenceContext { + FenceContext::now( + ServerIdentity { + build: "test".into(), + instance_id: Uuid::from_u128(0x99), + }, + Some(LOG), + ) + } + + #[tokio::test] + async fn seeded_from_load_fences_including_recovered_names() { + let tmp = tempdir().unwrap(); + { + let meta = open(tmp.path()).await; + for ns in ["fenced", "plain"] { + meta.handle(ns.into()) + .await + .unwrap() + .store(DatabaseConfig::default()) + .await + .unwrap(); + } + let controller = FenceController::unfenced("fenced".into()); + controller + .apply_command( + &meta, + FenceRequest { + namespace: "fenced".into(), + operation_id: OP, + command_id: Uuid::from_u128(1), + expected_state: FenceState::Unfenced, + expected_revision: 0, + command: FenceCommand::AcquireSourceWriteFence { + expected_log_id: LOG, + drain_policy: None, + }, + }, + ctx(), + ) + .await + .unwrap(); + meta.shutdown().await.unwrap(); + } + // A directory with an unreadable marker and no config: startup cannot recover it. + let lost = tmp.path().join("dbs").join("lost"); + std::fs::create_dir_all(&lost).unwrap(); + std::fs::write(lost.join(fence_store::MARKER_FILE_NAME), b"garbage").unwrap(); + + // Restart. + let meta = open(tmp.path()).await; + let registry = FenceRegistry::seeded(meta.load_fences().await.unwrap()); + assert_eq!(registry.len(), 2); + + let fenced = registry.get(&"fenced".into()).unwrap(); + let gate = fenced.gate(); + assert_eq!(gate.state(), FenceState::SourceDraining); + assert_eq!(gate.revision(), 1); + assert_eq!(gate.operation_id(), Some(OP)); + assert_eq!( + fenced + .permits(OperationClass::NormalWrite) + .unwrap_err() + .outcome(), + FenceOutcome::MigrationWriteFenced + ); + assert!(registry.check_available(&"fenced".into()).is_ok()); + + let lost = registry.get(&"lost".into()).unwrap(); + assert!(lost.gate().is_unavailable()); + let e = registry.check_available(&"lost".into()).unwrap_err(); + assert_eq!(e.outcome(), FenceOutcome::FenceStateUnavailable); + assert_eq!(e.detail(), Some(FenceDetail::CorruptRecord)); + for class in [OperationClass::NormalRead, OperationClass::NormalWrite] { + assert!(lost.permits(class).is_err()); + } + + // An ordinary namespace has no controller until it is used, then an UNFENCED one. + assert!(registry.get(&"plain".into()).is_none()); + let plain = registry.controller(&"plain".into()); + assert_eq!(plain.gate().state(), FenceState::Unfenced); + assert!(plain.permits(OperationClass::NormalWrite).is_ok()); + assert_eq!(registry.len(), 3); + } + + #[test] + fn controller_is_stable_until_removed() { + let registry = FenceRegistry::default(); + let a = registry.controller(&"ns".into()); + let b = registry.controller(&"ns".into()); + assert!(Arc::ptr_eq(&a, &b)); + assert!(Arc::ptr_eq(®istry.get(&"ns".into()).unwrap(), &a)); + assert!(registry.remove(&"ns".into()).is_some()); + assert!(!Arc::ptr_eq(®istry.controller(&"ns".into()), &a)); + } +} diff --git a/libsql-server/src/namespace/meta_store.rs b/libsql-server/src/namespace/meta_store.rs index 175fb24b55..6ed7b29fd5 100644 --- a/libsql-server/src/namespace/meta_store.rs +++ b/libsql-server/src/namespace/meta_store.rs @@ -804,6 +804,17 @@ fn not_primary() -> FenceError { .with_detail(FenceDetail::NotPrimary) } +/// A fence transaction whose `COMMIT` failed: whether it took effect is unknown, and the +/// controller keeps the namespace closed until the same command is replayed (section 8.4). +fn commit_indeterminate(e: rusqlite::Error) -> FenceStoreError { + FenceError::new( + FenceOutcome::FenceCommitIndeterminate, + format!("the metastore commit of a fence transition failed: {e}"), + ) + .with_detail(FenceDetail::IndeterminateCommit) + .into() +} + fn unavailable_receipt(e: impl std::fmt::Display) -> FenceError { FenceError::new( FenceOutcome::FenceStateUnavailable, @@ -918,7 +929,7 @@ fn apply_fence_command( .or(stored.record()) .map_or(request.operation_id, |r| r.operation_id); fence_store::prune_receipts(&tx, ns, owner, ctx.now_ms, inner.fence.receipt_retention)?; - tx.commit()?; + tx.commit().map_err(commit_indeterminate)?; // The command established the fence from the durable state; whatever startup could not // recover about this name is settled. inner.recovered.lock().remove(ns); @@ -1029,7 +1040,7 @@ fn complete_fence_drain( } fence_store::write_record(&tx, &next, fence_store::stored_revision(&tx, ns)?)?; fence_store::write_receipt(&tx, &final_receipt)?; - tx.commit()?; + tx.commit().map_err(commit_indeterminate)?; inner.recovered.lock().remove(ns); after_fence_commit(inner, &conn, Some(&next), true); diff --git a/libsql-server/src/namespace/mod.rs b/libsql-server/src/namespace/mod.rs index cba4030090..28ba60ea26 100644 --- a/libsql-server/src/namespace/mod.rs +++ b/libsql-server/src/namespace/mod.rs @@ -14,6 +14,7 @@ use crate::connection::Connection as _; use crate::database::Database; use crate::stats::Stats; +use self::fence::controller::FenceController; use self::meta_store::MetaStoreHandle; pub use self::name::NamespaceName; pub use self::store::NamespaceStore; @@ -66,6 +67,10 @@ pub struct Namespace { stats: Arc, db_config_store: MetaStoreHandle, path: Arc, + /// The namespace's fence controller, from the store's registry. Every connection, and the + /// dump and replication services that reach the namespace through the store, read its + /// gate. + fence: Arc, } impl Namespace { @@ -98,6 +103,13 @@ impl Namespace { Ok(()) } + // Read by the protocol layers that consult the gate outside a connection (dump, + // replication, lifecycle), which land later in this series. + #[allow(dead_code)] + pub(crate) fn fence(&self) -> &Arc { + &self.fence + } + pub fn config(&self) -> Arc { self.db_config_store.get() } diff --git a/libsql-server/src/namespace/store.rs b/libsql-server/src/namespace/store.rs index 1813ef5187..3a132cf10f 100644 --- a/libsql-server/src/namespace/store.rs +++ b/libsql-server/src/namespace/store.rs @@ -21,6 +21,7 @@ use crate::stats::Stats; use super::broadcasters::{BroadcasterHandle, BroadcasterRegistry}; use super::configurator::{DynConfigurator, NamespaceConfigurators}; +use super::fence::registry::FenceRegistry; use super::meta_store::{MetaStore, MetaStoreHandle}; use super::schema_lock::SchemaLocksRegistry; use super::{Namespace, ResetCb, ResetOp, ResolveNamespacePathFn, RestoreOption}; @@ -50,6 +51,9 @@ pub struct NamespaceStoreInner { broadcasters: BroadcasterRegistry, configurators: NamespaceConfigurators, db_kind: DatabaseKind, + /// Fence controllers, outside the cache: a namespace that is evicted and reloaded gets the + /// controller it had. + fences: FenceRegistry, } impl NamespaceStore { @@ -84,6 +88,13 @@ impl NamespaceStore { .time_to_idle(Duration::from_secs(86400)) .build(); + // Every namespace with fence state gets its controller before anything is served + // (section 8.5). + let fences = FenceRegistry::seeded(metadata.load_fences().await?); + if fences.len() > 0 { + tracing::info!("loaded {} namespace fence controllers", fences.len()); + } + Ok(Self { inner: Arc::new(NamespaceStoreInner { store, @@ -95,6 +106,7 @@ impl NamespaceStore { broadcasters: Default::default(), configurators, db_kind, + fences, }), }) } @@ -120,6 +132,8 @@ impl NamespaceStore { } }) .await??; + // The namespace's fence state went with it. + self.inner.fences.remove(&namespace); let mut bottomless_db_id_init = NamespaceBottomlessDbIdInit::FetchFromConfig; if let Some(ns) = self.inner.store.remove(&namespace).await { @@ -336,6 +350,9 @@ impl NamespaceStore { } }; + // A namespace whose fence state is unavailable is refused before any setup work. + self.inner.fences.check_available(&namespace)?; + // A lookup that cannot create: only the default namespace and lazy creation create a // namespace here, and those refuse a name whose fence state is not established. let handle = match self.inner.metadata.lookup(&namespace).await? { @@ -371,6 +388,10 @@ impl NamespaceStore { config: MetaStoreHandle, restore_option: RestoreOption, ) -> crate::Result { + // The controller is handed to the namespace before its first connection exists, so no + // connection is ever opened without a gate (section 8.5). + self.inner.fences.check_available(namespace)?; + let fence = self.inner.fences.controller(namespace); let ns = self .get_configurator(&config.get()) .setup( @@ -381,6 +402,7 @@ impl NamespaceStore { self.resolve_attach_fn(), self.clone(), self.broadcaster(namespace.clone()), + fence, ) .await?; @@ -535,3 +557,210 @@ impl NamespaceStore { .await } } + +#[cfg(test)] +mod fence_tests { + use std::path::Path; + + use libsql_sys::wal::Sqlite3WalManager; + use tempfile::tempdir; + use tokio::sync::Semaphore; + use uuid::Uuid; + + use super::*; + use crate::config::MetaStoreConfig; + use crate::namespace::configurator::{BaseNamespaceConfig, PrimaryConfig, PrimaryConfigurator}; + use crate::namespace::fence::command::{FenceCommand, FenceRequest}; + use crate::namespace::fence::outcome::{FenceDetail, FenceOutcome}; + use crate::namespace::fence::record::ServerIdentity; + use crate::namespace::fence::state::{FenceState, OperationClass}; + use crate::namespace::fence::store as fence_store; + use crate::namespace::meta_store::{metastore_connection_maker, FenceContext}; + + const LOG: Uuid = Uuid::from_u128(0x10); + const OP: Uuid = Uuid::from_u128(0xa); + + async fn open_store(dir: &Path) -> NamespaceStore { + let (maker, manager) = metastore_connection_maker(None, dir).await.unwrap(); + let meta = MetaStore::new( + MetaStoreConfig { + namespace_fence: true, + ..Default::default() + }, + dir, + maker().unwrap(), + manager, + DatabaseKind::Primary, + ) + .await + .unwrap(); + let mut configurators = NamespaceConfigurators::empty(); + configurators.with_primary(PrimaryConfigurator::new( + BaseNamespaceConfig { + base_path: dir.to_path_buf().into(), + extensions: Arc::new([]), + stats_sender: tokio::sync::mpsc::channel(1).0, + max_response_size: 100_000_000, + max_total_response_size: 100_000_000, + max_concurrent_connections: Arc::new(Semaphore::new(10)), + max_concurrent_requests: 10_000, + encryption_config: None, + connection_creation_timeout: None, + disable_intelligent_throttling: false, + }, + PrimaryConfig { + max_log_size: 1_000_000_000, + max_log_duration: None, + bottomless_replication: None, + scripted_backup: None, + checkpoint_interval: None, + }, + Arc::new(|| Sqlite3WalManager::default()), + )); + NamespaceStore::new(false, false, 10, meta, configurators, DatabaseKind::Primary) + .await + .unwrap() + } + + fn acquire(ns: &'static str) -> FenceRequest { + FenceRequest { + namespace: ns.into(), + operation_id: OP, + command_id: Uuid::from_u128(1), + expected_state: FenceState::Unfenced, + expected_revision: 0, + command: FenceCommand::AcquireSourceWriteFence { + expected_log_id: LOG, + drain_policy: None, + }, + } + } + + fn ctx() -> FenceContext { + FenceContext::now( + ServerIdentity { + build: "test".into(), + instance_id: Uuid::from_u128(0x99), + }, + Some(LOG), + ) + } + + #[tokio::test] + async fn evicted_namespace_reloads_with_the_same_controller() { + let tmp = tempdir().unwrap(); + let store = open_store(tmp.path()).await; + store + .create("ns".into(), RestoreOption::Latest, Default::default()) + .await + .unwrap(); + let (fence, stats) = store + .with("ns".into(), |ns| (ns.fence().clone(), ns.stats())) + .await + .unwrap(); + assert!(Arc::ptr_eq( + &fence, + &store.inner.fences.get(&"ns".into()).unwrap() + )); + fence + .apply_command(store.meta_store(), acquire("ns"), ctx()) + .await + .unwrap(); + let gate = fence.gate(); + assert_eq!(gate.state(), FenceState::SourceDraining); + + // Evict the namespace, as idle or capacity eviction does, and let it shut down. + store + .inner + .store + .invalidate(&NamespaceName::from("ns")) + .await; + store.inner.store.run_pending_tasks().await; + assert!(store + .inner + .store + .get(&NamespaceName::from("ns")) + .await + .is_none()); + + let (reloaded, reloaded_stats) = store + .with("ns".into(), |ns| (ns.fence().clone(), ns.stats())) + .await + .unwrap(); + // A new namespace instance, the same controller and gate. + assert!(!Arc::ptr_eq(&stats, &reloaded_stats)); + assert!(Arc::ptr_eq(&fence, &reloaded)); + assert_eq!(reloaded.gate(), gate); + assert!(reloaded.permits(OperationClass::NormalWrite).is_err()); + } + + #[tokio::test] + async fn restart_installs_the_durable_gate_before_serving() { + let tmp = tempdir().unwrap(); + { + let store = open_store(tmp.path()).await; + store + .create("ns".into(), RestoreOption::Latest, Default::default()) + .await + .unwrap(); + let fence = store.inner.fences.controller(&"ns".into()); + fence + .apply_command(store.meta_store(), acquire("ns"), ctx()) + .await + .unwrap(); + store.shutdown().await.unwrap(); + } + + let store = open_store(tmp.path()).await; + // The registry holds the durable gate before the namespace is loaded. + let fence = store.inner.fences.get(&"ns".into()).unwrap(); + assert_eq!(fence.gate().state(), FenceState::SourceDraining); + assert_eq!(fence.gate().revision(), 1); + let loaded = store + .with("ns".into(), |ns| ns.fence().clone()) + .await + .unwrap(); + assert!(Arc::ptr_eq(&fence, &loaded)); + } + + #[tokio::test] + async fn unavailable_namespace_is_refused_before_setup() { + let tmp = tempdir().unwrap(); + // An unreadable marker in a directory with no config: the fence state cannot be + // established. + let dir = tmp.path().join("dbs").join("lost"); + std::fs::create_dir_all(&dir).unwrap(); + std::fs::write(dir.join(fence_store::MARKER_FILE_NAME), b"garbage").unwrap(); + + let store = open_store(tmp.path()).await; + let r = store.with("lost".into(), |_| ()).await; + match r { + Err(Error::NamespaceFence(e)) => { + assert_eq!(e.outcome(), FenceOutcome::FenceStateUnavailable); + assert_eq!(e.detail(), Some(FenceDetail::CorruptRecord)); + } + other => panic!("expected a fence error, got {other:?}"), + } + // Nothing was set up: no database file, and the namespace is not cached. + assert!(!dir.join("data").exists()); + assert!(store + .inner + .store + .get(&NamespaceName::from("lost")) + .await + .is_none()); + } + + #[tokio::test] + async fn destroy_forgets_the_controller() { + let tmp = tempdir().unwrap(); + let store = open_store(tmp.path()).await; + store + .create("ns".into(), RestoreOption::Latest, Default::default()) + .await + .unwrap(); + assert!(store.inner.fences.get(&"ns".into()).is_some()); + store.destroy("ns".into(), false).await.unwrap(); + assert!(store.inner.fences.get(&"ns".into()).is_none()); + } +} From 5ca36aca380722e3cdf286a6fb093a7f2f8373a5 Mon Sep 17 00:00:00 2001 From: River Date: Tue, 29 Sep 2026 15:46:13 +0000 Subject: [PATCH 2/5] libsql-server: gate write transactions at the WAL by admission generation Make ManagedConnectionWalWrapper::begin_write_txn the authoritative namespace fence check. Before queueing for the write slot it requires the live gate to admit the connection's operation class and the program and its read transaction to have been admitted under the gate's current write generation. A refusal returns SQLITE_AUTH (not BUSY, so SQLite does not retry it, and before acquire(), so no slot is released that was never held) and leaves the typed outcome in the connection's FenceConnState. - FenceConnState gains begin_program, begin_read_txn and admit_write. CoreConnection::run, every with_raw call and vacuum_if_needed start a program; the WAL wrapper records the generation of each new read transaction before its snapshot is taken. - The Vm refuses Write and DDL statements early against the live gate and reports a WAL refusal as Error::NamespaceFence instead of SQLITE_AUTH. A plain SQLITE_AUTH from an authorizer is left unchanged. - New detail stale_transaction on MIGRATION_WRITE_FENCED for a transaction or program that began under an earlier generation. - Namespaces without a fence stay at generation 0 and behave as before. Tests cover a program parked between admission and its write while the fence is acquired and released, read-to-write upgrades, DDL, a header pragma, BEGIN IMMEDIATE, VACUUM and raw writes, stale transactions after release, and unchanged behaviour of unfenced namespaces. Co-authored-by: Tomasz Szymczyszyn --- .../src/connection/connection_core.rs | 21 +- .../src/connection/connection_manager.rs | 383 +++++++++++++++++- libsql-server/src/connection/legacy.rs | 10 +- libsql-server/src/connection/program.rs | 43 +- .../src/namespace/fence/controller.rs | 86 +++- libsql-server/src/namespace/fence/outcome.rs | 3 + 6 files changed, 528 insertions(+), 18 deletions(-) diff --git a/libsql-server/src/connection/connection_core.rs b/libsql-server/src/connection/connection_core.rs index 9025914c62..ad2075166b 100644 --- a/libsql-server/src/connection/connection_core.rs +++ b/libsql-server/src/connection/connection_core.rs @@ -11,6 +11,7 @@ use crate::connection::legacy::open_conn_active_checkpoint; use crate::error::Error; use crate::metrics::{PROGRAM_EXEC_COUNT, QUERY_CANCELED, VACUUM_COUNT, WAL_CHECKPOINT_COUNT}; use crate::namespace::broadcasters::BroadcasterHandle; +use crate::namespace::fence::controller::FenceConnState; use crate::namespace::meta_store::MetaStoreHandle; use crate::namespace::ResolveNamespacePathFn; use crate::query_analysis::StmtKind; @@ -37,6 +38,8 @@ pub(super) struct CoreConnection { broadcaster: BroadcasterHandle, hooked: bool, canceled: Arc, + /// Shared with this connection's WAL wrapper (`docs/NAMESPACE_FENCE.md` section 7.4). + fence: Arc, } fn update_stats( @@ -68,6 +71,7 @@ impl CoreConnection { get_current_frame_no: GetCurrentFrameNo, block_writes: Arc, resolve_attach_path: ResolveNamespacePathFn, + fence: Arc, ) -> Result { let conn = open_conn_active_checkpoint( path, @@ -113,6 +117,7 @@ impl CoreConnection { hooked: false, canceled, get_current_frame_no, + fence, }; for ext in extensions.iter() { @@ -188,17 +193,21 @@ impl CoreConnection { pgm: Program, mut builder: B, ) -> Result { - let (config, stats, block_writes, resolve_attach_path) = { + let (config, stats, block_writes, resolve_attach_path, fence) = { let mut lock = this.lock(); let config = lock.config_store.get(); let stats = lock.stats.clone(); let block_writes = lock.block_writes.clone(); let resolve_attach_path = lock.resolve_attach_path.clone(); + let fence = lock.fence.clone(); lock.update_hooks(); - (config, stats, block_writes, resolve_attach_path) + (config, stats, block_writes, resolve_attach_path, fence) }; + // The program is admitted under the gate's current write generation; a write + // transaction it opens must start under the same one (section 8.1). + fence.begin_program(); builder.init(&this.lock().builder_config)?; let mut vm = Vm::new( @@ -229,7 +238,8 @@ impl CoreConnection { update_stats(&stats, sql, rows_read, rows_written, mem_used, elapsed) }, resolve_attach_path, - ); + ) + .with_fence(fence); let mut has_timeout = false; while !vm.finished() { @@ -296,6 +306,7 @@ impl CoreConnection { } pub(super) fn vacuum_if_needed(&self) -> Result<()> { + self.fence.begin_program(); let page_count = self .conn .query_row("PRAGMA page_count", (), |row| row.get::<_, i64>(0))?; @@ -416,6 +427,10 @@ mod test { hooked: false, canceled: Arc::new(false.into()), get_current_frame_no: Arc::new(|| None), + fence: FenceConnState::new( + FenceController::unfenced(Default::default()), + crate::namespace::fence::state::OperationClass::NormalWrite, + ), }; let conn = Arc::new(Mutex::new(conn)); diff --git a/libsql-server/src/connection/connection_manager.rs b/libsql-server/src/connection/connection_manager.rs index 4baaa0ddc0..4cf4675092 100644 --- a/libsql-server/src/connection/connection_manager.rs +++ b/libsql-server/src/connection/connection_manager.rs @@ -120,9 +120,6 @@ pub struct ManagedConnectionWalWrapper { manager: ConnectionManager, /// The connection's fence state, which `begin_write_txn` checks against the namespace's /// gate (`docs/NAMESPACE_FENCE.md` section 8.1). - // Installed here so that no connection exists without it; the check itself lands in the - // next commit of this series. - #[allow(dead_code)] fence: Arc, } @@ -417,6 +414,13 @@ impl WrapWal for ManagedConnectionWalWrapper { #[tracing::instrument(skip_all, fields(id = self.id))] fn begin_write_txn(&mut self, wrapped: &mut InnerWal) -> libsql_sys::wal::Result<()> { tracing::debug!("begin write"); + // The authoritative fence check. It runs before `acquire()`, so a refusal holds no slot + // and releases none, and it returns `SQLITE_AUTH` rather than `SQLITE_BUSY`, so SQLite's + // busy handler does not retry it. The typed reason is left in the connection's fence + // state for the program layer. + if self.fence.admit_write().is_err() { + return Err(rusqlite::ffi::Error::new(rusqlite::ffi::SQLITE_AUTH)); + } self.acquire()?; match wrapped.begin_write_txn() { Ok(_) => { @@ -498,6 +502,9 @@ impl WrapWal for ManagedConnectionWalWrapper { #[tracing::instrument(skip_all, fields(id = self.id))] fn begin_read_txn(&mut self, wrapped: &mut InnerWal) -> libsql_sys::wal::Result { tracing::debug!("begin read txn"); + // Recorded before the snapshot is taken: a transition racing with it leaves the + // transaction with the older generation, which can only refuse a later upgrade. + self.fence.begin_read_txn(); wrapped.begin_read_txn() } @@ -563,3 +570,373 @@ impl WrapWal for ManagedConnectionWalWrapper { ret } } + +/// The WAL write gate (`docs/NAMESPACE_FENCE.md` section 8.1): tests on real connections of a +/// namespace whose fence controller goes through committed metastore transitions. +#[cfg(test)] +mod fence_tests { + use std::path::Path; + use std::sync::Arc; + + use libsql_sys::wal::wrapper::PassthroughWalWrapper; + use libsql_sys::wal::Sqlite3WalManager; + use rusqlite::functions::FunctionFlags; + use rusqlite::ErrorCode; + use tempfile::tempdir; + + use crate::connection::connection_core::CoreConnection; + use crate::connection::legacy::{LegacyConnection, MakeLegacyConnection}; + use crate::connection::program::Program; + use crate::connection::Connection as _; + use crate::error::Error; + use crate::namespace::fence::controller::tests::{ + create_namespace, ctx, fence_source, open_metastore, release, OP, + }; + use crate::namespace::fence::controller::FenceController; + use crate::namespace::fence::outcome::{FenceDetail, FenceError, FenceOutcome}; + use crate::namespace::fence::state::FenceState; + use crate::namespace::meta_store::{MetaStore, MetaStoreHandle}; + use crate::query_result_builder::test::{StepResult, TestBuilder}; + use crate::query_result_builder::QueryResultBuilder as _; + use crate::DEFAULT_AUTO_CHECKPOINT; + + type Conn = LegacyConnection; + + struct Harness { + _dir: tempfile::TempDir, + meta: MetaStore, + controller: Arc, + maker: MakeLegacyConnection, + } + + impl Harness { + async fn new() -> Self { + let dir = tempdir().unwrap(); + let meta_dir = dir.path().join("meta"); + let db_dir = dir.path().join("db"); + std::fs::create_dir_all(&meta_dir).unwrap(); + std::fs::create_dir_all(&db_dir).unwrap(); + let meta = open_metastore(&meta_dir).await; + create_namespace(&meta, "ns").await; + let controller = FenceController::unfenced("ns".into()); + let maker = make_connections(&db_dir, controller.clone()).await; + let this = Self { + _dir: dir, + meta, + controller, + maker, + }; + let conn = this.conn().await; + assert_ok(&run(&conn, &["create table t (x)"]).await); + this + } + + async fn conn(&self) -> Conn { + self.maker.make_connection().await.unwrap() + } + + /// UNFENCED -> SOURCE_DRAINING -> SOURCE_WRITE_FENCED. + async fn fence(&self) { + fence_source(&self.meta, &self.controller, OP).await; + assert_eq!( + self.controller.gate().state(), + FenceState::SourceWriteFenced + ); + } + + /// SOURCE_WRITE_FENCED -> RELEASED: writes are admitted again, under a new generation. + async fn release(&self) { + let revision = self.controller.gate().revision(); + self.controller + .apply_command(&self.meta, release("ns", OP, 2, revision), ctx()) + .await + .unwrap(); + assert_eq!(self.controller.gate().state(), FenceState::Released); + } + } + + async fn make_connections( + path: &Path, + fence: Arc, + ) -> MakeLegacyConnection { + MakeLegacyConnection::new( + path.into(), + PassthroughWalWrapper, + Default::default(), + Default::default(), + MetaStoreHandle::load(path).unwrap(), + Arc::new([]), + 100000000, + 100000000, + DEFAULT_AUTO_CHECKPOINT, + Arc::new(|| None), + None, + Default::default(), + Arc::new(|_| unreachable!()), + Arc::new(|| Sqlite3WalManager::default()), + fence, + ) + .await + .unwrap() + } + + async fn run(conn: &Conn, stmts: &[&'static str]) -> Vec { + let inner = conn.inner.clone(); + let stmts = stmts.to_vec(); + tokio::task::spawn_blocking(move || { + CoreConnection::run(inner, Program::seq(&stmts), TestBuilder::default()) + .unwrap() + .into_ret() + }) + .await + .unwrap() + } + + fn assert_ok(steps: &[StepResult]) { + for (i, step) in steps.iter().enumerate() { + assert!(step.is_ok(), "step {i} failed: {step:?}"); + } + } + + fn fence_error(step: &StepResult) -> &FenceError { + match step { + Err(Error::NamespaceFence(e)) => e, + other => panic!("expected a fence denial, got {other:?}"), + } + } + + async fn count(conn: &Conn) -> i64 { + conn.with_raw(|c| c.query_row("select count(*) from t", (), |r| r.get(0))) + .unwrap() + } + + /// A program is admitted, the fence is acquired and released while it runs, and the write it + /// then attempts is refused at the WAL although the live gate is open again: the program was + /// admitted under a generation that is no longer current. The race is held open by a SQL + /// function that parks the program between admission and its write. + #[tokio::test(flavor = "multi_thread")] + async fn wal_gate_rejects_program_admitted_before_fence() { + let h = Harness::new().await; + let conn = h.conn().await; + + let (reached_tx, mut reached_rx) = tokio::sync::mpsc::unbounded_channel::<()>(); + let (resume_tx, resume_rx) = std::sync::mpsc::channel::<()>(); + let parked = std::panic::AssertUnwindSafe((reached_tx, std::sync::Mutex::new(resume_rx))); + conn.with_raw(move |c| { + c.create_scalar_function("park", 0, FunctionFlags::SQLITE_UTF8, move |_| { + let (reached, resume) = &*parked; + reached.send(()).unwrap(); + resume.lock().unwrap().recv().unwrap(); + Ok(1) + }) + }) + .unwrap(); + + let admitted_at = h.controller.write_generation(); + let program = tokio::spawn({ + let conn = conn.clone(); + async move { run(&conn, &["select park()", "insert into t values (1)"]).await } + }); + reached_rx.recv().await.unwrap(); + assert_eq!(conn.fence.program_generation(), admitted_at); + + h.fence().await; + h.release().await; + assert!(h.controller.write_generation() > admitted_at); + resume_tx.send(()).unwrap(); + + let steps = program.await.unwrap(); + assert!(steps[0].is_ok()); + let e = fence_error(&steps[1]); + assert_eq!(e.outcome(), FenceOutcome::MigrationWriteFenced); + assert_eq!(e.detail(), Some(FenceDetail::StaleTransaction)); + assert_eq!(count(&conn).await, 0); + + // The next program is admitted under the current generation and writes. + assert_ok(&run(&conn, &["insert into t values (2)"]).await); + assert_eq!(count(&conn).await, 1); + } + + /// A read transaction opened before a transition cannot be upgraded to a write transaction + /// after it, whether the gate is still closed or open again. + #[tokio::test(flavor = "multi_thread")] + async fn fence_rejects_read_to_write_upgrade() { + let h = Harness::new().await; + let conn = h.conn().await; + + // Gate closed: refused before the write runs. + assert_ok(&run(&conn, &["begin", "select * from t"]).await); + h.fence().await; + let steps = run(&conn, &["insert into t values (1)"]).await; + assert_eq!( + fence_error(&steps[0]).outcome(), + FenceOutcome::MigrationWriteFenced + ); + + // Gate open again: the transaction still belongs to the old generation, and only the + // WAL gate can tell. + h.release().await; + let steps = run(&conn, &["insert into t values (1)"]).await; + let e = fence_error(&steps[0]); + assert_eq!(e.outcome(), FenceOutcome::MigrationWriteFenced); + assert_eq!(e.detail(), Some(FenceDetail::StaleTransaction)); + assert_ok(&run(&conn, &["rollback"]).await); + + // A fresh transaction writes. + assert_ok(&run(&conn, &["begin", "insert into t values (1)", "commit"]).await); + assert_eq!(count(&conn).await, 1); + } + + /// DDL, a pragma that writes the header, and `BEGIN IMMEDIATE` are refused while writes are + /// fenced; reads keep working, and nothing was written. + #[tokio::test(flavor = "multi_thread")] + async fn fence_rejects_ddl_and_pragma() { + let h = Harness::new().await; + let conn = h.conn().await; + h.fence().await; + + for stmt in [ + "create table u (x)", + "create index i on t (x)", + "drop table t", + "pragma user_version = 7", + "begin immediate", + ] { + let steps = run(&conn, &[stmt]).await; + assert_eq!( + fence_error(&steps[0]).outcome(), + FenceOutcome::MigrationWriteFenced, + "{stmt}" + ); + assert!(conn.inner.lock().is_autocommit(), "{stmt}"); + } + assert_ok(&run(&conn, &["select * from t"]).await); + + h.release().await; + let version: i64 = conn + .with_raw(|c| c.query_row("pragma user_version", (), |r| r.get(0))) + .unwrap(); + assert_eq!(version, 0); + let tables: i64 = conn + .with_raw(|c| { + c.query_row( + "select count(*) from sqlite_schema where name in ('u', 'i')", + (), + |r| r.get(0), + ) + }) + .unwrap(); + assert_eq!(tables, 0); + } + + /// `with_raw` users (admin shell, schema migration, dump load) bypass statement + /// classification but not the WAL: the write fails with `SQLITE_AUTH` and the typed reason + /// is in the connection's denial slot. + #[tokio::test(flavor = "multi_thread")] + async fn fence_rejects_raw_with_raw_write() { + let h = Harness::new().await; + let conn = h.conn().await; + h.fence().await; + + for sql in ["insert into t values (1)", "begin immediate", "vacuum"] { + let err = conn.with_raw(|c| c.execute_batch(sql)).unwrap_err(); + match err { + rusqlite::Error::SqliteFailure(e, _) => { + assert_eq!(e.code, ErrorCode::AuthorizationForStatementDenied, "{sql}") + } + e => panic!("{sql}: unexpected error {e}"), + } + assert_eq!( + conn.fence.take_denial().unwrap().outcome(), + FenceOutcome::MigrationWriteFenced, + "{sql}" + ); + } + assert_eq!(count(&conn).await, 0); + + h.release().await; + conn.with_raw(|c| c.execute_batch("insert into t values (1)")) + .unwrap(); + assert_eq!(count(&conn).await, 1); + } + + /// The fence acquired and released between a transaction's first read and its write: the + /// write is refused although the gate is open, while a connection that was idle across the + /// transition, and the same connection after a rollback, write normally. + #[tokio::test(flavor = "multi_thread")] + async fn stale_generation_cannot_write_after_release() { + let h = Harness::new().await; + let in_txn = h.conn().await; + let idle = h.conn().await; + + assert_ok(&run(&in_txn, &["begin", "select count(*) from t"]).await); + h.fence().await; + h.release().await; + assert!(h + .controller + .permits(crate::namespace::fence::state::OperationClass::NormalWrite) + .is_ok()); + + let steps = run(&in_txn, &["insert into t values (1)", "commit"]).await; + assert_eq!( + fence_error(&steps[0]).detail(), + Some(FenceDetail::StaleTransaction) + ); + // The commit that follows ends the transaction without having written anything. + assert!(steps[1].is_ok()); + assert!(in_txn.inner.lock().is_autocommit()); + assert_ok(&run(&idle, &["insert into t values (2)"]).await); + assert_ok(&run(&in_txn, &["insert into t values (3)"]).await); + assert_eq!(count(&idle).await, 2); + } + + /// A namespace that never had a fence behaves as before: generation 0 everywhere, every + /// kind of write works, and a plain `SQLITE_AUTH` from an authorizer is reported as the + /// SQLite error it is, not as a fence denial. + #[tokio::test(flavor = "multi_thread")] + async fn unfenced_namespace_is_unchanged() { + let h = Harness::new().await; + let conn = h.conn().await; + + assert_ok( + &run( + &conn, + &[ + "insert into t values (1)", + "begin", + "select * from t", + "insert into t values (2)", + "commit", + "create table u (x)", + "pragma user_version = 3", + ], + ) + .await, + ); + conn.with_raw(|c| c.execute_batch("insert into t values (3)")) + .unwrap(); + assert_eq!(count(&conn).await, 3); + assert_eq!(h.controller.write_generation(), 0); + assert_eq!(conn.fence.program_generation(), 0); + assert_eq!(conn.fence.txn_generation(), 0); + + conn.with_raw(|c| { + c.authorizer(Some(|ctx: rusqlite::hooks::AuthContext<'_>| { + match ctx.action { + rusqlite::hooks::AuthAction::Insert { .. } => { + rusqlite::hooks::Authorization::Deny + } + _ => rusqlite::hooks::Authorization::Allow, + } + })) + }); + let steps = run(&conn, &["insert into t values (4)"]).await; + match &steps[0] { + Err(Error::RusqliteErrorExtended(rusqlite::Error::SqliteFailure(e, _), _)) => { + assert_eq!(e.code, ErrorCode::AuthorizationForStatementDenied) + } + other => panic!("expected the authorizer's error, got {other:?}"), + } + assert!(conn.fence.take_denial().is_none()); + } +} diff --git a/libsql-server/src/connection/legacy.rs b/libsql-server/src/connection/legacy.rs index 29676d237a..93ab86cb49 100644 --- a/libsql-server/src/connection/legacy.rs +++ b/libsql-server/src/connection/legacy.rs @@ -173,9 +173,7 @@ where pub struct LegacyConnection { pub(super) inner: Arc>>>, - /// Shared with the connection's WAL wrapper. - // Read by the WAL gate and the program admission check in the next commit of this series. - #[allow(dead_code)] + /// Shared with the connection's WAL wrapper and its `CoreConnection`. pub(super) fence: Arc, } @@ -344,7 +342,7 @@ where let connection_manager = connection_manager.clone(); let fence = fence.clone(); move || -> crate::Result<_> { - let manager = ManagedConnectionWalWrapper::new(connection_manager, fence); + let manager = ManagedConnectionWalWrapper::new(connection_manager, fence.clone()); let id = manager.id(); let wal = make_wal().wrap(manager).wrap(wal_wrapper); @@ -359,6 +357,7 @@ where current_frame_no_receiver, block_writes, resolve_attach_path, + fence, )?; let namespace = path @@ -464,6 +463,9 @@ where fn with_raw(&self, f: impl FnOnce(&mut rusqlite::Connection) -> R) -> R { let mut inner = self.inner.lock(); + // A raw use of the connection is a program like any other: the WAL gate admits a write + // transaction it opens only under the generation it started under. + self.fence.begin_program(); f(inner.raw_mut()) } } diff --git a/libsql-server/src/connection/program.rs b/libsql-server/src/connection/program.rs index 4d5ada51ff..8cafe681e7 100644 --- a/libsql-server/src/connection/program.rs +++ b/libsql-server/src/connection/program.rs @@ -7,6 +7,7 @@ use rusqlite::StatementStatus; use crate::auth::Permission; use crate::error::Error; use crate::metrics::{READ_QUERY_COUNT, WRITE_QUERY_COUNT}; +use crate::namespace::fence::controller::FenceConnState; use crate::namespace::{NamespaceName, ResolveNamespacePathFn}; use crate::query::Query; use crate::query_analysis::StmtKind; @@ -104,6 +105,7 @@ pub struct Vm<'a, B, F, S> { should_block: F, update_stats: S, resolve_attach_path: ResolveNamespacePathFn, + fence: Option>, } impl<'a, B, F, S> Vm<'a, B, F, S> @@ -127,9 +129,19 @@ where should_block, update_stats, resolve_attach_path, + fence: None, } } + /// Check the namespace fence on this connection: a statement that can write is refused + /// before it runs while the live gate denies the connection's class, and a refusal by the + /// WAL gate is reported as the fence error rather than as `SQLITE_AUTH` + /// (`docs/NAMESPACE_FENCE.md` section 8.1). + pub fn with_fence(mut self, fence: Arc) -> Self { + self.fence = Some(fence); + self + } + #[inline] fn current_step(&self) -> &Step { &self.program.steps()[self.current_step] @@ -164,7 +176,9 @@ where // builder error interrupt the execution of query. we should exit immediately. Err(e @ Error::BuilderError(_)) => return Err(e), Err(mut e) => { - if let Error::RusqliteError(err) = e { + if let Some(denial) = self.take_fence_denial(&e) { + e = Error::NamespaceFence(denial); + } else if let Error::RusqliteError(err) = e { let extended_code = unsafe { rusqlite::ffi::sqlite3_extended_errcode(conn.handle()) }; @@ -200,12 +214,39 @@ where Ok(query) } + /// The fence denial behind `e`, when `e` is the `SQLITE_AUTH` the WAL gate returns. A plain + /// `SQLITE_AUTH` (from an authorizer) is left alone, and so is an empty denial slot. + fn take_fence_denial(&self, e: &Error) -> Option { + let fence = self.fence.as_ref()?; + match e { + Error::RusqliteError(rusqlite::Error::SqliteFailure( + rusqlite::ffi::Error { + code: rusqlite::ErrorCode::AuthorizationForStatementDenied, + .. + }, + _, + )) => fence.take_denial(), + _ => None, + } + } + fn execute_query(&mut self, conn: &rusqlite::Connection) -> crate::Result<(u64, Option)> { tracing::debug!("executing query: {}", self.current_step().query.stmt.stmt); increment_counter!("libsql_server_libsql_query_execute"); let start = Instant::now(); + // Early fence admission for statements that can write. The WAL gate is authoritative + // and catches everything this misses (a misclassified statement, a read-to-write + // upgrade, a gate change while the statement runs). + if let Some(fence) = &self.fence { + if matches!( + self.current_step().query.stmt.kind, + StmtKind::Write | StmtKind::DDL + ) { + fence.controller().permits(fence.class())?; + } + } let (blocked, reason) = (self.should_block)(&self.current_step().query.stmt.kind); if blocked { return Err(Error::Blocked(reason)); diff --git a/libsql-server/src/namespace/fence/controller.rs b/libsql-server/src/namespace/fence/controller.rs index 265e04e40f..a9a31dc03a 100644 --- a/libsql-server/src/namespace/fence/controller.rs +++ b/libsql-server/src/namespace/fence/controller.rs @@ -384,7 +384,14 @@ fn pending_indeterminate((operation_id, command_id): CommandKey) -> FenceError { } /// The fence state of one connection, shared by its WAL wrapper and its `CoreConnection` -/// (section 7.4). The WAL gate reads and writes it; this commit only installs it. +/// (section 7.4). +/// +/// - A program (a `CoreConnection::run`, a `with_raw` call, a vacuum) starts with +/// [`begin_program`](Self::begin_program), which records the generation it was admitted under. +/// - The WAL wrapper calls [`begin_read_txn`](Self::begin_read_txn) whenever SQLite opens a read +/// transaction, and [`admit_write`](Self::admit_write) in `begin_write_txn` before it queues for +/// the write slot. A refusal leaves its typed outcome in the denial slot, which the program +/// layer takes to report the fence error instead of the bare `SQLITE_AUTH` the WAL returns. #[derive(Debug)] pub struct FenceConnState { controller: Arc, @@ -425,13 +432,69 @@ impl FenceConnState { self.txn_generation.load(Ordering::Acquire) } + /// Start a program on this connection: record the generation it is admitted under and + /// forget any denial a previous program left behind. Returns that generation. + pub fn begin_program(&self) -> u64 { + let generation = self.controller.write_generation(); + self.program_generation.store(generation, Ordering::Release); + *self.denial.lock() = None; + generation + } + + /// SQLite is opening a new read transaction on this connection: record the generation it + /// is opened under. A later upgrade of that transaction to a write transaction must happen + /// under the same generation. + pub fn begin_read_txn(&self) { + self.txn_generation + .store(self.controller.write_generation(), Ordering::Release); + } + + /// The authoritative write admission (section 8.1, check 2): the live gate permits this + /// connection's class, and the program and its read transaction were both admitted under + /// the gate's current write generation. On refusal the typed outcome is left in the denial + /// slot and returned. + pub fn admit_write(&self) -> Result<(), FenceError> { + let result = { + let gate = self.controller.gate.borrow(); + gate.permits(self.class).and_then(|()| { + let current = gate.write_generation; + let program = self.program_generation(); + let txn = self.txn_generation(); + if program == current && txn == current { + Ok(()) + } else { + Err(FenceError::new( + FenceOutcome::MigrationWriteFenced, + format!( + "the namespace fence changed after this transaction began \ + (program admitted at generation {program}, transaction opened at \ + generation {txn}, current generation {current}); roll back and \ + begin a new transaction" + ), + ) + .with_detail(FenceDetail::StaleTransaction)) + } + }) + }; + if let Err(e) = &result { + tracing::debug!( + namespace = %self.controller.namespace, + class = ?self.class, + "write transaction refused by the namespace fence: {e}" + ); + *self.denial.lock() = Some(e.clone()); + } + result + } + + /// Take the typed outcome of the last refusal at the WAL, if any. pub fn take_denial(&self) -> Option { self.denial.lock().take() } } #[cfg(test)] -mod tests { +pub(crate) mod tests { use std::path::Path; use tempfile::tempdir; @@ -445,7 +508,7 @@ mod tests { use crate::namespace::meta_store::{metastore_connection_maker, FenceCommitKind}; const LOG: Uuid = Uuid::from_u128(0x10); - const OP: Uuid = Uuid::from_u128(0xa); + pub(crate) const OP: Uuid = Uuid::from_u128(0xa); const OTHER_OP: Uuid = Uuid::from_u128(0xb); pub(crate) async fn open_metastore(dir: &Path) -> MetaStore { @@ -474,7 +537,7 @@ mod tests { .unwrap(); } - fn ctx() -> FenceContext { + pub(crate) fn ctx() -> FenceContext { FenceContext::now( ServerIdentity { build: "test".into(), @@ -484,7 +547,7 @@ mod tests { ) } - fn acquire(ns: &'static str, op: Uuid, command_id: u128) -> FenceRequest { + pub(crate) fn acquire(ns: &'static str, op: Uuid, command_id: u128) -> FenceRequest { FenceRequest { namespace: ns.into(), operation_id: op, @@ -498,7 +561,12 @@ mod tests { } } - fn release(ns: &'static str, op: Uuid, command_id: u128, revision: u64) -> FenceRequest { + pub(crate) fn release( + ns: &'static str, + op: Uuid, + command_id: u128, + revision: u64, + ) -> FenceRequest { FenceRequest { namespace: ns.into(), operation_id: op, @@ -519,7 +587,11 @@ mod tests { /// Acquire and complete the drain directly, as the write drain will: the controller ends in /// `SOURCE_WRITE_FENCED`. - async fn fence_source(meta: &MetaStore, controller: &Arc, op: Uuid) { + pub(crate) async fn fence_source( + meta: &MetaStore, + controller: &Arc, + op: Uuid, + ) { let commit = controller .apply_command(meta, acquire("ns", op, 1), ctx()) .await diff --git a/libsql-server/src/namespace/fence/outcome.rs b/libsql-server/src/namespace/fence/outcome.rs index 6a9b6a8b57..cf4df0c9af 100644 --- a/libsql-server/src/namespace/fence/outcome.rs +++ b/libsql-server/src/namespace/fence/outcome.rs @@ -201,6 +201,8 @@ pub enum FenceDetail { IncompleteTargetCreation, MetastoreBehindMarker, IndeterminateCommit, + // MIGRATION_WRITE_FENCED + StaleTransaction, } impl FenceDetail { @@ -223,6 +225,7 @@ impl FenceDetail { FenceDetail::IncompleteTargetCreation => "incomplete_target_creation", FenceDetail::MetastoreBehindMarker => "metastore_behind_marker", FenceDetail::IndeterminateCommit => "indeterminate_commit", + FenceDetail::StaleTransaction => "stale_transaction", } } } From 2232892fbaf88224231891da11c7ec893f8923d6 Mon Sep 17 00:00:00 2001 From: River Date: Tue, 29 Sep 2026 16:03:47 +0000 Subject: [PATCH 3/5] libsql-server: class writer-queue entries and wake them on fence changes Queue entries and the write slot now carry the operation class. A write transaction waiting for the slot re-checks the connection's fence admission every time it takes the manager's lock, and the fence controller wakes every registered write queue after each change of the write generation, so a writer queued before a fence leaves the queue with MIGRATION_WRITE_FENCED instead of waiting for the slot and then writing. Checkpoints ask for the slot as maintenance: they are never refused and queue again when woken. The manager exposes what the positive write drain needs: the active writer and its class, a notification on every release, and abort_active(), which uses the registered rollback handle. Abort no longer panics when the connection has already closed. VACUUM is skipped, and reported as skipped rather than failed, while the fence denies normal writes, including when the fence closes between the check and the statement. TRUNCATE checkpoints run in every state. Co-authored-by: Tomasz Szymczyszyn --- .../src/connection/connection_core.rs | 43 +- .../src/connection/connection_manager.rs | 425 +++++++++++++++++- libsql-server/src/connection/legacy.rs | 12 +- .../src/namespace/fence/controller.rs | 94 +++- 4 files changed, 535 insertions(+), 39 deletions(-) diff --git a/libsql-server/src/connection/connection_core.rs b/libsql-server/src/connection/connection_core.rs index ad2075166b..212e5c2b3d 100644 --- a/libsql-server/src/connection/connection_core.rs +++ b/libsql-server/src/connection/connection_core.rs @@ -12,6 +12,7 @@ use crate::error::Error; use crate::metrics::{PROGRAM_EXEC_COUNT, QUERY_CANCELED, VACUUM_COUNT, WAL_CHECKPOINT_COUNT}; use crate::namespace::broadcasters::BroadcasterHandle; use crate::namespace::fence::controller::FenceConnState; +use crate::namespace::fence::state::OperationClass; use crate::namespace::meta_store::MetaStoreHandle; use crate::namespace::ResolveNamespacePathFn; use crate::query_analysis::StmtKind; @@ -25,6 +26,15 @@ use super::program::{DescribeCol, DescribeParam, DescribeResponse, Program, Vm}; pub type GetCurrentFrameNo = Arc Option + Send + Sync + 'static>; +/// What [`CoreConnection::vacuum_if_needed_above`] did. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(crate) enum VacuumOutcome { + Vacuumed, + NotNeeded, + /// Skipped because the namespace fence denies normal writes. + Fenced, +} + /// The base connection type, shared between legacy and libsql-wal implementations pub(super) struct CoreConnection { conn: libsql_sys::Connection, @@ -306,22 +316,45 @@ impl CoreConnection { } pub(super) fn vacuum_if_needed(&self) -> Result<()> { + // NOTICE: don't bother vacuuming if we don't have at least 256MiB of data + self.vacuum_if_needed_above(65536).map(|_| ()) + } + + /// `VACUUM` if the database has at least `min_pages` pages and more than half of them are + /// free. `VACUUM` is not maintenance: it takes a write transaction and produces replicated + /// frames, so it is skipped whenever the namespace fence denies normal writes + /// (`docs/NAMESPACE_FENCE.md` section 7.3), including when the fence closes between the + /// check and the `VACUUM` itself. + pub(super) fn vacuum_if_needed_above(&self, min_pages: i64) -> Result { self.fence.begin_program(); + if let Err(e) = self.fence.controller().permits(OperationClass::Vacuum) { + tracing::debug!("skipping vacuum: {e}"); + return Ok(VacuumOutcome::Fenced); + } let page_count = self .conn .query_row("PRAGMA page_count", (), |row| row.get::<_, i64>(0))?; let freelist_count = self .conn .query_row("PRAGMA freelist_count", (), |row| row.get::<_, i64>(0))?; - // NOTICE: don't bother vacuuming if we don't have at least 256MiB of data - if page_count >= 65536 && freelist_count * 2 > page_count { + let outcome = if page_count >= min_pages && freelist_count * 2 > page_count { tracing::info!("Vacuuming: pages={page_count} freelist={freelist_count}"); - self.conn.execute("VACUUM", ())?; + if let Err(e) = self.conn.execute("VACUUM", ()) { + return match self.fence.take_denial() { + Some(denial) => { + tracing::debug!("skipping vacuum: {denial}"); + Ok(VacuumOutcome::Fenced) + } + None => Err(e.into()), + }; + } + VacuumOutcome::Vacuumed } else { tracing::trace!("Not vacuuming: pages={page_count} freelist={freelist_count}"); - } + VacuumOutcome::NotNeeded + }; VACUUM_COUNT.increment(1); - Ok(()) + Ok(outcome) } pub(super) fn describe(&self, sql: &str) -> crate::Result { diff --git a/libsql-server/src/connection/connection_manager.rs b/libsql-server/src/connection/connection_manager.rs index 4cf4675092..009adfeb45 100644 --- a/libsql-server/src/connection/connection_manager.rs +++ b/libsql-server/src/connection/connection_manager.rs @@ -15,6 +15,7 @@ use rusqlite::ErrorCode; use super::connection_core::CoreConnection; use super::TXN_TIMEOUT; use crate::namespace::fence::controller::FenceConnState; +use crate::namespace::fence::state::OperationClass; pub type ConnId = u64; pub type InnerWalManager = Sqlite3WalManager; @@ -25,10 +26,16 @@ pub type ManagedConnectionWal = WrappedWal); @@ -36,10 +43,13 @@ impl Abort { fn from_conn(conn: &Arc>>) -> Self { let conn = Arc::downgrade(conn); Self(Arc::new(move || { - conn.upgrade() - .expect("connection still owns the slot, so it must exist") - .lock() - .force_rollback(); + // The connection can be closing concurrently: a drain aborting the active writer + // races with the client going away. A connection that is gone has already released + // its slot (or is about to, from `close`), so there is nothing left to roll back. + match conn.upgrade() { + Some(conn) => conn.lock().force_rollback(), + None => tracing::debug!("connection closed before it could be rolled back"), + } })) } @@ -62,6 +72,89 @@ impl ConnectionManager { let abort = Abort::from_conn(conn); self.inner.abort_handle.lock().insert(id, abort); } + + /// The connection holding the write slot, and the class it holds it for. A slot that has + /// been handed to a queued connection that has not taken it yet counts as held. + // Used by the positive source write drain (section 8.3). + #[cfg_attr(not(test), allow(dead_code))] + pub(crate) fn active_writer(&self) -> Option<(ConnId, OperationClass)> { + self.inner.current.lock().map(|slot| (slot.id, slot.class)) + } + + /// Notified (with `notify_waiters`) every time the write slot is released or handed on. A + /// waiter registers interest (`Notified::enable`) before it checks + /// [`active_writer`](Self::active_writer), so a release in between is not missed. + // Used by the positive source write drain (section 8.3). + #[cfg_attr(not(test), allow(dead_code))] + pub(crate) fn released(&self) -> &tokio::sync::Notify { + &self.inner.released + } + + /// Roll back the transaction of the connection holding the write slot, using the rollback + /// handle it registered. Returns the connection that was asked to roll back, if any. The + /// slot is released by the rollback itself (`end_read_txn`/`end_write_txn`), which notifies + /// [`released`](Self::released); a connection closing at the same time is tolerated. + // Used by the positive source write drain (section 8.3). + #[cfg_attr(not(test), allow(dead_code))] + pub(crate) fn abort_active(&self) -> Option { + let id = self.active_writer()?.0; + let handle = self.inner.abort_handle.lock().get(&id).cloned(); + match handle { + Some(handle) => { + tracing::debug!("aborting the active writer {id}"); + handle.abort(); + Some(id) + } + None => { + tracing::debug!("the active writer {id} is closing; nothing to abort"); + None + } + } + } + + /// A waker for the fence controller, which calls it after every change of the write + /// generation (`docs/NAMESPACE_FENCE.md` section 8.2): every connection waiting in the write + /// queue is woken and re-checks the fence. One the gate denies returns the typed denial + /// instead of waiting for the slot; one it admits (a checkpoint) queues again. The waker + /// holds the manager weakly and reports `false` once it is gone, so the controller can + /// forget it. + pub(crate) fn fence_waker(&self) -> Box bool + Send + Sync> { + let inner = Arc::downgrade(&self.inner); + Box::new(move || match inner.upgrade() { + Some(inner) => { + wake_queue_for_fence(&inner); + true + } + None => false, + }) + } + + #[cfg(test)] + pub(crate) fn queued_writers(&self) -> usize { + self.inner.write_queue.len() + } +} + +fn wake_queue_for_fence(inner: &ConnectionManagerInner) { + // Under the `current` lock, so that a waiter either queued before this (and is stolen and + // woken here) or observes the new token when it takes the lock. + let _current = inner.current.lock(); + inner.fence_token.fetch_add(1, Ordering::SeqCst); + let mut woken = 0; + loop { + match inner.write_queue.steal() { + Steal::Empty => break, + Steal::Success((id, _, unparker)) => { + tracing::debug!("fence changed, waking queued connection id={id}"); + unparker.unpark(); + woken += 1; + } + Steal::Retry => (), + } + } + if woken > 0 { + tracing::debug!("fence changed, woke {woken} queued connections"); + } } impl Deref for ConnectionManager { @@ -92,12 +185,17 @@ pub struct ConnectionManagerInner { abort_handle: Mutex>, /// threads waiting to acquire the lock /// todo: limit how many can be push - write_queue: crossbeam::deque::Injector<(ConnId, Unparker)>, + write_queue: crossbeam::deque::Injector, txn_timeout_duration: Duration, /// the time we are given to acquire a transaction after we were given a slot acquire_timeout_duration: Duration, next_conn_id: AtomicU64, sync_token: AtomicU64, + /// Incremented, under the `current` lock, every time the queue is woken for a fence change. + /// A waiter that sees it move knows it was taken off the queue. + fence_token: AtomicU64, + /// Notified whenever the write slot is released or handed on. + released: tokio::sync::Notify, } impl Default for ConnectionManagerInner { @@ -110,6 +208,8 @@ impl Default for ConnectionManagerInner { acquire_timeout_duration: Duration::from_millis(15), next_conn_id: Default::default(), sync_token: AtomicU64::new(0), + fence_token: AtomicU64::new(0), + released: Default::default(), } } } @@ -133,13 +233,37 @@ impl ManagedConnectionWalWrapper { self.id } - fn acquire(&self) -> libsql_sys::wal::Result<()> { + /// Wait for the write slot on behalf of work of `class`. `Maintenance` (checkpoints) is never + /// refused by the fence; every other class re-checks the connection's write admission each + /// time it takes the `current` lock, so a waiter queued before a fence change leaves the + /// queue with the typed denial (`docs/NAMESPACE_FENCE.md` section 8.2). + fn acquire(&self, class: OperationClass) -> libsql_sys::wal::Result<()> { let parker = Parker::new(); let mut enqueued = false; let enqueued_at = Instant::now(); let sync_token = self.manager.sync_token.load(Ordering::SeqCst); + let mut fence_token = self.manager.fence_token.load(Ordering::SeqCst); loop { let mut current = self.manager.current.lock(); + let ours = current.as_ref().map_or(false, |slot| slot.id == self.id); + if class != OperationClass::Maintenance && self.fence.admit_write().is_err() { + // The denial is in the connection's fence state. If the slot had already been + // handed to us, pass it on: we are not going to use it. + if ours { + self.hand_off(&mut current); + } + tracing::debug!("write slot request refused by the namespace fence"); + return Err(rusqlite::ffi::Error::new(rusqlite::ffi::SQLITE_AUTH)); + } + let observed = self.manager.fence_token.load(Ordering::SeqCst); + if observed != fence_token { + fence_token = observed; + // A fence change emptied the queue. Unless the slot was handed to us before + // that, we are no longer queued and must queue again. + if enqueued && !ours { + enqueued = false; + } + } // if current is not currently us, and we havent enqueued yet, then enqueue // current can be us in two cases: // - in previous iteration, the queue was empty, and we popped ourselves @@ -194,7 +318,7 @@ impl ManagedConnectionWalWrapper { if current.as_mut().map_or(true, |slot| slot.id != self.id) && !enqueued { self.manager .write_queue - .push((self.id, parker.unparker().clone())); + .push((self.id, class, parker.unparker().clone())); enqueued = true; tracing::debug!("enqueued"); } @@ -284,6 +408,7 @@ impl ManagedConnectionWalWrapper { None => { *current = Some(Slot { id: self.id, + class, started_at: Instant::now(), state: SlotState::Acquiring, }); @@ -321,6 +446,7 @@ impl ManagedConnectionWalWrapper { None => { *current = Some(Slot { id: self.id, + class, started_at: Instant::now(), state: SlotState::Acquiring, }) @@ -344,10 +470,11 @@ impl ManagedConnectionWalWrapper { }; match next { - Some((id, unpaker)) => { + Some((id, class, unpaker)) => { tracing::debug!(line = line!(), "unparking id={id}"); **current = Some(Slot { id, + class, started_at: Instant::now(), state: SlotState::Notified, }); @@ -369,12 +496,15 @@ impl ManagedConnectionWalWrapper { assert_eq!(slot.id, self.id); tracing::debug!("transaction finished after {:?}", slot.started_at.elapsed()); - match self.schedule_next(&mut current) { - Some(_) => (), - None => { - *current = None; - } - } + self.hand_off(&mut current); + } + + /// Give the (already vacated or ours) slot to the next queued connection, or leave it free, + /// and tell drain waiters. + fn hand_off(&self, current: &mut MutexGuard>) { + **current = None; + self.schedule_next(current); + self.manager.released.notify_waiters(); } } @@ -421,7 +551,7 @@ impl WrapWal for ManagedConnectionWalWrapper { if self.fence.admit_write().is_err() { return Err(rusqlite::ffi::Error::new(rusqlite::ffi::SQLITE_AUTH)); } - self.acquire()?; + self.acquire(self.fence.class())?; match wrapped.begin_write_txn() { Ok(_) => { tracing::debug!("transaction acquired"); @@ -460,7 +590,7 @@ impl WrapWal for ManagedConnectionWalWrapper { backfilled: Option<&mut i32>, ) -> libsql_sys::wal::Result<()> { let before = Instant::now(); - self.acquire()?; + self.acquire(OperationClass::Maintenance)?; self.manager.current.lock().as_mut().unwrap().state = SlotState::Acquired(SlotType::Checkpoint); @@ -473,11 +603,19 @@ impl WrapWal for ManagedConnectionWalWrapper { if mode as i32 >= CheckpointMode::Restart as i32 { tracing::debug!("forcing queue sync"); self.manager.sync_token.fetch_add(1, Ordering::SeqCst); - let queue_len = self.manager.write_queue.len(); - for _ in 0..queue_len { - let (id, unparker) = self.manager.write_queue.steal().success().unwrap(); - tracing::debug!("forcing queue sync for id={id}"); - unparker.unpark(); + // A fence change can empty the queue concurrently (`wake_queue_for_fence`), so an + // entry counted here may already be gone. + let mut queue_len = self.manager.write_queue.len(); + while queue_len > 0 { + match self.manager.write_queue.steal() { + Steal::Success((id, _, unparker)) => { + tracing::debug!("forcing queue sync for id={id}"); + unparker.unpark(); + queue_len -= 1; + } + Steal::Empty => break, + Steal::Retry => (), + } } } @@ -577,6 +715,7 @@ impl WrapWal for ManagedConnectionWalWrapper { mod fence_tests { use std::path::Path; use std::sync::Arc; + use std::time::Duration; use libsql_sys::wal::wrapper::PassthroughWalWrapper; use libsql_sys::wal::Sqlite3WalManager; @@ -585,6 +724,7 @@ mod fence_tests { use tempfile::tempdir; use crate::connection::connection_core::CoreConnection; + use crate::connection::connection_core::VacuumOutcome; use crate::connection::legacy::{LegacyConnection, MakeLegacyConnection}; use crate::connection::program::Program; use crate::connection::Connection as _; @@ -594,7 +734,7 @@ mod fence_tests { }; use crate::namespace::fence::controller::FenceController; use crate::namespace::fence::outcome::{FenceDetail, FenceError, FenceOutcome}; - use crate::namespace::fence::state::FenceState; + use crate::namespace::fence::state::{FenceState, OperationClass}; use crate::namespace::meta_store::{MetaStore, MetaStoreHandle}; use crate::query_result_builder::test::{StepResult, TestBuilder}; use crate::query_result_builder::QueryResultBuilder as _; @@ -611,6 +751,13 @@ mod fence_tests { impl Harness { async fn new() -> Self { + Self::with_txn_timeout(None).await + } + + /// A harness whose connections steal the write slot from a transaction only after + /// `txn_timeout` (the test default is 100 ms). Queue tests hold a writer open far longer + /// than that and must not have it stolen. + async fn with_txn_timeout(txn_timeout: Option) -> Self { let dir = tempdir().unwrap(); let meta_dir = dir.path().join("meta"); let db_dir = dir.path().join("db"); @@ -619,7 +766,13 @@ mod fence_tests { let meta = open_metastore(&meta_dir).await; create_namespace(&meta, "ns").await; let controller = FenceController::unfenced("ns".into()); - let maker = make_connections(&db_dir, controller.clone()).await; + let config = MetaStoreHandle::load(&db_dir).unwrap(); + if txn_timeout.is_some() { + let mut c = (*config.get()).clone(); + c.txn_timeout = txn_timeout; + config.store(c).await.unwrap(); + } + let maker = make_connections(&db_dir, config, controller.clone()).await; let this = Self { _dir: dir, meta, @@ -657,6 +810,7 @@ mod fence_tests { async fn make_connections( path: &Path, + config: MetaStoreHandle, fence: Arc, ) -> MakeLegacyConnection { MakeLegacyConnection::new( @@ -664,7 +818,7 @@ mod fence_tests { PassthroughWalWrapper, Default::default(), Default::default(), - MetaStoreHandle::load(path).unwrap(), + config, Arc::new([]), 100000000, 100000000, @@ -939,4 +1093,227 @@ mod fence_tests { } assert!(conn.fence.take_denial().is_none()); } + + /// Long enough that no test transaction is ever stolen by the manager's own timeout. + const LONG_TXN: Option = Some(Duration::from_secs(600)); + /// Upper bound for a test waiting on something that happens promptly; reaching it is a + /// failure, never the expected path. + const PROMPT: Duration = Duration::from_secs(30); + + /// Wait until `n` connections are parked in the write queue. This polls a condition; it + /// does not use elapsed time as evidence of anything. + async fn until_queued(h: &Harness, n: usize) { + tokio::time::timeout(PROMPT, async { + while h.maker.connection_manager().queued_writers() != n { + tokio::time::sleep(Duration::from_millis(1)).await; + } + }) + .await + .unwrap_or_else(|_| panic!("{n} connections never queued for the write slot")); + } + + fn freelist(conn: &Conn) -> i64 { + conn.with_raw(|c| c.query_row("pragma freelist_count", (), |r| r.get(0))) + .unwrap() + } + + /// A writer parked in the write queue behind an open transaction leaves the queue with the + /// typed denial as soon as the fence changes the write generation: it neither waits for the + /// slot nor, once the holder commits, writes. The holder itself, admitted before the fence, + /// keeps the slot and commits, which is what the positive drain waits for. + #[tokio::test(flavor = "multi_thread")] + async fn fence_rejects_queued_writer() { + let h = Harness::with_txn_timeout(LONG_TXN).await; + let manager = h.maker.connection_manager().clone(); + let holder = h.conn().await; + assert_ok(&run(&holder, &["begin immediate", "insert into t values (1)"]).await); + let (_, class) = manager + .active_writer() + .expect("the holder has the write slot"); + assert_eq!(class, OperationClass::NormalWrite); + + let queued = h.conn().await; + let waiting = tokio::spawn({ + let queued = queued.clone(); + async move { run(&queued, &["insert into t values (2)"]).await } + }); + until_queued(&h, 1).await; + assert!(!waiting.is_finished()); + + h.fence().await; + let steps = tokio::time::timeout(PROMPT, waiting) + .await + .expect("the queued writer was not woken by the fence") + .unwrap(); + assert_eq!( + fence_error(&steps[0]).outcome(), + FenceOutcome::MigrationWriteFenced + ); + assert_eq!(manager.queued_writers(), 0); + assert_eq!( + manager.active_writer().map(|(_, c)| c), + Some(OperationClass::NormalWrite) + ); + + assert_ok(&run(&holder, &["commit"]).await); + assert_eq!(manager.active_writer(), None); + assert_eq!(count(&holder).await, 1); + // The refused connection stays refused while the fence holds. + let steps = run(&queued, &["insert into t values (3)"]).await; + assert_eq!( + fence_error(&steps[0]).outcome(), + FenceOutcome::MigrationWriteFenced + ); + assert_eq!(count(&holder).await, 1); + } + + /// A checkpoint is maintenance: one queued behind an open transaction when the fence + /// changes queues again instead of being refused, and runs once the holder commits. + #[tokio::test(flavor = "multi_thread")] + async fn queued_checkpoint_survives_fence_wake() { + let h = Harness::with_txn_timeout(LONG_TXN).await; + let manager = h.maker.connection_manager().clone(); + let holder = h.conn().await; + assert_ok(&run(&holder, &["begin immediate", "insert into t values (1)"]).await); + + let checkpointer = h.conn().await; + let checkpoint = tokio::task::spawn_blocking({ + let inner = checkpointer.inner.clone(); + move || inner.lock().checkpoint() + }); + until_queued(&h, 1).await; + + h.fence().await; + // Woken by both transitions, it put itself back in the queue. + until_queued(&h, 1).await; + assert!(!checkpoint.is_finished()); + + assert_ok(&run(&holder, &["commit"]).await); + tokio::time::timeout(PROMPT, checkpoint) + .await + .expect("the checkpoint never got the slot") + .unwrap() + .unwrap(); + assert_eq!(manager.active_writer(), None); + assert_eq!(count(&holder).await, 1); + } + + /// `TRUNCATE` checkpoints run in every fence state. + #[tokio::test(flavor = "multi_thread")] + async fn checkpoint_allowed_while_fenced() { + let h = Harness::new().await; + let conn = h.conn().await; + assert_ok( + &run( + &conn, + &["insert into t values (1)", "insert into t values (2)"], + ) + .await, + ); + h.fence().await; + + conn.checkpoint().await.unwrap(); + let (busy, log, checkpointed): (i64, i64, i64) = conn + .with_raw(|c| { + c.query_row("pragma wal_checkpoint(truncate)", (), |r| { + Ok((r.get(0)?, r.get(1)?, r.get(2)?)) + }) + }) + .unwrap(); + assert_eq!((busy, log, checkpointed), (0, 0, 0)); + assert_eq!(count(&conn).await, 2); + } + + /// `VACUUM` is not maintenance. While normal writes are denied it is skipped, and reported as + /// skipped rather than failed; once writes are admitted again it runs. + #[tokio::test(flavor = "multi_thread")] + async fn vacuum_skipped_while_fenced() { + let h = Harness::new().await; + let conn = h.conn().await; + assert_ok( + &run( + &conn, + &[ + "insert into t select randomblob(4096) from \ + (with recursive n(i) as (select 1 union all select i + 1 from n where i < 200) \ + select i from n)", + "delete from t", + ], + ) + .await, + ); + let free = freelist(&conn); + assert!(free > 100, "freelist {free}"); + + h.fence().await; + let outcome = conn.inner.lock().vacuum_if_needed_above(0).unwrap(); + assert_eq!(outcome, VacuumOutcome::Fenced); + assert_eq!(freelist(&conn), free); + // The periodic path reports success, not a failed vacuum. + conn.vacuum_if_needed().await.unwrap(); + assert_eq!(freelist(&conn), free); + + h.release().await; + let outcome = conn.inner.lock().vacuum_if_needed_above(0).unwrap(); + assert_eq!(outcome, VacuumOutcome::Vacuumed); + assert_eq!(freelist(&conn), 0); + } + + /// `abort_active` rolls back the connection holding the write slot, which releases it and + /// notifies drain waiters; a rollback handle whose connection has already closed does + /// nothing instead of panicking, and there is nothing to abort without a writer. + #[tokio::test(flavor = "multi_thread")] + async fn abort_active_tolerates_closed_connection() { + let h = Harness::with_txn_timeout(LONG_TXN).await; + let manager = h.maker.connection_manager().clone(); + assert_eq!(manager.abort_active(), None); + + let holder = h.conn().await; + assert_ok(&run(&holder, &["begin immediate", "insert into t values (1)"]).await); + let (id, _) = manager.active_writer().unwrap(); + let released = manager.released().notified(); + tokio::pin!(released); + released.as_mut().enable(); + assert_eq!(manager.abort_active(), Some(id)); + tokio::time::timeout(PROMPT, released) + .await + .expect("the rollback did not notify drain waiters"); + assert_eq!(manager.active_writer(), None); + assert_eq!(count(&h.conn().await).await, 0); + + // A handle taken while its connection was open, used after the connection closed. + let closing = h.conn().await; + let closing_id = *manager.inner.abort_handle.lock().keys().max().unwrap(); + let handle = manager.inner.abort_handle.lock()[&closing_id].clone(); + drop(closing); + assert!(!manager.inner.abort_handle.lock().contains_key(&closing_id)); + handle.abort(); + assert_eq!(manager.abort_active(), None); + } + + /// Committing releases the slot and wakes a waiter that registered before it looked at the + /// active writer, which is the drain's wait (section 8.3 step 5). + #[tokio::test(flavor = "multi_thread")] + async fn release_notifies_drain_waiters() { + let h = Harness::with_txn_timeout(LONG_TXN).await; + let manager = h.maker.connection_manager().clone(); + let holder = h.conn().await; + assert_ok(&run(&holder, &["begin immediate", "insert into t values (1)"]).await); + h.fence().await; + + let released = manager.released().notified(); + tokio::pin!(released); + released.as_mut().enable(); + assert!(manager.active_writer().is_some()); + let commit = tokio::spawn({ + let holder = holder.clone(); + async move { run(&holder, &["commit"]).await } + }); + tokio::time::timeout(PROMPT, released) + .await + .expect("the commit did not notify drain waiters"); + assert_eq!(manager.active_writer(), None); + assert_ok(&commit.await.unwrap()); + assert_eq!(count(&holder).await, 1); + } } diff --git a/libsql-server/src/connection/legacy.rs b/libsql-server/src/connection/legacy.rs index 93ab86cb49..88e7115b7a 100644 --- a/libsql-server/src/connection/legacy.rs +++ b/libsql-server/src/connection/legacy.rs @@ -77,6 +77,9 @@ where fence: Arc, ) -> Result { let txn_timeout = config_store.get().txn_timeout.unwrap_or(TXN_TIMEOUT); + let connection_manager = ConnectionManager::new(txn_timeout); + // Queued writers re-check the fence whenever its write generation changes. + fence.register_write_queue(connection_manager.fence_waker()); let mut this = Self { db_path, @@ -93,7 +96,7 @@ where encryption_config, block_writes, resolve_attach_path, - connection_manager: ConnectionManager::new(txn_timeout), + connection_manager, make_wal_manager, fence, }; @@ -104,6 +107,13 @@ where Ok(this) } + /// The write-slot manager shared by every connection this maker opens. + // Used by the positive source write drain (section 8.3). + #[cfg_attr(not(test), allow(dead_code))] + pub(crate) fn connection_manager(&self) -> &ConnectionManager { + &self.connection_manager + } + /// Tries to create a database, retrying if the database is busy. async fn try_create_db(&self) -> Result> { // try 100 times to acquire initial db connection. diff --git a/libsql-server/src/namespace/fence/controller.rs b/libsql-server/src/namespace/fence/controller.rs index a9a31dc03a..02b9a3db8c 100644 --- a/libsql-server/src/namespace/fence/controller.rs +++ b/libsql-server/src/namespace/fence/controller.rs @@ -110,11 +110,17 @@ impl GateSnapshot { } } +/// Wakes one connection manager's write queue after a write-generation change; returns `false` +/// once the manager is gone. +pub type WriteQueueWaker = Box bool + Send + Sync>; + /// The fence controller of one namespace. pub struct FenceController { namespace: NamespaceName, transition_lock: Arc>, gate: watch::Sender, + /// The write queues of the namespace's connection managers (section 8.2). + write_queues: Mutex>, #[cfg(test)] hooks: FenceTestHooks, } @@ -136,6 +142,7 @@ impl FenceController { namespace, transition_lock: Default::default(), gate, + write_queues: Mutex::new(Vec::new()), #[cfg(test)] hooks: FenceTestHooks::default(), }) @@ -174,6 +181,13 @@ impl FenceController { self.gate.borrow().permits(class) } + /// Register a connection manager's write queue, to be woken after every change of the write + /// generation so that queued writers re-check the gate instead of waiting for the slot. + /// Wakers whose manager is gone are dropped on the next change. + pub fn register_write_queue(&self, waker: WriteQueueWaker) { + self.write_queues.lock().push(waker); + } + /// Take the namespace's transition lock. Every fence command on the namespace runs while /// holding it, from its first check to its response. pub async fn begin_transition(self: &Arc) -> Transition { @@ -224,6 +238,7 @@ impl FenceController { /// Publish a new gate. `fence: None` keeps the published fence. The write generation moves /// whenever the state, the owning operation or the indeterminate flag changes. fn publish(&self, fence: Option, indeterminate: Option) { + let mut generation_changed = false; self.gate.send_modify(|gate| { let fence = fence.unwrap_or_else(|| gate.fence.clone()); let changed = fence.state() != gate.fence.state() @@ -234,17 +249,24 @@ impl FenceController { gate.indeterminate = indeterminate; if changed { gate.write_generation += 1; + generation_changed = true; } }); - let gate = self.gate.borrow(); - tracing::debug!( - namespace = %self.namespace, - state = %gate.state(), - revision = gate.revision(), - write_generation = gate.write_generation, - indeterminate = gate.indeterminate.is_some(), - "published namespace fence gate" - ); + { + let gate = self.gate.borrow(); + tracing::debug!( + namespace = %self.namespace, + state = %gate.state(), + revision = gate.revision(), + write_generation = gate.write_generation, + indeterminate = gate.indeterminate.is_some(), + "published namespace fence gate" + ); + } + // After the gate is published, so that every woken writer re-checks against it. + if generation_changed { + self.write_queues.lock().retain(|wake| wake()); + } } } @@ -682,6 +704,60 @@ pub(crate) mod tests { assert_eq!(controller.gate(), before); } + /// Registered write queues are woken after every change of the write generation, and only + /// then; a waker whose manager is gone is forgotten. + #[tokio::test] + async fn write_queues_are_woken_on_every_generation_change() { + let tmp = tempdir().unwrap(); + let meta = open_metastore(tmp.path()).await; + create_namespace(&meta, "ns").await; + let controller = FenceController::unfenced("ns".into()); + + let woken = Arc::new(AtomicU64::new(0)); + let seen_generation = Arc::new(AtomicU64::new(0)); + controller.register_write_queue(Box::new({ + let woken = woken.clone(); + let seen_generation = seen_generation.clone(); + let controller = Arc::downgrade(&controller); + move || { + // The gate is already published when the queue is woken. + if let Some(c) = controller.upgrade() { + seen_generation.store(c.write_generation(), Ordering::SeqCst); + } + woken.fetch_add(1, Ordering::SeqCst); + true + } + })); + let gone = Arc::new(AtomicU64::new(0)); + controller.register_write_queue(Box::new({ + let gone = gone.clone(); + move || { + gone.fetch_add(1, Ordering::SeqCst); + false + } + })); + assert_eq!(controller.write_queues.lock().len(), 2); + + fence_source(&meta, &controller, OP).await; + assert_eq!(woken.load(Ordering::SeqCst), 2); + assert_eq!(seen_generation.load(Ordering::SeqCst), 2); + assert_eq!(gone.load(Ordering::SeqCst), 1); + assert_eq!(controller.write_queues.lock().len(), 1); + + // A replay does not move the generation and wakes nobody. + let revision = controller.gate().revision(); + controller + .apply_command(&meta, release("ns", OP, 2, revision), ctx()) + .await + .unwrap(); + assert_eq!(woken.load(Ordering::SeqCst), 3); + controller + .apply_command(&meta, release("ns", OP, 2, revision), ctx()) + .await + .unwrap(); + assert_eq!(woken.load(Ordering::SeqCst), 3); + } + #[tokio::test] async fn failed_before_commit_leaves_gate_unchanged() { let tmp = tempdir().unwrap(); From 7d5a6a8c7780cb90faa2bdd52dd07a886fc256b5 Mon Sep 17 00:00:00 2001 From: River Date: Tue, 29 Sep 2026 16:32:59 +0000 Subject: [PATCH 4/5] libsql-server: positive source write drain AcquireSourceWriteFence now runs the drain of docs/NAMESPACE_FENCE.md section 8.3 end to end, under the namespace's transition lock and on a task of its own: - an in-memory INSTALLING gate closes write admission (and moves the write generation, waking queued writers) before SOURCE_DRAINING is persisted; a command proven not to have committed removes it again; - the drain waits on the connection manager's release notification for the writer that held the slot when admission closed, never on elapsed time or the transaction timeout; at the deadline it answers DRAINING (admission stays closed) or, with force_rollback, rolls the writer back and waits for the actual release; - the frozen boundary (log id, last committed frame) is read under the write-slot lock once no writer holds it, and SOURCE_WRITE_FENCED is committed with it; - replaying a DRAINING command resumes the same drain. Primary connection makers register a write-drain source (their connection manager, held weakly, and their replication log) with the namespace's controller; NamespaceStore::execute_fence_command loads the namespace before an acquisition so that the source exists. The default drain deadline is --namespace-fence-default-write-drain-ms (30 s). FrozenBoundary.frame_no becomes optional, for a log without frames. Co-authored-by: Tomasz Szymczyszyn --- libsql-server/proto/namespace_fence.proto | 3 +- libsql-server/src/config.rs | 3 + .../src/connection/connection_manager.rs | 44 +- libsql-server/src/connection/legacy.rs | 2 - .../src/generated/namespace_fence.rs | 5 +- libsql-server/src/main.rs | 9 + .../src/namespace/configurator/helpers.rs | 47 +- .../src/namespace/fence/controller.rs | 144 +++- libsql-server/src/namespace/fence/drain.rs | 771 ++++++++++++++++++ libsql-server/src/namespace/fence/mod.rs | 6 +- libsql-server/src/namespace/fence/record.rs | 5 +- .../src/namespace/fence/transition.rs | 16 +- libsql-server/src/namespace/meta_store.rs | 23 +- libsql-server/src/namespace/store.rs | 35 +- 14 files changed, 1053 insertions(+), 60 deletions(-) create mode 100644 libsql-server/src/namespace/fence/drain.rs diff --git a/libsql-server/proto/namespace_fence.proto b/libsql-server/proto/namespace_fence.proto index efdee1ab81..73250dfb0d 100644 --- a/libsql-server/proto/namespace_fence.proto +++ b/libsql-server/proto/namespace_fence.proto @@ -75,7 +75,8 @@ message DrainPolicy { message FrozenBoundary { string log_id = 1; - uint64 frame_no = 2; + // Absent when the replication log has no frames. + optional uint64 frame_no = 2; } message LegacyBlocks { diff --git a/libsql-server/src/config.rs b/libsql-server/src/config.rs index 9ac7add98b..4457232d03 100644 --- a/libsql-server/src/config.rs +++ b/libsql-server/src/config.rs @@ -193,6 +193,9 @@ pub struct MetaStoreConfig { /// How long receipts of finished fence operations are kept. `None` is the default of /// 30 days. pub namespace_fence_receipt_retention: Option, + /// How long `AcquireSourceWriteFence` waits for active writers when the request names no + /// drain policy. `None` is the default of 30 seconds. + pub namespace_fence_default_write_drain: Option, } #[derive(Debug, Clone)] diff --git a/libsql-server/src/connection/connection_manager.rs b/libsql-server/src/connection/connection_manager.rs index 009adfeb45..aab548461b 100644 --- a/libsql-server/src/connection/connection_manager.rs +++ b/libsql-server/src/connection/connection_manager.rs @@ -75,17 +75,41 @@ impl ConnectionManager { /// The connection holding the write slot, and the class it holds it for. A slot that has /// been handed to a queued connection that has not taken it yet counts as held. - // Used by the positive source write drain (section 8.3). - #[cfg_attr(not(test), allow(dead_code))] pub(crate) fn active_writer(&self) -> Option<(ConnId, OperationClass)> { self.inner.current.lock().map(|slot| (slot.id, slot.class)) } + /// Whether a connection holds the write slot for a write transaction: any holder but a + /// checkpoint (`Maintenance`), which cannot change logical contents. + pub(crate) fn has_writer(&self) -> bool { + self.active_writer() + .is_some_and(|(_, class)| class != OperationClass::Maintenance) + } + + /// Run `f` under the write-slot lock if no connection holds the slot for a write + /// transaction (a checkpoint may). While `f` runs no write transaction can start or end on + /// this manager, so, once the fence has closed write admission, anything `f` reads about the + /// committed log is final (`docs/NAMESPACE_FENCE.md` section 8.3, step 6). Returns the + /// holder otherwise. + pub(crate) fn with_no_writer( + &self, + f: impl FnOnce() -> R, + ) -> Result { + let current = self.inner.current.lock(); + match *current { + Some(slot) if slot.class != OperationClass::Maintenance => Err((slot.id, slot.class)), + _ => Ok(f()), + } + } + + /// A handle that does not keep the manager alive. + pub(crate) fn downgrade(&self) -> WeakConnectionManager { + WeakConnectionManager(Arc::downgrade(&self.inner)) + } + /// Notified (with `notify_waiters`) every time the write slot is released or handed on. A /// waiter registers interest (`Notified::enable`) before it checks /// [`active_writer`](Self::active_writer), so a release in between is not missed. - // Used by the positive source write drain (section 8.3). - #[cfg_attr(not(test), allow(dead_code))] pub(crate) fn released(&self) -> &tokio::sync::Notify { &self.inner.released } @@ -94,8 +118,6 @@ impl ConnectionManager { /// handle it registered. Returns the connection that was asked to roll back, if any. The /// slot is released by the rollback itself (`end_read_txn`/`end_write_txn`), which notifies /// [`released`](Self::released); a connection closing at the same time is tolerated. - // Used by the positive source write drain (section 8.3). - #[cfg_attr(not(test), allow(dead_code))] pub(crate) fn abort_active(&self) -> Option { let id = self.active_writer()?.0; let handle = self.inner.abort_handle.lock().get(&id).cloned(); @@ -135,6 +157,16 @@ impl ConnectionManager { } } +/// A [`ConnectionManager`] that is not kept alive by this handle. +#[derive(Clone)] +pub(crate) struct WeakConnectionManager(std::sync::Weak); + +impl WeakConnectionManager { + pub(crate) fn upgrade(&self) -> Option { + self.0.upgrade().map(|inner| ConnectionManager { inner }) + } +} + fn wake_queue_for_fence(inner: &ConnectionManagerInner) { // Under the `current` lock, so that a waiter either queued before this (and is stolen and // woken here) or observes the new token when it takes the lock. diff --git a/libsql-server/src/connection/legacy.rs b/libsql-server/src/connection/legacy.rs index 88e7115b7a..71763e199e 100644 --- a/libsql-server/src/connection/legacy.rs +++ b/libsql-server/src/connection/legacy.rs @@ -108,8 +108,6 @@ where } /// The write-slot manager shared by every connection this maker opens. - // Used by the positive source write drain (section 8.3). - #[cfg_attr(not(test), allow(dead_code))] pub(crate) fn connection_manager(&self) -> &ConnectionManager { &self.connection_manager } diff --git a/libsql-server/src/generated/namespace_fence.rs b/libsql-server/src/generated/namespace_fence.rs index dcbbcafe96..7012a0dbee 100644 --- a/libsql-server/src/generated/namespace_fence.rs +++ b/libsql-server/src/generated/namespace_fence.rs @@ -12,8 +12,9 @@ pub struct DrainPolicy { pub struct FrozenBoundary { #[prost(string, tag = "1")] pub log_id: ::prost::alloc::string::String, - #[prost(uint64, tag = "2")] - pub frame_no: u64, + /// Absent when the replication log has no frames. + #[prost(uint64, optional, tag = "2")] + pub frame_no: ::core::option::Option, } #[allow(clippy::derive_partial_eq_without_eq)] #[derive(Clone, PartialEq, ::prost::Message)] diff --git a/libsql-server/src/main.rs b/libsql-server/src/main.rs index 5d738a1ed5..a1ab51c4c4 100644 --- a/libsql-server/src/main.rs +++ b/libsql-server/src/main.rs @@ -268,6 +268,12 @@ struct Cli { #[clap(long, env = "SQLD_NAMESPACE_FENCE_RECEIPT_RETENTION_S")] namespace_fence_receipt_retention_s: Option, + /// How long, in milliseconds, acquiring a namespace write fence waits for active writers + /// when the request names no drain policy (the deadline then answers `DRAINING`). + /// Defaults to 30 seconds. + #[clap(long, env = "SQLD_NAMESPACE_FENCE_DEFAULT_WRITE_DRAIN_MS")] + namespace_fence_default_write_drain_ms: Option, + /// Shutdown timeout duration in seconds, defaults to 30 seconds. #[clap(long, env = "SQLD_SHUTDOWN_TIMEOUT")] shutdown_timeout: Option, @@ -664,6 +670,9 @@ fn make_meta_store_config(config: &Cli) -> anyhow::Result { namespace_fence_receipt_retention: config .namespace_fence_receipt_retention_s .map(Duration::from_secs), + namespace_fence_default_write_drain: config + .namespace_fence_default_write_drain_ms + .map(Duration::from_millis), }) } diff --git a/libsql-server/src/namespace/configurator/helpers.rs b/libsql-server/src/namespace/configurator/helpers.rs index 1f2524cade..a10ef89d6b 100644 --- a/libsql-server/src/namespace/configurator/helpers.rs +++ b/libsql-server/src/namespace/configurator/helpers.rs @@ -24,7 +24,7 @@ use crate::connection::{Connection as _, MakeConnection, MakeThrottledConnection use crate::database::{PrimaryConnection, PrimaryConnectionMaker}; use crate::error::LoadDumpError; use crate::namespace::broadcasters::BroadcasterHandle; -use crate::namespace::fence::controller::FenceController; +use crate::namespace::fence::controller::{FenceController, WriteDrainSource}; use crate::namespace::meta_store::MetaStoreHandle; use crate::namespace::replication_wal::{make_replication_wal_wrapper, ReplicationWalWrapper}; use crate::namespace::{ @@ -160,26 +160,33 @@ pub(super) async fn make_primary_connection_maker( let rcv = logger.new_frame_notifier.subscribe(); move || *rcv.borrow() }); + let legacy_maker = MakeLegacyConnection::new( + db_path.to_path_buf(), + wal_wrapper.clone(), + stats.clone(), + broadcaster, + meta_store_handle.clone(), + base_config.extensions.clone(), + base_config.max_response_size, + base_config.max_total_response_size, + auto_checkpoint, + get_current_frame_no.clone(), + encryption_config, + block_writes, + resolve_attach_path, + make_wal_manager.clone(), + fence.clone(), + ) + .await?; + // The positive write drain waits on this maker's write slot and reads the frozen boundary + // from its replication log (`docs/NAMESPACE_FENCE.md` section 8.3). + fence.register_write_drain(WriteDrainSource::new( + legacy_maker.connection_manager(), + logger.log_id(), + get_current_frame_no, + )); let connection_maker = Arc::new( - MakeLegacyConnection::new( - db_path.to_path_buf(), - wal_wrapper.clone(), - stats.clone(), - broadcaster, - meta_store_handle.clone(), - base_config.extensions.clone(), - base_config.max_response_size, - base_config.max_total_response_size, - auto_checkpoint, - get_current_frame_no, - encryption_config, - block_writes, - resolve_attach_path, - make_wal_manager.clone(), - fence, - ) - .await? - .throttled( + legacy_maker.throttled( base_config.max_concurrent_connections.clone(), base_config .connection_creation_timeout diff --git a/libsql-server/src/namespace/fence/controller.rs b/libsql-server/src/namespace/fence/controller.rs index 02b9a3db8c..cc6a274195 100644 --- a/libsql-server/src/namespace/fence/controller.rs +++ b/libsql-server/src/namespace/fence/controller.rs @@ -17,9 +17,11 @@ use parking_lot::Mutex; use tokio::sync::{watch, OwnedMutexGuard}; use uuid::Uuid; +use crate::connection::connection_manager::{ConnectionManager, WeakConnectionManager}; use crate::error::Error; use crate::namespace::meta_store::{FenceCommit, FenceContext, MetaStore}; use crate::namespace::NamespaceName; +use crate::replication::FrameNo; use super::command::FenceRequest; #[cfg(test)] @@ -46,6 +48,10 @@ pub struct GateSnapshot { /// A command whose commit outcome is unknown. While set, every class except maintenance /// and observability is denied, and every other command is refused. pub indeterminate: Option, + /// The in-memory `INSTALLING` gate of a closing transition that is being persisted + /// (section 8.3, step 2): write admission is closed on top of whatever `fence` allows. + /// Never persisted. + pub installing: Option, } impl GateSnapshot { @@ -54,6 +60,7 @@ impl GateSnapshot { fence, write_generation: 0, indeterminate: None, + installing: None, } } @@ -90,7 +97,30 @@ impl GateSnapshot { .with_detail(FenceDetail::IndeterminateCommit)); } } - self.fence.permits(class) + self.fence.permits(class)?; + if let Some((operation_id, command_id)) = self.installing { + if matches!( + class, + OperationClass::NormalWrite + | OperationClass::Vacuum + | OperationClass::CapabilityImport + | OperationClass::Lifecycle + ) { + return Err(FenceError::new( + FenceOutcome::MigrationWriteFenced, + format!( + "{class:?} is not permitted: fence command {command_id} of operation \ + {operation_id} is closing write admission" + ), + )); + } + } + Ok(()) + } + + /// Whether a closing transition is being installed. + pub fn is_installing(&self) -> bool { + self.installing.is_some() } /// Normal write admission. @@ -114,6 +144,39 @@ impl GateSnapshot { /// once the manager is gone. pub type WriteQueueWaker = Box bool + Send + Sync>; +/// The last frame committed to a namespace's replication log, `None` while it has none. +pub type GetCurrentFrameNo = Arc Option + Send + Sync + 'static>; + +/// What the write drain needs from one primary connection maker of the namespace: its +/// connection manager (held weakly, so an evicted namespace's manager goes away with it) and +/// its replication log. +pub struct WriteDrainSource { + pub(crate) manager: WeakConnectionManager, + pub(crate) log_id: Uuid, + pub(crate) current_frame_no: GetCurrentFrameNo, +} + +impl WriteDrainSource { + pub(crate) fn new( + manager: &ConnectionManager, + log_id: Uuid, + current_frame_no: GetCurrentFrameNo, + ) -> Self { + Self { + manager: manager.downgrade(), + log_id, + current_frame_no, + } + } +} + +/// A [`WriteDrainSource`] whose manager is alive, held for the length of a drain. +pub(crate) struct LiveWriteDrain { + pub(crate) manager: ConnectionManager, + pub(crate) log_id: Uuid, + pub(crate) current_frame_no: GetCurrentFrameNo, +} + /// The fence controller of one namespace. pub struct FenceController { namespace: NamespaceName, @@ -121,6 +184,9 @@ pub struct FenceController { gate: watch::Sender, /// The write queues of the namespace's connection managers (section 8.2). write_queues: Mutex>, + /// What the write drain needs from each of the namespace's primary connection makers + /// (section 8.3). + write_drains: Mutex>, #[cfg(test)] hooks: FenceTestHooks, } @@ -143,6 +209,7 @@ impl FenceController { transition_lock: Default::default(), gate, write_queues: Mutex::new(Vec::new()), + write_drains: Mutex::new(Vec::new()), #[cfg(test)] hooks: FenceTestHooks::default(), }) @@ -188,6 +255,32 @@ impl FenceController { self.write_queues.lock().push(waker); } + /// Register what the write drain needs from a primary connection maker of this namespace: + /// its connection manager and its replication log. Sources whose manager is gone are + /// dropped the next time the drain looks. + pub fn register_write_drain(&self, source: WriteDrainSource) { + self.write_drains.lock().push(source); + } + + /// The write-drain sources whose manager is still alive, oldest first. The drain holds them + /// (and so their managers) for as long as it runs. + pub(crate) fn live_write_drains(&self) -> Vec { + let mut sources = self.write_drains.lock(); + let mut live = Vec::with_capacity(sources.len()); + sources.retain(|source| match source.manager.upgrade() { + Some(manager) => { + live.push(LiveWriteDrain { + manager, + log_id: source.log_id, + current_frame_no: source.current_frame_no.clone(), + }); + true + } + None => false, + }); + live + } + /// Take the namespace's transition lock. Every fence command on the namespace runs while /// holding it, from its first check to its response. pub async fn begin_transition(self: &Arc) -> Transition { @@ -236,17 +329,25 @@ impl FenceController { } /// Publish a new gate. `fence: None` keeps the published fence. The write generation moves - /// whenever the state, the owning operation or the indeterminate flag changes. - fn publish(&self, fence: Option, indeterminate: Option) { + /// whenever the state, the owning operation, the indeterminate flag or the installing gate + /// changes. + fn publish( + &self, + fence: Option, + indeterminate: Option, + installing: Option, + ) { let mut generation_changed = false; self.gate.send_modify(|gate| { let fence = fence.unwrap_or_else(|| gate.fence.clone()); let changed = fence.state() != gate.fence.state() || fence.record().map(|r| r.operation_id) != gate.fence.record().map(|r| r.operation_id) - || indeterminate != gate.indeterminate; + || indeterminate != gate.indeterminate + || installing != gate.installing; gate.fence = fence; gate.indeterminate = indeterminate; + gate.installing = installing; if changed { gate.write_generation += 1; generation_changed = true; @@ -260,6 +361,7 @@ impl FenceController { revision = gate.revision(), write_generation = gate.write_generation, indeterminate = gate.indeterminate.is_some(), + installing = gate.installing.is_some(), "published namespace fence gate" ); } @@ -282,6 +384,28 @@ impl Transition { &self.controller } + /// Publish the in-memory `INSTALLING` gate for the closing command `key` (section 8.3, + /// step 2): write admission closes and the write generation moves, which wakes the write + /// queues. It is replaced by whatever the command's commit publishes, or removed with + /// [`remove_installing`](Self::remove_installing) when the command is proven not to have + /// committed. + pub fn install_closing_gate(&mut self, key: CommandKey) { + let indeterminate = self.controller.gate.borrow().indeterminate; + self.controller.publish(None, indeterminate, Some(key)); + } + + /// Remove the `INSTALLING` gate of a command that was proven not to have committed. The + /// write generation moves again, so nothing admitted before it closed can write. + pub fn remove_installing(&mut self) { + let (indeterminate, installing) = { + let gate = self.controller.gate.borrow(); + (gate.indeterminate, gate.installing) + }; + if installing.is_some() { + self.controller.publish(None, indeterminate, None); + } + } + /// Commit `request` in the metastore and publish the result. pub async fn apply( &mut self, @@ -353,7 +477,7 @@ impl Transition { match result { Ok(commit) => { let _ = controller.hook(HookPoint::BeforeGatePublish).await; - controller.publish(commit.record.clone().map(StoredFence::Record), None); + controller.publish(commit.record.clone().map(StoredFence::Record), None, None); let _ = controller.hook(HookPoint::BeforeResponse).await; Ok(commit) } @@ -365,7 +489,7 @@ impl Transition { "fence commit outcome unknown; the namespace stays closed until the command \ is replayed: {e}" ); - controller.publish(None, Some(key)); + controller.publish(None, Some(key), None); Err(e.into()) } } @@ -627,7 +751,7 @@ pub(crate) mod tests { DrainCompletion::SourceWrites { boundary: FrozenBoundary { log_id: LOG, - frame_no: 0, + frame_no: Some(0), }, }, ctx(), @@ -867,7 +991,7 @@ pub(crate) mod tests { // Simulate an indeterminate outcome for a command that never reached the metastore: // the replay applies it. - controller.publish(None, Some((OP, Uuid::from_u128(1)))); + controller.publish(None, Some((OP, Uuid::from_u128(1))), None); assert!(controller.permits(OperationClass::NormalWrite).is_err()); let r = controller .apply_command(&meta, acquire("ns", OP, 1), ctx()) @@ -1001,8 +1125,8 @@ pub(crate) mod tests { #[test] fn conn_state_starts_at_the_current_generation() { let controller = FenceController::unfenced("ns".into()); - controller.publish(None, Some((OP, OP))); - controller.publish(None, None); + controller.publish(None, Some((OP, OP)), None); + controller.publish(None, None, None); let state = FenceConnState::new(controller.clone(), OperationClass::NormalWrite); assert_eq!(state.program_generation(), 2); assert_eq!(state.txn_generation(), 2); diff --git a/libsql-server/src/namespace/fence/drain.rs b/libsql-server/src/namespace/fence/drain.rs new file mode 100644 index 0000000000..4f77a95d4a --- /dev/null +++ b/libsql-server/src/namespace/fence/drain.rs @@ -0,0 +1,771 @@ +//! The positive source write drain (`docs/NAMESPACE_FENCE.md` sections 8.3 and 8.4). +//! +//! `AcquireSourceWriteFence` closes write admission, persists `SOURCE_DRAINING`, and then waits +//! for the writer that was already holding the write slot when admission closed to commit or +//! roll back. It waits on the connection manager's release notification: neither elapsed time +//! nor the transaction timeout is ever taken as evidence that a writer has finished. Once no +//! connection holds the slot for a write, it reads the frozen boundary (`log_id`, last committed +//! frame) under the slot lock and persists `SOURCE_WRITE_FENCED` with it. Every step runs under +//! the namespace's transition lock, on a task of its own, so a caller that goes away does not +//! interrupt it. + +use std::sync::Arc; +use std::time::Duration; + +use tokio::time::Instant; + +use crate::namespace::meta_store::{FenceCommit, FenceContext, MetaStore}; + +use super::command::{DrainPolicy, FenceCommand, FenceRequest, OnDeadline}; +use super::controller::{FenceController, LiveWriteDrain, Transition}; +use super::hooks::HookPoint; +use super::outcome::FenceOutcome; +use super::record::FrozenBoundary; +use super::state::OperationClass; +use super::transition::DrainCompletion; + +/// How long `AcquireSourceWriteFence` waits for active writers when neither the request nor +/// `--namespace-fence-default-write-drain-ms` names a deadline. +pub const DEFAULT_WRITE_DRAIN: Duration = Duration::from_secs(30); + +/// After a forced rollback, how long the drain waits at least for the rolled-back writer to +/// release the slot before it answers `DRAINING` (the request's own deadline, if longer, is +/// used instead). A rollback normally releases at once; it is delayed only while a program is +/// still running on the connection. Reaching this bound is never taken as proof of anything: +/// the answer is `DRAINING` and admission stays closed. +pub const FORCED_ROLLBACK_GRACE: Duration = Duration::from_secs(10); + +impl FenceController { + /// Run one fence command to completion under the namespace's transition lock, including + /// the drain it starts, and return its result. + /// + /// The command runs on its own task: a caller that goes away (a lost response) interrupts + /// neither the commit nor the drain, and the result can be recovered by replaying the same + /// command or inspecting the fence. + pub async fn execute( + self: &Arc, + meta: &MetaStore, + request: FenceRequest, + ctx: FenceContext, + ) -> crate::Result { + let this = self.clone(); + let meta = meta.clone(); + tokio::spawn(async move { + let mut transition = this.begin_transition().await; + match request.command { + FenceCommand::AcquireSourceWriteFence { .. } => { + acquire_source_write_fence(&mut transition, &meta, request, ctx).await + } + _ => transition.apply(&meta, request, ctx).await, + } + }) + .await? + } +} + +/// `AcquireSourceWriteFence`, steps 1 to 7 of section 8.3, under `transition`. +/// +/// Returns the `APPLIED` commit of `SOURCE_WRITE_FENCED` once the drain is proven, the +/// `DRAINING` commit of `SOURCE_DRAINING` when the deadline passes first (write admission stays +/// closed and a replay of the same command resumes the drain), or the stored result of a replay. +pub async fn acquire_source_write_fence( + transition: &mut Transition, + meta: &MetaStore, + request: FenceRequest, + mut ctx: FenceContext, +) -> crate::Result { + let controller = transition.controller().clone(); + let key = (request.operation_id, request.command_id); + let policy = match &request.command { + FenceCommand::AcquireSourceWriteFence { drain_policy, .. } => { + drain_policy.unwrap_or_else(|| meta.fence_default_write_drain()) + } + _ => return transition.apply(meta, request, ctx).await, + }; + + // Step 1 is the metastore's (replay, owner, identity, shared schema, expectation). The + // identity it checks is the namespace's own replication log id. + if let Some(source) = controller.live_write_drains().last() { + ctx.namespace_log_id = Some(source.log_id); + } + + // Steps 2 and 3: close write admission in memory. Publishing moves the write generation and + // wakes the write queues, whose waiters then fail with MIGRATION_WRITE_FENCED. Where writes + // are already closed (a resumed drain, an indeterminate commit being reconciled, a fenced + // namespace) there is nothing to install. + if controller.permits(OperationClass::NormalWrite).is_ok() { + transition.install_closing_gate(key); + let _ = controller.hook(HookPoint::AfterInstallingGate).await; + } + + // Step 4: persist SOURCE_DRAINING. The commit publishes it in place of the INSTALLING gate. A + // command proven not to have committed removes the INSTALLING gate again; one whose commit + // is unknown has already closed the gate as indeterminate. + let commit = match transition.apply(meta, request, ctx.clone()).await { + Ok(commit) => commit, + Err(e) => { + transition.remove_installing(); + return Err(e); + } + }; + if commit.receipt.outcome != FenceOutcome::Draining { + // A replay of a finished acquisition, or ALREADY_APPLIED. + return Ok(commit); + } + let drain_key = (commit.receipt.operation_id, commit.receipt.command_id); + + // Steps 5 and 6. + let boundary = match drain_writers(&controller, policy).await { + Some(boundary) => boundary, + None => return Ok(commit), + }; + + // Step 7. + ctx.now_ms = now_ms(); + transition + .complete_drain( + meta, + drain_key, + DrainCompletion::SourceWrites { boundary }, + ctx, + ) + .await +} + +/// Wait until no connection manager of the namespace has a writer holding its write slot, and +/// read the frozen boundary. `None` when the drain could not be proven within the policy: the +/// deadline passed (after a forced rollback, the same deadline again, or at least +/// [`FORCED_ROLLBACK_GRACE`]), or the namespace has no +/// loaded primary whose replication log could be read. +async fn drain_writers( + controller: &FenceController, + policy: DrainPolicy, +) -> Option { + let namespace = controller.namespace().clone(); + let deadline_after = Duration::from_millis(policy.deadline_ms); + let mut deadline = Instant::now() + deadline_after; + let mut forced = false; + loop { + // Every manager registered from now on belongs to a maker opened after write admission + // closed, so it has no pre-cutoff writer; sources are re-read on each round anyway. + let sources = controller.live_write_drains(); + if sources.is_empty() { + tracing::warn!( + %namespace, + "the namespace has no loaded primary; the write drain cannot read its replication \ + log and stays DRAINING until the command is replayed" + ); + return None; + } + + if !wait_for_writers(&sources, deadline).await { + match policy.on_deadline { + OnDeadline::ForceRollback if !forced => { + forced = true; + for source in &sources { + let manager = source.manager.clone(); + // The rollback takes the connection's lock, which a running program + // holds; it releases the slot when it happens, and that release is what + // the drain keeps waiting for. + tokio::task::spawn_blocking(move || { + if let Some(id) = manager.abort_active() { + tracing::info!( + connection = id, + "write drain deadline passed; rolling back the active writer" + ); + } + }); + } + deadline = Instant::now() + deadline_after.max(FORCED_ROLLBACK_GRACE); + continue; + } + _ => { + tracing::info!( + %namespace, + deadline_ms = policy.deadline_ms, + on_deadline = policy.on_deadline.as_str(), + forced, + "write drain deadline passed with a writer still active; answering DRAINING" + ); + return None; + } + } + } + + let _ = controller.hook(HookPoint::BeforeBoundaryCapture).await; + match capture_boundary(&sources) { + Ok(boundary) => { + tracing::info!( + %namespace, + log_id = %boundary.log_id, + frame_no = ?boundary.frame_no, + "write drain proven" + ); + return Some(boundary); + } + // Cannot happen while write admission is closed; wait again rather than guess. + Err((id, class)) => { + tracing::warn!( + %namespace, + connection = id, + ?class, + "a writer holds the write slot at boundary capture; waiting again" + ); + } + } + } +} + +/// Wait, on each manager's release notification, until none of them has a connection holding +/// the write slot for a write. `false` when `deadline` passes first. +/// +/// With write admission closed a manager that has been seen without a writer stays without one +/// (only checkpoints can take the slot), so the managers are waited for one after the other. +async fn wait_for_writers(sources: &[LiveWriteDrain], deadline: Instant) -> bool { + for source in sources { + loop { + let released = source.manager.released().notified(); + tokio::pin!(released); + // Registered before the check, so a release in between is not missed. + released.as_mut().enable(); + if !source.manager.has_writer() { + break; + } + tokio::select! { + _ = &mut released => {} + _ = tokio::time::sleep_until(deadline) => return false, + } + } + } + true +} + +/// Step 6: under each manager's write-slot lock, observe that no connection holds the slot for a +/// write, and read the last committed frame of the replication log. The replication logger +/// commits a transaction's frames and publishes its frame number before the transaction +/// releases the slot, so what is read here is final. +fn capture_boundary( + sources: &[LiveWriteDrain], +) -> Result< + FrozenBoundary, + ( + crate::connection::connection_manager::ConnId, + OperationClass, + ), +> { + let mut frame_no = None; + for source in sources { + let frame = source + .manager + .with_no_writer(|| (source.current_frame_no)())?; + frame_no = frame_no.max(frame); + } + let log_id = sources + .last() + .expect("capture_boundary is called with at least one source") + .log_id; + Ok(FrozenBoundary { log_id, frame_no }) +} + +fn now_ms() -> i64 { + std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .map_or(0, |d| i64::try_from(d.as_millis()).unwrap_or(i64::MAX)) +} + +#[cfg(test)] +mod tests { + use std::sync::Arc; + use std::time::Duration; + + use rusqlite::ErrorCode; + use tempfile::{tempdir, TempDir}; + use uuid::Uuid; + + use super::*; + use crate::connection::config::DatabaseConfig; + use crate::connection::Connection as _; + use crate::database::{Connection, Database}; + use crate::error::Error; + use crate::namespace::fence::hooks::HookPoint; + use crate::namespace::fence::record::ServerIdentity; + use crate::namespace::fence::state::FenceState; + use crate::namespace::meta_store::FenceCommitKind; + use crate::namespace::store::fence_tests::open_store; + use crate::namespace::store::NamespaceStore; + use crate::namespace::RestoreOption; + use crate::replication::primary::logger::ReplicationLogger; + + const OP: Uuid = Uuid::from_u128(0xa); + const OTHER_OP: Uuid = Uuid::from_u128(0xb); + /// Long enough that no test ever reaches it: a drain must never finish because of time. + const LONG: DrainPolicy = DrainPolicy { + deadline_ms: 600_000, + on_deadline: OnDeadline::Fail, + }; + const PROMPT: Duration = Duration::from_secs(30); + + /// A primary namespace `ns` with a table `t`, served by a real `NamespaceStore`, so that the + /// drain goes through the connection manager and replication logger the configurator + /// registered. + struct Source { + _dir: TempDir, + store: NamespaceStore, + fence: Arc, + logger: Arc, + } + + impl Source { + async fn new() -> Self { + let dir = tempdir().unwrap(); + let store = open_store(dir.path()).await; + store + .create( + "ns".into(), + RestoreOption::Latest, + DatabaseConfig { + // Held writers must never have the slot stolen by the timeout. + txn_timeout: Some(Duration::from_secs(600)), + ..Default::default() + }, + ) + .await + .unwrap(); + let (fence, logger) = store + .with("ns".into(), |ns| { + let logger = match &ns.db { + Database::Primary(p) => p.wal_wrapper.wrapper().logger(), + _ => unreachable!(), + }; + (ns.fence().clone(), logger) + }) + .await + .unwrap(); + let this = Self { + _dir: dir, + store, + fence, + logger, + }; + let conn = this.conn().await; + raw(&conn, "create table t (x)").await.unwrap(); + this + } + + async fn conn(&self) -> Arc { + let maker = self + .store + .with("ns".into(), |ns| ns.db.connection_maker()) + .await + .unwrap(); + Arc::new(maker.create().await.unwrap()) + } + + fn log_id(&self) -> Uuid { + self.logger.log_id() + } + + /// The last frame the replication log has committed. + fn frame_no(&self) -> Option { + *self.logger.new_frame_notifier.borrow() + } + + fn acquire(&self, op: Uuid, command_id: u128, policy: DrainPolicy) -> FenceRequest { + FenceRequest { + namespace: "ns".into(), + operation_id: op, + command_id: Uuid::from_u128(command_id), + expected_state: FenceState::Unfenced, + expected_revision: 0, + command: FenceCommand::AcquireSourceWriteFence { + expected_log_id: self.log_id(), + drain_policy: Some(policy), + }, + } + } + + /// Run `request` through the store, on a task of its own. + fn execute( + &self, + request: FenceRequest, + ) -> tokio::task::JoinHandle> { + let store = self.store.clone(); + tokio::spawn(async move { + store + .execute_fence_command( + request, + ServerIdentity { + build: "test".into(), + instance_id: Uuid::from_u128(0x99), + }, + ) + .await + }) + } + + /// Wait until the published gate is in `state`. + async fn until_state(&self, state: FenceState) { + let mut rx = self.fence.subscribe(); + tokio::time::timeout(PROMPT, rx.wait_for(|g| g.state() == state)) + .await + .expect("the gate never reached the state") + .unwrap(); + } + + async fn count(&self) -> i64 { + let conn = self.conn().await; + tokio::task::spawn_blocking(move || { + conn.with_raw(|c| c.query_row("select count(*) from t", (), |r| r.get(0))) + }) + .await + .unwrap() + .unwrap() + } + } + + /// Run `sql` as one raw program on `conn`, off the async runtime (it can block on the write + /// slot). + async fn raw(conn: &Arc, sql: &'static str) -> rusqlite::Result<()> { + let conn = conn.clone(); + tokio::task::spawn_blocking(move || conn.with_raw(|c| c.execute_batch(sql))) + .await + .unwrap() + } + + fn assert_fenced(result: rusqlite::Result<()>) { + match result { + Err(rusqlite::Error::SqliteFailure(e, _)) => { + assert_eq!(e.code, ErrorCode::AuthorizationForStatementDenied, "{e}") + } + other => panic!("expected the WAL gate to refuse the write, got {other:?}"), + } + } + + fn fence_outcome(result: &crate::Result) -> FenceOutcome { + match result { + Ok(c) => c.receipt.outcome, + Err(Error::NamespaceFence(e)) => e.outcome(), + Err(e) => panic!("unexpected error: {e}"), + } + } + + fn boundary(commit: &FenceCommit) -> FrozenBoundary { + commit + .record + .as_ref() + .unwrap() + .frozen_boundary + .expect("a write-fenced source has a frozen boundary") + } + + /// Two operations acquire the same namespace at once: exactly one owns it, and the other + /// gets the typed ownership conflict. The first is parked after closing write admission so + /// that the second is certainly waiting on the transition lock. + #[tokio::test(flavor = "multi_thread")] + async fn acquire_race_single_owner() { + let s = Source::new().await; + let paused = s.fence.hooks().pause_at(HookPoint::AfterInstallingGate); + let first = s.execute(s.acquire(OP, 1, LONG)); + paused.reached().await; + let second = s.execute(s.acquire(OTHER_OP, 2, LONG)); + paused.resume(); + + let first = first.await.unwrap(); + let second = second.await.unwrap(); + let outcomes = [fence_outcome(&first), fence_outcome(&second)]; + assert_eq!( + outcomes, + [ + FenceOutcome::Applied, + FenceOutcome::FenceOwnedByAnotherOperation + ] + ); + let gate = s.fence.gate(); + assert_eq!(gate.state(), FenceState::SourceWriteFenced); + assert_eq!(gate.operation_id(), Some(OP)); + assert!(!gate.is_installing()); + } + + /// A writer that holds the write slot when the fence arrives commits, and only then is the + /// freeze acknowledged, with a boundary that includes its commit. Writes attempted while + /// the drain waits are refused. + #[tokio::test(flavor = "multi_thread")] + async fn active_writer_commits_before_ack() { + let s = Source::new().await; + let holder = s.conn().await; + raw(&holder, "begin immediate; insert into t values (1);") + .await + .unwrap(); + + let acquire = s.execute(s.acquire(OP, 1, LONG)); + s.until_state(FenceState::SourceDraining).await; + // Admission is closed while the pre-cutoff writer is still active. + assert_fenced(raw(&s.conn().await, "insert into t values (2)").await); + assert!(!acquire.is_finished()); + + raw(&holder, "commit").await.unwrap(); + let committed_frame = s.frame_no(); + assert!(committed_frame.is_some()); + + let commit = acquire.await.unwrap().unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + assert_eq!(commit.receipt.state_after, FenceState::SourceWriteFenced); + assert_eq!( + boundary(&commit), + FrozenBoundary { + log_id: s.log_id(), + frame_no: committed_frame, + } + ); + assert_eq!(s.count().await, 1); + } + + /// Under `force_rollback`, a writer still active at the deadline is rolled back, and the + /// freeze is acknowledged only after its slot was actually released; its write is not in + /// the database or the boundary. + #[tokio::test(flavor = "multi_thread")] + async fn forced_rollback_before_ack() { + let s = Source::new().await; + let before = s.frame_no(); + let holder = s.conn().await; + raw(&holder, "begin immediate; insert into t values (1);") + .await + .unwrap(); + + let commit = s + .execute(s.acquire( + OP, + 1, + DrainPolicy { + deadline_ms: 0, + on_deadline: OnDeadline::ForceRollback, + }, + )) + .await + .unwrap() + .unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + assert_eq!( + boundary(&commit), + FrozenBoundary { + log_id: s.log_id(), + frame_no: before, + } + ); + assert_eq!(s.frame_no(), before); + assert_eq!(s.count().await, 0); + // The rolled-back transaction is gone; its connection cannot write either. + assert!(raw(&holder, "commit").await.is_err()); + assert_fenced(raw(&holder, "insert into t values (3)").await); + assert_eq!(s.frame_no(), before); + } + + /// After the acknowledgement nothing commits: the boundary is the last committed frame, and + /// autocommit writes, explicit transactions, DDL and a read transaction opened before the + /// fence that tries to upgrade are all refused without adding a frame. + #[tokio::test(flavor = "multi_thread")] + async fn no_commit_after_ack() { + let s = Source::new().await; + let writer = s.conn().await; + for _ in 0..3 { + raw(&writer, "insert into t values (1)").await.unwrap(); + } + let last = s.frame_no(); + // A reader whose transaction predates the fence. + let reader = s.conn().await; + raw(&reader, "begin; select count(*) from t;") + .await + .unwrap(); + + let commit = s.execute(s.acquire(OP, 1, LONG)).await.unwrap().unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + assert_eq!( + boundary(&commit), + FrozenBoundary { + log_id: s.log_id(), + frame_no: last, + } + ); + + assert_fenced(raw(&writer, "insert into t values (2)").await); + assert_fenced(raw(&writer, "begin immediate").await); + assert_fenced(raw(&writer, "create table u (y)").await); + assert_fenced(raw(&reader, "insert into t values (2)").await); + assert_fenced(raw(&s.conn().await, "insert into t values (2)").await); + assert_eq!(s.frame_no(), last); + assert_eq!(s.count().await, 3); + assert_eq!(boundary(&commit).frame_no, s.frame_no()); + } + + /// With `on_deadline: fail`, a writer still active at the deadline makes the command answer + /// `DRAINING`: the durable state is `SOURCE_DRAINING` and write admission stays closed. + #[tokio::test(flavor = "multi_thread")] + async fn deadline_returns_draining_and_stays_closed() { + let s = Source::new().await; + let holder = s.conn().await; + raw(&holder, "begin immediate; insert into t values (1);") + .await + .unwrap(); + + let policy = DrainPolicy { + deadline_ms: 0, + on_deadline: OnDeadline::Fail, + }; + let commit = s.execute(s.acquire(OP, 1, policy)).await.unwrap().unwrap(); + assert_eq!(commit.kind, FenceCommitKind::Committed); + assert_eq!(commit.receipt.outcome, FenceOutcome::Draining); + assert_eq!(s.fence.gate().state(), FenceState::SourceDraining); + let inspected = s + .store + .meta_store() + .inspect_fence("ns".into()) + .await + .unwrap(); + assert_eq!(inspected.fence.state(), FenceState::SourceDraining); + assert_fenced(raw(&s.conn().await, "insert into t values (2)").await); + // Nothing reopens by itself; the writer that was active before the fence is still the + // only one that can commit. + raw(&holder, "commit").await.unwrap(); + assert_fenced(raw(&holder, "insert into t values (3)").await); + assert_eq!(s.fence.gate().state(), FenceState::SourceDraining); + assert_eq!(s.count().await, 1); + } + + /// Replaying a command whose receipt is `DRAINING` resumes the same drain: it completes once + /// the writer has finished, and a further replay returns that stored result. + #[tokio::test(flavor = "multi_thread")] + async fn replay_of_draining_resumes_and_completes() { + let s = Source::new().await; + let holder = s.conn().await; + raw(&holder, "begin immediate; insert into t values (1);") + .await + .unwrap(); + let policy = DrainPolicy { + deadline_ms: 0, + on_deadline: OnDeadline::Fail, + }; + let first = s.execute(s.acquire(OP, 1, policy)).await.unwrap().unwrap(); + assert_eq!(first.receipt.outcome, FenceOutcome::Draining); + let draining_revision = s.fence.gate().revision(); + + // Replayed while the writer is still active: the same drain resumes, writes nothing and, + // with the same zero deadline, answers DRAINING again. + let replay = s.execute(s.acquire(OP, 1, policy)).await.unwrap().unwrap(); + assert_eq!(replay.kind, FenceCommitKind::Resumed); + assert_eq!(replay.receipt.outcome, FenceOutcome::Draining); + assert_eq!(s.fence.gate().revision(), draining_revision); + + raw(&holder, "commit").await.unwrap(); + let committed = s.frame_no(); + let done = s.execute(s.acquire(OP, 1, policy)).await.unwrap().unwrap(); + assert_eq!(done.receipt.outcome, FenceOutcome::Applied); + assert_eq!(done.receipt.command_id, Uuid::from_u128(1)); + assert_eq!( + boundary(&done), + FrozenBoundary { + log_id: s.log_id(), + frame_no: committed, + } + ); + assert_eq!(s.fence.gate().state(), FenceState::SourceWriteFenced); + assert_eq!(s.fence.gate().revision(), draining_revision + 1); + + let again = s.execute(s.acquire(OP, 1, policy)).await.unwrap().unwrap(); + assert_eq!(again.kind, FenceCommitKind::Replayed); + assert_eq!(again.receipt, done.receipt); + assert_eq!(boundary(&again), boundary(&done)); + } + + /// Releasing the write fence commits, publishes a new write generation and only then + /// answers: new programs write again, and a transaction that began under the fence cannot. + #[tokio::test(flavor = "multi_thread")] + async fn release_reopens_with_new_generation() { + let s = Source::new().await; + s.execute(s.acquire(OP, 1, LONG)).await.unwrap().unwrap(); + let fenced = s.fence.gate(); + let reader = s.conn().await; + raw(&reader, "begin; select count(*) from t;") + .await + .unwrap(); + + let release = FenceRequest { + namespace: "ns".into(), + operation_id: OP, + command_id: Uuid::from_u128(2), + expected_state: FenceState::SourceWriteFenced, + expected_revision: fenced.revision(), + command: FenceCommand::ReleaseSourceWriteFence, + }; + let paused = s.fence.hooks().pause_at(HookPoint::BeforeResponse); + let released = s.execute(release); + paused.reached().await; + // Published before the response is sent. + let gate = s.fence.gate(); + assert_eq!(gate.state(), FenceState::Released); + assert!(gate.write_generation > fenced.write_generation); + assert!(!released.is_finished()); + paused.resume(); + let commit = released.await.unwrap().unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + + raw(&s.conn().await, "insert into t values (1)") + .await + .unwrap(); + assert_fenced(raw(&reader, "insert into t values (2)").await); + raw(&reader, "rollback").await.unwrap(); + raw(&reader, "insert into t values (3)").await.unwrap(); + assert_eq!(s.count().await, 2); + } + + /// An acquisition refused by the metastore (here: the caller observed a different + /// replication log) removes the INSTALLING gate it had published; writes are admitted again + /// under a new generation. + #[tokio::test(flavor = "multi_thread")] + async fn refused_acquire_reopens_writes() { + let s = Source::new().await; + let before = s.fence.gate().write_generation; + let mut request = s.acquire(OP, 1, LONG); + request.command = FenceCommand::AcquireSourceWriteFence { + expected_log_id: Uuid::from_u128(0x77), + drain_policy: Some(LONG), + }; + let result = s.execute(request).await.unwrap(); + assert_eq!( + fence_outcome(&result), + FenceOutcome::FencePreconditionFailed + ); + let gate = s.fence.gate(); + assert_eq!(gate.state(), FenceState::Unfenced); + assert!(!gate.is_installing()); + assert_eq!(gate.write_generation, before + 2); + raw(&s.conn().await, "insert into t values (1)") + .await + .unwrap(); + } + + /// While the INSTALLING gate is up, before anything is persisted, writes are already + /// refused and reads are served. + #[tokio::test(flavor = "multi_thread")] + async fn installing_gate_closes_writes_before_persisting() { + let s = Source::new().await; + let paused = s.fence.hooks().pause_at(HookPoint::AfterInstallingGate); + let acquire = s.execute(s.acquire(OP, 1, LONG)); + paused.reached().await; + let gate = s.fence.gate(); + assert!(gate.is_installing()); + assert_eq!(gate.state(), FenceState::Unfenced); + let inspected = s + .store + .meta_store() + .inspect_fence("ns".into()) + .await + .unwrap(); + assert_eq!(inspected.fence.state(), FenceState::Unfenced); + assert_fenced(raw(&s.conn().await, "insert into t values (1)").await); + assert_eq!(s.count().await, 0); + paused.resume(); + let commit = acquire.await.unwrap().unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + assert!(!s.fence.gate().is_installing()); + } +} diff --git a/libsql-server/src/namespace/fence/mod.rs b/libsql-server/src/namespace/fence/mod.rs index d5ac933e2e..6b445a9ba6 100644 --- a/libsql-server/src/namespace/fence/mod.rs +++ b/libsql-server/src/namespace/fence/mod.rs @@ -8,8 +8,9 @@ //! markers with their strict durable encoding ([`record`]), the pure transition function //! ([`transition`]), the metastore tables, compare-and-swap and marker file that persist them //! ([`store`], driven by `MetaStore::apply_fence_command`), and the in-memory authority built -//! on them: the per-namespace [`controller`] with its gate, the [`registry`] that holds the -//! controllers outside the namespace cache, and the test [`hooks`] on their paths. +//! on them: the per-namespace [`controller`] with its gate, the positive write [`drain`], the +//! [`registry`] that holds the controllers outside the namespace cache, and the test [`hooks`] +//! on their paths. // The persistence, controller and protocol layers that consume these types land in the // following commits of this series; until then most of the module is unused by the rest of @@ -18,6 +19,7 @@ pub mod command; pub mod controller; +pub mod drain; pub mod hooks; pub mod outcome; pub mod record; diff --git a/libsql-server/src/namespace/fence/record.rs b/libsql-server/src/namespace/fence/record.rs index a2d9f7f5dd..247b481ad2 100644 --- a/libsql-server/src/namespace/fence/record.rs +++ b/libsql-server/src/namespace/fence/record.rs @@ -32,7 +32,8 @@ pub struct NamespaceIdentity { #[derive(Debug, Clone, Copy, PartialEq, Eq)] pub struct FrozenBoundary { pub log_id: Uuid, - pub frame_no: u64, + /// The last frame committed to the replication log, or `None` when the log has no frames. + pub frame_no: Option, } /// The pre-fence values of the legacy `block_*` configuration fields, restored when the @@ -700,7 +701,7 @@ pub(super) mod tests { drain_started_at_ms: Some(100), frozen_boundary: Some(FrozenBoundary { log_id: Uuid::from_u128(10), - frame_no: 1234, + frame_no: Some(1234), }), validation: None, legacy_blocks: LegacyBlocks { diff --git a/libsql-server/src/namespace/fence/transition.rs b/libsql-server/src/namespace/fence/transition.rs index f1de43f747..d589ce8859 100644 --- a/libsql-server/src/namespace/fence/transition.rs +++ b/libsql-server/src/namespace/fence/transition.rs @@ -903,7 +903,7 @@ mod tests { let boundary = DrainCompletion::SourceWrites { boundary: FrozenBoundary { log_id: LOG, - frame_no: 42, + frame_no: Some(42), }, }; let mut h = if state.role() == Some(Role::Target) || state == S::Absent { @@ -1083,14 +1083,14 @@ mod tests { h.complete(DrainCompletion::SourceWrites { boundary: FrozenBoundary { log_id: LOG, - frame_no: 7, + frame_no: Some(7), }, }); assert_eq!(h.state(), S::SourceWriteFenced); assert_eq!(h.revision(), 2); assert_eq!( h.record.as_ref().unwrap().frozen_boundary.unwrap().frame_no, - 7 + Some(7) ); let acquire_receipt = h .receipts @@ -1255,7 +1255,7 @@ mod tests { h.complete(DrainCompletion::SourceWrites { boundary: FrozenBoundary { log_id: LOG, - frame_no: 1, + frame_no: Some(1), }, }); h.run(OP, CommandKind::SetSourceReadFence).unwrap(); @@ -1464,7 +1464,7 @@ mod tests { DrainCompletion::SourceWrites { boundary: FrozenBoundary { log_id: LOG, - frame_no: 3, + frame_no: Some(3), }, }, &env(), @@ -1492,7 +1492,7 @@ mod tests { DrainCompletion::SourceWrites { boundary: FrozenBoundary { log_id: Uuid::from_u128(0x77), - frame_no: 1, + frame_no: Some(1), }, }, &env(), @@ -1506,7 +1506,7 @@ mod tests { let boundary = DrainCompletion::SourceWrites { boundary: FrozenBoundary { log_id: LOG, - frame_no: 1, + frame_no: Some(1), }, }; assert!(complete_drain(&record, &final_receipt, boundary, &env()).is_err()); @@ -1732,7 +1732,7 @@ mod tests { h.complete(DrainCompletion::SourceWrites { boundary: FrozenBoundary { log_id: LOG, - frame_no: 5, + frame_no: Some(5), }, }); assert_eq!(h.state(), S::SourceWriteFenced); diff --git a/libsql-server/src/namespace/meta_store.rs b/libsql-server/src/namespace/meta_store.rs index 6ed7b29fd5..1a16994c3f 100644 --- a/libsql-server/src/namespace/meta_store.rs +++ b/libsql-server/src/namespace/meta_store.rs @@ -31,7 +31,7 @@ use crate::{ config::MetaStoreConfig, connection::legacy::open_conn_active_checkpoint, error::Error, Result, }; -use super::fence::command::{FenceCommand, FenceRequest}; +use super::fence::command::{DrainPolicy, FenceCommand, FenceRequest, OnDeadline}; use super::fence::outcome::{FenceDetail, FenceError, FenceOutcome}; use super::fence::record::{ CommandReceipt, NamespaceFenceRecord, ServerIdentity, ValidationSnapshot, @@ -110,6 +110,8 @@ struct FenceSettings { /// namespace directory holds a marker. fail_closed: bool, receipt_retention: Duration, + /// The write drain deadline of an `AcquireSourceWriteFence` that names no drain policy. + default_write_drain: Duration, } fn setup_connection(conn: &rusqlite::Connection) -> Result<()> { @@ -235,6 +237,9 @@ impl MetaStoreInner { receipt_retention: config .namespace_fence_receipt_retention .unwrap_or(fence_store::DEFAULT_RECEIPT_RETENTION), + default_write_drain: config + .namespace_fence_default_write_drain + .unwrap_or(crate::namespace::fence::drain::DEFAULT_WRITE_DRAIN), }; let mut this = MetaStoreInner { @@ -1300,6 +1305,16 @@ impl MetaStore { self.inner.fence.enabled } + /// The drain policy of an `AcquireSourceWriteFence` that names none: the configured + /// deadline, then `DRAINING`. + pub fn fence_default_write_drain(&self) -> DrainPolicy { + DrainPolicy { + deadline_ms: u64::try_from(self.inner.fence.default_write_drain.as_millis()) + .unwrap_or(u64::MAX), + on_deadline: OnDeadline::Fail, + } + } + /// Whether this metastore holds fence state, so fences are loaded and enforced. pub fn fence_enforced(&self) -> bool { self.inner.fence.tables @@ -1749,7 +1764,7 @@ mod fence_tests { let boundary = FrozenBoundary { log_id: LOG, - frame_no: 42, + frame_no: Some(42), }; let commit = store .complete_fence_drain( @@ -2124,7 +2139,7 @@ mod fence_tests { DrainCompletion::SourceWrites { boundary: FrozenBoundary { log_id: LOG, - frame_no: 1, + frame_no: Some(1), }, }, ctx(2_000), @@ -2175,7 +2190,7 @@ mod fence_tests { DrainCompletion::SourceWrites { boundary: FrozenBoundary { log_id: LOG, - frame_no: 7, + frame_no: Some(7), }, }, ctx(2_000), diff --git a/libsql-server/src/namespace/store.rs b/libsql-server/src/namespace/store.rs index 3a132cf10f..af406dfd96 100644 --- a/libsql-server/src/namespace/store.rs +++ b/libsql-server/src/namespace/store.rs @@ -21,8 +21,10 @@ use crate::stats::Stats; use super::broadcasters::{BroadcasterHandle, BroadcasterRegistry}; use super::configurator::{DynConfigurator, NamespaceConfigurators}; +use super::fence::command::{FenceCommand, FenceRequest}; +use super::fence::record::ServerIdentity; use super::fence::registry::FenceRegistry; -use super::meta_store::{MetaStore, MetaStoreHandle}; +use super::meta_store::{FenceCommit, FenceContext, MetaStore, MetaStoreHandle}; use super::schema_lock::SchemaLocksRegistry; use super::{Namespace, ResetCb, ResetOp, ResolveNamespacePathFn, RestoreOption}; @@ -531,6 +533,33 @@ impl NamespaceStore { &self.inner.metadata } + /// Run one fence command on its namespace, including the drain it starts + /// (`docs/NAMESPACE_FENCE.md` sections 5.3 and 8). `AcquireSourceWriteFence` loads the + /// namespace first, so that its connection manager and replication log are registered with + /// the namespace's controller before the drain needs them. + // The admin routes that call this are not part of the server yet. + #[cfg_attr(not(test), allow(dead_code))] + pub(crate) async fn execute_fence_command( + &self, + request: FenceRequest, + server: ServerIdentity, + ) -> crate::Result { + let controller = match request.command { + FenceCommand::AcquireSourceWriteFence { .. } => { + self.with(request.namespace.clone(), |ns| ns.fence().clone()) + .await? + } + _ => self.inner.fences.controller(&request.namespace), + }; + controller + .execute( + &self.inner.metadata, + request, + FenceContext::now(server, None), + ) + .await + } + pub(crate) fn schema_locks(&self) -> &SchemaLocksRegistry { &self.inner.schema_locks } @@ -559,7 +588,7 @@ impl NamespaceStore { } #[cfg(test)] -mod fence_tests { +pub(crate) mod fence_tests { use std::path::Path; use libsql_sys::wal::Sqlite3WalManager; @@ -580,7 +609,7 @@ mod fence_tests { const LOG: Uuid = Uuid::from_u128(0x10); const OP: Uuid = Uuid::from_u128(0xa); - async fn open_store(dir: &Path) -> NamespaceStore { + pub(crate) async fn open_store(dir: &Path) -> NamespaceStore { let (maker, manager) = metastore_connection_maker(None, dir).await.unwrap(); let meta = MetaStore::new( MetaStoreConfig { From eb2159283d64bbb0b8742db607c5fd841d16240a Mon Sep 17 00:00:00 2001 From: River Date: Tue, 29 Sep 2026 16:53:59 +0000 Subject: [PATCH 5/5] libsql-server: reconcile indeterminate fence commits and restart at each boundary Add crash-restart tests for the namespace fence: each server lifetime runs on its own runtime and is ended without any shutdown code while a fence command is parked at a hook point, so the next start takes the real dirty-recovery path on the same directory. They cover every persistence boundary of AcquireSourceWriteFence and ReleaseSourceWriteFence (including a marker that lags the metastore commit), a restart in SOURCE_DRAINING with a writer active at the crash, indeterminate commits through the drain path (applied and not applied), and lost acquisition responses resolved by replay and inspection. The tests exposed that a source restarted while draining could never finish its drain: dirty recovery rebuilds the replication log under a new log id, and completing the drain refused a boundary on a log other than the one the fence was acquired against, leaving the namespace in SOURCE_DRAINING for good. The frozen boundary now names the log that is live when the drain is proven, the record's identity keeps the acquisition log id, and the server warns when the two differ. Write admission was durably closed throughout, so the data at the boundary is unchanged. The BeforeMetastoreCommit test hook can now report a commit as indeterminate without running it. The contract document describes the restart and log rebuild semantics. Co-authored-by: Tomasz Szymczyszyn --- .../src/namespace/fence/controller.rs | 14 +- libsql-server/src/namespace/fence/drain.rs | 10 + libsql-server/src/namespace/fence/hooks.rs | 4 +- libsql-server/src/namespace/fence/mod.rs | 3 + libsql-server/src/namespace/fence/tests.rs | 898 ++++++++++++++++++ .../src/namespace/fence/transition.rs | 34 +- 6 files changed, 942 insertions(+), 21 deletions(-) create mode 100644 libsql-server/src/namespace/fence/tests.rs diff --git a/libsql-server/src/namespace/fence/controller.rs b/libsql-server/src/namespace/fence/controller.rs index cc6a274195..7e718e5516 100644 --- a/libsql-server/src/namespace/fence/controller.rs +++ b/libsql-server/src/namespace/fence/controller.rs @@ -458,11 +458,17 @@ impl Transition { } } - if let HookOutcome::Fail(e) = controller.hook(HookPoint::BeforeMetastoreCommit).await { - return Err(e.into()); - } + let result = match controller.hook(HookPoint::BeforeMetastoreCommit).await { + HookOutcome::Continue => run.await, + HookOutcome::Fail(e) => return Err(e.into()), + // A commit that failed without applying, but whose outcome the controller cannot + // know (test hook). + HookOutcome::Indeterminate => { + Err(indeterminate(key, "the commit was not acknowledged (test hook)").into()) + } + }; - let result = match run.await { + let result = match result { Ok(commit) => match controller.hook(HookPoint::AfterMetastoreCommit).await { HookOutcome::Continue => Ok(commit), HookOutcome::Indeterminate | HookOutcome::Fail(_) => Err(indeterminate( diff --git a/libsql-server/src/namespace/fence/drain.rs b/libsql-server/src/namespace/fence/drain.rs index 4f77a95d4a..6165784f60 100644 --- a/libsql-server/src/namespace/fence/drain.rs +++ b/libsql-server/src/namespace/fence/drain.rs @@ -121,6 +121,16 @@ pub async fn acquire_source_write_fence( }; // Step 7. + let acquired_on = commit.record.as_ref().and_then(|r| r.identity.log_id); + if acquired_on.is_some_and(|log_id| log_id != boundary.log_id) { + tracing::warn!( + namespace = %controller.namespace(), + acquired_on = ?acquired_on, + boundary_log_id = %boundary.log_id, + "the replication log was rebuilt since the write fence was acquired (the source \ + restarted while draining); the frozen boundary names the rebuilt log" + ); + } ctx.now_ms = now_ms(); transition .complete_drain( diff --git a/libsql-server/src/namespace/fence/hooks.rs b/libsql-server/src/namespace/fence/hooks.rs index 90ef0d1685..28a00c673e 100644 --- a/libsql-server/src/namespace/fence/hooks.rs +++ b/libsql-server/src/namespace/fence/hooks.rs @@ -56,7 +56,9 @@ pub enum HookAction { /// Fail at this point with `error`, as if the step had failed before it took effect. Fail(FenceError), /// At `AfterMetastoreCommit`: report the commit as indeterminate even though it happened, - /// which is what a lost commit acknowledgement looks like to the controller. + /// which is what a lost commit acknowledgement looks like to the controller. At + /// `BeforeMetastoreCommit`: report it as indeterminate without running it, which is a + /// commit that failed without applying but whose outcome the controller cannot know. Indeterminate, } diff --git a/libsql-server/src/namespace/fence/mod.rs b/libsql-server/src/namespace/fence/mod.rs index 6b445a9ba6..7e64e0cd7d 100644 --- a/libsql-server/src/namespace/fence/mod.rs +++ b/libsql-server/src/namespace/fence/mod.rs @@ -28,6 +28,9 @@ pub mod state; pub mod store; pub mod transition; +#[cfg(test)] +mod tests; + #[allow(clippy::all)] pub(crate) mod proto { include!("../../generated/namespace_fence.rs"); diff --git a/libsql-server/src/namespace/fence/tests.rs b/libsql-server/src/namespace/fence/tests.rs new file mode 100644 index 0000000000..bb41fbeecd --- /dev/null +++ b/libsql-server/src/namespace/fence/tests.rs @@ -0,0 +1,898 @@ +//! Restart at every persistence boundary, indeterminate commits and lost responses +//! (`docs/NAMESPACE_FENCE.md` sections 8.4, 8.5 and 16; section 17 row 7). +//! +//! A restart here is a crash: each server lifetime runs on a runtime of its own, and ending it +//! shuts that runtime down without running any shutdown code and leaks the `NamespaceStore`, so +//! nothing is flushed, the namespace's `.sentinel` stays behind and the next start takes the +//! dirty-recovery path. The next lifetime opens a new `NamespaceStore` (and so a new +//! `MetaStore` and fence registry) on the same directory. The task running the fence command is +//! parked at a hook point when the crash happens, so the crash lands exactly on that boundary. + +use std::future::Future; +use std::path::Path; +use std::sync::Arc; +use std::time::Duration; + +use rusqlite::ErrorCode; +use tempfile::tempdir; +use tokio::runtime::Runtime; +use uuid::Uuid; + +use super::command::{DrainPolicy, FenceCommand, FenceRequest, OnDeadline}; +use super::controller::FenceController; +use super::hooks::{HookAction, HookPoint, Paused}; +use super::outcome::{FenceDetail, FenceError, FenceOutcome}; +use super::record::{FrozenBoundary, ServerIdentity}; +use super::state::{FenceState, OperationClass}; +use super::store as fence_store; +use crate::connection::config::DatabaseConfig; +use crate::connection::Connection as _; +use crate::database::{Connection, Database}; +use crate::error::Error; +use crate::namespace::meta_store::{FenceCommit, FenceCommitKind, FenceInspection}; +use crate::namespace::store::fence_tests::open_store; +use crate::namespace::store::NamespaceStore; +use crate::namespace::RestoreOption; + +const OP: Uuid = Uuid::from_u128(0xa); +const OTHER_OP: Uuid = Uuid::from_u128(0xb); +/// Long enough that no test reaches it: a drain must never finish because of time. +const LONG: DrainPolicy = DrainPolicy { + deadline_ms: 600_000, + on_deadline: OnDeadline::Fail, +}; +/// A deadline that has already passed: an acquisition with a writer still active answers +/// `DRAINING` at once. +const NOW: DrainPolicy = DrainPolicy { + deadline_ms: 0, + on_deadline: OnDeadline::Fail, +}; +/// Bound on waiting for a task to reach a hook point; a test that hits it has failed. +const PROMPT: Duration = Duration::from_secs(30); +/// Rows committed to `t` before any fence command runs. +const ROWS: i64 = 3; + +fn server_identity() -> ServerIdentity { + ServerIdentity { + build: "test".into(), + instance_id: Uuid::from_u128(0x99), + } +} + +/// One lifetime of a server process on `dir`. +struct Server { + rt: Runtime, + store: NamespaceStore, +} + +impl Server { + fn boot(dir: &Path) -> Self { + let rt = tokio::runtime::Builder::new_multi_thread() + .worker_threads(4) + .enable_all() + .build() + .unwrap(); + let store = rt.block_on(open_store(dir)); + Self { rt, store } + } + + /// End the lifetime the way a crash does: no shutdown code runs, nothing is flushed or + /// closed, and every task (including a fence command parked at a hook point) stops where + /// it is. + fn crash(self) { + let Self { rt, store } = self; + std::mem::forget(store); + rt.shutdown_background(); + } + + fn run(&self, f: F) -> F::Output { + self.rt.block_on(f) + } + + /// Create `ns` with a table `t` holding [`ROWS`] rows. + fn create_source(&self) { + self.run(async { + self.store + .create( + "ns".into(), + RestoreOption::Latest, + DatabaseConfig { + // Held writers must never have the slot stolen by the timeout. + txn_timeout: Some(Duration::from_secs(600)), + ..Default::default() + }, + ) + .await + .unwrap(); + let conn = self.conn().await; + raw(&conn, "create table t (x)").await.unwrap(); + for _ in 0..ROWS { + raw(&conn, "insert into t values (1)").await.unwrap(); + } + }) + } + + /// The namespace's controller, loading the namespace if it is not loaded. + async fn fence(&self) -> Arc { + self.store + .with("ns".into(), |ns| ns.fence().clone()) + .await + .unwrap() + } + + async fn conn(&self) -> Arc { + let maker = self + .store + .with("ns".into(), |ns| ns.db.connection_maker()) + .await + .unwrap(); + Arc::new(maker.create().await.unwrap()) + } + + /// The live replication log's id and last committed frame. + async fn log(&self) -> (Uuid, Option) { + self.store + .with("ns".into(), |ns| match &ns.db { + Database::Primary(p) => { + let logger = p.wal_wrapper.wrapper().logger(); + let frame_no = *logger.new_frame_notifier.borrow(); + (logger.log_id(), frame_no) + } + _ => unreachable!(), + }) + .await + .unwrap() + } + + fn execute( + &self, + request: FenceRequest, + ) -> tokio::task::JoinHandle> { + let store = self.store.clone(); + self.rt.spawn(async move { + store + .execute_fence_command(request, server_identity()) + .await + }) + } + + async fn inspect(&self) -> FenceInspection { + self.store + .meta_store() + .inspect_fence("ns".into()) + .await + .unwrap() + } + + async fn count(&self) -> i64 { + let conn = self.conn().await; + tokio::task::spawn_blocking(move || { + conn.with_raw(|c| c.query_row("select count(*) from t", (), |r| r.get(0))) + }) + .await + .unwrap() + .unwrap() + } + + /// Whether a new connection may begin a write transaction. Writes nothing. + async fn writes_admitted(&self) -> bool { + let conn = self.conn().await; + match raw(&conn, "begin immediate; rollback;").await { + Ok(()) => true, + Err(rusqlite::Error::SqliteFailure(e, _)) + if e.code == ErrorCode::AuthorizationForStatementDenied => + { + false + } + Err(e) => panic!("unexpected error probing write admission: {e}"), + } + } + + /// Acquire and complete the source write fence under `OP`, command 1. + fn fence_source(&self) -> FenceCommit { + self.run(async { + let (log_id, _) = self.log().await; + let commit = self + .execute(acquire(log_id, 1, LONG)) + .await + .unwrap() + .unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + commit + }) + } +} + +/// Run `sql` as one raw program on `conn`, off the async runtime (it can block on the write +/// slot). +async fn raw(conn: &Arc, sql: &'static str) -> rusqlite::Result<()> { + let conn = conn.clone(); + tokio::task::spawn_blocking(move || conn.with_raw(|c| c.execute_batch(sql))) + .await + .unwrap() +} + +fn acquire(log_id: Uuid, command_id: u128, policy: DrainPolicy) -> FenceRequest { + FenceRequest { + namespace: "ns".into(), + operation_id: OP, + command_id: Uuid::from_u128(command_id), + expected_state: FenceState::Unfenced, + expected_revision: 0, + command: FenceCommand::AcquireSourceWriteFence { + expected_log_id: log_id, + drain_policy: Some(policy), + }, + } +} + +fn release(command_id: u128, revision: u64) -> FenceRequest { + FenceRequest { + namespace: "ns".into(), + operation_id: OP, + command_id: Uuid::from_u128(command_id), + expected_state: FenceState::SourceWriteFenced, + expected_revision: revision, + command: FenceCommand::ReleaseSourceWriteFence, + } +} + +fn fence_error(result: &crate::Result) -> &FenceError { + match result { + Err(Error::NamespaceFence(e)) => e, + other => panic!("expected a fence error, got {other:?}"), + } +} + +fn boundary(commit: &FenceCommit) -> FrozenBoundary { + commit + .record + .as_ref() + .and_then(|r| r.frozen_boundary) + .expect("a write-fenced source has a frozen boundary") +} + +/// The marker file's bytes, if there is one. +fn read_marker_bytes(dbs: &Path) -> Option> { + match std::fs::read(fence_store::marker_path(dbs, &"ns".into())) { + Ok(bytes) => Some(bytes), + Err(e) if e.kind() == std::io::ErrorKind::NotFound => None, + Err(e) => panic!("{e}"), + } +} + +/// Put the marker file back to `bytes` (`None`: no marker). +fn restore_marker_bytes(dbs: &Path, bytes: Option<&[u8]>) { + let path = fence_store::marker_path(dbs, &"ns".into()); + match bytes { + Some(bytes) => std::fs::write(path, bytes).unwrap(), + None => std::fs::remove_file(path).unwrap(), + } +} + +async fn reached(paused: &Paused, case: &str, point: HookPoint) { + tokio::time::timeout(PROMPT, paused.reached()) + .await + .unwrap_or_else(|_| panic!("{case}: the command never reached {point:?}")); +} + +#[derive(Debug, Clone, Copy)] +enum Command { + Acquire, + Release, +} + +/// What the metastore holds for the command when the process dies. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +enum Durable { + /// Nothing of the command. + Nothing, + /// The `SOURCE_DRAINING` record and the command's `DRAINING` receipt (acquisition only). + Draining, + /// The command's final result. + Final, +} + +#[derive(Debug, Clone, Copy)] +struct Boundary { + name: &'static str, + command: Command, + point: HookPoint, + /// The point is the one reached on the acquisition's second commit (the completion of the + /// drain), not its first. + second_commit: bool, + /// The marker file is put back to what it held before the commit, as a crash between the + /// metastore commit and the marker write leaves it. + marker_lags: bool, + durable: Durable, +} + +const BOUNDARIES: &[Boundary] = &[ + Boundary { + name: "acquire/after-installing-gate", + command: Command::Acquire, + point: HookPoint::AfterInstallingGate, + second_commit: false, + marker_lags: false, + durable: Durable::Nothing, + }, + Boundary { + name: "acquire/before-draining-commit", + command: Command::Acquire, + point: HookPoint::BeforeMetastoreCommit, + second_commit: false, + marker_lags: false, + durable: Durable::Nothing, + }, + Boundary { + name: "acquire/after-draining-commit", + command: Command::Acquire, + point: HookPoint::AfterMetastoreCommit, + second_commit: false, + marker_lags: false, + durable: Durable::Draining, + }, + Boundary { + name: "acquire/after-draining-commit/marker-lags", + command: Command::Acquire, + point: HookPoint::AfterMetastoreCommit, + second_commit: false, + marker_lags: true, + durable: Durable::Draining, + }, + Boundary { + name: "acquire/before-draining-publish", + command: Command::Acquire, + point: HookPoint::BeforeGatePublish, + second_commit: false, + marker_lags: false, + durable: Durable::Draining, + }, + Boundary { + name: "acquire/draining-published", + command: Command::Acquire, + point: HookPoint::BeforeResponse, + second_commit: false, + marker_lags: false, + durable: Durable::Draining, + }, + Boundary { + name: "acquire/before-boundary-capture", + command: Command::Acquire, + point: HookPoint::BeforeBoundaryCapture, + second_commit: false, + marker_lags: false, + durable: Durable::Draining, + }, + Boundary { + name: "acquire/before-fenced-commit", + command: Command::Acquire, + point: HookPoint::BeforeMetastoreCommit, + second_commit: true, + marker_lags: false, + durable: Durable::Draining, + }, + Boundary { + name: "acquire/after-fenced-commit", + command: Command::Acquire, + point: HookPoint::AfterMetastoreCommit, + second_commit: true, + marker_lags: false, + durable: Durable::Final, + }, + Boundary { + name: "acquire/after-fenced-commit/marker-lags", + command: Command::Acquire, + point: HookPoint::AfterMetastoreCommit, + second_commit: true, + marker_lags: true, + durable: Durable::Final, + }, + Boundary { + name: "acquire/before-fenced-publish", + command: Command::Acquire, + point: HookPoint::BeforeGatePublish, + second_commit: true, + marker_lags: false, + durable: Durable::Final, + }, + Boundary { + name: "acquire/before-fenced-response", + command: Command::Acquire, + point: HookPoint::BeforeResponse, + second_commit: true, + marker_lags: false, + durable: Durable::Final, + }, + Boundary { + name: "release/before-commit", + command: Command::Release, + point: HookPoint::BeforeMetastoreCommit, + second_commit: false, + marker_lags: false, + durable: Durable::Nothing, + }, + Boundary { + name: "release/after-commit", + command: Command::Release, + point: HookPoint::AfterMetastoreCommit, + second_commit: false, + marker_lags: false, + durable: Durable::Final, + }, + Boundary { + name: "release/after-commit/marker-lags", + command: Command::Release, + point: HookPoint::AfterMetastoreCommit, + second_commit: false, + marker_lags: true, + durable: Durable::Final, + }, + Boundary { + name: "release/before-publish", + command: Command::Release, + point: HookPoint::BeforeGatePublish, + second_commit: false, + marker_lags: false, + durable: Durable::Final, + }, + Boundary { + name: "release/before-response", + command: Command::Release, + point: HookPoint::BeforeResponse, + second_commit: false, + marker_lags: false, + durable: Durable::Final, + }, +]; + +/// The state a restart recovers for `case`: the one before the command, or the one it +/// committed. Never anything else. +fn recovered_state(case: &Boundary) -> FenceState { + match (case.command, case.durable) { + (Command::Acquire, Durable::Nothing) => FenceState::Unfenced, + (Command::Acquire, Durable::Draining) => FenceState::SourceDraining, + (Command::Acquire, Durable::Final) => FenceState::SourceWriteFenced, + (Command::Release, Durable::Nothing) => FenceState::SourceWriteFenced, + (Command::Release, Durable::Final) => FenceState::Released, + (Command::Release, Durable::Draining) => unreachable!(), + } +} + +/// Kill the process at every point where a fence command persists, publishes or answers, and +/// restart it on the same directory. The restarted server recovers exactly the state before the +/// command or the state it committed, installs that gate before it serves the namespace (so a +/// namespace is never open unless an opening transition committed), and a replay of the same +/// command then finishes it with the stored or the expected result. +#[test] +fn restart_at_each_boundary() { + for case in BOUNDARIES { + restart_at(case); + } +} + +fn restart_at(case: &Boundary) { + let name = case.name; + let dir = tempdir().unwrap(); + let dbs = dir.path().join("dbs"); + + // First lifetime: run the command until it reaches the boundary, then crash. + let server = Server::boot(dir.path()); + server.create_source(); + let (log_before, _) = server.run(server.log()); + let (request, revision_before) = match case.command { + Command::Acquire => (acquire(log_before, 1, LONG), 0), + Command::Release => { + let fenced = server.fence_source(); + let revision = fenced.record.as_ref().unwrap().revision; + (release(2, revision), revision) + } + }; + let committed_boundary = server.run(async { + let fence = server.fence().await; + let hooks = fence.hooks(); + let mut marker_before = read_marker_bytes(&dbs); + let paused = if case.second_commit { + let capture = hooks.pause_at(HookPoint::BeforeBoundaryCapture); + let task = server.execute(request.clone()); + reached(&capture, name, HookPoint::BeforeBoundaryCapture).await; + marker_before = read_marker_bytes(&dbs); + let paused = hooks.pause_at(case.point); + capture.resume(); + reached(&paused, name, case.point).await; + drop(task); + paused + } else { + let paused = hooks.pause_at(case.point); + let task = server.execute(request.clone()); + reached(&paused, name, case.point).await; + drop(task); + paused + }; + if case.marker_lags { + restore_marker_bytes(&dbs, marker_before.as_deref()); + } + // What the metastore holds at the moment of the crash. + let inspected = server.inspect().await; + assert_eq!(inspected.fence.state(), recovered_state(case), "{name}"); + drop(paused); + inspected.fence.record().and_then(|r| r.frozen_boundary) + }); + server.crash(); + + // Second lifetime. + let server = Server::boot(dir.path()); + server.run(async { + let expected = recovered_state(case); + let fence = server.fence().await; + let gate = fence.gate(); + assert_eq!(gate.state(), expected, "{name}: recovered state"); + assert!( + gate.indeterminate.is_none() && !gate.is_installing(), + "{name}" + ); + let open = matches!(expected, FenceState::Unfenced | FenceState::Released); + assert_eq!( + server.writes_admitted().await, + open, + "{name}: write admission" + ); + // Committed data survived, nothing else was written. + assert_eq!(server.count().await, ROWS, "{name}"); + // The marker was repaired if it had fallen behind. + let marker = fence_store::read_marker(&dbs, &"ns".into()).unwrap(); + assert_eq!( + marker.and_then(|m| m.ok()).map(|m| m.record.revision), + gate.fence.record().map(|r| r.revision), + "{name}: marker" + ); + + // A crash leaves the namespace dirty, so its replication log was rebuilt from the + // database file under a new log id. + let (log_after, frame_after) = server.log().await; + assert_ne!(log_after, log_before, "{name}: the log was not rebuilt"); + + let replay = server.execute(request.clone()).await.unwrap(); + match (case.command, case.durable) { + (Command::Acquire, Durable::Nothing) => { + // Nothing was acknowledged and nothing was written, and the identity the caller + // observed is gone: the replay is refused before anything is written, and the + // caller acquires again under the identity it reads now. + let e = fence_error(&replay); + assert_eq!(e.outcome(), FenceOutcome::FencePreconditionFailed, "{name}"); + assert_eq!(e.detail(), Some(FenceDetail::NamespaceIdentityMismatch)); + assert_eq!(fence.gate().state(), FenceState::Unfenced, "{name}"); + assert!(server.writes_admitted().await, "{name}"); + let commit = server + .execute(acquire(log_after, 3, LONG)) + .await + .unwrap() + .unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied, "{name}"); + assert_eq!( + boundary(&commit), + FrozenBoundary { + log_id: log_after, + frame_no: frame_after, + }, + "{name}" + ); + } + (Command::Acquire, Durable::Draining) => { + // The drain that was requested resumes and completes at once: recovery + // discarded any uncommitted work. The boundary is on the live, rebuilt log. + let commit = replay.unwrap(); + assert_eq!(commit.kind, FenceCommitKind::Committed, "{name}"); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied, "{name}"); + assert_eq!(commit.receipt.command_id, request.command_id); + assert_eq!(commit.receipt.revision_after, 2, "{name}"); + let record = commit.record.as_ref().unwrap(); + assert_eq!(record.identity.log_id, Some(log_before), "{name}"); + assert_eq!( + boundary(&commit), + FrozenBoundary { + log_id: log_after, + frame_no: frame_after, + }, + "{name}" + ); + } + (Command::Acquire, Durable::Final) => { + // The stored result, boundary included: it names the log that was live when + // the drain was proven. + let commit = replay.unwrap(); + assert_eq!(commit.kind, FenceCommitKind::Replayed, "{name}"); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied, "{name}"); + assert_eq!(Some(boundary(&commit)), committed_boundary, "{name}"); + assert_eq!(boundary(&commit).log_id, log_before, "{name}"); + } + (Command::Release, durable) => { + let commit = replay.unwrap(); + let kind = if durable == Durable::Final { + FenceCommitKind::Replayed + } else { + FenceCommitKind::Committed + }; + assert_eq!(commit.kind, kind, "{name}"); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied, "{name}"); + assert_eq!(commit.receipt.revision_before, revision_before, "{name}"); + assert_eq!(commit.receipt.state_after, FenceState::Released, "{name}"); + } + } + + // Settled: the gate is the durable state, and a further replay answers the same. + let gate = fence.gate(); + let durable = server.inspect().await; + assert_eq!(gate.fence, durable.fence, "{name}"); + let open = gate.state() == FenceState::Released; + assert_eq!(server.writes_admitted().await, open, "{name}"); + assert_eq!(server.count().await, ROWS, "{name}"); + }); + server.crash(); +} + +/// After a restart in `SOURCE_DRAINING` with a writer that was active at the crash, nothing +/// advances by itself: the namespace stays closed, other commands cannot move it on, and only +/// the replay of the same acquisition completes the drain, at once. +#[test] +fn restart_in_draining_waits_for_the_same_command() { + let dir = tempdir().unwrap(); + + let server = Server::boot(dir.path()); + server.create_source(); + let (log_before, _) = server.run(server.log()); + let request = acquire(log_before, 1, NOW); + server.run(async { + let holder = server.conn().await; + raw(&holder, "begin immediate; insert into t values (2);") + .await + .unwrap(); + let commit = server.execute(request.clone()).await.unwrap().unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Draining); + // The writer never finishes: its transaction dies with the process (closing the + // connection rolls it back, which is what SQLite recovery does to it on disk). + drop(holder); + }); + server.crash(); + + let server = Server::boot(dir.path()); + server.run(async { + let fence = server.fence().await; + assert_eq!(fence.gate().state(), FenceState::SourceDraining); + assert!(!server.writes_admitted().await); + // The uncommitted write is gone. + assert_eq!(server.count().await, ROWS); + + // Reads are served and move nothing; other commands cannot advance the namespace. + let set_read_fence = FenceRequest { + namespace: "ns".into(), + operation_id: OP, + command_id: Uuid::from_u128(2), + expected_state: FenceState::SourceWriteFenced, + expected_revision: 2, + command: FenceCommand::SetSourceReadFence { drain_policy: None }, + }; + let r = server.execute(set_read_fence).await.unwrap(); + assert!(fence_error(&r).outcome() != FenceOutcome::Applied); + let (log_after, frame_after) = server.log().await; + let mut other = acquire(log_after, 3, LONG); + other.operation_id = OTHER_OP; + let r = server.execute(other).await.unwrap(); + assert_eq!( + fence_error(&r).outcome(), + FenceOutcome::FenceOwnedByAnotherOperation + ); + let inspected = server.inspect().await; + assert_eq!(inspected.fence.state(), FenceState::SourceDraining); + assert_eq!(inspected.fence.revision(), 1); + assert_eq!(fence.gate().state(), FenceState::SourceDraining); + + // The same command, with the same zero deadline, completes at once. + let commit = server.execute(request.clone()).await.unwrap().unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + assert_eq!(commit.receipt.command_id, request.command_id); + assert_eq!( + boundary(&commit), + FrozenBoundary { + log_id: log_after, + frame_no: frame_after, + } + ); + assert_eq!(fence.gate().state(), FenceState::SourceWriteFenced); + assert!(!server.writes_admitted().await); + assert_eq!(server.count().await, ROWS); + }); + server.crash(); +} + +/// A commit whose outcome is unknown closes the namespace (every class but maintenance and +/// observability), makes every other command answer `FENCE_COMMIT_INDETERMINATE`, and is +/// reconciled by replaying the same command from the durable row: both when the commit had +/// happened and when it had not. +#[test] +fn indeterminate_commit_keeps_gate_closed() { + for committed in [true, false] { + let dir = tempdir().unwrap(); + let server = Server::boot(dir.path()); + server.create_source(); + server.run(async { + let (log_id, frame_no) = server.log().await; + let fence = server.fence().await; + let request = acquire(log_id, 1, LONG); + fence.hooks().arm( + if committed { + HookPoint::AfterMetastoreCommit + } else { + HookPoint::BeforeMetastoreCommit + }, + HookAction::Indeterminate, + ); + let r = server.execute(request.clone()).await.unwrap(); + let e = fence_error(&r); + assert_eq!(e.outcome(), FenceOutcome::FenceCommitIndeterminate); + assert_eq!(e.detail(), Some(FenceDetail::IndeterminateCommit)); + + let durable = server.inspect().await.fence.state(); + assert_eq!( + durable, + if committed { + FenceState::SourceDraining + } else { + FenceState::Unfenced + } + ); + let gate = fence.gate(); + assert_eq!(gate.indeterminate, Some((OP, request.command_id))); + assert!(!gate.is_installing()); + for class in OperationClass::ALL { + let permitted = fence.permits(class); + match class { + OperationClass::Maintenance | OperationClass::Observability => { + assert!(permitted.is_ok()) + } + _ => assert_eq!( + permitted.unwrap_err().outcome(), + FenceOutcome::FenceStateUnavailable, + "{class:?}" + ), + } + } + assert!(!server.writes_admitted().await); + + // Any other command is refused with the indeterminate code, nothing is written. + let mut other = acquire(log_id, 2, LONG); + other.operation_id = OTHER_OP; + let r = server.execute(other).await.unwrap(); + assert_eq!( + fence_error(&r).outcome(), + FenceOutcome::FenceCommitIndeterminate + ); + let r = server.execute(acquire(log_id, 3, LONG)).await.unwrap(); + assert_eq!( + fence_error(&r).outcome(), + FenceOutcome::FenceCommitIndeterminate + ); + assert_eq!(server.inspect().await.fence.state(), durable); + + // The replay reconciles from the durable row: it resumes the drain that committed, + // or runs the acquisition that did not, and completes it. + let commit = server.execute(request.clone()).await.unwrap().unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + assert_eq!(commit.receipt.command_id, request.command_id); + assert_eq!(commit.receipt.revision_after, 2); + assert_eq!(boundary(&commit), FrozenBoundary { log_id, frame_no }); + let gate = fence.gate(); + assert_eq!(gate.indeterminate, None); + assert_eq!(gate.state(), FenceState::SourceWriteFenced); + assert_eq!(gate.fence, server.inspect().await.fence); + assert!(!server.writes_admitted().await); + // Maintenance and reads are served again. + assert_eq!(server.count().await, ROWS); + }); + server.crash(); + } +} + +/// The caller of an acquisition goes away before it hears the answer, once while the drain is +/// still waiting and once just before the final response. The acquisition finishes on its own +/// task regardless, `InspectFence` shows the result, and a replay of the same command returns +/// it. +#[test] +fn acquire_response_loss_resolved_by_replay_and_inspect() { + for lose_at in [HookPoint::AfterMetastoreCommit, HookPoint::BeforeResponse] { + let dir = tempdir().unwrap(); + let server = Server::boot(dir.path()); + server.create_source(); + server.run(async { + let (log_id, _) = server.log().await; + let fence = server.fence().await; + let request = acquire(log_id, 1, LONG); + let holder = server.conn().await; + raw(&holder, "begin immediate; insert into t values (2);") + .await + .unwrap(); + + let hooks = fence.hooks(); + let capture = hooks.pause_at(HookPoint::BeforeBoundaryCapture); + let first = (lose_at == HookPoint::AfterMetastoreCommit) + .then(|| hooks.pause_at(HookPoint::AfterMetastoreCommit)); + let caller = server.execute(request.clone()); + if let Some(first) = first { + // The caller is gone right after SOURCE_DRAINING commits. + reached(&first, "response loss", HookPoint::AfterMetastoreCommit).await; + caller.abort(); + first.resume(); + } + + // The drain waits for the writer, with or without a caller. + let mut rx = fence.subscribe(); + tokio::time::timeout( + PROMPT, + rx.wait_for(|g| g.state() == FenceState::SourceDraining), + ) + .await + .unwrap() + .unwrap(); + assert_eq!( + server.inspect().await.fence.state(), + FenceState::SourceDraining + ); + // A replay while it runs waits for the transition lock rather than racing it. + let replay_during = server.execute(request.clone()); + + raw(&holder, "commit").await.unwrap(); + let (_, committed) = server.log().await; + reached(&capture, "response loss", HookPoint::BeforeBoundaryCapture).await; + if lose_at == HookPoint::BeforeResponse { + // The caller is gone after the final commit was published, before the answer. + let last = hooks.pause_at(HookPoint::BeforeResponse); + capture.resume(); + reached(&last, "response loss", HookPoint::BeforeResponse).await; + assert_eq!(fence.gate().state(), FenceState::SourceWriteFenced); + caller.abort(); + last.resume(); + } else { + capture.resume(); + } + assert!(caller.await.unwrap_err().is_cancelled()); + tokio::time::timeout( + PROMPT, + rx.wait_for(|g| g.state() == FenceState::SourceWriteFenced), + ) + .await + .unwrap() + .unwrap(); + + let inspected = server.inspect().await; + assert_eq!(inspected.fence.state(), FenceState::SourceWriteFenced); + let record = inspected.fence.record().unwrap(); + assert_eq!( + record.frozen_boundary, + Some(FrozenBoundary { + log_id, + frame_no: committed, + }) + ); + let receipt = inspected + .receipts + .iter() + .filter_map(|r| r.receipt.as_ref().ok()) + .find(|r| r.command_id == request.command_id) + .unwrap() + .clone(); + assert_eq!(receipt.outcome, FenceOutcome::Applied); + + for replay in [ + replay_during.await.unwrap().unwrap(), + server.execute(request.clone()).await.unwrap().unwrap(), + ] { + assert_eq!(replay.kind, FenceCommitKind::Replayed); + assert_eq!(replay.receipt, receipt); + assert_eq!(replay.record.as_ref(), Some(record)); + } + assert_eq!(server.count().await, ROWS + 1); + assert!(!server.writes_admitted().await); + }); + server.crash(); + } +} diff --git a/libsql-server/src/namespace/fence/transition.rs b/libsql-server/src/namespace/fence/transition.rs index d589ce8859..9c664b320c 100644 --- a/libsql-server/src/namespace/fence/transition.rs +++ b/libsql-server/src/namespace/fence/transition.rs @@ -673,12 +673,13 @@ pub fn complete_drain( let mut next = record.clone(); if let DrainCompletion::SourceWrites { boundary } = completion { - if record.identity.log_id != Some(boundary.log_id) { - return Err(precondition( - FenceDetail::NamespaceIdentityMismatch, - "the frozen boundary belongs to a different replication log", - )); - } + // The boundary names the replication log that is live when the drain is proven. It is + // not the log the identity was captured on when the source was restarted while + // draining: crash recovery rebuilds the log from the database file under a new id + // (section 8.5). Write admission has been durably closed since `SOURCE_DRAINING` + // committed, and the lifecycle paths that could replace the database are denied, so + // the data is what was committed before the cutoff. The identity keeps the log id the + // caller acquired against. next.frozen_boundary = Some(boundary); } next.state = to; @@ -1485,20 +1486,21 @@ mod tests { assert!(complete_drain(&record, &receipt, DrainCompletion::SourceReads, &env()).is_err()); assert!(complete_drain(&record, &receipt, DrainCompletion::TargetImport, &env()).is_err()); - // A boundary from another log. - let err = complete_drain( + // A boundary on a log rebuilt since acquisition (a restart while draining) is recorded + // as it is; the identity keeps the log the caller acquired against. + let rebuilt = FrozenBoundary { + log_id: Uuid::from_u128(0x77), + frame_no: Some(1), + }; + let (next, _) = complete_drain( &record, &receipt, - DrainCompletion::SourceWrites { - boundary: FrozenBoundary { - log_id: Uuid::from_u128(0x77), - frame_no: Some(1), - }, - }, + DrainCompletion::SourceWrites { boundary: rebuilt }, &env(), ) - .unwrap_err(); - assert_eq!(err.detail(), Some(FenceDetail::NamespaceIdentityMismatch)); + .unwrap(); + assert_eq!(next.frozen_boundary, Some(rebuilt)); + assert_eq!(next.identity.log_id, Some(LOG)); // A final receipt, or another operation's. let mut final_receipt = receipt.clone();