diff --git a/libsql-server/proto/namespace_fence.proto b/libsql-server/proto/namespace_fence.proto index efdee1ab81..73250dfb0d 100644 --- a/libsql-server/proto/namespace_fence.proto +++ b/libsql-server/proto/namespace_fence.proto @@ -75,7 +75,8 @@ message DrainPolicy { message FrozenBoundary { string log_id = 1; - uint64 frame_no = 2; + // Absent when the replication log has no frames. + optional uint64 frame_no = 2; } message LegacyBlocks { diff --git a/libsql-server/src/config.rs b/libsql-server/src/config.rs index 9ac7add98b..4457232d03 100644 --- a/libsql-server/src/config.rs +++ b/libsql-server/src/config.rs @@ -193,6 +193,9 @@ pub struct MetaStoreConfig { /// How long receipts of finished fence operations are kept. `None` is the default of /// 30 days. pub namespace_fence_receipt_retention: Option, + /// How long `AcquireSourceWriteFence` waits for active writers when the request names no + /// drain policy. `None` is the default of 30 seconds. + pub namespace_fence_default_write_drain: Option, } #[derive(Debug, Clone)] diff --git a/libsql-server/src/connection/connection_core.rs b/libsql-server/src/connection/connection_core.rs index 17a6f52961..212e5c2b3d 100644 --- a/libsql-server/src/connection/connection_core.rs +++ b/libsql-server/src/connection/connection_core.rs @@ -11,6 +11,8 @@ use crate::connection::legacy::open_conn_active_checkpoint; use crate::error::Error; use crate::metrics::{PROGRAM_EXEC_COUNT, QUERY_CANCELED, VACUUM_COUNT, WAL_CHECKPOINT_COUNT}; use crate::namespace::broadcasters::BroadcasterHandle; +use crate::namespace::fence::controller::FenceConnState; +use crate::namespace::fence::state::OperationClass; use crate::namespace::meta_store::MetaStoreHandle; use crate::namespace::ResolveNamespacePathFn; use crate::query_analysis::StmtKind; @@ -24,6 +26,15 @@ use super::program::{DescribeCol, DescribeParam, DescribeResponse, Program, Vm}; pub type GetCurrentFrameNo = Arc Option + Send + Sync + 'static>; +/// What [`CoreConnection::vacuum_if_needed_above`] did. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(crate) enum VacuumOutcome { + Vacuumed, + NotNeeded, + /// Skipped because the namespace fence denies normal writes. + Fenced, +} + /// The base connection type, shared between legacy and libsql-wal implementations pub(super) struct CoreConnection { conn: libsql_sys::Connection, @@ -37,6 +48,8 @@ pub(super) struct CoreConnection { broadcaster: BroadcasterHandle, hooked: bool, canceled: Arc, + /// Shared with this connection's WAL wrapper (`docs/NAMESPACE_FENCE.md` section 7.4). + fence: Arc, } fn update_stats( @@ -68,6 +81,7 @@ impl CoreConnection { get_current_frame_no: GetCurrentFrameNo, block_writes: Arc, resolve_attach_path: ResolveNamespacePathFn, + fence: Arc, ) -> Result { let conn = open_conn_active_checkpoint( path, @@ -113,6 +127,7 @@ impl CoreConnection { hooked: false, canceled, get_current_frame_no, + fence, }; for ext in extensions.iter() { @@ -188,17 +203,21 @@ impl CoreConnection { pgm: Program, mut builder: B, ) -> Result { - let (config, stats, block_writes, resolve_attach_path) = { + let (config, stats, block_writes, resolve_attach_path, fence) = { let mut lock = this.lock(); let config = lock.config_store.get(); let stats = lock.stats.clone(); let block_writes = lock.block_writes.clone(); let resolve_attach_path = lock.resolve_attach_path.clone(); + let fence = lock.fence.clone(); lock.update_hooks(); - (config, stats, block_writes, resolve_attach_path) + (config, stats, block_writes, resolve_attach_path, fence) }; + // The program is admitted under the gate's current write generation; a write + // transaction it opens must start under the same one (section 8.1). + fence.begin_program(); builder.init(&this.lock().builder_config)?; let mut vm = Vm::new( @@ -229,7 +248,8 @@ impl CoreConnection { update_stats(&stats, sql, rows_read, rows_written, mem_used, elapsed) }, resolve_attach_path, - ); + ) + .with_fence(fence); let mut has_timeout = false; while !vm.finished() { @@ -296,21 +316,45 @@ impl CoreConnection { } pub(super) fn vacuum_if_needed(&self) -> Result<()> { + // NOTICE: don't bother vacuuming if we don't have at least 256MiB of data + self.vacuum_if_needed_above(65536).map(|_| ()) + } + + /// `VACUUM` if the database has at least `min_pages` pages and more than half of them are + /// free. `VACUUM` is not maintenance: it takes a write transaction and produces replicated + /// frames, so it is skipped whenever the namespace fence denies normal writes + /// (`docs/NAMESPACE_FENCE.md` section 7.3), including when the fence closes between the + /// check and the `VACUUM` itself. + pub(super) fn vacuum_if_needed_above(&self, min_pages: i64) -> Result { + self.fence.begin_program(); + if let Err(e) = self.fence.controller().permits(OperationClass::Vacuum) { + tracing::debug!("skipping vacuum: {e}"); + return Ok(VacuumOutcome::Fenced); + } let page_count = self .conn .query_row("PRAGMA page_count", (), |row| row.get::<_, i64>(0))?; let freelist_count = self .conn .query_row("PRAGMA freelist_count", (), |row| row.get::<_, i64>(0))?; - // NOTICE: don't bother vacuuming if we don't have at least 256MiB of data - if page_count >= 65536 && freelist_count * 2 > page_count { + let outcome = if page_count >= min_pages && freelist_count * 2 > page_count { tracing::info!("Vacuuming: pages={page_count} freelist={freelist_count}"); - self.conn.execute("VACUUM", ())?; + if let Err(e) = self.conn.execute("VACUUM", ()) { + return match self.fence.take_denial() { + Some(denial) => { + tracing::debug!("skipping vacuum: {denial}"); + Ok(VacuumOutcome::Fenced) + } + None => Err(e.into()), + }; + } + VacuumOutcome::Vacuumed } else { tracing::trace!("Not vacuuming: pages={page_count} freelist={freelist_count}"); - } + VacuumOutcome::NotNeeded + }; VACUUM_COUNT.increment(1); - Ok(()) + Ok(outcome) } pub(super) fn describe(&self, sql: &str) -> crate::Result { @@ -394,6 +438,7 @@ mod test { use crate::auth::Authenticated; use crate::connection::legacy::MakeLegacyConnection; use crate::connection::{Connection as _, RequestContext, TXN_TIMEOUT}; + use crate::namespace::fence::controller::FenceController; use crate::namespace::meta_store::{metastore_connection_maker, MetaStore}; use crate::namespace::NamespaceName; use crate::query_result_builder::test::{test_driver, TestBuilder}; @@ -415,6 +460,10 @@ mod test { hooked: false, canceled: Arc::new(false.into()), get_current_frame_no: Arc::new(|| None), + fence: FenceConnState::new( + FenceController::unfenced(Default::default()), + crate::namespace::fence::state::OperationClass::NormalWrite, + ), }; let conn = Arc::new(Mutex::new(conn)); @@ -454,6 +503,7 @@ mod test { Default::default(), Arc::new(|_| unreachable!()), Arc::new(|| Sqlite3WalManager::default()), + FenceController::unfenced(Default::default()), ) .await .unwrap(); @@ -500,6 +550,7 @@ mod test { Default::default(), Arc::new(|_| unreachable!()), Arc::new(|| Sqlite3WalManager::default()), + FenceController::unfenced(Default::default()), ) .await .unwrap(); @@ -551,6 +602,7 @@ mod test { Default::default(), Arc::new(|_| unreachable!()), Arc::new(|| Sqlite3WalManager::default()), + FenceController::unfenced(Default::default()), ) .await .unwrap(); @@ -634,6 +686,7 @@ mod test { Default::default(), Arc::new(|_| unreachable!()), Arc::new(|| Sqlite3WalManager::default()), + FenceController::unfenced(Default::default()), ) .await .unwrap(); @@ -727,6 +780,7 @@ mod test { Default::default(), Arc::new(|_| unreachable!()), Arc::new(|| Sqlite3WalManager::default()), + FenceController::unfenced(Default::default()), ) .await .unwrap(); diff --git a/libsql-server/src/connection/connection_manager.rs b/libsql-server/src/connection/connection_manager.rs index 9fa22fbb8e..aab548461b 100644 --- a/libsql-server/src/connection/connection_manager.rs +++ b/libsql-server/src/connection/connection_manager.rs @@ -14,6 +14,8 @@ use rusqlite::ErrorCode; use super::connection_core::CoreConnection; use super::TXN_TIMEOUT; +use crate::namespace::fence::controller::FenceConnState; +use crate::namespace::fence::state::OperationClass; pub type ConnId = u64; pub type InnerWalManager = Sqlite3WalManager; @@ -24,10 +26,16 @@ pub type ManagedConnectionWal = WrappedWal); @@ -35,10 +43,13 @@ impl Abort { fn from_conn(conn: &Arc>>) -> Self { let conn = Arc::downgrade(conn); Self(Arc::new(move || { - conn.upgrade() - .expect("connection still owns the slot, so it must exist") - .lock() - .force_rollback(); + // The connection can be closing concurrently: a drain aborting the active writer + // races with the client going away. A connection that is gone has already released + // its slot (or is about to, from `close`), so there is nothing left to roll back. + match conn.upgrade() { + Some(conn) => conn.lock().force_rollback(), + None => tracing::debug!("connection closed before it could be rolled back"), + } })) } @@ -61,6 +72,121 @@ impl ConnectionManager { let abort = Abort::from_conn(conn); self.inner.abort_handle.lock().insert(id, abort); } + + /// The connection holding the write slot, and the class it holds it for. A slot that has + /// been handed to a queued connection that has not taken it yet counts as held. + pub(crate) fn active_writer(&self) -> Option<(ConnId, OperationClass)> { + self.inner.current.lock().map(|slot| (slot.id, slot.class)) + } + + /// Whether a connection holds the write slot for a write transaction: any holder but a + /// checkpoint (`Maintenance`), which cannot change logical contents. + pub(crate) fn has_writer(&self) -> bool { + self.active_writer() + .is_some_and(|(_, class)| class != OperationClass::Maintenance) + } + + /// Run `f` under the write-slot lock if no connection holds the slot for a write + /// transaction (a checkpoint may). While `f` runs no write transaction can start or end on + /// this manager, so, once the fence has closed write admission, anything `f` reads about the + /// committed log is final (`docs/NAMESPACE_FENCE.md` section 8.3, step 6). Returns the + /// holder otherwise. + pub(crate) fn with_no_writer( + &self, + f: impl FnOnce() -> R, + ) -> Result { + let current = self.inner.current.lock(); + match *current { + Some(slot) if slot.class != OperationClass::Maintenance => Err((slot.id, slot.class)), + _ => Ok(f()), + } + } + + /// A handle that does not keep the manager alive. + pub(crate) fn downgrade(&self) -> WeakConnectionManager { + WeakConnectionManager(Arc::downgrade(&self.inner)) + } + + /// Notified (with `notify_waiters`) every time the write slot is released or handed on. A + /// waiter registers interest (`Notified::enable`) before it checks + /// [`active_writer`](Self::active_writer), so a release in between is not missed. + pub(crate) fn released(&self) -> &tokio::sync::Notify { + &self.inner.released + } + + /// Roll back the transaction of the connection holding the write slot, using the rollback + /// handle it registered. Returns the connection that was asked to roll back, if any. The + /// slot is released by the rollback itself (`end_read_txn`/`end_write_txn`), which notifies + /// [`released`](Self::released); a connection closing at the same time is tolerated. + pub(crate) fn abort_active(&self) -> Option { + let id = self.active_writer()?.0; + let handle = self.inner.abort_handle.lock().get(&id).cloned(); + match handle { + Some(handle) => { + tracing::debug!("aborting the active writer {id}"); + handle.abort(); + Some(id) + } + None => { + tracing::debug!("the active writer {id} is closing; nothing to abort"); + None + } + } + } + + /// A waker for the fence controller, which calls it after every change of the write + /// generation (`docs/NAMESPACE_FENCE.md` section 8.2): every connection waiting in the write + /// queue is woken and re-checks the fence. One the gate denies returns the typed denial + /// instead of waiting for the slot; one it admits (a checkpoint) queues again. The waker + /// holds the manager weakly and reports `false` once it is gone, so the controller can + /// forget it. + pub(crate) fn fence_waker(&self) -> Box bool + Send + Sync> { + let inner = Arc::downgrade(&self.inner); + Box::new(move || match inner.upgrade() { + Some(inner) => { + wake_queue_for_fence(&inner); + true + } + None => false, + }) + } + + #[cfg(test)] + pub(crate) fn queued_writers(&self) -> usize { + self.inner.write_queue.len() + } +} + +/// A [`ConnectionManager`] that is not kept alive by this handle. +#[derive(Clone)] +pub(crate) struct WeakConnectionManager(std::sync::Weak); + +impl WeakConnectionManager { + pub(crate) fn upgrade(&self) -> Option { + self.0.upgrade().map(|inner| ConnectionManager { inner }) + } +} + +fn wake_queue_for_fence(inner: &ConnectionManagerInner) { + // Under the `current` lock, so that a waiter either queued before this (and is stolen and + // woken here) or observes the new token when it takes the lock. + let _current = inner.current.lock(); + inner.fence_token.fetch_add(1, Ordering::SeqCst); + let mut woken = 0; + loop { + match inner.write_queue.steal() { + Steal::Empty => break, + Steal::Success((id, _, unparker)) => { + tracing::debug!("fence changed, waking queued connection id={id}"); + unparker.unpark(); + woken += 1; + } + Steal::Retry => (), + } + } + if woken > 0 { + tracing::debug!("fence changed, woke {woken} queued connections"); + } } impl Deref for ConnectionManager { @@ -91,12 +217,17 @@ pub struct ConnectionManagerInner { abort_handle: Mutex>, /// threads waiting to acquire the lock /// todo: limit how many can be push - write_queue: crossbeam::deque::Injector<(ConnId, Unparker)>, + write_queue: crossbeam::deque::Injector, txn_timeout_duration: Duration, /// the time we are given to acquire a transaction after we were given a slot acquire_timeout_duration: Duration, next_conn_id: AtomicU64, sync_token: AtomicU64, + /// Incremented, under the `current` lock, every time the queue is woken for a fence change. + /// A waiter that sees it move knows it was taken off the queue. + fence_token: AtomicU64, + /// Notified whenever the write slot is released or handed on. + released: tokio::sync::Notify, } impl Default for ConnectionManagerInner { @@ -109,6 +240,8 @@ impl Default for ConnectionManagerInner { acquire_timeout_duration: Duration::from_millis(15), next_conn_id: Default::default(), sync_token: AtomicU64::new(0), + fence_token: AtomicU64::new(0), + released: Default::default(), } } } @@ -117,25 +250,52 @@ impl Default for ConnectionManagerInner { pub struct ManagedConnectionWalWrapper { id: ConnId, manager: ConnectionManager, + /// The connection's fence state, which `begin_write_txn` checks against the namespace's + /// gate (`docs/NAMESPACE_FENCE.md` section 8.1). + fence: Arc, } impl ManagedConnectionWalWrapper { - pub(crate) fn new(manager: ConnectionManager) -> Self { + pub(crate) fn new(manager: ConnectionManager, fence: Arc) -> Self { let id = manager.inner.next_conn_id.fetch_add(1, Ordering::SeqCst); - Self { id, manager } + Self { id, manager, fence } } pub fn id(&self) -> ConnId { self.id } - fn acquire(&self) -> libsql_sys::wal::Result<()> { + /// Wait for the write slot on behalf of work of `class`. `Maintenance` (checkpoints) is never + /// refused by the fence; every other class re-checks the connection's write admission each + /// time it takes the `current` lock, so a waiter queued before a fence change leaves the + /// queue with the typed denial (`docs/NAMESPACE_FENCE.md` section 8.2). + fn acquire(&self, class: OperationClass) -> libsql_sys::wal::Result<()> { let parker = Parker::new(); let mut enqueued = false; let enqueued_at = Instant::now(); let sync_token = self.manager.sync_token.load(Ordering::SeqCst); + let mut fence_token = self.manager.fence_token.load(Ordering::SeqCst); loop { let mut current = self.manager.current.lock(); + let ours = current.as_ref().map_or(false, |slot| slot.id == self.id); + if class != OperationClass::Maintenance && self.fence.admit_write().is_err() { + // The denial is in the connection's fence state. If the slot had already been + // handed to us, pass it on: we are not going to use it. + if ours { + self.hand_off(&mut current); + } + tracing::debug!("write slot request refused by the namespace fence"); + return Err(rusqlite::ffi::Error::new(rusqlite::ffi::SQLITE_AUTH)); + } + let observed = self.manager.fence_token.load(Ordering::SeqCst); + if observed != fence_token { + fence_token = observed; + // A fence change emptied the queue. Unless the slot was handed to us before + // that, we are no longer queued and must queue again. + if enqueued && !ours { + enqueued = false; + } + } // if current is not currently us, and we havent enqueued yet, then enqueue // current can be us in two cases: // - in previous iteration, the queue was empty, and we popped ourselves @@ -190,7 +350,7 @@ impl ManagedConnectionWalWrapper { if current.as_mut().map_or(true, |slot| slot.id != self.id) && !enqueued { self.manager .write_queue - .push((self.id, parker.unparker().clone())); + .push((self.id, class, parker.unparker().clone())); enqueued = true; tracing::debug!("enqueued"); } @@ -280,6 +440,7 @@ impl ManagedConnectionWalWrapper { None => { *current = Some(Slot { id: self.id, + class, started_at: Instant::now(), state: SlotState::Acquiring, }); @@ -317,6 +478,7 @@ impl ManagedConnectionWalWrapper { None => { *current = Some(Slot { id: self.id, + class, started_at: Instant::now(), state: SlotState::Acquiring, }) @@ -340,10 +502,11 @@ impl ManagedConnectionWalWrapper { }; match next { - Some((id, unpaker)) => { + Some((id, class, unpaker)) => { tracing::debug!(line = line!(), "unparking id={id}"); **current = Some(Slot { id, + class, started_at: Instant::now(), state: SlotState::Notified, }); @@ -365,12 +528,15 @@ impl ManagedConnectionWalWrapper { assert_eq!(slot.id, self.id); tracing::debug!("transaction finished after {:?}", slot.started_at.elapsed()); - match self.schedule_next(&mut current) { - Some(_) => (), - None => { - *current = None; - } - } + self.hand_off(&mut current); + } + + /// Give the (already vacated or ours) slot to the next queued connection, or leave it free, + /// and tell drain waiters. + fn hand_off(&self, current: &mut MutexGuard>) { + **current = None; + self.schedule_next(current); + self.manager.released.notify_waiters(); } } @@ -410,7 +576,14 @@ impl WrapWal for ManagedConnectionWalWrapper { #[tracing::instrument(skip_all, fields(id = self.id))] fn begin_write_txn(&mut self, wrapped: &mut InnerWal) -> libsql_sys::wal::Result<()> { tracing::debug!("begin write"); - self.acquire()?; + // The authoritative fence check. It runs before `acquire()`, so a refusal holds no slot + // and releases none, and it returns `SQLITE_AUTH` rather than `SQLITE_BUSY`, so SQLite's + // busy handler does not retry it. The typed reason is left in the connection's fence + // state for the program layer. + if self.fence.admit_write().is_err() { + return Err(rusqlite::ffi::Error::new(rusqlite::ffi::SQLITE_AUTH)); + } + self.acquire(self.fence.class())?; match wrapped.begin_write_txn() { Ok(_) => { tracing::debug!("transaction acquired"); @@ -449,7 +622,7 @@ impl WrapWal for ManagedConnectionWalWrapper { backfilled: Option<&mut i32>, ) -> libsql_sys::wal::Result<()> { let before = Instant::now(); - self.acquire()?; + self.acquire(OperationClass::Maintenance)?; self.manager.current.lock().as_mut().unwrap().state = SlotState::Acquired(SlotType::Checkpoint); @@ -462,11 +635,19 @@ impl WrapWal for ManagedConnectionWalWrapper { if mode as i32 >= CheckpointMode::Restart as i32 { tracing::debug!("forcing queue sync"); self.manager.sync_token.fetch_add(1, Ordering::SeqCst); - let queue_len = self.manager.write_queue.len(); - for _ in 0..queue_len { - let (id, unparker) = self.manager.write_queue.steal().success().unwrap(); - tracing::debug!("forcing queue sync for id={id}"); - unparker.unpark(); + // A fence change can empty the queue concurrently (`wake_queue_for_fence`), so an + // entry counted here may already be gone. + let mut queue_len = self.manager.write_queue.len(); + while queue_len > 0 { + match self.manager.write_queue.steal() { + Steal::Success((id, _, unparker)) => { + tracing::debug!("forcing queue sync for id={id}"); + unparker.unpark(); + queue_len -= 1; + } + Steal::Empty => break, + Steal::Retry => (), + } } } @@ -491,6 +672,9 @@ impl WrapWal for ManagedConnectionWalWrapper { #[tracing::instrument(skip_all, fields(id = self.id))] fn begin_read_txn(&mut self, wrapped: &mut InnerWal) -> libsql_sys::wal::Result { tracing::debug!("begin read txn"); + // Recorded before the snapshot is taken: a transition racing with it leaves the + // transaction with the older generation, which can only refuse a later upgrade. + self.fence.begin_read_txn(); wrapped.begin_read_txn() } @@ -556,3 +740,612 @@ impl WrapWal for ManagedConnectionWalWrapper { ret } } + +/// The WAL write gate (`docs/NAMESPACE_FENCE.md` section 8.1): tests on real connections of a +/// namespace whose fence controller goes through committed metastore transitions. +#[cfg(test)] +mod fence_tests { + use std::path::Path; + use std::sync::Arc; + use std::time::Duration; + + use libsql_sys::wal::wrapper::PassthroughWalWrapper; + use libsql_sys::wal::Sqlite3WalManager; + use rusqlite::functions::FunctionFlags; + use rusqlite::ErrorCode; + use tempfile::tempdir; + + use crate::connection::connection_core::CoreConnection; + use crate::connection::connection_core::VacuumOutcome; + use crate::connection::legacy::{LegacyConnection, MakeLegacyConnection}; + use crate::connection::program::Program; + use crate::connection::Connection as _; + use crate::error::Error; + use crate::namespace::fence::controller::tests::{ + create_namespace, ctx, fence_source, open_metastore, release, OP, + }; + use crate::namespace::fence::controller::FenceController; + use crate::namespace::fence::outcome::{FenceDetail, FenceError, FenceOutcome}; + use crate::namespace::fence::state::{FenceState, OperationClass}; + use crate::namespace::meta_store::{MetaStore, MetaStoreHandle}; + use crate::query_result_builder::test::{StepResult, TestBuilder}; + use crate::query_result_builder::QueryResultBuilder as _; + use crate::DEFAULT_AUTO_CHECKPOINT; + + type Conn = LegacyConnection; + + struct Harness { + _dir: tempfile::TempDir, + meta: MetaStore, + controller: Arc, + maker: MakeLegacyConnection, + } + + impl Harness { + async fn new() -> Self { + Self::with_txn_timeout(None).await + } + + /// A harness whose connections steal the write slot from a transaction only after + /// `txn_timeout` (the test default is 100 ms). Queue tests hold a writer open far longer + /// than that and must not have it stolen. + async fn with_txn_timeout(txn_timeout: Option) -> Self { + let dir = tempdir().unwrap(); + let meta_dir = dir.path().join("meta"); + let db_dir = dir.path().join("db"); + std::fs::create_dir_all(&meta_dir).unwrap(); + std::fs::create_dir_all(&db_dir).unwrap(); + let meta = open_metastore(&meta_dir).await; + create_namespace(&meta, "ns").await; + let controller = FenceController::unfenced("ns".into()); + let config = MetaStoreHandle::load(&db_dir).unwrap(); + if txn_timeout.is_some() { + let mut c = (*config.get()).clone(); + c.txn_timeout = txn_timeout; + config.store(c).await.unwrap(); + } + let maker = make_connections(&db_dir, config, controller.clone()).await; + let this = Self { + _dir: dir, + meta, + controller, + maker, + }; + let conn = this.conn().await; + assert_ok(&run(&conn, &["create table t (x)"]).await); + this + } + + async fn conn(&self) -> Conn { + self.maker.make_connection().await.unwrap() + } + + /// UNFENCED -> SOURCE_DRAINING -> SOURCE_WRITE_FENCED. + async fn fence(&self) { + fence_source(&self.meta, &self.controller, OP).await; + assert_eq!( + self.controller.gate().state(), + FenceState::SourceWriteFenced + ); + } + + /// SOURCE_WRITE_FENCED -> RELEASED: writes are admitted again, under a new generation. + async fn release(&self) { + let revision = self.controller.gate().revision(); + self.controller + .apply_command(&self.meta, release("ns", OP, 2, revision), ctx()) + .await + .unwrap(); + assert_eq!(self.controller.gate().state(), FenceState::Released); + } + } + + async fn make_connections( + path: &Path, + config: MetaStoreHandle, + fence: Arc, + ) -> MakeLegacyConnection { + MakeLegacyConnection::new( + path.into(), + PassthroughWalWrapper, + Default::default(), + Default::default(), + config, + Arc::new([]), + 100000000, + 100000000, + DEFAULT_AUTO_CHECKPOINT, + Arc::new(|| None), + None, + Default::default(), + Arc::new(|_| unreachable!()), + Arc::new(|| Sqlite3WalManager::default()), + fence, + ) + .await + .unwrap() + } + + async fn run(conn: &Conn, stmts: &[&'static str]) -> Vec { + let inner = conn.inner.clone(); + let stmts = stmts.to_vec(); + tokio::task::spawn_blocking(move || { + CoreConnection::run(inner, Program::seq(&stmts), TestBuilder::default()) + .unwrap() + .into_ret() + }) + .await + .unwrap() + } + + fn assert_ok(steps: &[StepResult]) { + for (i, step) in steps.iter().enumerate() { + assert!(step.is_ok(), "step {i} failed: {step:?}"); + } + } + + fn fence_error(step: &StepResult) -> &FenceError { + match step { + Err(Error::NamespaceFence(e)) => e, + other => panic!("expected a fence denial, got {other:?}"), + } + } + + async fn count(conn: &Conn) -> i64 { + conn.with_raw(|c| c.query_row("select count(*) from t", (), |r| r.get(0))) + .unwrap() + } + + /// A program is admitted, the fence is acquired and released while it runs, and the write it + /// then attempts is refused at the WAL although the live gate is open again: the program was + /// admitted under a generation that is no longer current. The race is held open by a SQL + /// function that parks the program between admission and its write. + #[tokio::test(flavor = "multi_thread")] + async fn wal_gate_rejects_program_admitted_before_fence() { + let h = Harness::new().await; + let conn = h.conn().await; + + let (reached_tx, mut reached_rx) = tokio::sync::mpsc::unbounded_channel::<()>(); + let (resume_tx, resume_rx) = std::sync::mpsc::channel::<()>(); + let parked = std::panic::AssertUnwindSafe((reached_tx, std::sync::Mutex::new(resume_rx))); + conn.with_raw(move |c| { + c.create_scalar_function("park", 0, FunctionFlags::SQLITE_UTF8, move |_| { + let (reached, resume) = &*parked; + reached.send(()).unwrap(); + resume.lock().unwrap().recv().unwrap(); + Ok(1) + }) + }) + .unwrap(); + + let admitted_at = h.controller.write_generation(); + let program = tokio::spawn({ + let conn = conn.clone(); + async move { run(&conn, &["select park()", "insert into t values (1)"]).await } + }); + reached_rx.recv().await.unwrap(); + assert_eq!(conn.fence.program_generation(), admitted_at); + + h.fence().await; + h.release().await; + assert!(h.controller.write_generation() > admitted_at); + resume_tx.send(()).unwrap(); + + let steps = program.await.unwrap(); + assert!(steps[0].is_ok()); + let e = fence_error(&steps[1]); + assert_eq!(e.outcome(), FenceOutcome::MigrationWriteFenced); + assert_eq!(e.detail(), Some(FenceDetail::StaleTransaction)); + assert_eq!(count(&conn).await, 0); + + // The next program is admitted under the current generation and writes. + assert_ok(&run(&conn, &["insert into t values (2)"]).await); + assert_eq!(count(&conn).await, 1); + } + + /// A read transaction opened before a transition cannot be upgraded to a write transaction + /// after it, whether the gate is still closed or open again. + #[tokio::test(flavor = "multi_thread")] + async fn fence_rejects_read_to_write_upgrade() { + let h = Harness::new().await; + let conn = h.conn().await; + + // Gate closed: refused before the write runs. + assert_ok(&run(&conn, &["begin", "select * from t"]).await); + h.fence().await; + let steps = run(&conn, &["insert into t values (1)"]).await; + assert_eq!( + fence_error(&steps[0]).outcome(), + FenceOutcome::MigrationWriteFenced + ); + + // Gate open again: the transaction still belongs to the old generation, and only the + // WAL gate can tell. + h.release().await; + let steps = run(&conn, &["insert into t values (1)"]).await; + let e = fence_error(&steps[0]); + assert_eq!(e.outcome(), FenceOutcome::MigrationWriteFenced); + assert_eq!(e.detail(), Some(FenceDetail::StaleTransaction)); + assert_ok(&run(&conn, &["rollback"]).await); + + // A fresh transaction writes. + assert_ok(&run(&conn, &["begin", "insert into t values (1)", "commit"]).await); + assert_eq!(count(&conn).await, 1); + } + + /// DDL, a pragma that writes the header, and `BEGIN IMMEDIATE` are refused while writes are + /// fenced; reads keep working, and nothing was written. + #[tokio::test(flavor = "multi_thread")] + async fn fence_rejects_ddl_and_pragma() { + let h = Harness::new().await; + let conn = h.conn().await; + h.fence().await; + + for stmt in [ + "create table u (x)", + "create index i on t (x)", + "drop table t", + "pragma user_version = 7", + "begin immediate", + ] { + let steps = run(&conn, &[stmt]).await; + assert_eq!( + fence_error(&steps[0]).outcome(), + FenceOutcome::MigrationWriteFenced, + "{stmt}" + ); + assert!(conn.inner.lock().is_autocommit(), "{stmt}"); + } + assert_ok(&run(&conn, &["select * from t"]).await); + + h.release().await; + let version: i64 = conn + .with_raw(|c| c.query_row("pragma user_version", (), |r| r.get(0))) + .unwrap(); + assert_eq!(version, 0); + let tables: i64 = conn + .with_raw(|c| { + c.query_row( + "select count(*) from sqlite_schema where name in ('u', 'i')", + (), + |r| r.get(0), + ) + }) + .unwrap(); + assert_eq!(tables, 0); + } + + /// `with_raw` users (admin shell, schema migration, dump load) bypass statement + /// classification but not the WAL: the write fails with `SQLITE_AUTH` and the typed reason + /// is in the connection's denial slot. + #[tokio::test(flavor = "multi_thread")] + async fn fence_rejects_raw_with_raw_write() { + let h = Harness::new().await; + let conn = h.conn().await; + h.fence().await; + + for sql in ["insert into t values (1)", "begin immediate", "vacuum"] { + let err = conn.with_raw(|c| c.execute_batch(sql)).unwrap_err(); + match err { + rusqlite::Error::SqliteFailure(e, _) => { + assert_eq!(e.code, ErrorCode::AuthorizationForStatementDenied, "{sql}") + } + e => panic!("{sql}: unexpected error {e}"), + } + assert_eq!( + conn.fence.take_denial().unwrap().outcome(), + FenceOutcome::MigrationWriteFenced, + "{sql}" + ); + } + assert_eq!(count(&conn).await, 0); + + h.release().await; + conn.with_raw(|c| c.execute_batch("insert into t values (1)")) + .unwrap(); + assert_eq!(count(&conn).await, 1); + } + + /// The fence acquired and released between a transaction's first read and its write: the + /// write is refused although the gate is open, while a connection that was idle across the + /// transition, and the same connection after a rollback, write normally. + #[tokio::test(flavor = "multi_thread")] + async fn stale_generation_cannot_write_after_release() { + let h = Harness::new().await; + let in_txn = h.conn().await; + let idle = h.conn().await; + + assert_ok(&run(&in_txn, &["begin", "select count(*) from t"]).await); + h.fence().await; + h.release().await; + assert!(h + .controller + .permits(crate::namespace::fence::state::OperationClass::NormalWrite) + .is_ok()); + + let steps = run(&in_txn, &["insert into t values (1)", "commit"]).await; + assert_eq!( + fence_error(&steps[0]).detail(), + Some(FenceDetail::StaleTransaction) + ); + // The commit that follows ends the transaction without having written anything. + assert!(steps[1].is_ok()); + assert!(in_txn.inner.lock().is_autocommit()); + assert_ok(&run(&idle, &["insert into t values (2)"]).await); + assert_ok(&run(&in_txn, &["insert into t values (3)"]).await); + assert_eq!(count(&idle).await, 2); + } + + /// A namespace that never had a fence behaves as before: generation 0 everywhere, every + /// kind of write works, and a plain `SQLITE_AUTH` from an authorizer is reported as the + /// SQLite error it is, not as a fence denial. + #[tokio::test(flavor = "multi_thread")] + async fn unfenced_namespace_is_unchanged() { + let h = Harness::new().await; + let conn = h.conn().await; + + assert_ok( + &run( + &conn, + &[ + "insert into t values (1)", + "begin", + "select * from t", + "insert into t values (2)", + "commit", + "create table u (x)", + "pragma user_version = 3", + ], + ) + .await, + ); + conn.with_raw(|c| c.execute_batch("insert into t values (3)")) + .unwrap(); + assert_eq!(count(&conn).await, 3); + assert_eq!(h.controller.write_generation(), 0); + assert_eq!(conn.fence.program_generation(), 0); + assert_eq!(conn.fence.txn_generation(), 0); + + conn.with_raw(|c| { + c.authorizer(Some(|ctx: rusqlite::hooks::AuthContext<'_>| { + match ctx.action { + rusqlite::hooks::AuthAction::Insert { .. } => { + rusqlite::hooks::Authorization::Deny + } + _ => rusqlite::hooks::Authorization::Allow, + } + })) + }); + let steps = run(&conn, &["insert into t values (4)"]).await; + match &steps[0] { + Err(Error::RusqliteErrorExtended(rusqlite::Error::SqliteFailure(e, _), _)) => { + assert_eq!(e.code, ErrorCode::AuthorizationForStatementDenied) + } + other => panic!("expected the authorizer's error, got {other:?}"), + } + assert!(conn.fence.take_denial().is_none()); + } + + /// Long enough that no test transaction is ever stolen by the manager's own timeout. + const LONG_TXN: Option = Some(Duration::from_secs(600)); + /// Upper bound for a test waiting on something that happens promptly; reaching it is a + /// failure, never the expected path. + const PROMPT: Duration = Duration::from_secs(30); + + /// Wait until `n` connections are parked in the write queue. This polls a condition; it + /// does not use elapsed time as evidence of anything. + async fn until_queued(h: &Harness, n: usize) { + tokio::time::timeout(PROMPT, async { + while h.maker.connection_manager().queued_writers() != n { + tokio::time::sleep(Duration::from_millis(1)).await; + } + }) + .await + .unwrap_or_else(|_| panic!("{n} connections never queued for the write slot")); + } + + fn freelist(conn: &Conn) -> i64 { + conn.with_raw(|c| c.query_row("pragma freelist_count", (), |r| r.get(0))) + .unwrap() + } + + /// A writer parked in the write queue behind an open transaction leaves the queue with the + /// typed denial as soon as the fence changes the write generation: it neither waits for the + /// slot nor, once the holder commits, writes. The holder itself, admitted before the fence, + /// keeps the slot and commits, which is what the positive drain waits for. + #[tokio::test(flavor = "multi_thread")] + async fn fence_rejects_queued_writer() { + let h = Harness::with_txn_timeout(LONG_TXN).await; + let manager = h.maker.connection_manager().clone(); + let holder = h.conn().await; + assert_ok(&run(&holder, &["begin immediate", "insert into t values (1)"]).await); + let (_, class) = manager + .active_writer() + .expect("the holder has the write slot"); + assert_eq!(class, OperationClass::NormalWrite); + + let queued = h.conn().await; + let waiting = tokio::spawn({ + let queued = queued.clone(); + async move { run(&queued, &["insert into t values (2)"]).await } + }); + until_queued(&h, 1).await; + assert!(!waiting.is_finished()); + + h.fence().await; + let steps = tokio::time::timeout(PROMPT, waiting) + .await + .expect("the queued writer was not woken by the fence") + .unwrap(); + assert_eq!( + fence_error(&steps[0]).outcome(), + FenceOutcome::MigrationWriteFenced + ); + assert_eq!(manager.queued_writers(), 0); + assert_eq!( + manager.active_writer().map(|(_, c)| c), + Some(OperationClass::NormalWrite) + ); + + assert_ok(&run(&holder, &["commit"]).await); + assert_eq!(manager.active_writer(), None); + assert_eq!(count(&holder).await, 1); + // The refused connection stays refused while the fence holds. + let steps = run(&queued, &["insert into t values (3)"]).await; + assert_eq!( + fence_error(&steps[0]).outcome(), + FenceOutcome::MigrationWriteFenced + ); + assert_eq!(count(&holder).await, 1); + } + + /// A checkpoint is maintenance: one queued behind an open transaction when the fence + /// changes queues again instead of being refused, and runs once the holder commits. + #[tokio::test(flavor = "multi_thread")] + async fn queued_checkpoint_survives_fence_wake() { + let h = Harness::with_txn_timeout(LONG_TXN).await; + let manager = h.maker.connection_manager().clone(); + let holder = h.conn().await; + assert_ok(&run(&holder, &["begin immediate", "insert into t values (1)"]).await); + + let checkpointer = h.conn().await; + let checkpoint = tokio::task::spawn_blocking({ + let inner = checkpointer.inner.clone(); + move || inner.lock().checkpoint() + }); + until_queued(&h, 1).await; + + h.fence().await; + // Woken by both transitions, it put itself back in the queue. + until_queued(&h, 1).await; + assert!(!checkpoint.is_finished()); + + assert_ok(&run(&holder, &["commit"]).await); + tokio::time::timeout(PROMPT, checkpoint) + .await + .expect("the checkpoint never got the slot") + .unwrap() + .unwrap(); + assert_eq!(manager.active_writer(), None); + assert_eq!(count(&holder).await, 1); + } + + /// `TRUNCATE` checkpoints run in every fence state. + #[tokio::test(flavor = "multi_thread")] + async fn checkpoint_allowed_while_fenced() { + let h = Harness::new().await; + let conn = h.conn().await; + assert_ok( + &run( + &conn, + &["insert into t values (1)", "insert into t values (2)"], + ) + .await, + ); + h.fence().await; + + conn.checkpoint().await.unwrap(); + let (busy, log, checkpointed): (i64, i64, i64) = conn + .with_raw(|c| { + c.query_row("pragma wal_checkpoint(truncate)", (), |r| { + Ok((r.get(0)?, r.get(1)?, r.get(2)?)) + }) + }) + .unwrap(); + assert_eq!((busy, log, checkpointed), (0, 0, 0)); + assert_eq!(count(&conn).await, 2); + } + + /// `VACUUM` is not maintenance. While normal writes are denied it is skipped, and reported as + /// skipped rather than failed; once writes are admitted again it runs. + #[tokio::test(flavor = "multi_thread")] + async fn vacuum_skipped_while_fenced() { + let h = Harness::new().await; + let conn = h.conn().await; + assert_ok( + &run( + &conn, + &[ + "insert into t select randomblob(4096) from \ + (with recursive n(i) as (select 1 union all select i + 1 from n where i < 200) \ + select i from n)", + "delete from t", + ], + ) + .await, + ); + let free = freelist(&conn); + assert!(free > 100, "freelist {free}"); + + h.fence().await; + let outcome = conn.inner.lock().vacuum_if_needed_above(0).unwrap(); + assert_eq!(outcome, VacuumOutcome::Fenced); + assert_eq!(freelist(&conn), free); + // The periodic path reports success, not a failed vacuum. + conn.vacuum_if_needed().await.unwrap(); + assert_eq!(freelist(&conn), free); + + h.release().await; + let outcome = conn.inner.lock().vacuum_if_needed_above(0).unwrap(); + assert_eq!(outcome, VacuumOutcome::Vacuumed); + assert_eq!(freelist(&conn), 0); + } + + /// `abort_active` rolls back the connection holding the write slot, which releases it and + /// notifies drain waiters; a rollback handle whose connection has already closed does + /// nothing instead of panicking, and there is nothing to abort without a writer. + #[tokio::test(flavor = "multi_thread")] + async fn abort_active_tolerates_closed_connection() { + let h = Harness::with_txn_timeout(LONG_TXN).await; + let manager = h.maker.connection_manager().clone(); + assert_eq!(manager.abort_active(), None); + + let holder = h.conn().await; + assert_ok(&run(&holder, &["begin immediate", "insert into t values (1)"]).await); + let (id, _) = manager.active_writer().unwrap(); + let released = manager.released().notified(); + tokio::pin!(released); + released.as_mut().enable(); + assert_eq!(manager.abort_active(), Some(id)); + tokio::time::timeout(PROMPT, released) + .await + .expect("the rollback did not notify drain waiters"); + assert_eq!(manager.active_writer(), None); + assert_eq!(count(&h.conn().await).await, 0); + + // A handle taken while its connection was open, used after the connection closed. + let closing = h.conn().await; + let closing_id = *manager.inner.abort_handle.lock().keys().max().unwrap(); + let handle = manager.inner.abort_handle.lock()[&closing_id].clone(); + drop(closing); + assert!(!manager.inner.abort_handle.lock().contains_key(&closing_id)); + handle.abort(); + assert_eq!(manager.abort_active(), None); + } + + /// Committing releases the slot and wakes a waiter that registered before it looked at the + /// active writer, which is the drain's wait (section 8.3 step 5). + #[tokio::test(flavor = "multi_thread")] + async fn release_notifies_drain_waiters() { + let h = Harness::with_txn_timeout(LONG_TXN).await; + let manager = h.maker.connection_manager().clone(); + let holder = h.conn().await; + assert_ok(&run(&holder, &["begin immediate", "insert into t values (1)"]).await); + h.fence().await; + + let released = manager.released().notified(); + tokio::pin!(released); + released.as_mut().enable(); + assert!(manager.active_writer().is_some()); + let commit = tokio::spawn({ + let holder = holder.clone(); + async move { run(&holder, &["commit"]).await } + }); + tokio::time::timeout(PROMPT, released) + .await + .expect("the commit did not notify drain waiters"); + assert_eq!(manager.active_writer(), None); + assert_ok(&commit.await.unwrap()); + assert_eq!(count(&holder).await, 1); + } +} diff --git a/libsql-server/src/connection/legacy.rs b/libsql-server/src/connection/legacy.rs index ae5addd70d..71763e199e 100644 --- a/libsql-server/src/connection/legacy.rs +++ b/libsql-server/src/connection/legacy.rs @@ -14,6 +14,8 @@ use tokio::time::Duration; use crate::error::Error; use crate::metrics::DESCRIBE_COUNT; use crate::namespace::broadcasters::BroadcasterHandle; +use crate::namespace::fence::controller::{FenceConnState, FenceController}; +use crate::namespace::fence::state::OperationClass; use crate::namespace::meta_store::MetaStoreHandle; use crate::namespace::ResolveNamespacePathFn; use crate::query_result_builder::{QueryBuilderConfig, QueryResultBuilder}; @@ -47,6 +49,9 @@ pub struct MakeLegacyConnection { block_writes: Arc, resolve_attach_path: ResolveNamespacePathFn, make_wal_manager: Arc InnerWalManager + Sync + Send + 'static>, + /// The namespace's fence controller. Every connection this maker opens, starting with the + /// held `_db` connection, carries a fence state bound to it. + fence: Arc, } impl MakeLegacyConnection @@ -69,8 +74,12 @@ where block_writes: Arc, resolve_attach_path: ResolveNamespacePathFn, make_wal_manager: Arc InnerWalManager + Sync + Send + 'static>, + fence: Arc, ) -> Result { let txn_timeout = config_store.get().txn_timeout.unwrap_or(TXN_TIMEOUT); + let connection_manager = ConnectionManager::new(txn_timeout); + // Queued writers re-check the fence whenever its write generation changes. + fence.register_write_queue(connection_manager.fence_waker()); let mut this = Self { db_path, @@ -87,8 +96,9 @@ where encryption_config, block_writes, resolve_attach_path, - connection_manager: ConnectionManager::new(txn_timeout), + connection_manager, make_wal_manager, + fence, }; let db = this.try_create_db().await?; @@ -97,6 +107,11 @@ where Ok(this) } + /// The write-slot manager shared by every connection this maker opens. + pub(crate) fn connection_manager(&self) -> &ConnectionManager { + &self.connection_manager + } + /// Tries to create a database, retrying if the database is busy. async fn try_create_db(&self) -> Result> { // try 100 times to acquire initial db connection. @@ -146,6 +161,7 @@ where self.resolve_attach_path.clone(), self.connection_manager.clone(), self.make_wal_manager.clone(), + FenceConnState::new(self.fence.clone(), OperationClass::NormalWrite), ) .await } @@ -165,6 +181,8 @@ where pub struct LegacyConnection { pub(super) inner: Arc>>>, + /// Shared with the connection's WAL wrapper and its `CoreConnection`. + pub(super) fence: Arc, } #[cfg(test)] @@ -185,6 +203,10 @@ impl LegacyConnection { Arc::new(|_| unreachable!()), ConnectionManager::new(TXN_TIMEOUT), Arc::new(|| Sqlite3WalManager::default()), + FenceConnState::new( + FenceController::unfenced(Default::default()), + OperationClass::NormalWrite, + ), ) .await .unwrap() @@ -195,6 +217,7 @@ impl Clone for LegacyConnection { fn clone(&self) -> Self { Self { inner: self.inner.clone(), + fence: self.fence.clone(), } } } @@ -321,11 +344,13 @@ where resolve_attach_path: ResolveNamespacePathFn, connection_manager: ConnectionManager, make_wal: Arc InnerWalManager + Sync + Send + 'static>, + fence: Arc, ) -> crate::Result { let (conn, id) = tokio::task::spawn_blocking({ let connection_manager = connection_manager.clone(); + let fence = fence.clone(); move || -> crate::Result<_> { - let manager = ManagedConnectionWalWrapper::new(connection_manager); + let manager = ManagedConnectionWalWrapper::new(connection_manager, fence.clone()); let id = manager.id(); let wal = make_wal().wrap(manager).wrap(wal_wrapper); @@ -340,6 +365,7 @@ where current_frame_no_receiver, block_writes, resolve_attach_path, + fence, )?; let namespace = path @@ -366,7 +392,7 @@ where connection_manager.register_connection(&inner, id); - Ok(Self { inner }) + Ok(Self { inner, fence }) } pub async fn execute( @@ -445,6 +471,9 @@ where fn with_raw(&self, f: impl FnOnce(&mut rusqlite::Connection) -> R) -> R { let mut inner = self.inner.lock(); + // A raw use of the connection is a program like any other: the WAL gate admits a write + // transaction it opens only under the generation it started under. + self.fence.begin_program(); f(inner.raw_mut()) } } diff --git a/libsql-server/src/connection/program.rs b/libsql-server/src/connection/program.rs index 4d5ada51ff..8cafe681e7 100644 --- a/libsql-server/src/connection/program.rs +++ b/libsql-server/src/connection/program.rs @@ -7,6 +7,7 @@ use rusqlite::StatementStatus; use crate::auth::Permission; use crate::error::Error; use crate::metrics::{READ_QUERY_COUNT, WRITE_QUERY_COUNT}; +use crate::namespace::fence::controller::FenceConnState; use crate::namespace::{NamespaceName, ResolveNamespacePathFn}; use crate::query::Query; use crate::query_analysis::StmtKind; @@ -104,6 +105,7 @@ pub struct Vm<'a, B, F, S> { should_block: F, update_stats: S, resolve_attach_path: ResolveNamespacePathFn, + fence: Option>, } impl<'a, B, F, S> Vm<'a, B, F, S> @@ -127,9 +129,19 @@ where should_block, update_stats, resolve_attach_path, + fence: None, } } + /// Check the namespace fence on this connection: a statement that can write is refused + /// before it runs while the live gate denies the connection's class, and a refusal by the + /// WAL gate is reported as the fence error rather than as `SQLITE_AUTH` + /// (`docs/NAMESPACE_FENCE.md` section 8.1). + pub fn with_fence(mut self, fence: Arc) -> Self { + self.fence = Some(fence); + self + } + #[inline] fn current_step(&self) -> &Step { &self.program.steps()[self.current_step] @@ -164,7 +176,9 @@ where // builder error interrupt the execution of query. we should exit immediately. Err(e @ Error::BuilderError(_)) => return Err(e), Err(mut e) => { - if let Error::RusqliteError(err) = e { + if let Some(denial) = self.take_fence_denial(&e) { + e = Error::NamespaceFence(denial); + } else if let Error::RusqliteError(err) = e { let extended_code = unsafe { rusqlite::ffi::sqlite3_extended_errcode(conn.handle()) }; @@ -200,12 +214,39 @@ where Ok(query) } + /// The fence denial behind `e`, when `e` is the `SQLITE_AUTH` the WAL gate returns. A plain + /// `SQLITE_AUTH` (from an authorizer) is left alone, and so is an empty denial slot. + fn take_fence_denial(&self, e: &Error) -> Option { + let fence = self.fence.as_ref()?; + match e { + Error::RusqliteError(rusqlite::Error::SqliteFailure( + rusqlite::ffi::Error { + code: rusqlite::ErrorCode::AuthorizationForStatementDenied, + .. + }, + _, + )) => fence.take_denial(), + _ => None, + } + } + fn execute_query(&mut self, conn: &rusqlite::Connection) -> crate::Result<(u64, Option)> { tracing::debug!("executing query: {}", self.current_step().query.stmt.stmt); increment_counter!("libsql_server_libsql_query_execute"); let start = Instant::now(); + // Early fence admission for statements that can write. The WAL gate is authoritative + // and catches everything this misses (a misclassified statement, a read-to-write + // upgrade, a gate change while the statement runs). + if let Some(fence) = &self.fence { + if matches!( + self.current_step().query.stmt.kind, + StmtKind::Write | StmtKind::DDL + ) { + fence.controller().permits(fence.class())?; + } + } let (blocked, reason) = (self.should_block)(&self.current_step().query.stmt.kind); if blocked { return Err(Error::Blocked(reason)); diff --git a/libsql-server/src/generated/namespace_fence.rs b/libsql-server/src/generated/namespace_fence.rs index dcbbcafe96..7012a0dbee 100644 --- a/libsql-server/src/generated/namespace_fence.rs +++ b/libsql-server/src/generated/namespace_fence.rs @@ -12,8 +12,9 @@ pub struct DrainPolicy { pub struct FrozenBoundary { #[prost(string, tag = "1")] pub log_id: ::prost::alloc::string::String, - #[prost(uint64, tag = "2")] - pub frame_no: u64, + /// Absent when the replication log has no frames. + #[prost(uint64, optional, tag = "2")] + pub frame_no: ::core::option::Option, } #[allow(clippy::derive_partial_eq_without_eq)] #[derive(Clone, PartialEq, ::prost::Message)] diff --git a/libsql-server/src/main.rs b/libsql-server/src/main.rs index 5d738a1ed5..a1ab51c4c4 100644 --- a/libsql-server/src/main.rs +++ b/libsql-server/src/main.rs @@ -268,6 +268,12 @@ struct Cli { #[clap(long, env = "SQLD_NAMESPACE_FENCE_RECEIPT_RETENTION_S")] namespace_fence_receipt_retention_s: Option, + /// How long, in milliseconds, acquiring a namespace write fence waits for active writers + /// when the request names no drain policy (the deadline then answers `DRAINING`). + /// Defaults to 30 seconds. + #[clap(long, env = "SQLD_NAMESPACE_FENCE_DEFAULT_WRITE_DRAIN_MS")] + namespace_fence_default_write_drain_ms: Option, + /// Shutdown timeout duration in seconds, defaults to 30 seconds. #[clap(long, env = "SQLD_SHUTDOWN_TIMEOUT")] shutdown_timeout: Option, @@ -664,6 +670,9 @@ fn make_meta_store_config(config: &Cli) -> anyhow::Result { namespace_fence_receipt_retention: config .namespace_fence_receipt_retention_s .map(Duration::from_secs), + namespace_fence_default_write_drain: config + .namespace_fence_default_write_drain_ms + .map(Duration::from_millis), }) } diff --git a/libsql-server/src/namespace/configurator/helpers.rs b/libsql-server/src/namespace/configurator/helpers.rs index 599320783d..a10ef89d6b 100644 --- a/libsql-server/src/namespace/configurator/helpers.rs +++ b/libsql-server/src/namespace/configurator/helpers.rs @@ -24,6 +24,7 @@ use crate::connection::{Connection as _, MakeConnection, MakeThrottledConnection use crate::database::{PrimaryConnection, PrimaryConnectionMaker}; use crate::error::LoadDumpError; use crate::namespace::broadcasters::BroadcasterHandle; +use crate::namespace::fence::controller::{FenceController, WriteDrainSource}; use crate::namespace::meta_store::MetaStoreHandle; use crate::namespace::replication_wal::{make_replication_wal_wrapper, ReplicationWalWrapper}; use crate::namespace::{ @@ -50,6 +51,7 @@ pub(super) async fn make_primary_connection_maker( broadcaster: BroadcasterHandle, make_wal_manager: Arc InnerWalManager + Sync + Send + 'static>, encryption_config: Option, + fence: Arc, ) -> crate::Result<( Arc, ReplicationWalWrapper, @@ -158,25 +160,33 @@ pub(super) async fn make_primary_connection_maker( let rcv = logger.new_frame_notifier.subscribe(); move || *rcv.borrow() }); + let legacy_maker = MakeLegacyConnection::new( + db_path.to_path_buf(), + wal_wrapper.clone(), + stats.clone(), + broadcaster, + meta_store_handle.clone(), + base_config.extensions.clone(), + base_config.max_response_size, + base_config.max_total_response_size, + auto_checkpoint, + get_current_frame_no.clone(), + encryption_config, + block_writes, + resolve_attach_path, + make_wal_manager.clone(), + fence.clone(), + ) + .await?; + // The positive write drain waits on this maker's write slot and reads the frozen boundary + // from its replication log (`docs/NAMESPACE_FENCE.md` section 8.3). + fence.register_write_drain(WriteDrainSource::new( + legacy_maker.connection_manager(), + logger.log_id(), + get_current_frame_no, + )); let connection_maker = Arc::new( - MakeLegacyConnection::new( - db_path.to_path_buf(), - wal_wrapper.clone(), - stats.clone(), - broadcaster, - meta_store_handle.clone(), - base_config.extensions.clone(), - base_config.max_response_size, - base_config.max_total_response_size, - auto_checkpoint, - get_current_frame_no, - encryption_config, - block_writes, - resolve_attach_path, - make_wal_manager.clone(), - ) - .await? - .throttled( + legacy_maker.throttled( base_config.max_concurrent_connections.clone(), base_config .connection_creation_timeout diff --git a/libsql-server/src/namespace/configurator/mod.rs b/libsql-server/src/namespace/configurator/mod.rs index 517b21ca5a..029ab0b3ce 100644 --- a/libsql-server/src/namespace/configurator/mod.rs +++ b/libsql-server/src/namespace/configurator/mod.rs @@ -13,6 +13,7 @@ use crate::replication::script_backup_manager::ScriptBackupManager; use crate::StatsSender; use super::broadcasters::BroadcasterHandle; +use super::fence::controller::FenceController; use super::meta_store::MetaStoreHandle; use super::{ Namespace, NamespaceBottomlessDbIdInit, NamespaceName, NamespaceStore, ResetCb, @@ -119,6 +120,7 @@ pub trait ConfigureNamespace { resolve_attach_path: ResolveNamespacePathFn, store: NamespaceStore, broadcaster: BroadcasterHandle, + fence: Arc, ) -> Pin> + Send + 'a>>; fn cleanup<'a>( diff --git a/libsql-server/src/namespace/configurator/primary.rs b/libsql-server/src/namespace/configurator/primary.rs index f68405fad6..7b63671ed1 100644 --- a/libsql-server/src/namespace/configurator/primary.rs +++ b/libsql-server/src/namespace/configurator/primary.rs @@ -13,6 +13,7 @@ use crate::connection::{Connection as _, MakeConnection}; use crate::database::{Database, PrimaryDatabase}; use crate::namespace::broadcasters::BroadcasterHandle; use crate::namespace::configurator::helpers::make_primary_connection_maker; +use crate::namespace::fence::controller::FenceController; use crate::namespace::meta_store::MetaStoreHandle; use crate::namespace::{ Namespace, NamespaceBottomlessDbIdInit, NamespaceName, NamespaceStore, ResetCb, @@ -53,6 +54,7 @@ impl PrimaryConfigurator { db_path: Arc, broadcaster: BroadcasterHandle, encryption_config: Option, + fence: Arc, ) -> crate::Result { let mut join_set = JoinSet::new(); @@ -72,6 +74,7 @@ impl PrimaryConfigurator { broadcaster, self.make_wal_manager.clone(), encryption_config, + fence.clone(), ) .await?; @@ -112,6 +115,7 @@ impl PrimaryConfigurator { stats, db_config_store: meta_store_handle, path: db_path.into(), + fence, }) } } @@ -126,6 +130,7 @@ impl ConfigureNamespace for PrimaryConfigurator { resolve_attach_path: ResolveNamespacePathFn, _store: NamespaceStore, broadcaster: BroadcasterHandle, + fence: Arc, ) -> Pin> + Send + 'a>> { Box::pin(async move { let db_path: Arc = self.base.base_path.join("dbs").join(name.as_str()).into(); @@ -140,6 +145,7 @@ impl ConfigureNamespace for PrimaryConfigurator { db_path.clone(), broadcaster, self.base.encryption_config.clone(), + fence, ) .await { diff --git a/libsql-server/src/namespace/configurator/replica.rs b/libsql-server/src/namespace/configurator/replica.rs index b1a108af73..adea0fd406 100644 --- a/libsql-server/src/namespace/configurator/replica.rs +++ b/libsql-server/src/namespace/configurator/replica.rs @@ -18,6 +18,7 @@ use crate::connection::MakeConnection; use crate::database::{Database, ReplicaDatabase}; use crate::namespace::broadcasters::BroadcasterHandle; use crate::namespace::configurator::helpers::{make_stats, run_storage_monitor}; +use crate::namespace::fence::controller::FenceController; use crate::namespace::meta_store::MetaStoreHandle; use crate::namespace::{Namespace, NamespaceBottomlessDbIdInit, RestoreOption}; use crate::namespace::{NamespaceName, NamespaceStore, ResetCb, ResetOp, ResolveNamespacePathFn}; @@ -60,6 +61,7 @@ impl ConfigureNamespace for ReplicaConfigurator { resolve_attach_path: ResolveNamespacePathFn, store: NamespaceStore, broadcaster: BroadcasterHandle, + fence: Arc, ) -> Pin> + Send + 'a>> { Box::pin(async move { tracing::debug!("creating replica namespace"); @@ -104,6 +106,7 @@ impl ConfigureNamespace for ReplicaConfigurator { resolve_attach_path, store, broadcaster, + fence, ) .await; } @@ -220,6 +223,7 @@ impl ConfigureNamespace for ReplicaConfigurator { Arc::new(AtomicBool::new(false)), // this is always false for write proxy resolve_attach_path, self.make_wal_manager.clone(), + fence.clone(), ) .await?; @@ -274,6 +278,7 @@ impl ConfigureNamespace for ReplicaConfigurator { stats, db_config_store: meta_store_handle, path: db_path.into(), + fence, }) }) } diff --git a/libsql-server/src/namespace/configurator/schema.rs b/libsql-server/src/namespace/configurator/schema.rs index 275fd71e93..411ec91271 100644 --- a/libsql-server/src/namespace/configurator/schema.rs +++ b/libsql-server/src/namespace/configurator/schema.rs @@ -7,6 +7,7 @@ use crate::connection::config::DatabaseConfig; use crate::connection::connection_manager::InnerWalManager; use crate::database::{Database, SchemaDatabase}; use crate::namespace::broadcasters::BroadcasterHandle; +use crate::namespace::fence::controller::FenceController; use crate::namespace::meta_store::MetaStoreHandle; use crate::namespace::{ Namespace, NamespaceName, NamespaceStore, ResetCb, ResolveNamespacePathFn, RestoreOption, @@ -49,6 +50,7 @@ impl ConfigureNamespace for SchemaConfigurator { resolve_attach_path: ResolveNamespacePathFn, _store: NamespaceStore, broadcaster: BroadcasterHandle, + fence: Arc, ) -> std::pin::Pin> + Send + 'a>> { Box::pin(async move { let mut join_set = JoinSet::new(); @@ -69,6 +71,7 @@ impl ConfigureNamespace for SchemaConfigurator { broadcaster, self.make_wal_manager.clone(), self.base.encryption_config.clone(), + fence.clone(), ) .await?; @@ -90,6 +93,7 @@ impl ConfigureNamespace for SchemaConfigurator { stats, db_config_store: db_config.clone(), path: db_path.into(), + fence, }) }) } diff --git a/libsql-server/src/namespace/fence/controller.rs b/libsql-server/src/namespace/fence/controller.rs new file mode 100644 index 0000000000..7e718e5516 --- /dev/null +++ b/libsql-server/src/namespace/fence/controller.rs @@ -0,0 +1,1143 @@ +//! The per-namespace fence controller (`docs/NAMESPACE_FENCE.md` sections 7 and 8.4). +//! +//! A [`FenceController`] is the in-memory authority for one namespace's fence. It owns the +//! transition lock that serialises fence commands on the namespace, and the gate that every +//! admission path reads: a `watch` of [`GateSnapshot`]. The gate changes only after the +//! metastore has committed (commit → publish → respond), except that a commit whose outcome is +//! unknown closes it until the same command is replayed. +//! +//! Controllers live in the [`FenceRegistry`](super::registry::FenceRegistry), not in the +//! namespace cache, so evicting and reloading a namespace hands the reloaded namespace the +//! same controller. + +use std::sync::atomic::{AtomicU64, Ordering}; +use std::sync::Arc; + +use parking_lot::Mutex; +use tokio::sync::{watch, OwnedMutexGuard}; +use uuid::Uuid; + +use crate::connection::connection_manager::{ConnectionManager, WeakConnectionManager}; +use crate::error::Error; +use crate::namespace::meta_store::{FenceCommit, FenceContext, MetaStore}; +use crate::namespace::NamespaceName; +use crate::replication::FrameNo; + +use super::command::FenceRequest; +#[cfg(test)] +use super::hooks::FenceTestHooks; +use super::hooks::{HookOutcome, HookPoint}; +use super::outcome::{FenceDetail, FenceError, FenceOutcome}; +use super::state::{Admission, FenceState, OperationClass}; +use super::store::StoredFence; +use super::transition::DrainCompletion; + +/// `(operation_id, command_id)` of a fence command. +pub type CommandKey = (Uuid, Uuid); + +/// What every admission path reads: the fence as last published, the write-admission +/// generation, and whether a commit is indeterminate. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct GateSnapshot { + /// The durable fence as of the last publication (or as loaded at startup). + pub fence: StoredFence, + /// Incremented on every published change of the fence state, of its owning operation, or + /// of the indeterminate flag. A write transaction may only start when the generation its + /// program and its read transaction were admitted under equals this one (section 8.1). + pub write_generation: u64, + /// A command whose commit outcome is unknown. While set, every class except maintenance + /// and observability is denied, and every other command is refused. + pub indeterminate: Option, + /// The in-memory `INSTALLING` gate of a closing transition that is being persisted + /// (section 8.3, step 2): write admission is closed on top of whatever `fence` allows. + /// Never persisted. + pub installing: Option, +} + +impl GateSnapshot { + fn new(fence: StoredFence) -> Self { + Self { + fence, + write_generation: 0, + indeterminate: None, + installing: None, + } + } + + pub fn state(&self) -> FenceState { + self.fence.state() + } + + pub fn revision(&self) -> u64 { + self.fence.revision() + } + + pub fn operation_id(&self) -> Option { + self.fence.record().map(|r| r.operation_id) + } + + pub fn is_unavailable(&self) -> bool { + matches!(self.fence, StoredFence::Unavailable { .. }) + } + + /// The gate's decision for work of `class`. + pub fn permits(&self, class: OperationClass) -> Result<(), FenceError> { + if let Some((operation_id, command_id)) = self.indeterminate { + if !matches!( + class, + OperationClass::Maintenance | OperationClass::Observability + ) { + return Err(FenceError::new( + FenceOutcome::FenceStateUnavailable, + format!( + "the outcome of fence command {command_id} of operation {operation_id} \ + is not known yet" + ), + ) + .with_detail(FenceDetail::IndeterminateCommit)); + } + } + self.fence.permits(class)?; + if let Some((operation_id, command_id)) = self.installing { + if matches!( + class, + OperationClass::NormalWrite + | OperationClass::Vacuum + | OperationClass::CapabilityImport + | OperationClass::Lifecycle + ) { + return Err(FenceError::new( + FenceOutcome::MigrationWriteFenced, + format!( + "{class:?} is not permitted: fence command {command_id} of operation \ + {operation_id} is closing write admission" + ), + )); + } + } + Ok(()) + } + + /// Whether a closing transition is being installed. + pub fn is_installing(&self) -> bool { + self.installing.is_some() + } + + /// Normal write admission. + pub fn write(&self) -> Admission { + Admission::from( + self.permits(OperationClass::NormalWrite) + .map_err(|e| e.outcome()), + ) + } + + /// Normal read admission. + pub fn read(&self) -> Admission { + Admission::from( + self.permits(OperationClass::NormalRead) + .map_err(|e| e.outcome()), + ) + } +} + +/// Wakes one connection manager's write queue after a write-generation change; returns `false` +/// once the manager is gone. +pub type WriteQueueWaker = Box bool + Send + Sync>; + +/// The last frame committed to a namespace's replication log, `None` while it has none. +pub type GetCurrentFrameNo = Arc Option + Send + Sync + 'static>; + +/// What the write drain needs from one primary connection maker of the namespace: its +/// connection manager (held weakly, so an evicted namespace's manager goes away with it) and +/// its replication log. +pub struct WriteDrainSource { + pub(crate) manager: WeakConnectionManager, + pub(crate) log_id: Uuid, + pub(crate) current_frame_no: GetCurrentFrameNo, +} + +impl WriteDrainSource { + pub(crate) fn new( + manager: &ConnectionManager, + log_id: Uuid, + current_frame_no: GetCurrentFrameNo, + ) -> Self { + Self { + manager: manager.downgrade(), + log_id, + current_frame_no, + } + } +} + +/// A [`WriteDrainSource`] whose manager is alive, held for the length of a drain. +pub(crate) struct LiveWriteDrain { + pub(crate) manager: ConnectionManager, + pub(crate) log_id: Uuid, + pub(crate) current_frame_no: GetCurrentFrameNo, +} + +/// The fence controller of one namespace. +pub struct FenceController { + namespace: NamespaceName, + transition_lock: Arc>, + gate: watch::Sender, + /// The write queues of the namespace's connection managers (section 8.2). + write_queues: Mutex>, + /// What the write drain needs from each of the namespace's primary connection makers + /// (section 8.3). + write_drains: Mutex>, + #[cfg(test)] + hooks: FenceTestHooks, +} + +impl std::fmt::Debug for FenceController { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("FenceController") + .field("namespace", &self.namespace) + .field("gate", &*self.gate.borrow()) + .finish_non_exhaustive() + } +} + +impl FenceController { + /// A controller whose gate starts from `fence`, as established by the metastore. + pub fn new(namespace: NamespaceName, fence: StoredFence) -> Arc { + let (gate, _) = watch::channel(GateSnapshot::new(fence)); + Arc::new(Self { + namespace, + transition_lock: Default::default(), + gate, + write_queues: Mutex::new(Vec::new()), + write_drains: Mutex::new(Vec::new()), + #[cfg(test)] + hooks: FenceTestHooks::default(), + }) + } + + /// A controller for a namespace without fence state. + pub fn unfenced(namespace: NamespaceName) -> Arc { + Self::new( + namespace, + StoredFence::None { + namespace_exists: true, + }, + ) + } + + pub fn namespace(&self) -> &NamespaceName { + &self.namespace + } + + /// A copy of the current gate. + pub fn gate(&self) -> GateSnapshot { + self.gate.borrow().clone() + } + + /// A receiver that observes every publication. + pub fn subscribe(&self) -> watch::Receiver { + self.gate.subscribe() + } + + pub fn write_generation(&self) -> u64 { + self.gate.borrow().write_generation + } + + /// The live gate's decision for work of `class`. + pub fn permits(&self, class: OperationClass) -> Result<(), FenceError> { + self.gate.borrow().permits(class) + } + + /// Register a connection manager's write queue, to be woken after every change of the write + /// generation so that queued writers re-check the gate instead of waiting for the slot. + /// Wakers whose manager is gone are dropped on the next change. + pub fn register_write_queue(&self, waker: WriteQueueWaker) { + self.write_queues.lock().push(waker); + } + + /// Register what the write drain needs from a primary connection maker of this namespace: + /// its connection manager and its replication log. Sources whose manager is gone are + /// dropped the next time the drain looks. + pub fn register_write_drain(&self, source: WriteDrainSource) { + self.write_drains.lock().push(source); + } + + /// The write-drain sources whose manager is still alive, oldest first. The drain holds them + /// (and so their managers) for as long as it runs. + pub(crate) fn live_write_drains(&self) -> Vec { + let mut sources = self.write_drains.lock(); + let mut live = Vec::with_capacity(sources.len()); + sources.retain(|source| match source.manager.upgrade() { + Some(manager) => { + live.push(LiveWriteDrain { + manager, + log_id: source.log_id, + current_frame_no: source.current_frame_no.clone(), + }); + true + } + None => false, + }); + live + } + + /// Take the namespace's transition lock. Every fence command on the namespace runs while + /// holding it, from its first check to its response. + pub async fn begin_transition(self: &Arc) -> Transition { + let guard = self.transition_lock.clone().lock_owned().await; + Transition { + controller: self.clone(), + _guard: guard, + } + } + + /// Run one fence command to completion: take the transition lock, commit it in the + /// metastore, publish the result to the gate and return it. + /// + /// The work runs on its own task, so a caller that goes away (a lost response) does not + /// stop the publication of a command that committed. + pub async fn apply_command( + self: &Arc, + meta: &MetaStore, + request: FenceRequest, + ctx: FenceContext, + ) -> crate::Result { + let this = self.clone(); + let meta = meta.clone(); + tokio::spawn(async move { + let mut transition = this.begin_transition().await; + transition.apply(&meta, request, ctx).await + }) + .await? + } + + #[cfg(test)] + pub fn hooks(&self) -> &FenceTestHooks { + &self.hooks + } + + /// Reach a test hook point. Outside the library's own test build this does nothing. + #[cfg(test)] + pub(crate) async fn hook(&self, point: HookPoint) -> HookOutcome { + self.hooks.hit(point).await + } + + #[cfg(not(test))] + #[inline(always)] + pub(crate) async fn hook(&self, _point: HookPoint) -> HookOutcome { + HookOutcome::Continue + } + + /// Publish a new gate. `fence: None` keeps the published fence. The write generation moves + /// whenever the state, the owning operation, the indeterminate flag or the installing gate + /// changes. + fn publish( + &self, + fence: Option, + indeterminate: Option, + installing: Option, + ) { + let mut generation_changed = false; + self.gate.send_modify(|gate| { + let fence = fence.unwrap_or_else(|| gate.fence.clone()); + let changed = fence.state() != gate.fence.state() + || fence.record().map(|r| r.operation_id) + != gate.fence.record().map(|r| r.operation_id) + || indeterminate != gate.indeterminate + || installing != gate.installing; + gate.fence = fence; + gate.indeterminate = indeterminate; + gate.installing = installing; + if changed { + gate.write_generation += 1; + generation_changed = true; + } + }); + { + let gate = self.gate.borrow(); + tracing::debug!( + namespace = %self.namespace, + state = %gate.state(), + revision = gate.revision(), + write_generation = gate.write_generation, + indeterminate = gate.indeterminate.is_some(), + installing = gate.installing.is_some(), + "published namespace fence gate" + ); + } + // After the gate is published, so that every woken writer re-checks against it. + if generation_changed { + self.write_queues.lock().retain(|wake| wake()); + } + } +} + +/// A fence command in progress on one namespace. Holds the namespace's transition lock until +/// dropped. +pub struct Transition { + controller: Arc, + _guard: OwnedMutexGuard<()>, +} + +impl Transition { + pub fn controller(&self) -> &Arc { + &self.controller + } + + /// Publish the in-memory `INSTALLING` gate for the closing command `key` (section 8.3, + /// step 2): write admission closes and the write generation moves, which wakes the write + /// queues. It is replaced by whatever the command's commit publishes, or removed with + /// [`remove_installing`](Self::remove_installing) when the command is proven not to have + /// committed. + pub fn install_closing_gate(&mut self, key: CommandKey) { + let indeterminate = self.controller.gate.borrow().indeterminate; + self.controller.publish(None, indeterminate, Some(key)); + } + + /// Remove the `INSTALLING` gate of a command that was proven not to have committed. The + /// write generation moves again, so nothing admitted before it closed can write. + pub fn remove_installing(&mut self) { + let (indeterminate, installing) = { + let gate = self.controller.gate.borrow(); + (gate.indeterminate, gate.installing) + }; + if installing.is_some() { + self.controller.publish(None, indeterminate, None); + } + } + + /// Commit `request` in the metastore and publish the result. + pub async fn apply( + &mut self, + meta: &MetaStore, + request: FenceRequest, + ctx: FenceContext, + ) -> crate::Result { + debug_assert_eq!(&request.namespace, self.controller.namespace()); + let key = (request.operation_id, request.command_id); + self.commit( + key, + async move { meta.apply_fence_command(request, ctx).await }, + ) + .await + } + + /// Complete the drain that the `DRAINING` receipt `key` started, once the caller has + /// proven `completion`, and publish the result. + pub async fn complete_drain( + &mut self, + meta: &MetaStore, + key: CommandKey, + completion: DrainCompletion, + ctx: FenceContext, + ) -> crate::Result { + let namespace = self.controller.namespace().clone(); + self.commit(key, async move { + meta.complete_fence_drain(namespace, key.0, key.1, completion, ctx) + .await + }) + .await + } + + /// The commit → publish → respond sequence of section 8.4. + /// + /// - An error that proves nothing was committed leaves the gate exactly as it was. + /// - A commit whose outcome is unknown closes the gate (every class but maintenance and + /// observability) and marks `key` indeterminate: other commands get + /// `FENCE_COMMIT_INDETERMINATE` until `key` is replayed, and the replay, which the + /// metastore answers from the durable row, reopens it to whatever is durable. + /// - A commit is published before it is answered. + async fn commit(&mut self, key: CommandKey, run: F) -> crate::Result + where + F: std::future::Future>, + { + let controller = self.controller.clone(); + if let Some(pending) = controller.gate.borrow().indeterminate { + if pending != key { + return Err(pending_indeterminate(pending).into()); + } + } + + let result = match controller.hook(HookPoint::BeforeMetastoreCommit).await { + HookOutcome::Continue => run.await, + HookOutcome::Fail(e) => return Err(e.into()), + // A commit that failed without applying, but whose outcome the controller cannot + // know (test hook). + HookOutcome::Indeterminate => { + Err(indeterminate(key, "the commit was not acknowledged (test hook)").into()) + } + }; + + let result = match result { + Ok(commit) => match controller.hook(HookPoint::AfterMetastoreCommit).await { + HookOutcome::Continue => Ok(commit), + HookOutcome::Indeterminate | HookOutcome::Fail(_) => Err(indeterminate( + key, + "the commit was not acknowledged (test hook)", + )), + }, + Err(e) if is_indeterminate(&e) => Err(indeterminate(key, &e.to_string())), + Err(e) => return Err(e), + }; + + match result { + Ok(commit) => { + let _ = controller.hook(HookPoint::BeforeGatePublish).await; + controller.publish(commit.record.clone().map(StoredFence::Record), None, None); + let _ = controller.hook(HookPoint::BeforeResponse).await; + Ok(commit) + } + Err(e) => { + tracing::error!( + namespace = %controller.namespace, + operation_id = %key.0, + command_id = %key.1, + "fence commit outcome unknown; the namespace stays closed until the command \ + is replayed: {e}" + ); + controller.publish(None, Some(key), None); + Err(e.into()) + } + } + } +} + +/// Whether a metastore error leaves the commit's outcome unknown: the commit itself failed, or +/// the task running it died. +fn is_indeterminate(e: &Error) -> bool { + match e { + Error::NamespaceFence(f) => f.outcome() == FenceOutcome::FenceCommitIndeterminate, + Error::RuntimeTaskJoinError(_) => true, + _ => false, + } +} + +fn indeterminate(key: CommandKey, why: &str) -> FenceError { + FenceError::new( + FenceOutcome::FenceCommitIndeterminate, + format!( + "whether fence command {} of operation {} was committed is unknown ({why}); replay \ + the same command to reconcile", + key.1, key.0 + ), + ) + .with_detail(FenceDetail::IndeterminateCommit) +} + +fn pending_indeterminate((operation_id, command_id): CommandKey) -> FenceError { + FenceError::new( + FenceOutcome::FenceCommitIndeterminate, + format!( + "fence command {command_id} of operation {operation_id} has an unknown outcome; \ + only a replay of that command is accepted until it is reconciled" + ), + ) + .with_detail(FenceDetail::IndeterminateCommit) +} + +/// The fence state of one connection, shared by its WAL wrapper and its `CoreConnection` +/// (section 7.4). +/// +/// - A program (a `CoreConnection::run`, a `with_raw` call, a vacuum) starts with +/// [`begin_program`](Self::begin_program), which records the generation it was admitted under. +/// - The WAL wrapper calls [`begin_read_txn`](Self::begin_read_txn) whenever SQLite opens a read +/// transaction, and [`admit_write`](Self::admit_write) in `begin_write_txn` before it queues for +/// the write slot. A refusal leaves its typed outcome in the denial slot, which the program +/// layer takes to report the fence error instead of the bare `SQLITE_AUTH` the WAL returns. +#[derive(Debug)] +pub struct FenceConnState { + controller: Arc, + class: OperationClass, + /// The write generation the current program was admitted under. + program_generation: AtomicU64, + /// The write generation the current read transaction was opened under. + txn_generation: AtomicU64, + /// The typed outcome of the last refusal at the WAL. + denial: Mutex>, +} + +impl FenceConnState { + pub fn new(controller: Arc, class: OperationClass) -> Arc { + let generation = controller.write_generation(); + Arc::new(Self { + controller, + class, + program_generation: AtomicU64::new(generation), + txn_generation: AtomicU64::new(generation), + denial: Mutex::new(None), + }) + } + + pub fn controller(&self) -> &Arc { + &self.controller + } + + pub fn class(&self) -> OperationClass { + self.class + } + + pub fn program_generation(&self) -> u64 { + self.program_generation.load(Ordering::Acquire) + } + + pub fn txn_generation(&self) -> u64 { + self.txn_generation.load(Ordering::Acquire) + } + + /// Start a program on this connection: record the generation it is admitted under and + /// forget any denial a previous program left behind. Returns that generation. + pub fn begin_program(&self) -> u64 { + let generation = self.controller.write_generation(); + self.program_generation.store(generation, Ordering::Release); + *self.denial.lock() = None; + generation + } + + /// SQLite is opening a new read transaction on this connection: record the generation it + /// is opened under. A later upgrade of that transaction to a write transaction must happen + /// under the same generation. + pub fn begin_read_txn(&self) { + self.txn_generation + .store(self.controller.write_generation(), Ordering::Release); + } + + /// The authoritative write admission (section 8.1, check 2): the live gate permits this + /// connection's class, and the program and its read transaction were both admitted under + /// the gate's current write generation. On refusal the typed outcome is left in the denial + /// slot and returned. + pub fn admit_write(&self) -> Result<(), FenceError> { + let result = { + let gate = self.controller.gate.borrow(); + gate.permits(self.class).and_then(|()| { + let current = gate.write_generation; + let program = self.program_generation(); + let txn = self.txn_generation(); + if program == current && txn == current { + Ok(()) + } else { + Err(FenceError::new( + FenceOutcome::MigrationWriteFenced, + format!( + "the namespace fence changed after this transaction began \ + (program admitted at generation {program}, transaction opened at \ + generation {txn}, current generation {current}); roll back and \ + begin a new transaction" + ), + ) + .with_detail(FenceDetail::StaleTransaction)) + } + }) + }; + if let Err(e) = &result { + tracing::debug!( + namespace = %self.controller.namespace, + class = ?self.class, + "write transaction refused by the namespace fence: {e}" + ); + *self.denial.lock() = Some(e.clone()); + } + result + } + + /// Take the typed outcome of the last refusal at the WAL, if any. + pub fn take_denial(&self) -> Option { + self.denial.lock().take() + } +} + +#[cfg(test)] +pub(crate) mod tests { + use std::path::Path; + + use tempfile::tempdir; + + use super::*; + use crate::config::MetaStoreConfig; + use crate::connection::config::DatabaseConfig; + use crate::database::DatabaseKind; + use crate::namespace::fence::command::FenceCommand; + use crate::namespace::fence::record::{FrozenBoundary, ServerIdentity}; + use crate::namespace::meta_store::{metastore_connection_maker, FenceCommitKind}; + + const LOG: Uuid = Uuid::from_u128(0x10); + pub(crate) const OP: Uuid = Uuid::from_u128(0xa); + const OTHER_OP: Uuid = Uuid::from_u128(0xb); + + pub(crate) async fn open_metastore(dir: &Path) -> MetaStore { + let (maker, manager) = metastore_connection_maker(None, dir).await.unwrap(); + let conn = maker().unwrap(); + MetaStore::new( + MetaStoreConfig { + namespace_fence: true, + ..Default::default() + }, + dir, + conn, + manager, + DatabaseKind::Primary, + ) + .await + .unwrap() + } + + pub(crate) async fn create_namespace(meta: &MetaStore, ns: &'static str) { + meta.handle(ns.into()) + .await + .unwrap() + .store(DatabaseConfig::default()) + .await + .unwrap(); + } + + pub(crate) fn ctx() -> FenceContext { + FenceContext::now( + ServerIdentity { + build: "test".into(), + instance_id: Uuid::from_u128(0x99), + }, + Some(LOG), + ) + } + + pub(crate) fn acquire(ns: &'static str, op: Uuid, command_id: u128) -> FenceRequest { + FenceRequest { + namespace: ns.into(), + operation_id: op, + command_id: Uuid::from_u128(command_id), + expected_state: FenceState::Unfenced, + expected_revision: 0, + command: FenceCommand::AcquireSourceWriteFence { + expected_log_id: LOG, + drain_policy: None, + }, + } + } + + pub(crate) fn release( + ns: &'static str, + op: Uuid, + command_id: u128, + revision: u64, + ) -> FenceRequest { + FenceRequest { + namespace: ns.into(), + operation_id: op, + command_id: Uuid::from_u128(command_id), + expected_state: FenceState::SourceWriteFenced, + expected_revision: revision, + command: FenceCommand::ReleaseSourceWriteFence, + } + } + + fn outcome(r: &crate::Result) -> FenceOutcome { + match r { + Ok(c) => c.receipt.outcome, + Err(Error::NamespaceFence(e)) => e.outcome(), + Err(e) => panic!("unexpected error: {e}"), + } + } + + /// Acquire and complete the drain directly, as the write drain will: the controller ends in + /// `SOURCE_WRITE_FENCED`. + pub(crate) async fn fence_source( + meta: &MetaStore, + controller: &Arc, + op: Uuid, + ) { + let commit = controller + .apply_command(meta, acquire("ns", op, 1), ctx()) + .await + .unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Draining); + let mut t = controller.begin_transition().await; + let commit = t + .complete_drain( + meta, + (op, Uuid::from_u128(1)), + DrainCompletion::SourceWrites { + boundary: FrozenBoundary { + log_id: LOG, + frame_no: Some(0), + }, + }, + ctx(), + ) + .await + .unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + } + + #[tokio::test] + async fn committed_command_publishes_new_revision_and_generation() { + let tmp = tempdir().unwrap(); + let meta = open_metastore(tmp.path()).await; + create_namespace(&meta, "ns").await; + let controller = FenceController::unfenced("ns".into()); + let mut rx = controller.subscribe(); + assert!(controller.permits(OperationClass::NormalWrite).is_ok()); + assert_eq!(controller.write_generation(), 0); + + let commit = controller + .apply_command(&meta, acquire("ns", OP, 1), ctx()) + .await + .unwrap(); + assert_eq!(commit.kind, FenceCommitKind::Committed); + assert!(rx.has_changed().unwrap()); + let gate = rx.borrow_and_update().clone(); + assert_eq!(gate.state(), FenceState::SourceDraining); + assert_eq!(gate.revision(), 1); + assert_eq!(gate.operation_id(), Some(OP)); + assert_eq!(gate.write_generation, 1); + assert_eq!( + controller + .permits(OperationClass::NormalWrite) + .unwrap_err() + .outcome(), + FenceOutcome::MigrationWriteFenced + ); + assert!(controller.permits(OperationClass::NormalRead).is_ok()); + assert!(controller.permits(OperationClass::Maintenance).is_ok()); + + // The published gate is the durable one. + let inspected = meta.inspect_fence("ns".into()).await.unwrap(); + assert_eq!(inspected.fence, gate.fence); + } + + #[tokio::test] + async fn write_generation_bumps_on_every_write_admission_change() { + let tmp = tempdir().unwrap(); + let meta = open_metastore(tmp.path()).await; + create_namespace(&meta, "ns").await; + let controller = FenceController::unfenced("ns".into()); + + let mut generations = vec![controller.write_generation()]; + fence_source(&meta, &controller, OP).await; + // UNFENCED -> SOURCE_DRAINING -> SOURCE_WRITE_FENCED: two changes of state. + generations.push(controller.write_generation()); + let revision = controller.gate().revision(); + controller + .apply_command(&meta, release("ns", OP, 2, revision), ctx()) + .await + .unwrap(); + assert_eq!(controller.gate().state(), FenceState::Released); + assert!(controller.permits(OperationClass::NormalWrite).is_ok()); + generations.push(controller.write_generation()); + assert_eq!(generations, vec![0, 2, 3]); + + // A replay publishes the same durable state and does not move the generation. + let before = controller.gate(); + let replay = controller + .apply_command(&meta, release("ns", OP, 2, revision), ctx()) + .await + .unwrap(); + assert_eq!(replay.kind, FenceCommitKind::Replayed); + assert_eq!(controller.gate(), before); + } + + /// Registered write queues are woken after every change of the write generation, and only + /// then; a waker whose manager is gone is forgotten. + #[tokio::test] + async fn write_queues_are_woken_on_every_generation_change() { + let tmp = tempdir().unwrap(); + let meta = open_metastore(tmp.path()).await; + create_namespace(&meta, "ns").await; + let controller = FenceController::unfenced("ns".into()); + + let woken = Arc::new(AtomicU64::new(0)); + let seen_generation = Arc::new(AtomicU64::new(0)); + controller.register_write_queue(Box::new({ + let woken = woken.clone(); + let seen_generation = seen_generation.clone(); + let controller = Arc::downgrade(&controller); + move || { + // The gate is already published when the queue is woken. + if let Some(c) = controller.upgrade() { + seen_generation.store(c.write_generation(), Ordering::SeqCst); + } + woken.fetch_add(1, Ordering::SeqCst); + true + } + })); + let gone = Arc::new(AtomicU64::new(0)); + controller.register_write_queue(Box::new({ + let gone = gone.clone(); + move || { + gone.fetch_add(1, Ordering::SeqCst); + false + } + })); + assert_eq!(controller.write_queues.lock().len(), 2); + + fence_source(&meta, &controller, OP).await; + assert_eq!(woken.load(Ordering::SeqCst), 2); + assert_eq!(seen_generation.load(Ordering::SeqCst), 2); + assert_eq!(gone.load(Ordering::SeqCst), 1); + assert_eq!(controller.write_queues.lock().len(), 1); + + // A replay does not move the generation and wakes nobody. + let revision = controller.gate().revision(); + controller + .apply_command(&meta, release("ns", OP, 2, revision), ctx()) + .await + .unwrap(); + assert_eq!(woken.load(Ordering::SeqCst), 3); + controller + .apply_command(&meta, release("ns", OP, 2, revision), ctx()) + .await + .unwrap(); + assert_eq!(woken.load(Ordering::SeqCst), 3); + } + + #[tokio::test] + async fn failed_before_commit_leaves_gate_unchanged() { + let tmp = tempdir().unwrap(); + let meta = open_metastore(tmp.path()).await; + create_namespace(&meta, "ns").await; + let controller = FenceController::unfenced("ns".into()); + let before = controller.gate(); + + // A refusal by the transition function. + let mut wrong = acquire("ns", OP, 1); + wrong.expected_revision = 7; + let r = controller.apply_command(&meta, wrong, ctx()).await; + assert_eq!(outcome(&r), FenceOutcome::FenceRevisionMismatch); + assert_eq!(controller.gate(), before); + + // An injected failure before the metastore transaction. + controller.hooks().fail_at( + HookPoint::BeforeMetastoreCommit, + FenceError::new(FenceOutcome::FencePreconditionFailed, "injected"), + ); + let r = controller + .apply_command(&meta, acquire("ns", OP, 1), ctx()) + .await; + assert_eq!(outcome(&r), FenceOutcome::FencePreconditionFailed); + assert_eq!(controller.gate(), before); + + // The metastore is busy (another connection holds its write lock): nothing is written + // and the gate does not move. + let (maker, _) = metastore_connection_maker(None, tmp.path()).await.unwrap(); + let mut other = maker().unwrap(); + let lock = other + .transaction_with_behavior(rusqlite::TransactionBehavior::Immediate) + .unwrap(); + let r = controller + .apply_command(&meta, acquire("ns", OP, 1), ctx()) + .await; + assert!( + matches!(r, Err(Error::RusqliteError(_))), + "expected a busy metastore, got {r:?}" + ); + assert_eq!(controller.gate(), before); + drop(lock); + let inspected = meta.inspect_fence("ns".into()).await.unwrap(); + assert!(matches!(inspected.fence, StoredFence::None { .. })); + assert!(inspected.receipts.is_empty()); + } + + #[tokio::test] + async fn indeterminate_commit_keeps_writes_closed_until_replayed() { + let tmp = tempdir().unwrap(); + let meta = open_metastore(tmp.path()).await; + create_namespace(&meta, "ns").await; + let controller = FenceController::unfenced("ns".into()); + fence_source(&meta, &controller, OP).await; + let fenced = controller.gate(); + let revision = fenced.revision(); + + // Release commits, but its acknowledgement is lost. + controller.hooks().arm( + HookPoint::AfterMetastoreCommit, + super::super::hooks::HookAction::Indeterminate, + ); + let r = controller + .apply_command(&meta, release("ns", OP, 2, revision), ctx()) + .await; + assert_eq!(outcome(&r), FenceOutcome::FenceCommitIndeterminate); + let gate = controller.gate(); + assert_eq!(gate.indeterminate, Some((OP, Uuid::from_u128(2)))); + assert!(gate.write_generation > fenced.write_generation); + // The durable state says released, but nothing is admitted until it is reconciled. + for class in [ + OperationClass::NormalWrite, + OperationClass::NormalRead, + OperationClass::Stream, + OperationClass::Lifecycle, + ] { + let e = controller.permits(class).unwrap_err(); + assert_eq!(e.outcome(), FenceOutcome::FenceStateUnavailable); + assert_eq!(e.detail(), Some(FenceDetail::IndeterminateCommit)); + } + assert!(controller.permits(OperationClass::Maintenance).is_ok()); + + // Any other command is refused with the indeterminate code. + let r = controller + .apply_command(&meta, release("ns", OTHER_OP, 3, revision), ctx()) + .await; + assert_eq!(outcome(&r), FenceOutcome::FenceCommitIndeterminate); + + // The replay reconciles from the durable row and publishes it. + let r = controller + .apply_command(&meta, release("ns", OP, 2, revision), ctx()) + .await + .unwrap(); + assert_eq!(r.kind, FenceCommitKind::Replayed); + let gate = controller.gate(); + assert_eq!(gate.indeterminate, None); + assert_eq!(gate.state(), FenceState::Released); + assert!(controller.permits(OperationClass::NormalWrite).is_ok()); + } + + #[tokio::test] + async fn indeterminate_commit_that_did_not_apply_is_retried_by_replay() { + let tmp = tempdir().unwrap(); + let meta = open_metastore(tmp.path()).await; + create_namespace(&meta, "ns").await; + let controller = FenceController::unfenced("ns".into()); + + // Simulate an indeterminate outcome for a command that never reached the metastore: + // the replay applies it. + controller.publish(None, Some((OP, Uuid::from_u128(1))), None); + assert!(controller.permits(OperationClass::NormalWrite).is_err()); + let r = controller + .apply_command(&meta, acquire("ns", OP, 1), ctx()) + .await + .unwrap(); + assert_eq!(r.kind, FenceCommitKind::Committed); + let gate = controller.gate(); + assert_eq!(gate.indeterminate, None); + assert_eq!(gate.state(), FenceState::SourceDraining); + } + + #[tokio::test] + async fn publication_happens_before_the_response() { + let tmp = tempdir().unwrap(); + let meta = open_metastore(tmp.path()).await; + create_namespace(&meta, "ns").await; + let controller = FenceController::unfenced("ns".into()); + let before_publish = controller.hooks().pause_at(HookPoint::BeforeGatePublish); + + let task = tokio::spawn({ + let controller = controller.clone(); + let meta = meta.clone(); + async move { + controller + .apply_command(&meta, acquire("ns", OP, 1), ctx()) + .await + } + }); + before_publish.reached().await; + // Committed, not yet published: the gate still shows the old state. + assert_eq!(controller.gate().state(), FenceState::Unfenced); + assert!(matches!( + meta.inspect_fence("ns".into()).await.unwrap().fence, + StoredFence::Record(_) + )); + let before_response = controller.hooks().pause_at(HookPoint::BeforeResponse); + before_publish.resume(); + before_response.reached().await; + assert_eq!(controller.gate().state(), FenceState::SourceDraining); + assert!(!task.is_finished()); + before_response.resume(); + assert_eq!(outcome(&task.await.unwrap()), FenceOutcome::Draining); + } + + #[tokio::test] + async fn committed_command_is_published_when_the_caller_goes_away() { + let tmp = tempdir().unwrap(); + let meta = open_metastore(tmp.path()).await; + create_namespace(&meta, "ns").await; + let controller = FenceController::unfenced("ns".into()); + let after_commit = controller.hooks().pause_at(HookPoint::AfterMetastoreCommit); + + let caller = tokio::spawn({ + let controller = controller.clone(); + let meta = meta.clone(); + async move { + controller + .apply_command(&meta, acquire("ns", OP, 1), ctx()) + .await + } + }); + after_commit.reached().await; + // The response is lost: the caller is cancelled after the commit. + caller.abort(); + assert!(caller.await.unwrap_err().is_cancelled()); + let published = controller.hooks().pause_at(HookPoint::BeforeResponse); + after_commit.resume(); + published.reached().await; + assert_eq!(controller.gate().state(), FenceState::SourceDraining); + published.resume(); + + // The transition lock is free again and a replay answers from the receipt. + let replay = controller + .apply_command(&meta, acquire("ns", OP, 1), ctx()) + .await + .unwrap(); + assert_eq!(replay.receipt.outcome, FenceOutcome::Draining); + } + + #[tokio::test] + async fn transition_lock_serialises_commands() { + let tmp = tempdir().unwrap(); + let meta = open_metastore(tmp.path()).await; + create_namespace(&meta, "ns").await; + let controller = FenceController::unfenced("ns".into()); + + let held = controller.begin_transition().await; + let second = tokio::spawn({ + let controller = controller.clone(); + let meta = meta.clone(); + async move { + controller + .apply_command(&meta, acquire("ns", OP, 1), ctx()) + .await + } + }); + tokio::task::yield_now().await; + assert!(!second.is_finished()); + assert!(controller.transition_lock.try_lock().is_err()); + drop(held); + assert_eq!(outcome(&second.await.unwrap()), FenceOutcome::Draining); + } + + #[test] + fn unavailable_gate_denies_every_class_but_maintenance() { + let controller = FenceController::new( + "ns".into(), + StoredFence::Unavailable { + detail: FenceDetail::CorruptRecord, + reason: "test".into(), + marker: None, + }, + ); + let gate = controller.gate(); + assert!(gate.is_unavailable()); + for class in OperationClass::ALL { + let r = gate.permits(class); + match class { + OperationClass::Maintenance | OperationClass::Observability => { + assert!(r.is_ok()) + } + _ => { + let e = r.unwrap_err(); + assert_eq!(e.outcome(), FenceOutcome::FenceStateUnavailable); + assert_eq!(e.detail(), Some(FenceDetail::CorruptRecord)); + } + } + } + } + + #[test] + fn conn_state_starts_at_the_current_generation() { + let controller = FenceController::unfenced("ns".into()); + controller.publish(None, Some((OP, OP)), None); + controller.publish(None, None, None); + let state = FenceConnState::new(controller.clone(), OperationClass::NormalWrite); + assert_eq!(state.program_generation(), 2); + assert_eq!(state.txn_generation(), 2); + assert!(Arc::ptr_eq(state.controller(), &controller)); + assert_eq!(state.class(), OperationClass::NormalWrite); + assert!(state.take_denial().is_none()); + } +} diff --git a/libsql-server/src/namespace/fence/drain.rs b/libsql-server/src/namespace/fence/drain.rs new file mode 100644 index 0000000000..6165784f60 --- /dev/null +++ b/libsql-server/src/namespace/fence/drain.rs @@ -0,0 +1,781 @@ +//! The positive source write drain (`docs/NAMESPACE_FENCE.md` sections 8.3 and 8.4). +//! +//! `AcquireSourceWriteFence` closes write admission, persists `SOURCE_DRAINING`, and then waits +//! for the writer that was already holding the write slot when admission closed to commit or +//! roll back. It waits on the connection manager's release notification: neither elapsed time +//! nor the transaction timeout is ever taken as evidence that a writer has finished. Once no +//! connection holds the slot for a write, it reads the frozen boundary (`log_id`, last committed +//! frame) under the slot lock and persists `SOURCE_WRITE_FENCED` with it. Every step runs under +//! the namespace's transition lock, on a task of its own, so a caller that goes away does not +//! interrupt it. + +use std::sync::Arc; +use std::time::Duration; + +use tokio::time::Instant; + +use crate::namespace::meta_store::{FenceCommit, FenceContext, MetaStore}; + +use super::command::{DrainPolicy, FenceCommand, FenceRequest, OnDeadline}; +use super::controller::{FenceController, LiveWriteDrain, Transition}; +use super::hooks::HookPoint; +use super::outcome::FenceOutcome; +use super::record::FrozenBoundary; +use super::state::OperationClass; +use super::transition::DrainCompletion; + +/// How long `AcquireSourceWriteFence` waits for active writers when neither the request nor +/// `--namespace-fence-default-write-drain-ms` names a deadline. +pub const DEFAULT_WRITE_DRAIN: Duration = Duration::from_secs(30); + +/// After a forced rollback, how long the drain waits at least for the rolled-back writer to +/// release the slot before it answers `DRAINING` (the request's own deadline, if longer, is +/// used instead). A rollback normally releases at once; it is delayed only while a program is +/// still running on the connection. Reaching this bound is never taken as proof of anything: +/// the answer is `DRAINING` and admission stays closed. +pub const FORCED_ROLLBACK_GRACE: Duration = Duration::from_secs(10); + +impl FenceController { + /// Run one fence command to completion under the namespace's transition lock, including + /// the drain it starts, and return its result. + /// + /// The command runs on its own task: a caller that goes away (a lost response) interrupts + /// neither the commit nor the drain, and the result can be recovered by replaying the same + /// command or inspecting the fence. + pub async fn execute( + self: &Arc, + meta: &MetaStore, + request: FenceRequest, + ctx: FenceContext, + ) -> crate::Result { + let this = self.clone(); + let meta = meta.clone(); + tokio::spawn(async move { + let mut transition = this.begin_transition().await; + match request.command { + FenceCommand::AcquireSourceWriteFence { .. } => { + acquire_source_write_fence(&mut transition, &meta, request, ctx).await + } + _ => transition.apply(&meta, request, ctx).await, + } + }) + .await? + } +} + +/// `AcquireSourceWriteFence`, steps 1 to 7 of section 8.3, under `transition`. +/// +/// Returns the `APPLIED` commit of `SOURCE_WRITE_FENCED` once the drain is proven, the +/// `DRAINING` commit of `SOURCE_DRAINING` when the deadline passes first (write admission stays +/// closed and a replay of the same command resumes the drain), or the stored result of a replay. +pub async fn acquire_source_write_fence( + transition: &mut Transition, + meta: &MetaStore, + request: FenceRequest, + mut ctx: FenceContext, +) -> crate::Result { + let controller = transition.controller().clone(); + let key = (request.operation_id, request.command_id); + let policy = match &request.command { + FenceCommand::AcquireSourceWriteFence { drain_policy, .. } => { + drain_policy.unwrap_or_else(|| meta.fence_default_write_drain()) + } + _ => return transition.apply(meta, request, ctx).await, + }; + + // Step 1 is the metastore's (replay, owner, identity, shared schema, expectation). The + // identity it checks is the namespace's own replication log id. + if let Some(source) = controller.live_write_drains().last() { + ctx.namespace_log_id = Some(source.log_id); + } + + // Steps 2 and 3: close write admission in memory. Publishing moves the write generation and + // wakes the write queues, whose waiters then fail with MIGRATION_WRITE_FENCED. Where writes + // are already closed (a resumed drain, an indeterminate commit being reconciled, a fenced + // namespace) there is nothing to install. + if controller.permits(OperationClass::NormalWrite).is_ok() { + transition.install_closing_gate(key); + let _ = controller.hook(HookPoint::AfterInstallingGate).await; + } + + // Step 4: persist SOURCE_DRAINING. The commit publishes it in place of the INSTALLING gate. A + // command proven not to have committed removes the INSTALLING gate again; one whose commit + // is unknown has already closed the gate as indeterminate. + let commit = match transition.apply(meta, request, ctx.clone()).await { + Ok(commit) => commit, + Err(e) => { + transition.remove_installing(); + return Err(e); + } + }; + if commit.receipt.outcome != FenceOutcome::Draining { + // A replay of a finished acquisition, or ALREADY_APPLIED. + return Ok(commit); + } + let drain_key = (commit.receipt.operation_id, commit.receipt.command_id); + + // Steps 5 and 6. + let boundary = match drain_writers(&controller, policy).await { + Some(boundary) => boundary, + None => return Ok(commit), + }; + + // Step 7. + let acquired_on = commit.record.as_ref().and_then(|r| r.identity.log_id); + if acquired_on.is_some_and(|log_id| log_id != boundary.log_id) { + tracing::warn!( + namespace = %controller.namespace(), + acquired_on = ?acquired_on, + boundary_log_id = %boundary.log_id, + "the replication log was rebuilt since the write fence was acquired (the source \ + restarted while draining); the frozen boundary names the rebuilt log" + ); + } + ctx.now_ms = now_ms(); + transition + .complete_drain( + meta, + drain_key, + DrainCompletion::SourceWrites { boundary }, + ctx, + ) + .await +} + +/// Wait until no connection manager of the namespace has a writer holding its write slot, and +/// read the frozen boundary. `None` when the drain could not be proven within the policy: the +/// deadline passed (after a forced rollback, the same deadline again, or at least +/// [`FORCED_ROLLBACK_GRACE`]), or the namespace has no +/// loaded primary whose replication log could be read. +async fn drain_writers( + controller: &FenceController, + policy: DrainPolicy, +) -> Option { + let namespace = controller.namespace().clone(); + let deadline_after = Duration::from_millis(policy.deadline_ms); + let mut deadline = Instant::now() + deadline_after; + let mut forced = false; + loop { + // Every manager registered from now on belongs to a maker opened after write admission + // closed, so it has no pre-cutoff writer; sources are re-read on each round anyway. + let sources = controller.live_write_drains(); + if sources.is_empty() { + tracing::warn!( + %namespace, + "the namespace has no loaded primary; the write drain cannot read its replication \ + log and stays DRAINING until the command is replayed" + ); + return None; + } + + if !wait_for_writers(&sources, deadline).await { + match policy.on_deadline { + OnDeadline::ForceRollback if !forced => { + forced = true; + for source in &sources { + let manager = source.manager.clone(); + // The rollback takes the connection's lock, which a running program + // holds; it releases the slot when it happens, and that release is what + // the drain keeps waiting for. + tokio::task::spawn_blocking(move || { + if let Some(id) = manager.abort_active() { + tracing::info!( + connection = id, + "write drain deadline passed; rolling back the active writer" + ); + } + }); + } + deadline = Instant::now() + deadline_after.max(FORCED_ROLLBACK_GRACE); + continue; + } + _ => { + tracing::info!( + %namespace, + deadline_ms = policy.deadline_ms, + on_deadline = policy.on_deadline.as_str(), + forced, + "write drain deadline passed with a writer still active; answering DRAINING" + ); + return None; + } + } + } + + let _ = controller.hook(HookPoint::BeforeBoundaryCapture).await; + match capture_boundary(&sources) { + Ok(boundary) => { + tracing::info!( + %namespace, + log_id = %boundary.log_id, + frame_no = ?boundary.frame_no, + "write drain proven" + ); + return Some(boundary); + } + // Cannot happen while write admission is closed; wait again rather than guess. + Err((id, class)) => { + tracing::warn!( + %namespace, + connection = id, + ?class, + "a writer holds the write slot at boundary capture; waiting again" + ); + } + } + } +} + +/// Wait, on each manager's release notification, until none of them has a connection holding +/// the write slot for a write. `false` when `deadline` passes first. +/// +/// With write admission closed a manager that has been seen without a writer stays without one +/// (only checkpoints can take the slot), so the managers are waited for one after the other. +async fn wait_for_writers(sources: &[LiveWriteDrain], deadline: Instant) -> bool { + for source in sources { + loop { + let released = source.manager.released().notified(); + tokio::pin!(released); + // Registered before the check, so a release in between is not missed. + released.as_mut().enable(); + if !source.manager.has_writer() { + break; + } + tokio::select! { + _ = &mut released => {} + _ = tokio::time::sleep_until(deadline) => return false, + } + } + } + true +} + +/// Step 6: under each manager's write-slot lock, observe that no connection holds the slot for a +/// write, and read the last committed frame of the replication log. The replication logger +/// commits a transaction's frames and publishes its frame number before the transaction +/// releases the slot, so what is read here is final. +fn capture_boundary( + sources: &[LiveWriteDrain], +) -> Result< + FrozenBoundary, + ( + crate::connection::connection_manager::ConnId, + OperationClass, + ), +> { + let mut frame_no = None; + for source in sources { + let frame = source + .manager + .with_no_writer(|| (source.current_frame_no)())?; + frame_no = frame_no.max(frame); + } + let log_id = sources + .last() + .expect("capture_boundary is called with at least one source") + .log_id; + Ok(FrozenBoundary { log_id, frame_no }) +} + +fn now_ms() -> i64 { + std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .map_or(0, |d| i64::try_from(d.as_millis()).unwrap_or(i64::MAX)) +} + +#[cfg(test)] +mod tests { + use std::sync::Arc; + use std::time::Duration; + + use rusqlite::ErrorCode; + use tempfile::{tempdir, TempDir}; + use uuid::Uuid; + + use super::*; + use crate::connection::config::DatabaseConfig; + use crate::connection::Connection as _; + use crate::database::{Connection, Database}; + use crate::error::Error; + use crate::namespace::fence::hooks::HookPoint; + use crate::namespace::fence::record::ServerIdentity; + use crate::namespace::fence::state::FenceState; + use crate::namespace::meta_store::FenceCommitKind; + use crate::namespace::store::fence_tests::open_store; + use crate::namespace::store::NamespaceStore; + use crate::namespace::RestoreOption; + use crate::replication::primary::logger::ReplicationLogger; + + const OP: Uuid = Uuid::from_u128(0xa); + const OTHER_OP: Uuid = Uuid::from_u128(0xb); + /// Long enough that no test ever reaches it: a drain must never finish because of time. + const LONG: DrainPolicy = DrainPolicy { + deadline_ms: 600_000, + on_deadline: OnDeadline::Fail, + }; + const PROMPT: Duration = Duration::from_secs(30); + + /// A primary namespace `ns` with a table `t`, served by a real `NamespaceStore`, so that the + /// drain goes through the connection manager and replication logger the configurator + /// registered. + struct Source { + _dir: TempDir, + store: NamespaceStore, + fence: Arc, + logger: Arc, + } + + impl Source { + async fn new() -> Self { + let dir = tempdir().unwrap(); + let store = open_store(dir.path()).await; + store + .create( + "ns".into(), + RestoreOption::Latest, + DatabaseConfig { + // Held writers must never have the slot stolen by the timeout. + txn_timeout: Some(Duration::from_secs(600)), + ..Default::default() + }, + ) + .await + .unwrap(); + let (fence, logger) = store + .with("ns".into(), |ns| { + let logger = match &ns.db { + Database::Primary(p) => p.wal_wrapper.wrapper().logger(), + _ => unreachable!(), + }; + (ns.fence().clone(), logger) + }) + .await + .unwrap(); + let this = Self { + _dir: dir, + store, + fence, + logger, + }; + let conn = this.conn().await; + raw(&conn, "create table t (x)").await.unwrap(); + this + } + + async fn conn(&self) -> Arc { + let maker = self + .store + .with("ns".into(), |ns| ns.db.connection_maker()) + .await + .unwrap(); + Arc::new(maker.create().await.unwrap()) + } + + fn log_id(&self) -> Uuid { + self.logger.log_id() + } + + /// The last frame the replication log has committed. + fn frame_no(&self) -> Option { + *self.logger.new_frame_notifier.borrow() + } + + fn acquire(&self, op: Uuid, command_id: u128, policy: DrainPolicy) -> FenceRequest { + FenceRequest { + namespace: "ns".into(), + operation_id: op, + command_id: Uuid::from_u128(command_id), + expected_state: FenceState::Unfenced, + expected_revision: 0, + command: FenceCommand::AcquireSourceWriteFence { + expected_log_id: self.log_id(), + drain_policy: Some(policy), + }, + } + } + + /// Run `request` through the store, on a task of its own. + fn execute( + &self, + request: FenceRequest, + ) -> tokio::task::JoinHandle> { + let store = self.store.clone(); + tokio::spawn(async move { + store + .execute_fence_command( + request, + ServerIdentity { + build: "test".into(), + instance_id: Uuid::from_u128(0x99), + }, + ) + .await + }) + } + + /// Wait until the published gate is in `state`. + async fn until_state(&self, state: FenceState) { + let mut rx = self.fence.subscribe(); + tokio::time::timeout(PROMPT, rx.wait_for(|g| g.state() == state)) + .await + .expect("the gate never reached the state") + .unwrap(); + } + + async fn count(&self) -> i64 { + let conn = self.conn().await; + tokio::task::spawn_blocking(move || { + conn.with_raw(|c| c.query_row("select count(*) from t", (), |r| r.get(0))) + }) + .await + .unwrap() + .unwrap() + } + } + + /// Run `sql` as one raw program on `conn`, off the async runtime (it can block on the write + /// slot). + async fn raw(conn: &Arc, sql: &'static str) -> rusqlite::Result<()> { + let conn = conn.clone(); + tokio::task::spawn_blocking(move || conn.with_raw(|c| c.execute_batch(sql))) + .await + .unwrap() + } + + fn assert_fenced(result: rusqlite::Result<()>) { + match result { + Err(rusqlite::Error::SqliteFailure(e, _)) => { + assert_eq!(e.code, ErrorCode::AuthorizationForStatementDenied, "{e}") + } + other => panic!("expected the WAL gate to refuse the write, got {other:?}"), + } + } + + fn fence_outcome(result: &crate::Result) -> FenceOutcome { + match result { + Ok(c) => c.receipt.outcome, + Err(Error::NamespaceFence(e)) => e.outcome(), + Err(e) => panic!("unexpected error: {e}"), + } + } + + fn boundary(commit: &FenceCommit) -> FrozenBoundary { + commit + .record + .as_ref() + .unwrap() + .frozen_boundary + .expect("a write-fenced source has a frozen boundary") + } + + /// Two operations acquire the same namespace at once: exactly one owns it, and the other + /// gets the typed ownership conflict. The first is parked after closing write admission so + /// that the second is certainly waiting on the transition lock. + #[tokio::test(flavor = "multi_thread")] + async fn acquire_race_single_owner() { + let s = Source::new().await; + let paused = s.fence.hooks().pause_at(HookPoint::AfterInstallingGate); + let first = s.execute(s.acquire(OP, 1, LONG)); + paused.reached().await; + let second = s.execute(s.acquire(OTHER_OP, 2, LONG)); + paused.resume(); + + let first = first.await.unwrap(); + let second = second.await.unwrap(); + let outcomes = [fence_outcome(&first), fence_outcome(&second)]; + assert_eq!( + outcomes, + [ + FenceOutcome::Applied, + FenceOutcome::FenceOwnedByAnotherOperation + ] + ); + let gate = s.fence.gate(); + assert_eq!(gate.state(), FenceState::SourceWriteFenced); + assert_eq!(gate.operation_id(), Some(OP)); + assert!(!gate.is_installing()); + } + + /// A writer that holds the write slot when the fence arrives commits, and only then is the + /// freeze acknowledged, with a boundary that includes its commit. Writes attempted while + /// the drain waits are refused. + #[tokio::test(flavor = "multi_thread")] + async fn active_writer_commits_before_ack() { + let s = Source::new().await; + let holder = s.conn().await; + raw(&holder, "begin immediate; insert into t values (1);") + .await + .unwrap(); + + let acquire = s.execute(s.acquire(OP, 1, LONG)); + s.until_state(FenceState::SourceDraining).await; + // Admission is closed while the pre-cutoff writer is still active. + assert_fenced(raw(&s.conn().await, "insert into t values (2)").await); + assert!(!acquire.is_finished()); + + raw(&holder, "commit").await.unwrap(); + let committed_frame = s.frame_no(); + assert!(committed_frame.is_some()); + + let commit = acquire.await.unwrap().unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + assert_eq!(commit.receipt.state_after, FenceState::SourceWriteFenced); + assert_eq!( + boundary(&commit), + FrozenBoundary { + log_id: s.log_id(), + frame_no: committed_frame, + } + ); + assert_eq!(s.count().await, 1); + } + + /// Under `force_rollback`, a writer still active at the deadline is rolled back, and the + /// freeze is acknowledged only after its slot was actually released; its write is not in + /// the database or the boundary. + #[tokio::test(flavor = "multi_thread")] + async fn forced_rollback_before_ack() { + let s = Source::new().await; + let before = s.frame_no(); + let holder = s.conn().await; + raw(&holder, "begin immediate; insert into t values (1);") + .await + .unwrap(); + + let commit = s + .execute(s.acquire( + OP, + 1, + DrainPolicy { + deadline_ms: 0, + on_deadline: OnDeadline::ForceRollback, + }, + )) + .await + .unwrap() + .unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + assert_eq!( + boundary(&commit), + FrozenBoundary { + log_id: s.log_id(), + frame_no: before, + } + ); + assert_eq!(s.frame_no(), before); + assert_eq!(s.count().await, 0); + // The rolled-back transaction is gone; its connection cannot write either. + assert!(raw(&holder, "commit").await.is_err()); + assert_fenced(raw(&holder, "insert into t values (3)").await); + assert_eq!(s.frame_no(), before); + } + + /// After the acknowledgement nothing commits: the boundary is the last committed frame, and + /// autocommit writes, explicit transactions, DDL and a read transaction opened before the + /// fence that tries to upgrade are all refused without adding a frame. + #[tokio::test(flavor = "multi_thread")] + async fn no_commit_after_ack() { + let s = Source::new().await; + let writer = s.conn().await; + for _ in 0..3 { + raw(&writer, "insert into t values (1)").await.unwrap(); + } + let last = s.frame_no(); + // A reader whose transaction predates the fence. + let reader = s.conn().await; + raw(&reader, "begin; select count(*) from t;") + .await + .unwrap(); + + let commit = s.execute(s.acquire(OP, 1, LONG)).await.unwrap().unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + assert_eq!( + boundary(&commit), + FrozenBoundary { + log_id: s.log_id(), + frame_no: last, + } + ); + + assert_fenced(raw(&writer, "insert into t values (2)").await); + assert_fenced(raw(&writer, "begin immediate").await); + assert_fenced(raw(&writer, "create table u (y)").await); + assert_fenced(raw(&reader, "insert into t values (2)").await); + assert_fenced(raw(&s.conn().await, "insert into t values (2)").await); + assert_eq!(s.frame_no(), last); + assert_eq!(s.count().await, 3); + assert_eq!(boundary(&commit).frame_no, s.frame_no()); + } + + /// With `on_deadline: fail`, a writer still active at the deadline makes the command answer + /// `DRAINING`: the durable state is `SOURCE_DRAINING` and write admission stays closed. + #[tokio::test(flavor = "multi_thread")] + async fn deadline_returns_draining_and_stays_closed() { + let s = Source::new().await; + let holder = s.conn().await; + raw(&holder, "begin immediate; insert into t values (1);") + .await + .unwrap(); + + let policy = DrainPolicy { + deadline_ms: 0, + on_deadline: OnDeadline::Fail, + }; + let commit = s.execute(s.acquire(OP, 1, policy)).await.unwrap().unwrap(); + assert_eq!(commit.kind, FenceCommitKind::Committed); + assert_eq!(commit.receipt.outcome, FenceOutcome::Draining); + assert_eq!(s.fence.gate().state(), FenceState::SourceDraining); + let inspected = s + .store + .meta_store() + .inspect_fence("ns".into()) + .await + .unwrap(); + assert_eq!(inspected.fence.state(), FenceState::SourceDraining); + assert_fenced(raw(&s.conn().await, "insert into t values (2)").await); + // Nothing reopens by itself; the writer that was active before the fence is still the + // only one that can commit. + raw(&holder, "commit").await.unwrap(); + assert_fenced(raw(&holder, "insert into t values (3)").await); + assert_eq!(s.fence.gate().state(), FenceState::SourceDraining); + assert_eq!(s.count().await, 1); + } + + /// Replaying a command whose receipt is `DRAINING` resumes the same drain: it completes once + /// the writer has finished, and a further replay returns that stored result. + #[tokio::test(flavor = "multi_thread")] + async fn replay_of_draining_resumes_and_completes() { + let s = Source::new().await; + let holder = s.conn().await; + raw(&holder, "begin immediate; insert into t values (1);") + .await + .unwrap(); + let policy = DrainPolicy { + deadline_ms: 0, + on_deadline: OnDeadline::Fail, + }; + let first = s.execute(s.acquire(OP, 1, policy)).await.unwrap().unwrap(); + assert_eq!(first.receipt.outcome, FenceOutcome::Draining); + let draining_revision = s.fence.gate().revision(); + + // Replayed while the writer is still active: the same drain resumes, writes nothing and, + // with the same zero deadline, answers DRAINING again. + let replay = s.execute(s.acquire(OP, 1, policy)).await.unwrap().unwrap(); + assert_eq!(replay.kind, FenceCommitKind::Resumed); + assert_eq!(replay.receipt.outcome, FenceOutcome::Draining); + assert_eq!(s.fence.gate().revision(), draining_revision); + + raw(&holder, "commit").await.unwrap(); + let committed = s.frame_no(); + let done = s.execute(s.acquire(OP, 1, policy)).await.unwrap().unwrap(); + assert_eq!(done.receipt.outcome, FenceOutcome::Applied); + assert_eq!(done.receipt.command_id, Uuid::from_u128(1)); + assert_eq!( + boundary(&done), + FrozenBoundary { + log_id: s.log_id(), + frame_no: committed, + } + ); + assert_eq!(s.fence.gate().state(), FenceState::SourceWriteFenced); + assert_eq!(s.fence.gate().revision(), draining_revision + 1); + + let again = s.execute(s.acquire(OP, 1, policy)).await.unwrap().unwrap(); + assert_eq!(again.kind, FenceCommitKind::Replayed); + assert_eq!(again.receipt, done.receipt); + assert_eq!(boundary(&again), boundary(&done)); + } + + /// Releasing the write fence commits, publishes a new write generation and only then + /// answers: new programs write again, and a transaction that began under the fence cannot. + #[tokio::test(flavor = "multi_thread")] + async fn release_reopens_with_new_generation() { + let s = Source::new().await; + s.execute(s.acquire(OP, 1, LONG)).await.unwrap().unwrap(); + let fenced = s.fence.gate(); + let reader = s.conn().await; + raw(&reader, "begin; select count(*) from t;") + .await + .unwrap(); + + let release = FenceRequest { + namespace: "ns".into(), + operation_id: OP, + command_id: Uuid::from_u128(2), + expected_state: FenceState::SourceWriteFenced, + expected_revision: fenced.revision(), + command: FenceCommand::ReleaseSourceWriteFence, + }; + let paused = s.fence.hooks().pause_at(HookPoint::BeforeResponse); + let released = s.execute(release); + paused.reached().await; + // Published before the response is sent. + let gate = s.fence.gate(); + assert_eq!(gate.state(), FenceState::Released); + assert!(gate.write_generation > fenced.write_generation); + assert!(!released.is_finished()); + paused.resume(); + let commit = released.await.unwrap().unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + + raw(&s.conn().await, "insert into t values (1)") + .await + .unwrap(); + assert_fenced(raw(&reader, "insert into t values (2)").await); + raw(&reader, "rollback").await.unwrap(); + raw(&reader, "insert into t values (3)").await.unwrap(); + assert_eq!(s.count().await, 2); + } + + /// An acquisition refused by the metastore (here: the caller observed a different + /// replication log) removes the INSTALLING gate it had published; writes are admitted again + /// under a new generation. + #[tokio::test(flavor = "multi_thread")] + async fn refused_acquire_reopens_writes() { + let s = Source::new().await; + let before = s.fence.gate().write_generation; + let mut request = s.acquire(OP, 1, LONG); + request.command = FenceCommand::AcquireSourceWriteFence { + expected_log_id: Uuid::from_u128(0x77), + drain_policy: Some(LONG), + }; + let result = s.execute(request).await.unwrap(); + assert_eq!( + fence_outcome(&result), + FenceOutcome::FencePreconditionFailed + ); + let gate = s.fence.gate(); + assert_eq!(gate.state(), FenceState::Unfenced); + assert!(!gate.is_installing()); + assert_eq!(gate.write_generation, before + 2); + raw(&s.conn().await, "insert into t values (1)") + .await + .unwrap(); + } + + /// While the INSTALLING gate is up, before anything is persisted, writes are already + /// refused and reads are served. + #[tokio::test(flavor = "multi_thread")] + async fn installing_gate_closes_writes_before_persisting() { + let s = Source::new().await; + let paused = s.fence.hooks().pause_at(HookPoint::AfterInstallingGate); + let acquire = s.execute(s.acquire(OP, 1, LONG)); + paused.reached().await; + let gate = s.fence.gate(); + assert!(gate.is_installing()); + assert_eq!(gate.state(), FenceState::Unfenced); + let inspected = s + .store + .meta_store() + .inspect_fence("ns".into()) + .await + .unwrap(); + assert_eq!(inspected.fence.state(), FenceState::Unfenced); + assert_fenced(raw(&s.conn().await, "insert into t values (1)").await); + assert_eq!(s.count().await, 0); + paused.resume(); + let commit = acquire.await.unwrap().unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + assert!(!s.fence.gate().is_installing()); + } +} diff --git a/libsql-server/src/namespace/fence/hooks.rs b/libsql-server/src/namespace/fence/hooks.rs new file mode 100644 index 0000000000..28a00c673e --- /dev/null +++ b/libsql-server/src/namespace/fence/hooks.rs @@ -0,0 +1,143 @@ +//! Test hooks for fence race tests (`docs/NAMESPACE_FENCE.md` section 16). +//! +//! The crate has no failpoint library. Instead, a [`FenceController`] built by the library's +//! own test build carries a `FenceTestHooks`: named points on the transition path where a +//! test can park the task until it releases it, or inject a failure. Race tests are written +//! against these points and never against elapsed time. Outside `cfg(test)` only the point +//! names exist, and reaching a point does nothing. +//! +//! [`FenceController`]: super::controller::FenceController + +#[cfg(test)] +use std::collections::HashMap; +#[cfg(test)] +use std::sync::Arc; + +#[cfg(test)] +use parking_lot::Mutex; +#[cfg(test)] +use tokio::sync::Notify; + +use super::outcome::FenceError; + +/// A named point on a fence transition or a gated path. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] +pub enum HookPoint { + /// The in-memory `INSTALLING` gate of a closing transition has been published. + AfterInstallingGate, + /// Under the transition lock, immediately before the metastore transaction runs. + BeforeMetastoreCommit, + /// The metastore transaction returned a committed result. + AfterMetastoreCommit, + /// The committed result is about to be published to the gate. + BeforeGatePublish, + /// In `begin_write_txn`, after the gate check admitted the transaction. + InBeginWriteTxnAfterCheck, + /// The connection manager released the write slot. + AfterManagerRelease, + /// A drain is about to read the frozen boundary. + BeforeBoundaryCapture, + /// The rows of a quarantined target were committed; the target is not published yet. + AfterTargetRowsCommitted, + /// The gate is published; the answer is about to be returned. + BeforeResponse, +} + +#[cfg(test)] +/// What happens when a task reaches an armed point. Every action fires once: reaching the +/// point disarms it. +#[derive(Debug, Clone)] +pub enum HookAction { + /// Signal `reached`, then wait until `resume` is notified. + Pause { + reached: Arc, + resume: Arc, + }, + /// Fail at this point with `error`, as if the step had failed before it took effect. + Fail(FenceError), + /// At `AfterMetastoreCommit`: report the commit as indeterminate even though it happened, + /// which is what a lost commit acknowledgement looks like to the controller. At + /// `BeforeMetastoreCommit`: report it as indeterminate without running it, which is a + /// commit that failed without applying but whose outcome the controller cannot know. + Indeterminate, +} + +/// What the task that reached a point has to do next. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum HookOutcome { + Continue, + Fail(FenceError), + Indeterminate, +} + +#[cfg(test)] +/// The armed points of one controller. +#[derive(Debug, Default)] +pub struct FenceTestHooks { + armed: Mutex>, +} + +#[cfg(test)] +/// The handles a test uses to follow a paused task. +#[derive(Debug, Clone)] +pub struct Paused { + pub reached: Arc, + pub resume: Arc, +} + +#[cfg(test)] +impl Paused { + /// Wait until the task reaches the point. + pub async fn reached(&self) { + self.reached.notified().await + } + + /// Let the task continue. + pub fn resume(&self) { + self.resume.notify_one() + } +} + +#[cfg(test)] +impl FenceTestHooks { + pub fn arm(&self, point: HookPoint, action: HookAction) { + self.armed.lock().insert(point, action); + } + + /// Arm `point` to pause, returning the handles to wait for it and to release it. + pub fn pause_at(&self, point: HookPoint) -> Paused { + let paused = Paused { + reached: Arc::new(Notify::new()), + resume: Arc::new(Notify::new()), + }; + self.arm( + point, + HookAction::Pause { + reached: paused.reached.clone(), + resume: paused.resume.clone(), + }, + ); + paused + } + + pub fn fail_at(&self, point: HookPoint, error: FenceError) { + self.arm(point, HookAction::Fail(error)); + } + + /// Called by the code under test when it reaches `point`. + pub async fn hit(&self, point: HookPoint) -> HookOutcome { + let action = self.armed.lock().remove(&point); + match action { + None => HookOutcome::Continue, + Some(HookAction::Pause { reached, resume }) => { + // `notify_one` stores a permit, so a test that starts waiting after the task + // got here still sees it. + reached.notify_one(); + resume.notified().await; + HookOutcome::Continue + } + Some(HookAction::Fail(e)) => HookOutcome::Fail(e), + Some(HookAction::Indeterminate) => HookOutcome::Indeterminate, + } + } +} diff --git a/libsql-server/src/namespace/fence/mod.rs b/libsql-server/src/namespace/fence/mod.rs index 672b0565e8..7e64e0cd7d 100644 --- a/libsql-server/src/namespace/fence/mod.rs +++ b/libsql-server/src/namespace/fence/mod.rs @@ -2,12 +2,15 @@ //! example, moving a database between servers) uses as the data-plane authority boundary for //! one namespace. //! -//! `docs/NAMESPACE_FENCE.md` is the contract and the design. This module holds the parts with -//! no I/O: the states and permission matrix ([`state`]), the stable outcome codes and their -//! protocol mappings ([`outcome`]), commands and their canonical fingerprint ([`command`]), -//! records, receipts and markers with their strict durable encoding ([`record`]), the pure -//! transition function ([`transition`]), and the metastore tables, compare-and-swap and marker -//! file that persist them ([`store`], driven by `MetaStore::apply_fence_command`). +//! `docs/NAMESPACE_FENCE.md` is the contract and the design. This module holds the states and +//! permission matrix ([`state`]), the stable outcome codes and their protocol mappings +//! ([`outcome`]), commands and their canonical fingerprint ([`command`]), records, receipts and +//! markers with their strict durable encoding ([`record`]), the pure transition function +//! ([`transition`]), the metastore tables, compare-and-swap and marker file that persist them +//! ([`store`], driven by `MetaStore::apply_fence_command`), and the in-memory authority built +//! on them: the per-namespace [`controller`] with its gate, the positive write [`drain`], the +//! [`registry`] that holds the controllers outside the namespace cache, and the test [`hooks`] +//! on their paths. // The persistence, controller and protocol layers that consume these types land in the // following commits of this series; until then most of the module is unused by the rest of @@ -15,12 +18,19 @@ #![allow(dead_code)] pub mod command; +pub mod controller; +pub mod drain; +pub mod hooks; pub mod outcome; pub mod record; +pub mod registry; pub mod state; pub mod store; pub mod transition; +#[cfg(test)] +mod tests; + #[allow(clippy::all)] pub(crate) mod proto { include!("../../generated/namespace_fence.rs"); diff --git a/libsql-server/src/namespace/fence/outcome.rs b/libsql-server/src/namespace/fence/outcome.rs index 6a9b6a8b57..cf4df0c9af 100644 --- a/libsql-server/src/namespace/fence/outcome.rs +++ b/libsql-server/src/namespace/fence/outcome.rs @@ -201,6 +201,8 @@ pub enum FenceDetail { IncompleteTargetCreation, MetastoreBehindMarker, IndeterminateCommit, + // MIGRATION_WRITE_FENCED + StaleTransaction, } impl FenceDetail { @@ -223,6 +225,7 @@ impl FenceDetail { FenceDetail::IncompleteTargetCreation => "incomplete_target_creation", FenceDetail::MetastoreBehindMarker => "metastore_behind_marker", FenceDetail::IndeterminateCommit => "indeterminate_commit", + FenceDetail::StaleTransaction => "stale_transaction", } } } diff --git a/libsql-server/src/namespace/fence/record.rs b/libsql-server/src/namespace/fence/record.rs index a2d9f7f5dd..247b481ad2 100644 --- a/libsql-server/src/namespace/fence/record.rs +++ b/libsql-server/src/namespace/fence/record.rs @@ -32,7 +32,8 @@ pub struct NamespaceIdentity { #[derive(Debug, Clone, Copy, PartialEq, Eq)] pub struct FrozenBoundary { pub log_id: Uuid, - pub frame_no: u64, + /// The last frame committed to the replication log, or `None` when the log has no frames. + pub frame_no: Option, } /// The pre-fence values of the legacy `block_*` configuration fields, restored when the @@ -700,7 +701,7 @@ pub(super) mod tests { drain_started_at_ms: Some(100), frozen_boundary: Some(FrozenBoundary { log_id: Uuid::from_u128(10), - frame_no: 1234, + frame_no: Some(1234), }), validation: None, legacy_blocks: LegacyBlocks { diff --git a/libsql-server/src/namespace/fence/registry.rs b/libsql-server/src/namespace/fence/registry.rs new file mode 100644 index 0000000000..610189677b --- /dev/null +++ b/libsql-server/src/namespace/fence/registry.rs @@ -0,0 +1,221 @@ +//! The fence registry (`docs/NAMESPACE_FENCE.md` section 7.1). +//! +//! One [`FenceController`] per namespace, held outside the namespace cache so that eviction and +//! lazy reload hand a reloaded namespace the controller it had, with the same gate, revision +//! and write generation. The registry is seeded from the metastore before the namespace store +//! serves anything, and namespaces without fence state get an `UNFENCED` controller on first +//! use. + +use std::collections::HashMap; +use std::sync::Arc; + +use parking_lot::Mutex; + +use crate::namespace::NamespaceName; + +use super::controller::FenceController; +use super::outcome::FenceError; +use super::state::OperationClass; +use super::store::StoredFence; + +#[derive(Debug, Default)] +pub struct FenceRegistry { + controllers: Mutex>>, +} + +impl FenceRegistry { + /// A registry holding a controller for every namespace with fence state, as returned by + /// `MetaStore::load_fences` (which includes the namespaces startup could not recover). + pub fn seeded(fences: impl IntoIterator) -> Self { + let controllers = fences + .into_iter() + .map(|(ns, fence)| { + if let StoredFence::Unavailable { detail, reason, .. } = &fence { + tracing::error!( + namespace = %ns, + %detail, + "namespace fence state is unavailable; the namespace is not served: {reason}" + ); + } + let controller = FenceController::new(ns.clone(), fence); + (ns, controller) + }) + .collect(); + Self { + controllers: Mutex::new(controllers), + } + } + + /// The controller of `namespace`, creating an `UNFENCED` one if it has none yet. + pub fn controller(&self, namespace: &NamespaceName) -> Arc { + self.controllers + .lock() + .entry(namespace.clone()) + .or_insert_with(|| FenceController::unfenced(namespace.clone())) + .clone() + } + + /// The controller of `namespace`, if it has one. + pub fn get(&self, namespace: &NamespaceName) -> Option> { + self.controllers.lock().get(namespace).cloned() + } + + /// Forget `namespace`'s controller. Only for a namespace that was deleted together with its + /// fence state. + pub fn remove(&self, namespace: &NamespaceName) -> Option> { + self.controllers.lock().remove(namespace) + } + + /// Refuse a namespace whose fence state is `UNKNOWN_UNAVAILABLE`, before any work is done + /// to serve it. + pub fn check_available(&self, namespace: &NamespaceName) -> Result<(), FenceError> { + match self.get(namespace) { + Some(controller) => { + let gate = controller.gate(); + if gate.is_unavailable() { + gate.permits(OperationClass::NormalRead) + } else { + Ok(()) + } + } + None => Ok(()), + } + } + + pub fn len(&self) -> usize { + self.controllers.lock().len() + } +} + +#[cfg(test)] +mod tests { + use tempfile::tempdir; + use uuid::Uuid; + + use super::*; + use crate::config::MetaStoreConfig; + use crate::connection::config::DatabaseConfig; + use crate::database::DatabaseKind; + use crate::namespace::fence::command::{FenceCommand, FenceRequest}; + use crate::namespace::fence::outcome::{FenceDetail, FenceOutcome}; + use crate::namespace::fence::record::ServerIdentity; + use crate::namespace::fence::state::FenceState; + use crate::namespace::fence::store as fence_store; + use crate::namespace::meta_store::{metastore_connection_maker, FenceContext, MetaStore}; + + const LOG: Uuid = Uuid::from_u128(0x10); + const OP: Uuid = Uuid::from_u128(0xa); + + async fn open(dir: &std::path::Path) -> MetaStore { + let (maker, manager) = metastore_connection_maker(None, dir).await.unwrap(); + let conn = maker().unwrap(); + MetaStore::new( + MetaStoreConfig { + namespace_fence: true, + ..Default::default() + }, + dir, + conn, + manager, + DatabaseKind::Primary, + ) + .await + .unwrap() + } + + fn ctx() -> FenceContext { + FenceContext::now( + ServerIdentity { + build: "test".into(), + instance_id: Uuid::from_u128(0x99), + }, + Some(LOG), + ) + } + + #[tokio::test] + async fn seeded_from_load_fences_including_recovered_names() { + let tmp = tempdir().unwrap(); + { + let meta = open(tmp.path()).await; + for ns in ["fenced", "plain"] { + meta.handle(ns.into()) + .await + .unwrap() + .store(DatabaseConfig::default()) + .await + .unwrap(); + } + let controller = FenceController::unfenced("fenced".into()); + controller + .apply_command( + &meta, + FenceRequest { + namespace: "fenced".into(), + operation_id: OP, + command_id: Uuid::from_u128(1), + expected_state: FenceState::Unfenced, + expected_revision: 0, + command: FenceCommand::AcquireSourceWriteFence { + expected_log_id: LOG, + drain_policy: None, + }, + }, + ctx(), + ) + .await + .unwrap(); + meta.shutdown().await.unwrap(); + } + // A directory with an unreadable marker and no config: startup cannot recover it. + let lost = tmp.path().join("dbs").join("lost"); + std::fs::create_dir_all(&lost).unwrap(); + std::fs::write(lost.join(fence_store::MARKER_FILE_NAME), b"garbage").unwrap(); + + // Restart. + let meta = open(tmp.path()).await; + let registry = FenceRegistry::seeded(meta.load_fences().await.unwrap()); + assert_eq!(registry.len(), 2); + + let fenced = registry.get(&"fenced".into()).unwrap(); + let gate = fenced.gate(); + assert_eq!(gate.state(), FenceState::SourceDraining); + assert_eq!(gate.revision(), 1); + assert_eq!(gate.operation_id(), Some(OP)); + assert_eq!( + fenced + .permits(OperationClass::NormalWrite) + .unwrap_err() + .outcome(), + FenceOutcome::MigrationWriteFenced + ); + assert!(registry.check_available(&"fenced".into()).is_ok()); + + let lost = registry.get(&"lost".into()).unwrap(); + assert!(lost.gate().is_unavailable()); + let e = registry.check_available(&"lost".into()).unwrap_err(); + assert_eq!(e.outcome(), FenceOutcome::FenceStateUnavailable); + assert_eq!(e.detail(), Some(FenceDetail::CorruptRecord)); + for class in [OperationClass::NormalRead, OperationClass::NormalWrite] { + assert!(lost.permits(class).is_err()); + } + + // An ordinary namespace has no controller until it is used, then an UNFENCED one. + assert!(registry.get(&"plain".into()).is_none()); + let plain = registry.controller(&"plain".into()); + assert_eq!(plain.gate().state(), FenceState::Unfenced); + assert!(plain.permits(OperationClass::NormalWrite).is_ok()); + assert_eq!(registry.len(), 3); + } + + #[test] + fn controller_is_stable_until_removed() { + let registry = FenceRegistry::default(); + let a = registry.controller(&"ns".into()); + let b = registry.controller(&"ns".into()); + assert!(Arc::ptr_eq(&a, &b)); + assert!(Arc::ptr_eq(®istry.get(&"ns".into()).unwrap(), &a)); + assert!(registry.remove(&"ns".into()).is_some()); + assert!(!Arc::ptr_eq(®istry.controller(&"ns".into()), &a)); + } +} diff --git a/libsql-server/src/namespace/fence/tests.rs b/libsql-server/src/namespace/fence/tests.rs new file mode 100644 index 0000000000..bb41fbeecd --- /dev/null +++ b/libsql-server/src/namespace/fence/tests.rs @@ -0,0 +1,898 @@ +//! Restart at every persistence boundary, indeterminate commits and lost responses +//! (`docs/NAMESPACE_FENCE.md` sections 8.4, 8.5 and 16; section 17 row 7). +//! +//! A restart here is a crash: each server lifetime runs on a runtime of its own, and ending it +//! shuts that runtime down without running any shutdown code and leaks the `NamespaceStore`, so +//! nothing is flushed, the namespace's `.sentinel` stays behind and the next start takes the +//! dirty-recovery path. The next lifetime opens a new `NamespaceStore` (and so a new +//! `MetaStore` and fence registry) on the same directory. The task running the fence command is +//! parked at a hook point when the crash happens, so the crash lands exactly on that boundary. + +use std::future::Future; +use std::path::Path; +use std::sync::Arc; +use std::time::Duration; + +use rusqlite::ErrorCode; +use tempfile::tempdir; +use tokio::runtime::Runtime; +use uuid::Uuid; + +use super::command::{DrainPolicy, FenceCommand, FenceRequest, OnDeadline}; +use super::controller::FenceController; +use super::hooks::{HookAction, HookPoint, Paused}; +use super::outcome::{FenceDetail, FenceError, FenceOutcome}; +use super::record::{FrozenBoundary, ServerIdentity}; +use super::state::{FenceState, OperationClass}; +use super::store as fence_store; +use crate::connection::config::DatabaseConfig; +use crate::connection::Connection as _; +use crate::database::{Connection, Database}; +use crate::error::Error; +use crate::namespace::meta_store::{FenceCommit, FenceCommitKind, FenceInspection}; +use crate::namespace::store::fence_tests::open_store; +use crate::namespace::store::NamespaceStore; +use crate::namespace::RestoreOption; + +const OP: Uuid = Uuid::from_u128(0xa); +const OTHER_OP: Uuid = Uuid::from_u128(0xb); +/// Long enough that no test reaches it: a drain must never finish because of time. +const LONG: DrainPolicy = DrainPolicy { + deadline_ms: 600_000, + on_deadline: OnDeadline::Fail, +}; +/// A deadline that has already passed: an acquisition with a writer still active answers +/// `DRAINING` at once. +const NOW: DrainPolicy = DrainPolicy { + deadline_ms: 0, + on_deadline: OnDeadline::Fail, +}; +/// Bound on waiting for a task to reach a hook point; a test that hits it has failed. +const PROMPT: Duration = Duration::from_secs(30); +/// Rows committed to `t` before any fence command runs. +const ROWS: i64 = 3; + +fn server_identity() -> ServerIdentity { + ServerIdentity { + build: "test".into(), + instance_id: Uuid::from_u128(0x99), + } +} + +/// One lifetime of a server process on `dir`. +struct Server { + rt: Runtime, + store: NamespaceStore, +} + +impl Server { + fn boot(dir: &Path) -> Self { + let rt = tokio::runtime::Builder::new_multi_thread() + .worker_threads(4) + .enable_all() + .build() + .unwrap(); + let store = rt.block_on(open_store(dir)); + Self { rt, store } + } + + /// End the lifetime the way a crash does: no shutdown code runs, nothing is flushed or + /// closed, and every task (including a fence command parked at a hook point) stops where + /// it is. + fn crash(self) { + let Self { rt, store } = self; + std::mem::forget(store); + rt.shutdown_background(); + } + + fn run(&self, f: F) -> F::Output { + self.rt.block_on(f) + } + + /// Create `ns` with a table `t` holding [`ROWS`] rows. + fn create_source(&self) { + self.run(async { + self.store + .create( + "ns".into(), + RestoreOption::Latest, + DatabaseConfig { + // Held writers must never have the slot stolen by the timeout. + txn_timeout: Some(Duration::from_secs(600)), + ..Default::default() + }, + ) + .await + .unwrap(); + let conn = self.conn().await; + raw(&conn, "create table t (x)").await.unwrap(); + for _ in 0..ROWS { + raw(&conn, "insert into t values (1)").await.unwrap(); + } + }) + } + + /// The namespace's controller, loading the namespace if it is not loaded. + async fn fence(&self) -> Arc { + self.store + .with("ns".into(), |ns| ns.fence().clone()) + .await + .unwrap() + } + + async fn conn(&self) -> Arc { + let maker = self + .store + .with("ns".into(), |ns| ns.db.connection_maker()) + .await + .unwrap(); + Arc::new(maker.create().await.unwrap()) + } + + /// The live replication log's id and last committed frame. + async fn log(&self) -> (Uuid, Option) { + self.store + .with("ns".into(), |ns| match &ns.db { + Database::Primary(p) => { + let logger = p.wal_wrapper.wrapper().logger(); + let frame_no = *logger.new_frame_notifier.borrow(); + (logger.log_id(), frame_no) + } + _ => unreachable!(), + }) + .await + .unwrap() + } + + fn execute( + &self, + request: FenceRequest, + ) -> tokio::task::JoinHandle> { + let store = self.store.clone(); + self.rt.spawn(async move { + store + .execute_fence_command(request, server_identity()) + .await + }) + } + + async fn inspect(&self) -> FenceInspection { + self.store + .meta_store() + .inspect_fence("ns".into()) + .await + .unwrap() + } + + async fn count(&self) -> i64 { + let conn = self.conn().await; + tokio::task::spawn_blocking(move || { + conn.with_raw(|c| c.query_row("select count(*) from t", (), |r| r.get(0))) + }) + .await + .unwrap() + .unwrap() + } + + /// Whether a new connection may begin a write transaction. Writes nothing. + async fn writes_admitted(&self) -> bool { + let conn = self.conn().await; + match raw(&conn, "begin immediate; rollback;").await { + Ok(()) => true, + Err(rusqlite::Error::SqliteFailure(e, _)) + if e.code == ErrorCode::AuthorizationForStatementDenied => + { + false + } + Err(e) => panic!("unexpected error probing write admission: {e}"), + } + } + + /// Acquire and complete the source write fence under `OP`, command 1. + fn fence_source(&self) -> FenceCommit { + self.run(async { + let (log_id, _) = self.log().await; + let commit = self + .execute(acquire(log_id, 1, LONG)) + .await + .unwrap() + .unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + commit + }) + } +} + +/// Run `sql` as one raw program on `conn`, off the async runtime (it can block on the write +/// slot). +async fn raw(conn: &Arc, sql: &'static str) -> rusqlite::Result<()> { + let conn = conn.clone(); + tokio::task::spawn_blocking(move || conn.with_raw(|c| c.execute_batch(sql))) + .await + .unwrap() +} + +fn acquire(log_id: Uuid, command_id: u128, policy: DrainPolicy) -> FenceRequest { + FenceRequest { + namespace: "ns".into(), + operation_id: OP, + command_id: Uuid::from_u128(command_id), + expected_state: FenceState::Unfenced, + expected_revision: 0, + command: FenceCommand::AcquireSourceWriteFence { + expected_log_id: log_id, + drain_policy: Some(policy), + }, + } +} + +fn release(command_id: u128, revision: u64) -> FenceRequest { + FenceRequest { + namespace: "ns".into(), + operation_id: OP, + command_id: Uuid::from_u128(command_id), + expected_state: FenceState::SourceWriteFenced, + expected_revision: revision, + command: FenceCommand::ReleaseSourceWriteFence, + } +} + +fn fence_error(result: &crate::Result) -> &FenceError { + match result { + Err(Error::NamespaceFence(e)) => e, + other => panic!("expected a fence error, got {other:?}"), + } +} + +fn boundary(commit: &FenceCommit) -> FrozenBoundary { + commit + .record + .as_ref() + .and_then(|r| r.frozen_boundary) + .expect("a write-fenced source has a frozen boundary") +} + +/// The marker file's bytes, if there is one. +fn read_marker_bytes(dbs: &Path) -> Option> { + match std::fs::read(fence_store::marker_path(dbs, &"ns".into())) { + Ok(bytes) => Some(bytes), + Err(e) if e.kind() == std::io::ErrorKind::NotFound => None, + Err(e) => panic!("{e}"), + } +} + +/// Put the marker file back to `bytes` (`None`: no marker). +fn restore_marker_bytes(dbs: &Path, bytes: Option<&[u8]>) { + let path = fence_store::marker_path(dbs, &"ns".into()); + match bytes { + Some(bytes) => std::fs::write(path, bytes).unwrap(), + None => std::fs::remove_file(path).unwrap(), + } +} + +async fn reached(paused: &Paused, case: &str, point: HookPoint) { + tokio::time::timeout(PROMPT, paused.reached()) + .await + .unwrap_or_else(|_| panic!("{case}: the command never reached {point:?}")); +} + +#[derive(Debug, Clone, Copy)] +enum Command { + Acquire, + Release, +} + +/// What the metastore holds for the command when the process dies. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +enum Durable { + /// Nothing of the command. + Nothing, + /// The `SOURCE_DRAINING` record and the command's `DRAINING` receipt (acquisition only). + Draining, + /// The command's final result. + Final, +} + +#[derive(Debug, Clone, Copy)] +struct Boundary { + name: &'static str, + command: Command, + point: HookPoint, + /// The point is the one reached on the acquisition's second commit (the completion of the + /// drain), not its first. + second_commit: bool, + /// The marker file is put back to what it held before the commit, as a crash between the + /// metastore commit and the marker write leaves it. + marker_lags: bool, + durable: Durable, +} + +const BOUNDARIES: &[Boundary] = &[ + Boundary { + name: "acquire/after-installing-gate", + command: Command::Acquire, + point: HookPoint::AfterInstallingGate, + second_commit: false, + marker_lags: false, + durable: Durable::Nothing, + }, + Boundary { + name: "acquire/before-draining-commit", + command: Command::Acquire, + point: HookPoint::BeforeMetastoreCommit, + second_commit: false, + marker_lags: false, + durable: Durable::Nothing, + }, + Boundary { + name: "acquire/after-draining-commit", + command: Command::Acquire, + point: HookPoint::AfterMetastoreCommit, + second_commit: false, + marker_lags: false, + durable: Durable::Draining, + }, + Boundary { + name: "acquire/after-draining-commit/marker-lags", + command: Command::Acquire, + point: HookPoint::AfterMetastoreCommit, + second_commit: false, + marker_lags: true, + durable: Durable::Draining, + }, + Boundary { + name: "acquire/before-draining-publish", + command: Command::Acquire, + point: HookPoint::BeforeGatePublish, + second_commit: false, + marker_lags: false, + durable: Durable::Draining, + }, + Boundary { + name: "acquire/draining-published", + command: Command::Acquire, + point: HookPoint::BeforeResponse, + second_commit: false, + marker_lags: false, + durable: Durable::Draining, + }, + Boundary { + name: "acquire/before-boundary-capture", + command: Command::Acquire, + point: HookPoint::BeforeBoundaryCapture, + second_commit: false, + marker_lags: false, + durable: Durable::Draining, + }, + Boundary { + name: "acquire/before-fenced-commit", + command: Command::Acquire, + point: HookPoint::BeforeMetastoreCommit, + second_commit: true, + marker_lags: false, + durable: Durable::Draining, + }, + Boundary { + name: "acquire/after-fenced-commit", + command: Command::Acquire, + point: HookPoint::AfterMetastoreCommit, + second_commit: true, + marker_lags: false, + durable: Durable::Final, + }, + Boundary { + name: "acquire/after-fenced-commit/marker-lags", + command: Command::Acquire, + point: HookPoint::AfterMetastoreCommit, + second_commit: true, + marker_lags: true, + durable: Durable::Final, + }, + Boundary { + name: "acquire/before-fenced-publish", + command: Command::Acquire, + point: HookPoint::BeforeGatePublish, + second_commit: true, + marker_lags: false, + durable: Durable::Final, + }, + Boundary { + name: "acquire/before-fenced-response", + command: Command::Acquire, + point: HookPoint::BeforeResponse, + second_commit: true, + marker_lags: false, + durable: Durable::Final, + }, + Boundary { + name: "release/before-commit", + command: Command::Release, + point: HookPoint::BeforeMetastoreCommit, + second_commit: false, + marker_lags: false, + durable: Durable::Nothing, + }, + Boundary { + name: "release/after-commit", + command: Command::Release, + point: HookPoint::AfterMetastoreCommit, + second_commit: false, + marker_lags: false, + durable: Durable::Final, + }, + Boundary { + name: "release/after-commit/marker-lags", + command: Command::Release, + point: HookPoint::AfterMetastoreCommit, + second_commit: false, + marker_lags: true, + durable: Durable::Final, + }, + Boundary { + name: "release/before-publish", + command: Command::Release, + point: HookPoint::BeforeGatePublish, + second_commit: false, + marker_lags: false, + durable: Durable::Final, + }, + Boundary { + name: "release/before-response", + command: Command::Release, + point: HookPoint::BeforeResponse, + second_commit: false, + marker_lags: false, + durable: Durable::Final, + }, +]; + +/// The state a restart recovers for `case`: the one before the command, or the one it +/// committed. Never anything else. +fn recovered_state(case: &Boundary) -> FenceState { + match (case.command, case.durable) { + (Command::Acquire, Durable::Nothing) => FenceState::Unfenced, + (Command::Acquire, Durable::Draining) => FenceState::SourceDraining, + (Command::Acquire, Durable::Final) => FenceState::SourceWriteFenced, + (Command::Release, Durable::Nothing) => FenceState::SourceWriteFenced, + (Command::Release, Durable::Final) => FenceState::Released, + (Command::Release, Durable::Draining) => unreachable!(), + } +} + +/// Kill the process at every point where a fence command persists, publishes or answers, and +/// restart it on the same directory. The restarted server recovers exactly the state before the +/// command or the state it committed, installs that gate before it serves the namespace (so a +/// namespace is never open unless an opening transition committed), and a replay of the same +/// command then finishes it with the stored or the expected result. +#[test] +fn restart_at_each_boundary() { + for case in BOUNDARIES { + restart_at(case); + } +} + +fn restart_at(case: &Boundary) { + let name = case.name; + let dir = tempdir().unwrap(); + let dbs = dir.path().join("dbs"); + + // First lifetime: run the command until it reaches the boundary, then crash. + let server = Server::boot(dir.path()); + server.create_source(); + let (log_before, _) = server.run(server.log()); + let (request, revision_before) = match case.command { + Command::Acquire => (acquire(log_before, 1, LONG), 0), + Command::Release => { + let fenced = server.fence_source(); + let revision = fenced.record.as_ref().unwrap().revision; + (release(2, revision), revision) + } + }; + let committed_boundary = server.run(async { + let fence = server.fence().await; + let hooks = fence.hooks(); + let mut marker_before = read_marker_bytes(&dbs); + let paused = if case.second_commit { + let capture = hooks.pause_at(HookPoint::BeforeBoundaryCapture); + let task = server.execute(request.clone()); + reached(&capture, name, HookPoint::BeforeBoundaryCapture).await; + marker_before = read_marker_bytes(&dbs); + let paused = hooks.pause_at(case.point); + capture.resume(); + reached(&paused, name, case.point).await; + drop(task); + paused + } else { + let paused = hooks.pause_at(case.point); + let task = server.execute(request.clone()); + reached(&paused, name, case.point).await; + drop(task); + paused + }; + if case.marker_lags { + restore_marker_bytes(&dbs, marker_before.as_deref()); + } + // What the metastore holds at the moment of the crash. + let inspected = server.inspect().await; + assert_eq!(inspected.fence.state(), recovered_state(case), "{name}"); + drop(paused); + inspected.fence.record().and_then(|r| r.frozen_boundary) + }); + server.crash(); + + // Second lifetime. + let server = Server::boot(dir.path()); + server.run(async { + let expected = recovered_state(case); + let fence = server.fence().await; + let gate = fence.gate(); + assert_eq!(gate.state(), expected, "{name}: recovered state"); + assert!( + gate.indeterminate.is_none() && !gate.is_installing(), + "{name}" + ); + let open = matches!(expected, FenceState::Unfenced | FenceState::Released); + assert_eq!( + server.writes_admitted().await, + open, + "{name}: write admission" + ); + // Committed data survived, nothing else was written. + assert_eq!(server.count().await, ROWS, "{name}"); + // The marker was repaired if it had fallen behind. + let marker = fence_store::read_marker(&dbs, &"ns".into()).unwrap(); + assert_eq!( + marker.and_then(|m| m.ok()).map(|m| m.record.revision), + gate.fence.record().map(|r| r.revision), + "{name}: marker" + ); + + // A crash leaves the namespace dirty, so its replication log was rebuilt from the + // database file under a new log id. + let (log_after, frame_after) = server.log().await; + assert_ne!(log_after, log_before, "{name}: the log was not rebuilt"); + + let replay = server.execute(request.clone()).await.unwrap(); + match (case.command, case.durable) { + (Command::Acquire, Durable::Nothing) => { + // Nothing was acknowledged and nothing was written, and the identity the caller + // observed is gone: the replay is refused before anything is written, and the + // caller acquires again under the identity it reads now. + let e = fence_error(&replay); + assert_eq!(e.outcome(), FenceOutcome::FencePreconditionFailed, "{name}"); + assert_eq!(e.detail(), Some(FenceDetail::NamespaceIdentityMismatch)); + assert_eq!(fence.gate().state(), FenceState::Unfenced, "{name}"); + assert!(server.writes_admitted().await, "{name}"); + let commit = server + .execute(acquire(log_after, 3, LONG)) + .await + .unwrap() + .unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied, "{name}"); + assert_eq!( + boundary(&commit), + FrozenBoundary { + log_id: log_after, + frame_no: frame_after, + }, + "{name}" + ); + } + (Command::Acquire, Durable::Draining) => { + // The drain that was requested resumes and completes at once: recovery + // discarded any uncommitted work. The boundary is on the live, rebuilt log. + let commit = replay.unwrap(); + assert_eq!(commit.kind, FenceCommitKind::Committed, "{name}"); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied, "{name}"); + assert_eq!(commit.receipt.command_id, request.command_id); + assert_eq!(commit.receipt.revision_after, 2, "{name}"); + let record = commit.record.as_ref().unwrap(); + assert_eq!(record.identity.log_id, Some(log_before), "{name}"); + assert_eq!( + boundary(&commit), + FrozenBoundary { + log_id: log_after, + frame_no: frame_after, + }, + "{name}" + ); + } + (Command::Acquire, Durable::Final) => { + // The stored result, boundary included: it names the log that was live when + // the drain was proven. + let commit = replay.unwrap(); + assert_eq!(commit.kind, FenceCommitKind::Replayed, "{name}"); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied, "{name}"); + assert_eq!(Some(boundary(&commit)), committed_boundary, "{name}"); + assert_eq!(boundary(&commit).log_id, log_before, "{name}"); + } + (Command::Release, durable) => { + let commit = replay.unwrap(); + let kind = if durable == Durable::Final { + FenceCommitKind::Replayed + } else { + FenceCommitKind::Committed + }; + assert_eq!(commit.kind, kind, "{name}"); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied, "{name}"); + assert_eq!(commit.receipt.revision_before, revision_before, "{name}"); + assert_eq!(commit.receipt.state_after, FenceState::Released, "{name}"); + } + } + + // Settled: the gate is the durable state, and a further replay answers the same. + let gate = fence.gate(); + let durable = server.inspect().await; + assert_eq!(gate.fence, durable.fence, "{name}"); + let open = gate.state() == FenceState::Released; + assert_eq!(server.writes_admitted().await, open, "{name}"); + assert_eq!(server.count().await, ROWS, "{name}"); + }); + server.crash(); +} + +/// After a restart in `SOURCE_DRAINING` with a writer that was active at the crash, nothing +/// advances by itself: the namespace stays closed, other commands cannot move it on, and only +/// the replay of the same acquisition completes the drain, at once. +#[test] +fn restart_in_draining_waits_for_the_same_command() { + let dir = tempdir().unwrap(); + + let server = Server::boot(dir.path()); + server.create_source(); + let (log_before, _) = server.run(server.log()); + let request = acquire(log_before, 1, NOW); + server.run(async { + let holder = server.conn().await; + raw(&holder, "begin immediate; insert into t values (2);") + .await + .unwrap(); + let commit = server.execute(request.clone()).await.unwrap().unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Draining); + // The writer never finishes: its transaction dies with the process (closing the + // connection rolls it back, which is what SQLite recovery does to it on disk). + drop(holder); + }); + server.crash(); + + let server = Server::boot(dir.path()); + server.run(async { + let fence = server.fence().await; + assert_eq!(fence.gate().state(), FenceState::SourceDraining); + assert!(!server.writes_admitted().await); + // The uncommitted write is gone. + assert_eq!(server.count().await, ROWS); + + // Reads are served and move nothing; other commands cannot advance the namespace. + let set_read_fence = FenceRequest { + namespace: "ns".into(), + operation_id: OP, + command_id: Uuid::from_u128(2), + expected_state: FenceState::SourceWriteFenced, + expected_revision: 2, + command: FenceCommand::SetSourceReadFence { drain_policy: None }, + }; + let r = server.execute(set_read_fence).await.unwrap(); + assert!(fence_error(&r).outcome() != FenceOutcome::Applied); + let (log_after, frame_after) = server.log().await; + let mut other = acquire(log_after, 3, LONG); + other.operation_id = OTHER_OP; + let r = server.execute(other).await.unwrap(); + assert_eq!( + fence_error(&r).outcome(), + FenceOutcome::FenceOwnedByAnotherOperation + ); + let inspected = server.inspect().await; + assert_eq!(inspected.fence.state(), FenceState::SourceDraining); + assert_eq!(inspected.fence.revision(), 1); + assert_eq!(fence.gate().state(), FenceState::SourceDraining); + + // The same command, with the same zero deadline, completes at once. + let commit = server.execute(request.clone()).await.unwrap().unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + assert_eq!(commit.receipt.command_id, request.command_id); + assert_eq!( + boundary(&commit), + FrozenBoundary { + log_id: log_after, + frame_no: frame_after, + } + ); + assert_eq!(fence.gate().state(), FenceState::SourceWriteFenced); + assert!(!server.writes_admitted().await); + assert_eq!(server.count().await, ROWS); + }); + server.crash(); +} + +/// A commit whose outcome is unknown closes the namespace (every class but maintenance and +/// observability), makes every other command answer `FENCE_COMMIT_INDETERMINATE`, and is +/// reconciled by replaying the same command from the durable row: both when the commit had +/// happened and when it had not. +#[test] +fn indeterminate_commit_keeps_gate_closed() { + for committed in [true, false] { + let dir = tempdir().unwrap(); + let server = Server::boot(dir.path()); + server.create_source(); + server.run(async { + let (log_id, frame_no) = server.log().await; + let fence = server.fence().await; + let request = acquire(log_id, 1, LONG); + fence.hooks().arm( + if committed { + HookPoint::AfterMetastoreCommit + } else { + HookPoint::BeforeMetastoreCommit + }, + HookAction::Indeterminate, + ); + let r = server.execute(request.clone()).await.unwrap(); + let e = fence_error(&r); + assert_eq!(e.outcome(), FenceOutcome::FenceCommitIndeterminate); + assert_eq!(e.detail(), Some(FenceDetail::IndeterminateCommit)); + + let durable = server.inspect().await.fence.state(); + assert_eq!( + durable, + if committed { + FenceState::SourceDraining + } else { + FenceState::Unfenced + } + ); + let gate = fence.gate(); + assert_eq!(gate.indeterminate, Some((OP, request.command_id))); + assert!(!gate.is_installing()); + for class in OperationClass::ALL { + let permitted = fence.permits(class); + match class { + OperationClass::Maintenance | OperationClass::Observability => { + assert!(permitted.is_ok()) + } + _ => assert_eq!( + permitted.unwrap_err().outcome(), + FenceOutcome::FenceStateUnavailable, + "{class:?}" + ), + } + } + assert!(!server.writes_admitted().await); + + // Any other command is refused with the indeterminate code, nothing is written. + let mut other = acquire(log_id, 2, LONG); + other.operation_id = OTHER_OP; + let r = server.execute(other).await.unwrap(); + assert_eq!( + fence_error(&r).outcome(), + FenceOutcome::FenceCommitIndeterminate + ); + let r = server.execute(acquire(log_id, 3, LONG)).await.unwrap(); + assert_eq!( + fence_error(&r).outcome(), + FenceOutcome::FenceCommitIndeterminate + ); + assert_eq!(server.inspect().await.fence.state(), durable); + + // The replay reconciles from the durable row: it resumes the drain that committed, + // or runs the acquisition that did not, and completes it. + let commit = server.execute(request.clone()).await.unwrap().unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + assert_eq!(commit.receipt.command_id, request.command_id); + assert_eq!(commit.receipt.revision_after, 2); + assert_eq!(boundary(&commit), FrozenBoundary { log_id, frame_no }); + let gate = fence.gate(); + assert_eq!(gate.indeterminate, None); + assert_eq!(gate.state(), FenceState::SourceWriteFenced); + assert_eq!(gate.fence, server.inspect().await.fence); + assert!(!server.writes_admitted().await); + // Maintenance and reads are served again. + assert_eq!(server.count().await, ROWS); + }); + server.crash(); + } +} + +/// The caller of an acquisition goes away before it hears the answer, once while the drain is +/// still waiting and once just before the final response. The acquisition finishes on its own +/// task regardless, `InspectFence` shows the result, and a replay of the same command returns +/// it. +#[test] +fn acquire_response_loss_resolved_by_replay_and_inspect() { + for lose_at in [HookPoint::AfterMetastoreCommit, HookPoint::BeforeResponse] { + let dir = tempdir().unwrap(); + let server = Server::boot(dir.path()); + server.create_source(); + server.run(async { + let (log_id, _) = server.log().await; + let fence = server.fence().await; + let request = acquire(log_id, 1, LONG); + let holder = server.conn().await; + raw(&holder, "begin immediate; insert into t values (2);") + .await + .unwrap(); + + let hooks = fence.hooks(); + let capture = hooks.pause_at(HookPoint::BeforeBoundaryCapture); + let first = (lose_at == HookPoint::AfterMetastoreCommit) + .then(|| hooks.pause_at(HookPoint::AfterMetastoreCommit)); + let caller = server.execute(request.clone()); + if let Some(first) = first { + // The caller is gone right after SOURCE_DRAINING commits. + reached(&first, "response loss", HookPoint::AfterMetastoreCommit).await; + caller.abort(); + first.resume(); + } + + // The drain waits for the writer, with or without a caller. + let mut rx = fence.subscribe(); + tokio::time::timeout( + PROMPT, + rx.wait_for(|g| g.state() == FenceState::SourceDraining), + ) + .await + .unwrap() + .unwrap(); + assert_eq!( + server.inspect().await.fence.state(), + FenceState::SourceDraining + ); + // A replay while it runs waits for the transition lock rather than racing it. + let replay_during = server.execute(request.clone()); + + raw(&holder, "commit").await.unwrap(); + let (_, committed) = server.log().await; + reached(&capture, "response loss", HookPoint::BeforeBoundaryCapture).await; + if lose_at == HookPoint::BeforeResponse { + // The caller is gone after the final commit was published, before the answer. + let last = hooks.pause_at(HookPoint::BeforeResponse); + capture.resume(); + reached(&last, "response loss", HookPoint::BeforeResponse).await; + assert_eq!(fence.gate().state(), FenceState::SourceWriteFenced); + caller.abort(); + last.resume(); + } else { + capture.resume(); + } + assert!(caller.await.unwrap_err().is_cancelled()); + tokio::time::timeout( + PROMPT, + rx.wait_for(|g| g.state() == FenceState::SourceWriteFenced), + ) + .await + .unwrap() + .unwrap(); + + let inspected = server.inspect().await; + assert_eq!(inspected.fence.state(), FenceState::SourceWriteFenced); + let record = inspected.fence.record().unwrap(); + assert_eq!( + record.frozen_boundary, + Some(FrozenBoundary { + log_id, + frame_no: committed, + }) + ); + let receipt = inspected + .receipts + .iter() + .filter_map(|r| r.receipt.as_ref().ok()) + .find(|r| r.command_id == request.command_id) + .unwrap() + .clone(); + assert_eq!(receipt.outcome, FenceOutcome::Applied); + + for replay in [ + replay_during.await.unwrap().unwrap(), + server.execute(request.clone()).await.unwrap().unwrap(), + ] { + assert_eq!(replay.kind, FenceCommitKind::Replayed); + assert_eq!(replay.receipt, receipt); + assert_eq!(replay.record.as_ref(), Some(record)); + } + assert_eq!(server.count().await, ROWS + 1); + assert!(!server.writes_admitted().await); + }); + server.crash(); + } +} diff --git a/libsql-server/src/namespace/fence/transition.rs b/libsql-server/src/namespace/fence/transition.rs index f1de43f747..9c664b320c 100644 --- a/libsql-server/src/namespace/fence/transition.rs +++ b/libsql-server/src/namespace/fence/transition.rs @@ -673,12 +673,13 @@ pub fn complete_drain( let mut next = record.clone(); if let DrainCompletion::SourceWrites { boundary } = completion { - if record.identity.log_id != Some(boundary.log_id) { - return Err(precondition( - FenceDetail::NamespaceIdentityMismatch, - "the frozen boundary belongs to a different replication log", - )); - } + // The boundary names the replication log that is live when the drain is proven. It is + // not the log the identity was captured on when the source was restarted while + // draining: crash recovery rebuilds the log from the database file under a new id + // (section 8.5). Write admission has been durably closed since `SOURCE_DRAINING` + // committed, and the lifecycle paths that could replace the database are denied, so + // the data is what was committed before the cutoff. The identity keeps the log id the + // caller acquired against. next.frozen_boundary = Some(boundary); } next.state = to; @@ -903,7 +904,7 @@ mod tests { let boundary = DrainCompletion::SourceWrites { boundary: FrozenBoundary { log_id: LOG, - frame_no: 42, + frame_no: Some(42), }, }; let mut h = if state.role() == Some(Role::Target) || state == S::Absent { @@ -1083,14 +1084,14 @@ mod tests { h.complete(DrainCompletion::SourceWrites { boundary: FrozenBoundary { log_id: LOG, - frame_no: 7, + frame_no: Some(7), }, }); assert_eq!(h.state(), S::SourceWriteFenced); assert_eq!(h.revision(), 2); assert_eq!( h.record.as_ref().unwrap().frozen_boundary.unwrap().frame_no, - 7 + Some(7) ); let acquire_receipt = h .receipts @@ -1255,7 +1256,7 @@ mod tests { h.complete(DrainCompletion::SourceWrites { boundary: FrozenBoundary { log_id: LOG, - frame_no: 1, + frame_no: Some(1), }, }); h.run(OP, CommandKind::SetSourceReadFence).unwrap(); @@ -1464,7 +1465,7 @@ mod tests { DrainCompletion::SourceWrites { boundary: FrozenBoundary { log_id: LOG, - frame_no: 3, + frame_no: Some(3), }, }, &env(), @@ -1485,20 +1486,21 @@ mod tests { assert!(complete_drain(&record, &receipt, DrainCompletion::SourceReads, &env()).is_err()); assert!(complete_drain(&record, &receipt, DrainCompletion::TargetImport, &env()).is_err()); - // A boundary from another log. - let err = complete_drain( + // A boundary on a log rebuilt since acquisition (a restart while draining) is recorded + // as it is; the identity keeps the log the caller acquired against. + let rebuilt = FrozenBoundary { + log_id: Uuid::from_u128(0x77), + frame_no: Some(1), + }; + let (next, _) = complete_drain( &record, &receipt, - DrainCompletion::SourceWrites { - boundary: FrozenBoundary { - log_id: Uuid::from_u128(0x77), - frame_no: 1, - }, - }, + DrainCompletion::SourceWrites { boundary: rebuilt }, &env(), ) - .unwrap_err(); - assert_eq!(err.detail(), Some(FenceDetail::NamespaceIdentityMismatch)); + .unwrap(); + assert_eq!(next.frozen_boundary, Some(rebuilt)); + assert_eq!(next.identity.log_id, Some(LOG)); // A final receipt, or another operation's. let mut final_receipt = receipt.clone(); @@ -1506,7 +1508,7 @@ mod tests { let boundary = DrainCompletion::SourceWrites { boundary: FrozenBoundary { log_id: LOG, - frame_no: 1, + frame_no: Some(1), }, }; assert!(complete_drain(&record, &final_receipt, boundary, &env()).is_err()); @@ -1732,7 +1734,7 @@ mod tests { h.complete(DrainCompletion::SourceWrites { boundary: FrozenBoundary { log_id: LOG, - frame_no: 5, + frame_no: Some(5), }, }); assert_eq!(h.state(), S::SourceWriteFenced); diff --git a/libsql-server/src/namespace/meta_store.rs b/libsql-server/src/namespace/meta_store.rs index 175fb24b55..1a16994c3f 100644 --- a/libsql-server/src/namespace/meta_store.rs +++ b/libsql-server/src/namespace/meta_store.rs @@ -31,7 +31,7 @@ use crate::{ config::MetaStoreConfig, connection::legacy::open_conn_active_checkpoint, error::Error, Result, }; -use super::fence::command::{FenceCommand, FenceRequest}; +use super::fence::command::{DrainPolicy, FenceCommand, FenceRequest, OnDeadline}; use super::fence::outcome::{FenceDetail, FenceError, FenceOutcome}; use super::fence::record::{ CommandReceipt, NamespaceFenceRecord, ServerIdentity, ValidationSnapshot, @@ -110,6 +110,8 @@ struct FenceSettings { /// namespace directory holds a marker. fail_closed: bool, receipt_retention: Duration, + /// The write drain deadline of an `AcquireSourceWriteFence` that names no drain policy. + default_write_drain: Duration, } fn setup_connection(conn: &rusqlite::Connection) -> Result<()> { @@ -235,6 +237,9 @@ impl MetaStoreInner { receipt_retention: config .namespace_fence_receipt_retention .unwrap_or(fence_store::DEFAULT_RECEIPT_RETENTION), + default_write_drain: config + .namespace_fence_default_write_drain + .unwrap_or(crate::namespace::fence::drain::DEFAULT_WRITE_DRAIN), }; let mut this = MetaStoreInner { @@ -804,6 +809,17 @@ fn not_primary() -> FenceError { .with_detail(FenceDetail::NotPrimary) } +/// A fence transaction whose `COMMIT` failed: whether it took effect is unknown, and the +/// controller keeps the namespace closed until the same command is replayed (section 8.4). +fn commit_indeterminate(e: rusqlite::Error) -> FenceStoreError { + FenceError::new( + FenceOutcome::FenceCommitIndeterminate, + format!("the metastore commit of a fence transition failed: {e}"), + ) + .with_detail(FenceDetail::IndeterminateCommit) + .into() +} + fn unavailable_receipt(e: impl std::fmt::Display) -> FenceError { FenceError::new( FenceOutcome::FenceStateUnavailable, @@ -918,7 +934,7 @@ fn apply_fence_command( .or(stored.record()) .map_or(request.operation_id, |r| r.operation_id); fence_store::prune_receipts(&tx, ns, owner, ctx.now_ms, inner.fence.receipt_retention)?; - tx.commit()?; + tx.commit().map_err(commit_indeterminate)?; // The command established the fence from the durable state; whatever startup could not // recover about this name is settled. inner.recovered.lock().remove(ns); @@ -1029,7 +1045,7 @@ fn complete_fence_drain( } fence_store::write_record(&tx, &next, fence_store::stored_revision(&tx, ns)?)?; fence_store::write_receipt(&tx, &final_receipt)?; - tx.commit()?; + tx.commit().map_err(commit_indeterminate)?; inner.recovered.lock().remove(ns); after_fence_commit(inner, &conn, Some(&next), true); @@ -1289,6 +1305,16 @@ impl MetaStore { self.inner.fence.enabled } + /// The drain policy of an `AcquireSourceWriteFence` that names none: the configured + /// deadline, then `DRAINING`. + pub fn fence_default_write_drain(&self) -> DrainPolicy { + DrainPolicy { + deadline_ms: u64::try_from(self.inner.fence.default_write_drain.as_millis()) + .unwrap_or(u64::MAX), + on_deadline: OnDeadline::Fail, + } + } + /// Whether this metastore holds fence state, so fences are loaded and enforced. pub fn fence_enforced(&self) -> bool { self.inner.fence.tables @@ -1738,7 +1764,7 @@ mod fence_tests { let boundary = FrozenBoundary { log_id: LOG, - frame_no: 42, + frame_no: Some(42), }; let commit = store .complete_fence_drain( @@ -2113,7 +2139,7 @@ mod fence_tests { DrainCompletion::SourceWrites { boundary: FrozenBoundary { log_id: LOG, - frame_no: 1, + frame_no: Some(1), }, }, ctx(2_000), @@ -2164,7 +2190,7 @@ mod fence_tests { DrainCompletion::SourceWrites { boundary: FrozenBoundary { log_id: LOG, - frame_no: 7, + frame_no: Some(7), }, }, ctx(2_000), diff --git a/libsql-server/src/namespace/mod.rs b/libsql-server/src/namespace/mod.rs index cba4030090..28ba60ea26 100644 --- a/libsql-server/src/namespace/mod.rs +++ b/libsql-server/src/namespace/mod.rs @@ -14,6 +14,7 @@ use crate::connection::Connection as _; use crate::database::Database; use crate::stats::Stats; +use self::fence::controller::FenceController; use self::meta_store::MetaStoreHandle; pub use self::name::NamespaceName; pub use self::store::NamespaceStore; @@ -66,6 +67,10 @@ pub struct Namespace { stats: Arc, db_config_store: MetaStoreHandle, path: Arc, + /// The namespace's fence controller, from the store's registry. Every connection, and the + /// dump and replication services that reach the namespace through the store, read its + /// gate. + fence: Arc, } impl Namespace { @@ -98,6 +103,13 @@ impl Namespace { Ok(()) } + // Read by the protocol layers that consult the gate outside a connection (dump, + // replication, lifecycle), which land later in this series. + #[allow(dead_code)] + pub(crate) fn fence(&self) -> &Arc { + &self.fence + } + pub fn config(&self) -> Arc { self.db_config_store.get() } diff --git a/libsql-server/src/namespace/store.rs b/libsql-server/src/namespace/store.rs index 1813ef5187..af406dfd96 100644 --- a/libsql-server/src/namespace/store.rs +++ b/libsql-server/src/namespace/store.rs @@ -21,7 +21,10 @@ use crate::stats::Stats; use super::broadcasters::{BroadcasterHandle, BroadcasterRegistry}; use super::configurator::{DynConfigurator, NamespaceConfigurators}; -use super::meta_store::{MetaStore, MetaStoreHandle}; +use super::fence::command::{FenceCommand, FenceRequest}; +use super::fence::record::ServerIdentity; +use super::fence::registry::FenceRegistry; +use super::meta_store::{FenceCommit, FenceContext, MetaStore, MetaStoreHandle}; use super::schema_lock::SchemaLocksRegistry; use super::{Namespace, ResetCb, ResetOp, ResolveNamespacePathFn, RestoreOption}; @@ -50,6 +53,9 @@ pub struct NamespaceStoreInner { broadcasters: BroadcasterRegistry, configurators: NamespaceConfigurators, db_kind: DatabaseKind, + /// Fence controllers, outside the cache: a namespace that is evicted and reloaded gets the + /// controller it had. + fences: FenceRegistry, } impl NamespaceStore { @@ -84,6 +90,13 @@ impl NamespaceStore { .time_to_idle(Duration::from_secs(86400)) .build(); + // Every namespace with fence state gets its controller before anything is served + // (section 8.5). + let fences = FenceRegistry::seeded(metadata.load_fences().await?); + if fences.len() > 0 { + tracing::info!("loaded {} namespace fence controllers", fences.len()); + } + Ok(Self { inner: Arc::new(NamespaceStoreInner { store, @@ -95,6 +108,7 @@ impl NamespaceStore { broadcasters: Default::default(), configurators, db_kind, + fences, }), }) } @@ -120,6 +134,8 @@ impl NamespaceStore { } }) .await??; + // The namespace's fence state went with it. + self.inner.fences.remove(&namespace); let mut bottomless_db_id_init = NamespaceBottomlessDbIdInit::FetchFromConfig; if let Some(ns) = self.inner.store.remove(&namespace).await { @@ -336,6 +352,9 @@ impl NamespaceStore { } }; + // A namespace whose fence state is unavailable is refused before any setup work. + self.inner.fences.check_available(&namespace)?; + // A lookup that cannot create: only the default namespace and lazy creation create a // namespace here, and those refuse a name whose fence state is not established. let handle = match self.inner.metadata.lookup(&namespace).await? { @@ -371,6 +390,10 @@ impl NamespaceStore { config: MetaStoreHandle, restore_option: RestoreOption, ) -> crate::Result { + // The controller is handed to the namespace before its first connection exists, so no + // connection is ever opened without a gate (section 8.5). + self.inner.fences.check_available(namespace)?; + let fence = self.inner.fences.controller(namespace); let ns = self .get_configurator(&config.get()) .setup( @@ -381,6 +404,7 @@ impl NamespaceStore { self.resolve_attach_fn(), self.clone(), self.broadcaster(namespace.clone()), + fence, ) .await?; @@ -509,6 +533,33 @@ impl NamespaceStore { &self.inner.metadata } + /// Run one fence command on its namespace, including the drain it starts + /// (`docs/NAMESPACE_FENCE.md` sections 5.3 and 8). `AcquireSourceWriteFence` loads the + /// namespace first, so that its connection manager and replication log are registered with + /// the namespace's controller before the drain needs them. + // The admin routes that call this are not part of the server yet. + #[cfg_attr(not(test), allow(dead_code))] + pub(crate) async fn execute_fence_command( + &self, + request: FenceRequest, + server: ServerIdentity, + ) -> crate::Result { + let controller = match request.command { + FenceCommand::AcquireSourceWriteFence { .. } => { + self.with(request.namespace.clone(), |ns| ns.fence().clone()) + .await? + } + _ => self.inner.fences.controller(&request.namespace), + }; + controller + .execute( + &self.inner.metadata, + request, + FenceContext::now(server, None), + ) + .await + } + pub(crate) fn schema_locks(&self) -> &SchemaLocksRegistry { &self.inner.schema_locks } @@ -535,3 +586,210 @@ impl NamespaceStore { .await } } + +#[cfg(test)] +pub(crate) mod fence_tests { + use std::path::Path; + + use libsql_sys::wal::Sqlite3WalManager; + use tempfile::tempdir; + use tokio::sync::Semaphore; + use uuid::Uuid; + + use super::*; + use crate::config::MetaStoreConfig; + use crate::namespace::configurator::{BaseNamespaceConfig, PrimaryConfig, PrimaryConfigurator}; + use crate::namespace::fence::command::{FenceCommand, FenceRequest}; + use crate::namespace::fence::outcome::{FenceDetail, FenceOutcome}; + use crate::namespace::fence::record::ServerIdentity; + use crate::namespace::fence::state::{FenceState, OperationClass}; + use crate::namespace::fence::store as fence_store; + use crate::namespace::meta_store::{metastore_connection_maker, FenceContext}; + + const LOG: Uuid = Uuid::from_u128(0x10); + const OP: Uuid = Uuid::from_u128(0xa); + + pub(crate) async fn open_store(dir: &Path) -> NamespaceStore { + let (maker, manager) = metastore_connection_maker(None, dir).await.unwrap(); + let meta = MetaStore::new( + MetaStoreConfig { + namespace_fence: true, + ..Default::default() + }, + dir, + maker().unwrap(), + manager, + DatabaseKind::Primary, + ) + .await + .unwrap(); + let mut configurators = NamespaceConfigurators::empty(); + configurators.with_primary(PrimaryConfigurator::new( + BaseNamespaceConfig { + base_path: dir.to_path_buf().into(), + extensions: Arc::new([]), + stats_sender: tokio::sync::mpsc::channel(1).0, + max_response_size: 100_000_000, + max_total_response_size: 100_000_000, + max_concurrent_connections: Arc::new(Semaphore::new(10)), + max_concurrent_requests: 10_000, + encryption_config: None, + connection_creation_timeout: None, + disable_intelligent_throttling: false, + }, + PrimaryConfig { + max_log_size: 1_000_000_000, + max_log_duration: None, + bottomless_replication: None, + scripted_backup: None, + checkpoint_interval: None, + }, + Arc::new(|| Sqlite3WalManager::default()), + )); + NamespaceStore::new(false, false, 10, meta, configurators, DatabaseKind::Primary) + .await + .unwrap() + } + + fn acquire(ns: &'static str) -> FenceRequest { + FenceRequest { + namespace: ns.into(), + operation_id: OP, + command_id: Uuid::from_u128(1), + expected_state: FenceState::Unfenced, + expected_revision: 0, + command: FenceCommand::AcquireSourceWriteFence { + expected_log_id: LOG, + drain_policy: None, + }, + } + } + + fn ctx() -> FenceContext { + FenceContext::now( + ServerIdentity { + build: "test".into(), + instance_id: Uuid::from_u128(0x99), + }, + Some(LOG), + ) + } + + #[tokio::test] + async fn evicted_namespace_reloads_with_the_same_controller() { + let tmp = tempdir().unwrap(); + let store = open_store(tmp.path()).await; + store + .create("ns".into(), RestoreOption::Latest, Default::default()) + .await + .unwrap(); + let (fence, stats) = store + .with("ns".into(), |ns| (ns.fence().clone(), ns.stats())) + .await + .unwrap(); + assert!(Arc::ptr_eq( + &fence, + &store.inner.fences.get(&"ns".into()).unwrap() + )); + fence + .apply_command(store.meta_store(), acquire("ns"), ctx()) + .await + .unwrap(); + let gate = fence.gate(); + assert_eq!(gate.state(), FenceState::SourceDraining); + + // Evict the namespace, as idle or capacity eviction does, and let it shut down. + store + .inner + .store + .invalidate(&NamespaceName::from("ns")) + .await; + store.inner.store.run_pending_tasks().await; + assert!(store + .inner + .store + .get(&NamespaceName::from("ns")) + .await + .is_none()); + + let (reloaded, reloaded_stats) = store + .with("ns".into(), |ns| (ns.fence().clone(), ns.stats())) + .await + .unwrap(); + // A new namespace instance, the same controller and gate. + assert!(!Arc::ptr_eq(&stats, &reloaded_stats)); + assert!(Arc::ptr_eq(&fence, &reloaded)); + assert_eq!(reloaded.gate(), gate); + assert!(reloaded.permits(OperationClass::NormalWrite).is_err()); + } + + #[tokio::test] + async fn restart_installs_the_durable_gate_before_serving() { + let tmp = tempdir().unwrap(); + { + let store = open_store(tmp.path()).await; + store + .create("ns".into(), RestoreOption::Latest, Default::default()) + .await + .unwrap(); + let fence = store.inner.fences.controller(&"ns".into()); + fence + .apply_command(store.meta_store(), acquire("ns"), ctx()) + .await + .unwrap(); + store.shutdown().await.unwrap(); + } + + let store = open_store(tmp.path()).await; + // The registry holds the durable gate before the namespace is loaded. + let fence = store.inner.fences.get(&"ns".into()).unwrap(); + assert_eq!(fence.gate().state(), FenceState::SourceDraining); + assert_eq!(fence.gate().revision(), 1); + let loaded = store + .with("ns".into(), |ns| ns.fence().clone()) + .await + .unwrap(); + assert!(Arc::ptr_eq(&fence, &loaded)); + } + + #[tokio::test] + async fn unavailable_namespace_is_refused_before_setup() { + let tmp = tempdir().unwrap(); + // An unreadable marker in a directory with no config: the fence state cannot be + // established. + let dir = tmp.path().join("dbs").join("lost"); + std::fs::create_dir_all(&dir).unwrap(); + std::fs::write(dir.join(fence_store::MARKER_FILE_NAME), b"garbage").unwrap(); + + let store = open_store(tmp.path()).await; + let r = store.with("lost".into(), |_| ()).await; + match r { + Err(Error::NamespaceFence(e)) => { + assert_eq!(e.outcome(), FenceOutcome::FenceStateUnavailable); + assert_eq!(e.detail(), Some(FenceDetail::CorruptRecord)); + } + other => panic!("expected a fence error, got {other:?}"), + } + // Nothing was set up: no database file, and the namespace is not cached. + assert!(!dir.join("data").exists()); + assert!(store + .inner + .store + .get(&NamespaceName::from("lost")) + .await + .is_none()); + } + + #[tokio::test] + async fn destroy_forgets_the_controller() { + let tmp = tempdir().unwrap(); + let store = open_store(tmp.path()).await; + store + .create("ns".into(), RestoreOption::Latest, Default::default()) + .await + .unwrap(); + assert!(store.inner.fences.get(&"ns".into()).is_some()); + store.destroy("ns".into(), false).await.unwrap(); + assert!(store.inner.fences.get(&"ns".into()).is_none()); + } +}