From 3f0c00efc3705b6b1bddf09bae9096d147b37117 Mon Sep 17 00:00:00 2001 From: Pekka Enberg Date: Wed, 25 Mar 2026 10:15:03 +0200 Subject: [PATCH 01/33] Update README.md --- README.md | 10 ++++++++-- 1 file changed, 8 insertions(+), 2 deletions(-) diff --git a/README.md b/README.md index a94930b0df..f0c0936a62 100644 --- a/README.md +++ b/README.md @@ -33,8 +33,14 @@ --- -> [!NOTE] -> This repository contains libSQL, a fork of SQLite developed by Turso. For the full SQLite rewriten in Rust (also by Turso), please visit [tursodatabase/turso](https://github.com/tursodatabase/turso). +> [!IMPORTANT] +> **Turso database and libSQL are two different projects from the same team.** +> +> **libSQL** (this repository) is an open-source fork of SQLite. It extends SQLite with features like embedded replicas and remote access, but inherits SQLite's fundamental limitations such as the single-writer model. +> +> **[Turso](https://github.com/tursodatabase/turso) database** is a SQLite-compatible database rewritten from scratch in Rust. It is **not** a fork of SQLite — it is a completely new implementation that goes beyond what any SQLite fork can offer, including concurrent writes and bi-directional sync with offline support. Turso is currently in beta. +> +> **If you're starting a new project, you probably want to look into [Turso](https://github.com/tursodatabase/turso).** libSQL is actively maintained, but new features are being developed in Turso. ## Documentation From e4beacaa266fba930b637515e2082b42c2d6a817 Mon Sep 17 00:00:00 2001 From: Pekka Enberg Date: Wed, 25 Mar 2026 10:17:01 +0200 Subject: [PATCH 02/33] Fix typo --- README.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/README.md b/README.md index f0c0936a62..da827825b0 100644 --- a/README.md +++ b/README.md @@ -38,7 +38,7 @@ > > **libSQL** (this repository) is an open-source fork of SQLite. It extends SQLite with features like embedded replicas and remote access, but inherits SQLite's fundamental limitations such as the single-writer model. > -> **[Turso](https://github.com/tursodatabase/turso) database** is a SQLite-compatible database rewritten from scratch in Rust. It is **not** a fork of SQLite — it is a completely new implementation that goes beyond what any SQLite fork can offer, including concurrent writes and bi-directional sync with offline support. Turso is currently in beta. +> **[Turso database](https://github.com/tursodatabase/turso)** is a SQLite-compatible database rewritten from scratch in Rust. It is **not** a fork of SQLite — it is a completely new implementation that goes beyond what any SQLite fork can offer, including concurrent writes and bi-directional sync with offline support. Turso is currently in beta. > > **If you're starting a new project, you probably want to look into [Turso](https://github.com/tursodatabase/turso).** libSQL is actively maintained, but new features are being developed in Turso. From 6a0db201678c63b3c5e24953115a6d23279d9405 Mon Sep 17 00:00:00 2001 From: River Date: Tue, 29 Sep 2026 13:50:19 +0000 Subject: [PATCH 03/33] libsql-server: document namespace fence contract Add docs/NAMESPACE_FENCE.md, the contract and design for a durable, operation-owned namespace fence: source and target state machines and permission matrix, admin API, stable outcome codes and their mapping to HTTP, Hrana, gRPC, the write proxy and replication, metastore schema and CAS semantics, write admission generations and positive drain, read fence, quarantined target lifecycle and the reusable import capability, restart and eviction rules, fail-closed metastore recovery, deployment flag and legacy-binary protection, code-path coverage, test strategy and the acceptance-test map. Co-authored-by: Tomasz Szymczyszyn --- docs/NAMESPACE_FENCE.md | 776 ++++++++++++++++++++++++++++++++++++++++ 1 file changed, 776 insertions(+) create mode 100644 docs/NAMESPACE_FENCE.md diff --git a/docs/NAMESPACE_FENCE.md b/docs/NAMESPACE_FENCE.md new file mode 100644 index 0000000000..c2964dfbf2 --- /dev/null +++ b/docs/NAMESPACE_FENCE.md @@ -0,0 +1,776 @@ +# Namespace fence: contract and design + +This document describes the **namespace fence** in `libsql-server`: a durable, operation-owned control record that an external operation (for example, a tool that moves a database from one server to another) uses as the data-plane authority boundary for one namespace. + +It is both the contract a client of the fence can rely on and the design the implementation follows. Sections marked **Contract** are behaviour clients may depend on. Sections marked **Design** describe how the server delivers it and may change without notice. + +Status: the contract below is the target of the implementation series that starts with this document. The *Code-path coverage* table (section 14) and the *Acceptance tests* map (section 17) are kept in step with the code as it lands. + +## 1. What the fence is for + +A move of a namespace from a **source** server to a **target** server needs one place that decides, durably and inspectably, who may read and write each copy. The existing `block_reads` / `block_writes` / `block_reason` fields of `DatabaseConfig` cannot be that place: + +- they are process configuration, not owned by any operation, so two operators can overwrite each other; +- `CoreConnection::run` snapshots them once per program, so a program admitted before the flag changed can still write; +- the connection manager exposes no admission cutoff and no positive "no pre-cutoff writer remains" signal; +- the metastore's `MetaStoreHandle::version()` resets to zero on restart and cannot be a compare-and-swap revision; +- lifecycle operations (create, delete, fork, reset, restore, config), `/dump`, the admin shell, schema migrations and replication streams are not covered by statement blocking. + +The fence lets one operation: + +1. stop new source mutations and positively prove that no transaction admitted before the cutoff can still commit; +2. create a target that is quarantined from its first externally visible instant, and import into it through a server-created capability; +3. make the validated target readable while its writes stay closed during routing convergence; +4. stop source SQL, dump and replication reads before target write authority is published; and +5. publish target writes with an idempotent compare-and-swap transition whose result can be inspected after a restart or a lost response. + +The fence is the data-plane authority. Routing and client-side controls remain defence in depth. + +### Non-goals + +- Orchestrating a move: selecting a target, advancing phases, routing, pausing change streams, network policy. +- Choosing or implementing a bulk copy format. The fence provides the quarantine and import capability a copy uses (section 11); streamed export and memory-bounded import are separate work. +- Deleting the source, routing back after target publication, or cleaning up an abandoned target. +- Treating connection drain, `block_writes`, the schema scheduler's `block_writes` flag, `txn_timeout` or a single rejected probe as proof of drain. +- Revoking frames already held by embedded replicas that are offline. Proving that every client has converged on the new route is the operation's responsibility, not the server's. + +## 2. Safety invariants (Contract) + +1. **Single owner.** At most one `operation_id` owns a namespace fence. Ownership has no TTL and never fails open. Recovery resumes the same operation; another operation cannot steal or overwrite it, except through the audited adoption command (section 12), which keeps every gate closed. +2. **Live authoritative gate.** Every logical write is checked at the WAL write-transaction boundary. Statement classification may reject earlier but is never the authority. +3. **Positive source drain.** Once acquisition reports `SOURCE_WRITE_FENCED`, no transaction admitted before or after the cutoff can subsequently commit a logical mutation. +4. **No deferred writes.** A queued program or a read transaction created under an older admission generation can never become a writer, including after the fence is released or target writes are enabled. It must fail and begin a fresh transaction. +5. **Quarantine from birth.** A target's namespace row and its `TARGET_QUARANTINED` fence row commit in one metastore transaction before any connection, dump, replication stream or lifecycle operation can observe the namespace. +6. **No generic bypass.** Import and validation use a server-created `MigrationCapability` bound to `operation_id`, purpose and fence revision, never a caller-supplied header, the ordinary admin credential or the admin shell. +7. **Durable publication.** State, revision and receipt commit before a transition's response. Startup and in-process reload install the gate before the namespace is exposed. Unknown, corrupt or indeterminate control state fails closed. +8. **Irreversible target publication.** `TARGET_WRITABLE` has no reverse transition. A lost `EnableTargetWrites` response is resolved by replaying the same command or inspecting; if neither is possible the caller must treat the result as unknown and must not route back. +9. **Lifecycle coverage.** Config mutation, delete, reset, fork, restore, import, schema migration and every other logical mutation path are denied in states that deny them. +10. **Read-fence completeness.** A `SOURCE_READ_FENCED` acknowledgement means new normal SQL, dump and replication reads are denied, and every data-serving transaction or stream that was open before has ended or been explicitly aborted. + +A probe is evidence of a fence only when the server returns the expected fence code (section 6). Authentication failures, timeouts, `404`s and connection errors prove nothing. + +## 3. State model (Contract) + +A fence record exists per namespace once any operation has acted on it. A namespace with no record is `UNFENCED` (a source) or `ABSENT` (no namespace). + +### 3.1 States + +| State | Role | Durable | Meaning | +|---|---|---|---| +| `UNFENCED` | — | no row | Ordinary namespace. | +| `SOURCE_DRAINING` | source | yes | Write admission closed; waiting for pre-cutoff writers to finish. | +| `SOURCE_WRITE_FENCED` | source | yes | No write can commit; frozen replication boundary recorded; reads still served. | +| `SOURCE_READ_DRAINING` | source | yes | Reads closed for new work; waiting for old readers and streams to end. | +| `SOURCE_READ_FENCED` | source | yes | No read, dump or replication is served. | +| `RELEASED` | source | yes | Precommit rollback finished; the namespace is ordinary again. Terminal for the operation. | +| `TARGET_QUARANTINED` | target | yes | Created by the operation; only its import capability may write. | +| `TARGET_IMPORT_DRAINING` | target | yes | Import sealed; no new import work; waiting for import writers to finish. | +| `TARGET_VALIDATING` | target | yes | Import finished; only the operation's read-only validation is served. | +| `TARGET_WRITE_FENCED` | target | yes | Validated; readable and replicable; writes closed. | +| `TARGET_WRITABLE` | target | yes | Published. Terminal for the operation; no reverse transition. | +| `TARGET_ABORTED` | target | yes | Abandoned before publication. Normal traffic stays denied until a separately authorised cleanup. Terminal for the operation. | +| `UNKNOWN_UNAVAILABLE` | any | derived | The server cannot establish the control state (corrupt or unsupported payload, incomplete target creation, metastore rollback detected, indeterminate commit after restart). Every data-plane and lifecycle operation is denied. Never written as a normal transition. | + +`INSTALLING` is an in-memory gate state, never persisted and never reported as a durable state: it closes write admission while a closing transition is being persisted (section 8.1). + +### 3.2 Transitions + +```text +Source: + UNFENCED | RELEASED | TARGET_WRITABLE (finished op) + --AcquireSourceWriteFence--> SOURCE_DRAINING --(drain proven)--> SOURCE_WRITE_FENCED + SOURCE_WRITE_FENCED --SetSourceReadFence--> SOURCE_READ_DRAINING --(leases zero)--> SOURCE_READ_FENCED + SOURCE_READ_DRAINING | SOURCE_READ_FENCED --ClearSourceReadFence--> SOURCE_WRITE_FENCED + SOURCE_DRAINING | SOURCE_WRITE_FENCED --ReleaseSourceWriteFence--> RELEASED + +Target: + ABSENT --CreateTargetQuarantined--> TARGET_QUARANTINED + TARGET_QUARANTINED --SealTargetImport--> TARGET_IMPORT_DRAINING --(import writers zero)--> TARGET_VALIDATING + TARGET_VALIDATING --RecordTargetValidation--> TARGET_VALIDATING (stores the validation receipt) + TARGET_VALIDATING (with a successful validation receipt) --PublishTargetReadableWriteFenced--> TARGET_WRITE_FENCED + TARGET_WRITE_FENCED --EnableTargetWrites--> TARGET_WRITABLE + TARGET_QUARANTINED | TARGET_IMPORT_DRAINING | TARGET_VALIDATING | TARGET_WRITE_FENCED + --AbortQuarantinedTarget--> TARGET_ABORTED + +Any state owned by an unfinished operation --AdoptFence--> same state, new owner +``` + +Rules: + +- `TARGET_WRITABLE` never transitions to a frozen, aborted or absent state for the same operation. A *new* operation may later acquire the namespace as a source, which is a new move, not a reversal. +- `RELEASED` and `TARGET_WRITABLE` finish the operation. `TARGET_ABORTED` finishes the operation but does not free the namespace: only cleanup, which is outside this contract, may remove it. +- Once `SealTargetImport` has been applied, import can never resume for that target. +- `ReleaseSourceWriteFence` is a precommit rollback. After a caller has dispatched `EnableTargetWrites` on the target, it must never release the source: the source server cannot know whether a transition committed on another server, and the no-route-back rule is the operation's to enforce. +- Nothing expires. No state advances or releases because an operator, a router or the caller is unavailable. +- A restart in `SOURCE_DRAINING` or `TARGET_IMPORT_DRAINING` may complete only the drain that was already requested, and only when the same command is replayed (section 8.4). It never advances further. + +### 3.3 Permission matrix + +| State | Normal SQL read | Dump / replication | Normal logical write | Generic lifecycle / config | Operation-owned work | +|---|---|---|---|---|---| +| `UNFENCED`, `RELEASED` | allow | allow | allow | existing policy | acquire only | +| `SOURCE_DRAINING` | allow | allow | deny | deny | status, drain, release | +| `SOURCE_WRITE_FENCED` | allow | allow | deny | deny | export/validation reads, status, read fence, release | +| `SOURCE_READ_DRAINING`, `SOURCE_READ_FENCED` | deny | deny; open streams terminated | deny | deny | status, clear read fence | +| `TARGET_QUARANTINED` | deny | deny | deny | deny | import capability, status, seal, abort | +| `TARGET_IMPORT_DRAINING` | deny | deny | deny | deny | existing import writers finish; no new capability | +| `TARGET_VALIDATING` | deny | deny | deny | deny | read-only validation capability, status | +| `TARGET_WRITE_FENCED` | allow | allow | deny | deny | read-only validation, status, enable writes | +| `TARGET_WRITABLE` | allow | allow | allow | existing policy | status and receipts | +| `TARGET_ABORTED` | deny | deny | deny | deny except separately authorised cleanup | status | +| `UNKNOWN_UNAVAILABLE` | deny | deny | deny | deny | status, adoption | + +"Deny" on the lifecycle column covers: `POST /v1/namespaces/:ns/config`, delete, reset, fork (as source or destination), create over an existing record, restore of any kind, dump load, shared-schema linking and schema migration. Maintenance that cannot change logical contents is a separate class and continues in every state: WAL `TRUNCATE` checkpoint, bottomless WAL upload, the storage monitor, the replication logger's own connection and log compaction. **`VACUUM` is not maintenance** (it takes a write transaction and produces replicated frames) and is skipped in every state whose normal-write column is "deny". + +## 4. Admin API (Contract) + +The fence API is served on the admin listener (`--admin-listen-addr`) under the existing admin authentication. It is present only when the server is started with `--enable-namespace-fence` (section 13). There is no generic "set state" endpoint: each command is its own route. + +### 4.1 Authentication and preconditions + +- Every fence route requires admin authentication. **Mutating fence routes refuse to run when no admin auth key is configured** (`FENCE_PRECONDITION_FAILED`, reason `admin_auth_required`), because without a key the admin listener is unauthenticated. `operation_id` is ownership identity, not authentication. +- Authentication is checked first. Then, under the namespace's transition lock, the server looks up `(operation_id, command_id)`; replay handling (section 5.3) precedes every owner, role, state and revision check. +- Namespaces that are, or are linked to, a shared schema are rejected with `FENCE_PRECONDITION_FAILED`, reason `shared_schema_unsupported`. +- On a replica-kind server every fence route returns `FENCE_PRECONDITION_FAILED`, reason `not_primary`. Fences live on the primary that owns the WAL. + +### 4.2 Common request fields + +Every mutating request body is JSON and carries: + +```json +{ + "operation_id": "5f0c8a1e-...-uuid", + "command_id": "a41d...-uuid", + "expected_state": "SOURCE_WRITE_FENCED", + "expected_revision": 3 +} +``` + +`expected_state` for a namespace with no record is `UNFENCED` (source) or `ABSENT` (target); `expected_revision` is then `0`. + +### 4.3 Common response + +Success (`APPLIED`, `ALREADY_APPLIED`) is `200`; `DRAINING` is `202`. Every response, success or error, carries the current fence view: + +```json +{ + "outcome": "APPLIED", + "replayed": false, + "fence": { + "namespace": "db1", + "role": "SOURCE", + "state": "SOURCE_WRITE_FENCED", + "revision": 4, + "operation_id": "5f0c8a1e-...", + "incarnation": { "log_id": "…uuid…", "target_incarnation_id": null }, + "admission": { "write": "closed", "read": "open", "generation": 7 }, + "frozen_boundary": { "log_id": "…uuid…", "frame_no": 1234 }, + "drain_policy": { "deadline_ms": 30000, "on_deadline": "fail" }, + "created_at": "2026-01-01T00:00:00Z", + "last_transition_at": "2026-01-01T00:00:05Z", + "server": { "build": " ()", "instance_id": "…uuid…" }, + "provenance": { "metastore_restored_from_backup": false, "marker": "consistent" } + }, + "receipt": { + "operation_id": "5f0c8a1e-...", + "command_id": "a41d...", + "command": "AcquireSourceWriteFence", + "fingerprint": "sha256:…", + "outcome": "APPLIED", + "revision_before": 3, + "revision_after": 4, + "applied_at": "2026-01-01T00:00:05Z" + }, + "drain": { "active_writers": 0, "read_leases": { "sql": 0, "dump": 0, "replication": 0 }, "import_writers": 0 } +} +``` + +Errors use the same shape with `"outcome": ""`, plus `"error": ""` and, where useful, `"detail"` (a bounded reason string such as `role_mismatch` or `namespace_identity_mismatch`). + +### 4.4 Routes + +```HTTP +GET /v1/fence/capabilities +``` + +Read-only; served whenever the admin API is, even when the fence is disabled, so preflight can reject a server that cannot take part. Returns: + +```json +{ + "fence_protocol_version": 1, + "enabled": true, + "commands": ["InspectFence", "AcquireSourceWriteFence", "..."], + "states": ["SOURCE_DRAINING", "..."], + "proxy_stable_code": true, + "server": { "build": "…", "instance_id": "…" }, + "active_fences": 2, + "metastore": { "restored_from_backup": false, "restored_generation": null } +} +``` + +`active_fences` counts records not in `UNFENCED`, `RELEASED` or `TARGET_WRITABLE`. Deployment tooling must refuse to roll back to a server without fence support while it is non-zero. + +```HTTP +GET /v1/namespaces/:namespace/fence +``` + +`InspectFence`. Returns the persisted record, the last receipts of the owning operation (and, with `?receipts=all`, every retained receipt), and the live drain counters. Read-only: it never completes a drain, advances a state or refreshes anything. `404` if the namespace does not exist and has no record. + +```HTTP +POST /v1/namespaces/:namespace/fence/source/acquire-write-fence +``` + +`AcquireSourceWriteFence`. Extra body fields: `expected_namespace_identity: { "log_id": "" }` (the replication log id the caller observed) and `drain_policy: { "deadline_ms": , "on_deadline": "fail" | "force_rollback" }`. Returns `APPLIED` with `SOURCE_WRITE_FENCED` and the frozen boundary, or `DRAINING` with `SOURCE_DRAINING` if the deadline passed under `fail`, or if the request was cut short. Replaying the same command resumes the same drain. + +```HTTP +POST /v1/namespaces/:namespace/fence/source/set-read-fence +POST /v1/namespaces/:namespace/fence/source/clear-read-fence +POST /v1/namespaces/:namespace/fence/source/release-write-fence +``` + +`SetSourceReadFence` (extra: `drain_policy` for readers; at the deadline, SQL work is cancelled and streams are terminated, see section 9), `ClearSourceReadFence`, `ReleaseSourceWriteFence`. + +```HTTP +POST /v1/namespaces/:namespace/fence/target/create-quarantined +POST /v1/namespaces/:namespace/fence/target/seal-import +POST /v1/namespaces/:namespace/fence/target/validation-receipt +POST /v1/namespaces/:namespace/fence/target/publish-readable +POST /v1/namespaces/:namespace/fence/target/enable-writes +POST /v1/namespaces/:namespace/fence/target/abort +``` + +`CreateTargetQuarantined` (extra: the subset of namespace configuration a create accepts — `max_db_size`, `jwt_key`, `txn_timeout_s`, `allow_attach`, `durability_mode`, `bottomless_db_id`; a `dump_url` or any restore option is rejected, because import goes through the capability), `SealTargetImport` (extra: `drain_policy` for import writers), `RecordTargetValidation` (extra: `result: "ok" | "failed"`, `summary` — an opaque caller string of at most 4 KiB kept in the receipt; the server adds the target's current `log_id`, `frame_no` and page count), `PublishTargetReadableWriteFenced`, `EnableTargetWrites`, `AbortQuarantinedTarget`. + +```HTTP +POST /v1/namespaces/:namespace/fence/target/validation-query +``` + +Runs one read-only SQL program under the operation's validation capability in `TARGET_VALIDATING` or `TARGET_WRITE_FENCED`. Body: common fields without `command_id`, plus `stmts` in the `/v1/execute` shape. The connection is opened with `PRAGMA query_only=1` and holds a validation capability; a write is denied at the WAL. + +```HTTP +POST /v1/namespaces/:namespace/fence/adopt +``` + +`AdoptFence` (section 12). + +Import itself is not an admin route in this series. It is an internal API (section 11) that the bulk import work exposes over its own route. + +## 5. Durable state (Design, with contract points marked) + +### 5.1 Metastore schema + +Two additive tables are created in the metastore by `setup_connection` when the fence is enabled, or found and loaded whenever they exist: + +```sql +CREATE TABLE IF NOT EXISTS namespace_fences ( + namespace TEXT NOT NULL PRIMARY KEY, + format_version INTEGER NOT NULL, + revision INTEGER NOT NULL, + record BLOB NOT NULL, + FOREIGN KEY (namespace) REFERENCES namespace_configs (namespace) + ON DELETE RESTRICT ON UPDATE RESTRICT +); + +CREATE TABLE IF NOT EXISTS namespace_fence_receipts ( + namespace TEXT NOT NULL, + operation_id TEXT NOT NULL, + command_id TEXT NOT NULL, + format_version INTEGER NOT NULL, + revision_after INTEGER NOT NULL, + applied_at INTEGER NOT NULL, + receipt BLOB NOT NULL, + PRIMARY KEY (namespace, operation_id, command_id) +); +``` + +- `record` and `receipt` are protobuf messages defined in `libsql-server/proto/namespace_fence.proto` and generated into `libsql-server/src/generated/` by the existing `tests/bootstrap.rs` pattern. `format_version` is `1`. A reader that meets an unknown `format_version`, an undecodable payload, or a `revision` column that disagrees with the payload marks the namespace `UNKNOWN_UNAVAILABLE` and logs an operator error; it never guesses. +- The foreign key makes the row a guard even for code that does not know about fences: the metastore connection runs with `PRAGMA foreign_keys=ON`, so any `DELETE FROM namespace_configs` of a fenced namespace fails. +- **Contract:** `revision` starts at `1` on the first transition and increases by one on every applied transition, including `RecordTargetValidation` and `AdoptFence`. It is stored, so it survives restart. `MetaStoreHandle::version()` is not used. + +### 5.2 Record contents + +`NamespaceFenceRecord`: namespace name; role; state; revision; `operation_id`; identity (`log_id` for a source, captured at acquisition and checked against `expected_namespace_identity`; for a target, a server-generated `target_incarnation_id` plus the `log_id` once the namespace exists); admission summary; drain policy and deadline; frozen boundary (`log_id`, `frame_no`) once written; the last successful validation receipt (target); the pre-fence values of `block_reads`, `block_writes` and `block_reason` (for the legacy mirror, section 13.2); the operation's command history pointer; creation and last-transition timestamps; server build and the `instance_id` of the process that wrote it; adoption history. + +`CommandReceipt`: `operation_id`, `command_id`, command kind, canonical request fingerprint, outcome, revision before and after, timestamp, server instance id, and for adoption the approvers and incident reference. + +### 5.3 Command processing order (Contract) + +For every mutating command, under the per-namespace transition lock: + +1. Authenticate (admin auth) and check the deployment flag. +2. Compute the fingerprint: SHA-256 over the deterministic protobuf encoding of the command *including* namespace, `operation_id`, command kind and every argument, *excluding* `command_id`. +3. Look up `(namespace, operation_id, command_id)`: + - found, same fingerprint, final outcome: return the stored result with `"replayed": true`. This holds even though the revision has since advanced. + - found, same fingerprint, in-progress (`DRAINING`, or an indeterminate commit being reconciled): resume that same command (section 8.4). + - found, different fingerprint: `FENCE_COMMAND_CONFLICT`. Nothing changes. +4. Only a non-replay proceeds: owner check (`FENCE_OWNED_BY_ANOTHER_OPERATION`), role and transition check (`INVALID_FENCE_TRANSITION`, with `detail: role_mismatch` where the role is wrong), `expected_state` and `expected_revision` (`FENCE_REVISION_MISMATCH`), then command-specific preconditions (`FENCE_PRECONDITION_FAILED`). +5. A command from the owner that asks for the state the record is already in (for example `EnableTargetWrites` when already `TARGET_WRITABLE`) returns `ALREADY_APPLIED`, records a receipt so its own replay is stable, and does not change the revision. + +The pure part of this — steps 3 to 5 as `apply(record, receipts, command) -> (next record, receipt, outcome)` — is a function with no I/O and is unit-tested exhaustively. + +### 5.4 Transaction domain + +All namespace-config writes already pass through the metastore's single `MetaStoreInner.conn`. The schema scheduler holds a second metastore connection, so the real serialisation point is SQLite's write lock. A fence transition therefore: + +1. takes `inner.conn` (same lock order as today: `conn` before `configs`); +2. `BEGIN IMMEDIATE`; +3. reads the fence row, the relevant receipts and the config row; +4. runs `apply`; +5. writes the fence row, the receipt, and the legacy mirror of `block_*` into the config row; +6. `COMMIT`; +7. only then updates the in-memory registry and publishes the gate. + +Ordinary config writes (`try_process`) and `remove` read the fence row inside their own transaction and refuse when the state denies lifecycle operations. Because both take `BEGIN IMMEDIATE` on the same database, a config write cannot interleave with a fence transition. + +`try_process` today publishes the new config to the in-memory watch even when persisting failed. That is fixed for all config writes: the watch is updated only after commit, and the error is returned. + +### 5.5 Receipt retention (Contract) + +Receipts of the operation that currently owns a record are never pruned. Receipts of finished operations (`RELEASED`, `TARGET_WRITABLE`, or superseded by adoption) are kept for at least `--namespace-fence-receipt-retention` (default 30 days) and are pruned only inside a later transition on the same namespace. Delete of a namespace in `UNFENCED`, `RELEASED` or `TARGET_WRITABLE` removes its fence row and receipts in the same transaction and logs them. + +### 5.6 On-disk marker + +Each fenced namespace directory holds a small file `dbs//.fence` containing `format_version`, `operation_id`, role, state and revision. It is written (and fsynced) **after** the metastore commit of each transition, and **before** the metastore transaction for `CreateTargetQuarantined` (section 10.1). It exists so that recovery paths that lose or roll back the metastore can tell a fenced namespace from a legacy one: + +- metastore row present, marker absent or older: the metastore is authoritative; the marker is rewritten on load (a crash between commit and marker write); +- marker present, metastore row absent or at a lower revision: the metastore was lost or rolled back; the namespace is `UNKNOWN_UNAVAILABLE`, with provenance `metastore_behind_marker`; +- marker says `TARGET_QUARANTINED` revision 1 and no metastore rows exist: an incomplete target creation; `UNKNOWN_UNAVAILABLE`, reconciled only by replaying the same `CreateTargetQuarantined`. + +## 6. Outcome codes (Contract) + +Stable, machine-readable codes. Clients match on the code, never on the message. + +| Code | Kind | Admin HTTP | User HTTP (`/`, `/v1`, `/v2`, `/v3`, `/dump`) | Hrana error `code` | gRPC (RPC, proxy connect, replication) | Proxy `Error.stable_code` | +|---|---|---|---|---|---|---| +| `APPLIED` | success | 200 | — | — | — | — | +| `ALREADY_APPLIED` | success | 200 | — | — | — | — | +| `DRAINING` | in progress | 202 | — | — | — | — | +| `MIGRATION_WRITE_FENCED` | data plane | 423 | 423 | `MIGRATION_WRITE_FENCED` | `FAILED_PRECONDITION` | `MIGRATION_WRITE_FENCED` | +| `MIGRATION_READ_FENCED` | data plane | 423 | 423 | `MIGRATION_READ_FENCED` | `FAILED_PRECONDITION` | `MIGRATION_READ_FENCED` | +| `MIGRATION_TARGET_QUARANTINED` | data plane | 423 | 423 | `MIGRATION_TARGET_QUARANTINED` | `FAILED_PRECONDITION` | `MIGRATION_TARGET_QUARANTINED` | +| `FENCE_STATE_UNAVAILABLE` | data plane / control | 423 | 423 | `FENCE_STATE_UNAVAILABLE` | `FAILED_PRECONDITION` | `FENCE_STATE_UNAVAILABLE` | +| `OPERATION_CAPABILITY_REQUIRED` | control | 403 | — | — | `FAILED_PRECONDITION` (admin shell) | — | +| `FENCE_OWNED_BY_ANOTHER_OPERATION` | control | 409 | — | — | — | — | +| `FENCE_REVISION_MISMATCH` | control | 409 | — | — | — | — | +| `INVALID_FENCE_TRANSITION` | control | 409 | — | — | — | — | +| `FENCE_COMMAND_CONFLICT` | control | 409 | — | — | — | — | +| `FENCE_COMMIT_INDETERMINATE` | control | 409 | — | — | — | — | +| `FENCE_PRECONDITION_FAILED` | control | 412 | — | — | — | — | + +Notes: + +- `FENCE_COMMAND_CONFLICT`: a `command_id` reused with a different request fingerprint. +- `FENCE_COMMIT_INDETERMINATE`: the server could not establish whether its own commit landed (for example an I/O error on `COMMIT`). Admission stays closed; only a replay of the same `command_id` reconciles it; every other command on the namespace receives this code until then. +- `FENCE_PRECONDITION_FAILED` carries `detail`, one of: `admin_auth_required`, `fence_disabled`, `not_primary`, `shared_schema_unsupported`, `namespace_identity_mismatch`, `namespace_exists`, `validation_receipt_required`, `restore_not_allowed`, `adoption_not_authorised`. +- **Data-plane denials are never `500`, `503`, `429` or gRPC `UNAVAILABLE`.** `423 Locked` is chosen because common HTTP clients do not retry it. The JSON error body of the user HTTP API gains an additive `"code"` field (`{"error": "...", "code": "MIGRATION_WRITE_FENCED"}`); the existing `Blocked` error (from `block_reads`/`block_writes`) keeps its current mapping. +- gRPC statuses carry the code in the `x-libsql-fence-code` metadata entry and as the message prefix `": "`. +- Authentication (`401`), missing namespace (`404`), timeouts and transport errors are distinct from all of the above. + +### 6.1 Proxy protocol addition + +`libsql-replication/proto/proxy.proto`: + +```proto +message Error { + enum ErrorCode { SQL_ERROR = 0; TX_BUSY = 1; TX_TIMEOUT = 2; INTERNAL = 3; } + ErrorCode code = 1; + string message = 2; + int32 extended_code = 3; + // Stable machine-readable outcome, e.g. "MIGRATION_WRITE_FENCED". Absent from older servers. + optional string stable_code = 4; +} +``` + +This is additive in proto3: older peers skip the unknown field; a newer replica treats an absent field as "no typed outcome". For fence denials the primary sets `code = SQL_ERROR` and `stable_code`. The replica threads `stable_code` through `Error::RpcQueryError` into the same user HTTP status and Hrana code the primary would have returned. A fence denial at proxy connection creation is returned as `FAILED_PRECONDITION` with the metadata above, never `UNAVAILABLE`, so the replica's write-proxy reconnect loop (which retries `UNAVAILABLE` without bound) does not spin on it. + +### 6.2 Replicated configuration addition + +`libsql-replication/proto/metadata.proto` `DatabaseConfig` gains `optional ReplicatedFence fence = 14;` with `state` (string) and `revision` (uint64). The primary fills it from the gate in `hello`; a newer replica server uses it to deny local reads in states whose normal-read column is "deny". Older replicas ignore it and still see the legacy `block_*` mirror. + +## 7. FenceController (Design) + +### 7.1 Registry + +`FenceRegistry` (in `NamespaceStore`, outside the moka cache) maps `NamespaceName -> Arc`. It is loaded from the metastore (and markers) at startup, before any namespace is served, and changes only after a durable commit. Namespaces without a record get an `UNFENCED` controller lazily. Because the registry is not the cache value, cache eviction and lazy reload reinstall the same controller (section 8.5). + +### 7.2 Controller state + +Per namespace: + +- `transition_lock`: a `tokio::sync::Mutex` serialising commands on this namespace. +- `gate`: a `tokio::sync::watch` of `GateSnapshot { state, revision, write: Open | Closed(code), read: Open | Closed(code), write_generation, capabilities }`. The WAL wrapper, `CoreConnection`, dump, replication and lifecycle code read it without locks. +- `write_generation: u64`, stored in the snapshot and **incremented on every transition that closes or opens write admission** (acquire, release, create, seal, publish, enable, abort, adopt). +- writer tracking, provided by the namespace's `ManagedConnectionWalManager` (section 8.2). +- `read_leases`: counters and cancel handles per lease class (`sql`, `dump`, `replication`), with a `Notify` on every release. +- `capabilities`: the live `MigrationCapability` set and an import-writer counter. +- in `cfg(test)` builds only, an optional `FenceTestHooks` (section 16). + +### 7.3 Operation classes + +Every write-transaction request at the WAL, every read lease and every lifecycle call names a class: + +| Class | Examples | Allowed when | +|---|---|---| +| `NormalWrite` | SQL over HTTP, Hrana, RPC, proxy; admin shell; schema migration; dump load outside the capability | normal-write column allows | +| `Maintenance` | `TRUNCATE` checkpoint, the manager's checkpoint slot, storage monitor | always | +| `Vacuum` | `vacuum_if_needed`, `Namespace::checkpoint`, snapshot at shutdown | normal-write column allows; otherwise skipped with a debug log | +| `CapabilityImport` | import session writes | `TARGET_QUARANTINED`, matching `operation_id`, capability revision equal to the record's, capability not invalidated | +| `CapabilityValidate` | validation reads (never writes) | `TARGET_VALIDATING`, `TARGET_WRITE_FENCED` | +| `NormalRead` | SQL programs, Hrana cursors, `/beta/listen`, ATTACH of this namespace | normal-read column allows | +| `Stream` | `/dump`, `hello`, `log_entries`, `batch_log_entries`, `snapshot` | dump/replication column allows | +| `Observability` | stats, `/v1/jobs`, metrics | always; never counted as a read lease | + +### 7.4 Per-connection fence state + +Every `LegacyConnection` receives a `FenceConnState` shared by its `ManagedConnectionWalWrapper` and its `CoreConnection`: the connection's class and capability (if any), `program_generation` (captured at program start), `txn_generation` (captured at `begin_read_txn` when a new read transaction starts), and a `denial` slot for the typed outcome of the last WAL refusal. + +## 8. Write admission and positive drain (Design) + +### 8.1 The two checks + +1. **Program and lifecycle admission (early).** `CoreConnection::run` reads the *live* gate at program start. If the program contains a statement that can write and the gate denies it, it fails at once with the typed error. It records `program_generation`. Lifecycle entry points check the gate the same way. +2. **WAL `begin_write_txn` (authoritative).** In `ManagedConnectionWalWrapper::begin_write_txn`, **before** `acquire()`, the wrapper requires: the gate permits the connection's class (and capability), and `program_generation == txn_generation == gate.write_generation`. On refusal it writes the typed outcome into the `denial` slot and returns `SQLITE_AUTH` — a non-`BUSY` code, so SQLite's busy handler does not retry it, and before `acquire()`, so no slot is released that was never held. `Vm::try_step` turns an `SQLITE_AUTH` with a filled `denial` slot into `Error::Fence(outcome)`; without a filled slot the SQLite error is returned unchanged. + +This makes the WAL gate independent of statement classification: DDL, misclassified PRAGMAs, `with_raw` users (admin shell, schema migrations, dump load), a program that snapshotted config before the fence, and read-to-write upgrades all converge on `begin_write_txn`. + +**No deferred writes.** A read transaction opened at generation *g* cannot upgrade after any transition, because the generation has moved on. An explicit transaction opened in one program and continued in a later program fails when the later program tries to write if a transition happened in between, and the client must begin a fresh transaction. + +### 8.2 Connection manager changes + +- Queue entries carry the operation class (`NormalWrite`, `Maintenance`, `CapabilityImport`) and the connection id. `acquire()` for checkpoints becomes `acquire(Maintenance)`. +- On every write-generation change the controller wakes the whole write queue using the existing `sync_token` shape; each woken waiter re-checks the gate and, if denied, returns the typed error instead of waiting for a release. +- The manager exposes `active_writer() -> Option<(ConnId, OperationClass)>`, notifies a `Notify` from `release()`, and offers `abort_active()` that uses the registered rollback handle. `Abort::abort` currently panics if the connection is gone; the drain path tolerates a concurrently closing connection. +- The active writer's lease lasts until `release()`. Because `ReplicationLoggerWalWrapper::insert_frames` commits the log and publishes the new frame number before `end_write_txn` → `release()`, observing "no `NormalWrite` or `CapabilityImport` holder" under the manager's `current` lock means the committed `log_id` and `frame_no` are final. + +### 8.3 Source write drain, step by step + +`AcquireSourceWriteFence`, under the transition lock: + +1. Replay handling and checks (section 5.3), including `expected_namespace_identity.log_id == ReplicationLogger::log_id()` and the shared-schema rejection. +2. Publish the in-memory `INSTALLING` gate: write admission closed, `write_generation += 1`. New write admissions now fail with `MIGRATION_WRITE_FENCED`. +3. Wake the write queue (section 8.2). Queued writers fail with `MIGRATION_WRITE_FENCED`. +4. CAS `SOURCE_DRAINING` in the metastore with receipt outcome `DRAINING`. + - Committed: continue. + - Proven not committed (the transaction failed before `COMMIT` for a precondition or constraint reason): remove the `INSTALLING` gate, bump the generation, return the error. + - Unknown (error on `COMMIT`, task cancelled, timeout): the gate stays closed, the controller enters `Indeterminate(command_id)`, the response is `FENCE_COMMIT_INDETERMINATE`. It is never treated as not applied. +5. Wait for the active pre-cutoff writer to commit or roll back, on the manager's release notification. Elapsed time and `txn_timeout` are never evidence. + - Deadline reached with `on_deadline: fail`: respond `DRAINING`. The durable state stays `SOURCE_DRAINING` and admission stays closed. + - Deadline reached with `on_deadline: force_rollback`: `abort_active()`, then keep waiting for the release notification. +6. Under the manager's `current` lock, observe no writer holding the slot and read the frozen boundary (`log_id`, current `frame_no`). +7. CAS `SOURCE_WRITE_FENCED` with the boundary; update the receipt to `APPLIED`; write the marker; respond. + +### 8.4 Reconciliation and resumption + +- Replay of a command whose receipt says `DRAINING` resumes at step 5. After a restart there is no pre-cutoff writer (SQLite recovery discards uncommitted work), so it completes at once. +- Replay of a command held `Indeterminate` re-reads the metastore: if the row shows the command applied, it continues from that durable point; if it shows it did not, it retries the same CAS. Other commands receive `FENCE_COMMIT_INDETERMINATE` until then. After a restart the gate reflects whatever is durable, which by definition was never acknowledged as open. +- Opening transitions (`ReleaseSourceWriteFence`, `EnableTargetWrites`) follow **commit → publish the exact revision to the gate → respond `APPLIED`**. A crash after commit and before publication sends no success, and startup recovers the committed gate before exposing the namespace. + +### 8.5 Restart and eviction + +- At startup the registry is built from the metastore and markers before `NamespaceStore` serves anything. `make_namespace` takes the controller from the registry and passes it into the configurator's `setup()`, down to `MakeLegacyConnection::new` and every `LegacyConnection`, **before** the first connection (the maker's held `_db` connection) is created. The replication logger, dump and replication services get the same controller. +- `NamespaceStore::with` checks the registry before `handle()` or `load_namespace`: `UNKNOWN_UNAVAILABLE` is refused before any setup work. +- Idle or capacity eviction shuts the namespace down but leaves the controller in the registry; a lazy reload reinstalls the identical gate, revision and generation. A drain waiter that holds the evicted manager sees its connections close and is notified. +- Namespaces in `SOURCE_DRAINING` or `TARGET_IMPORT_DRAINING` after a restart stay closed until the same command is replayed. Nothing advances in the background. + +## 9. Source read fence (Design) + +`SetSourceReadFence`, under the transition lock: + +1. Checks (section 5.3). +2. Close read admission in memory. New SQL programs, dump requests, replication calls and ATTACHes of this namespace fail with `MIGRATION_READ_FENCED`. +3. CAS `SOURCE_READ_DRAINING` (receipt `DRAINING`). +4. Wait for all read leases to be released: + - **SQL:** a lease is held for the duration of each running program (including a Hrana cursor that is still producing rows). A connection that is idle with an open transaction holds no lease; its next program consults the live gate, fails, and rolls the transaction back. Idle upgraded Hrana WebSocket and HTTP streams may therefore stay open. + - **Dump:** a lease is held for the dump stream. The exporter checks a cancel flag between rows. A cancelled dump aborts the HTTP body (the chunked transfer is not completed), so a client never receives a dump that looks complete; the dump text also never reaches its final `COMMIT;`. + - **Replication:** `log_entries` and `snapshot` streams register a lease when created. The stream wrapper selects on the gate; on read fence it yields a terminal `FAILED_PRECONDITION` status carrying `MIGRATION_READ_FENCED` and ends. On cancel the wrapper drops the inner stream synchronously, so the lease is released even if the peer never reads again. `hello` and `batch_log_entries` are unary and are simply denied. + - At the deadline, SQL programs are cancelled through the connection's existing progress-handler cancel flag, dumps are cancelled, and streams are terminated. The command keeps waiting for the actual releases; if a lease does not release, the result stays `DRAINING`. +5. CAS `SOURCE_READ_FENCED`; respond. + +`ClearSourceReadFence` reopens reads (writes stay fenced) with a new revision. + +Covered surfaces: HTTP (`/`, `/v1/execute`, `/v1/batch`), Hrana over HTTP (`/v2`, `/v3`, cursors), Hrana over WebSocket, the gRPC proxy (`execute`, `stream_exec`, `describe`), the admin shell (which runs raw SQL and is checked per query), `/beta/listen`, ATTACH from other namespaces, `/dump`, and both replication services (the internal one used by replica servers and the external one on the user port). A replica server that receives the terminal status installs a local read denial for that namespace, so it stops serving its local copy. + +Transport keepalive: when the fence is enabled, the RPC server and the user-port gRPC service set HTTP/2 keepalive (`--namespace-fence-keepalive-interval`, default 30 s; timeout 20 s) so dead peers are detected; lease release does not depend on it. + +Denied replication calls from old replicas are logged at most once per namespace per minute and counted, so repeated reconnects are observable rather than noisy. Newer replicas back off on the typed code (capped exponential, at most 60 s). + +Internal work that must keep running is classed `Maintenance` or `Observability` and holds no read lease: bottomless WAL upload, the storage monitor's read transaction, stats and metrics. + +This fence stops future service from the source. It cannot recall bytes a peer has already received, frames stored by an embedded replica, or data served by a replica server that is partitioned from the primary; the operation must prove client and replica convergence separately. + +## 10. Target lifecycle (Design) + +### 10.1 CreateTargetQuarantined + +1. Checks (section 5.3). The name must have no config row, no fence row and no marker (`FENCE_PRECONDITION_FAILED`, `namespace_exists`). +2. Create `dbs//` and write the marker (`TARGET_QUARANTINED`, revision 1). +3. One metastore transaction: insert the config row (with the legacy mirror `block_reads = block_writes = true`), the fence row (`TARGET_QUARANTINED`, revision 1, new `target_incarnation_id`) and the receipt. +4. Install the controller in the registry with the quarantine gate. +5. Only now insert the config into the metastore's in-memory map (which is what makes `exists()` true) and call `load_namespace`. The first connection maker is created with the quarantine gate already in place. +6. Record the new `log_id` in the record (same revision, informational) and respond `APPLIED`. + +A crash after step 2 leaves a marker with no rows: `UNKNOWN_UNAVAILABLE` until the same command is replayed, which completes it. A crash after step 3 recovers `TARGET_QUARANTINED` from the metastore. + +### 10.2 SealTargetImport + +CAS `TARGET_IMPORT_DRAINING` (revision + 1, so every issued import capability is invalidated and no new one can be issued); wait for the import-writer count and the manager's `CapabilityImport` holder to reach zero (same mechanism as section 8.3, with the drain policy from the request); CAS `TARGET_VALIDATING`. A timeout leaves `TARGET_IMPORT_DRAINING` durable and closed; only a replay of the same command resumes it. + +### 10.3 Validation and publication + +In `TARGET_VALIDATING` the operation reads through the validation capability (`validation-query`, or the internal API). `RecordTargetValidation` stores the result in the record and receipt. `PublishTargetReadableWriteFenced` requires that the most recent `RecordTargetValidation` of the owning operation has `result: ok`; otherwise `FENCE_PRECONDITION_FAILED`, `validation_receipt_required`. Publication clears the legacy `block_reads` mirror. + +`EnableTargetWrites` is CAS from `TARGET_WRITE_FENCED`; commit, publish the gate with a new generation, respond. The legacy mirror is restored to the values given at creation. Replays return the stored result; a new command asking for the same thing returns `ALREADY_APPLIED`. Every other command on a `TARGET_WRITABLE` record from the same operation is `INVALID_FENCE_TRANSITION`. + +`AbortQuarantinedTarget` moves to `TARGET_ABORTED`; all normal traffic stays denied. + +## 11. Reusable import capability API (Design, for bulk import) + +The bulk import work consumes this internal Rust API, which does not depend on any HTTP route: + +```rust +// namespace::fence +pub enum TargetState { Quarantined, ImportDraining, Validating, WriteFenced, Writable, Aborted } + +pub struct MigrationCapability { // server-created, not constructible outside the module + id: Uuid, + namespace: NamespaceName, + operation_id: Uuid, + purpose: CapabilityPurpose, // Import | Validate + fence_revision: u64, +} + +impl NamespaceStore { + /// CreateTargetQuarantined, atomic with namespace creation. + pub async fn create_target_quarantined(&self, req: CreateTargetRequest) + -> Result; + + /// Issue an import capability and a capability-bearing connection. Valid only in + /// TARGET_QUARANTINED for the owning operation at `expected_revision`. + pub async fn open_import_session(&self, ns: NamespaceName, operation_id: Uuid, + expected_revision: u64) -> Result; + + /// Read-only validation connection (`query_only`), TARGET_VALIDATING or TARGET_WRITE_FENCED. + pub async fn open_validation_session(&self, ns: NamespaceName, operation_id: Uuid, + expected_revision: u64) -> Result; + + pub async fn apply_fence_command(&self, ns: NamespaceName, cmd: FenceCommand) + -> Result; + + pub async fn inspect_fence(&self, ns: NamespaceName) -> Result; +} + +pub struct ImportSession { /* capability, connection, import-writer guard */ } +impl ImportSession { + pub fn capability(&self) -> &MigrationCapability; + /// Run a closure with the raw connection inside the capability; writes are admitted by the + /// WAL only while the capability is valid. + pub async fn with_raw(&mut self, + f: impl FnOnce(&mut rusqlite::Connection) -> R + Send + 'static) -> Result; +} +``` + +`FenceError` carries a `FenceOutcome` (section 6) and converts into the server's `Error`, so a route built on top returns the same codes. Dropping an `ImportSession` decrements the import-writer count and wakes a waiting seal. The existing dump loader (`load_dump`) can run inside `ImportSession::with_raw`; this series includes a test that imports a small dump into a quarantined target that way. Streaming and memory bounds are not part of this series. + +## 12. Incident adoption (Contract) + +`AdoptFence` transfers ownership of an unfinished operation's record to a new `operation_id` when the original control-plane record is lost. It: + +- requires the admin credential **and** the separate adoption key configured with `--namespace-fence-adoption-key` (absent: adoption is disabled, `FENCE_PRECONDITION_FAILED`, `adoption_not_authorised`), passed in the `x-libsql-fence-adoption-key` header; +- requires `approvers`: two distinct, non-empty identity strings, an `incident_ref` and a `reason`, all stored in the receipt and emitted in a structured audit log line; +- requires `expected_state`, `expected_revision` and the current `operation_id` of the record; +- changes the owner and appends to the adoption history, with revision + 1, and **changes nothing else**: gates stay exactly as they were. It cannot open source writes, move out of any state, or act on `TARGET_WRITABLE` or `TARGET_ABORTED`; +- also applies to `UNKNOWN_UNAVAILABLE` caused by a metastore rollback, where it re-establishes the record in the state the marker last recorded (the marker is written only after a commit), with the adopting operation as owner. + +The server cannot verify who the approvers are: the admin API has one shared key and no principal. "Two-person" is enforced as a separate secret plus a recorded two-approver request; real two-person control belongs to whatever holds those secrets. + +## 13. Deployment and compatibility (Contract) + +### 13.1 Deployment flag + +`--enable-namespace-fence` (env `SQLD_ENABLE_NAMESPACE_FENCE`), **default off**. The flag controls *use*, not enforcement: + +- Off, and no fence tables exist: behaviour is unchanged, except the `try_process` persistence fix (section 5.4). The capability endpoint reports `enabled: false`; fence routes return `404`. +- Off, but fence tables or markers exist (the flag was turned off after use): fences are still loaded and enforced; mutating routes are disabled. +- On: tables are created, routes are served, and the fail-closed recovery rules of section 13.3 apply. + +Related flags: `--namespace-fence-receipt-retention`, `--namespace-fence-adoption-key`, `--namespace-fence-keepalive-interval`, and default drain deadlines `--namespace-fence-default-write-drain-ms` and `--namespace-fence-default-read-drain-ms` (used when a request has no `drain_policy`). + +Upgrade order: deploy a binary with capability discovery and proxy `stable_code` support on every primary and replica; confirm with `GET /v1/fence/capabilities`; then enable the flag; then use fences. Rollback to a binary without fence support is refused by deployment tooling while `active_fences > 0`. + +### 13.2 Protection against an older binary + +An older binary does not know the fence tables. While a record is active: + +- the config row's `block_reads`, `block_writes` and `block_reason` are mirrored from the fence state in the same transaction (`block_writes` in every state that denies writes, `block_reads` in every state that denies reads, `block_reason = "namespace fence: (operation )"`), and restored when the operation finishes; +- the foreign key from `namespace_fences` to `namespace_configs` makes an older binary's namespace delete fail. + +This is best effort. An older binary applies `block_*` at statement level only, lets its admin shell and `/dump` bypass them, and would let a config update overwrite them. It is a mitigation for an accidental rollback, not a guarantee; the guarantee is the deployment order above. + +### 13.3 Fail-closed metastore recovery + +With the flag on, or whenever fence tables or markers exist: + +- `MetaStore::handle()` no longer default-creates entries on read paths. `NamespaceStore::with`, `checkpoint`, ATTACH authorisation and `check_program_auth` use a non-creating lookup; only create, fork destination and replica lazy creation create, and they refuse names that have a fence record or marker. +- `restore()`: an undecodable namespace name, config or fence row marks that namespace `UNKNOWN_UNAVAILABLE` (when the name is decodable) or fails startup with an operator error (when it is not), instead of skipping the row. +- `maybe_recover_from_fs`: a directory with a marker is registered `UNKNOWN_UNAVAILABLE`; directories without markers keep today's behaviour (they are legacy, unfenced namespaces). +- `destroy_on_error`: the broken metastore is renamed aside rather than deleted, and directories with markers are registered `UNKNOWN_UNAVAILABLE` after the rebuild. +- Metastore restore from backup: the provenance (`restored_from_backup`, backup generation) is surfaced in the capability endpoint, in `InspectFence`, as a metric and in a startup log line; the marker comparison (section 5.6) makes any namespace whose record went backwards `UNKNOWN_UNAVAILABLE`. A restored record is never trusted over a newer marker. +- Replica-kind servers: lazy creation of a name refused by the primary with a fence code does not create a local default namespace. + +### 13.4 Shared schema + +v1 does not fence shared-schema databases or namespaces linked to one, because schema migration fan-out would have to obey the fence on every linked namespace. Acquisition and target creation reject them with `FENCE_PRECONDITION_FAILED`, `shared_schema_unsupported`. While a fence is active, config mutation (which includes linking to a shared schema) is denied. + +## 14. Code-path coverage + +How each path that can reach namespace data or lifecycle is covered. File references are to `libsql-server/src/`. + +| Path | Coverage | +|---|---| +| `connection/connection_core.rs` `CoreConnection::run` | Early live-gate check and `program_generation` capture at program start; SQL read lease for the program; `Error::Fence` from the WAL denial slot in `Vm::try_step`. | +| `connection_core.rs` `checkpoint`, `vacuum_if_needed`, `force_rollback` | Checkpoint is `Maintenance`; vacuum is `Vacuum` and skipped while writes are fenced; `force_rollback` is the drain's abort. | +| `connection/connection_manager.rs` | Authoritative gate in `begin_write_txn` before `acquire()`, generation check, non-`BUSY` refusal; classed queue entries; queue wake on generation change; active-writer query, release notification, `abort_active()`. | +| `connection/legacy.rs` | `FenceConnState` wired into every `LegacyConnection`; the controller is passed to `MakeLegacyConnection::new` before the first connection. `with_raw` users are covered by the WAL gate. | +| HTTP `/`, `/v1/execute`, `/v1/batch`, Hrana `/v2`, `/v3`, cursors, WebSocket, dev route | Core checks and WAL gate; typed status and `code` field; Hrana codes. | +| `rpc/proxy.rs`, `rpc/streaming_exec.rs`, `rpc/replica_proxy.rs`, `connection/write_proxy.rs` | `stable_code` set on the primary and carried back on the replica; connection-creation denials are `FAILED_PRECONDITION`, not `UNAVAILABLE`. | +| `namespace/meta_store.rs` `handle()`, `restore()`, `maybe_recover_from_fs`, `destroy_on_error`, `process`/`try_process`, `remove`, bottomless metastore restore | Non-creating lookups; fail-closed decoding; marker-aware recovery; rename-aside; publish only after commit; fence check inside the config and remove transactions; restore provenance surfaced. | +| `namespace/store.rs` `with`, `load_namespace`, `make_namespace`, eviction | Registry check before setup; controller passed into setup; eviction keeps the registry entry. | +| `store.rs` `create`, `destroy`, `reset`, `fork`, `checkpoint`, restore options | Create refuses names with a record; `CreateTargetQuarantined` is the atomic quarantined create; destroy, reset, fork (either side) and any restore are denied while lifecycle is denied; checkpoint uses a non-creating lookup and skips vacuum. | +| `http/admin/mod.rs` config, create, fork, delete, checkpoint, stats | Config POST, create, fork, delete follow the lifecycle column; config GET, stats and checkpoint are allowed. Fence routes live in `http/admin/fence.rs`. | +| `http/user/dump.rs` | Gate check before connection creation (typed, no panic on create error); dump stream lease; cancel flag in the exporter; aborted body on termination. | +| `rpc/replication/replication_log.rs` `hello`, `log_entries`, `batch_log_entries`, `snapshot` | Denied at request start; stream leases; typed terminal status for open streams; `ReplicatedFence` in `hello`'s config. | +| `admin_shell.rs` | Writes denied at the WAL (no capability); reads checked against the gate per query. | +| `schema/scheduler.rs`, `database/schema.rs` | Shared schema excluded from fencing; migration writes are WAL-gated; the scheduler's `block_writes` flag is not treated as drain evidence. | +| `namespace/configurator/helpers.rs` `load_dump`, `http/admin/mod.rs` `dump_stream_from_url` | Restore options and dump URLs are refused for fenced namespaces; import goes through `ImportSession`. | +| `connection/program.rs` ATTACH resolution | The attached namespace's gate is checked (`NormalRead`), through a non-creating lookup. | +| `http/user/listen.rs` `/beta/listen` | `NormalRead`; denied where reads are denied; ended by the read fence. | +| Raw internal connections (storage monitor, periodic checkpoint, shutdown checkpoint, replication logger, `checkpoint_db`, bottomless) | `Maintenance` / `Observability`; not leases; never blocked. | +| DDL, autocommit, explicit transactions, batches, PRAGMAs, read-to-write upgrades | All converge on `begin_write_txn`; classification is only the early check. | + +## 15. Observability (Contract) + +Metrics (labels are bounded; namespace, operation id, command id, revision and caller never appear as labels): + +- `libsql_server_fence_transitions_total{command, outcome}` +- `libsql_server_fence_drain_duration_seconds{kind = write | read | import}` (histogram) +- `libsql_server_fence_forced_total{kind = rollback | sql_cancel | dump_cancel | stream_termination}` +- `libsql_server_fence_replays_total{result = replay | conflict}` +- `libsql_server_fence_denials_total{code, surface = http | hrana | rpc | proxy | dump | replication | admin_shell | lifecycle}` +- `libsql_server_fence_namespaces{role, state}` (gauge) +- `libsql_server_fence_oldest_active_age_seconds` (gauge) +- `libsql_server_fence_adoptions_total` +- `libsql_server_metastore_restored_from_backup` (gauge, 0 or 1) + +Every transition emits one structured log event (target `libsql_server::fence::audit`) with namespace, operation id, command id, command, outcome, revisions, state before and after, drain duration and forced actions, replay/conflict, server instance, and for adoption the approvers and incident reference. + +## 16. Test strategy (Design) + +- **No new dependency.** The crate has no failpoint library. Race tests use `#[cfg(test)]` hooks: `FenceTestHooks` holds named points (`AfterInstallingGate`, `BeforeMetastoreCommit`, `AfterMetastoreCommit`, `BeforeGatePublish`, `InBeginWriteTxnAfterCheck`, `AfterManagerRelease`, `BeforeBoundaryCapture`, `AfterTargetRowsCommitted`, `BeforeResponse`), each able to park the task on a `tokio::sync::Barrier` or `Notify` or to inject an error or an indeterminate commit. Hooks compile only in the library's own test build, so they cost nothing in release builds; integration tests under `tests/` cover protocol behaviour and do not rely on hooks. +- **Restart** at a boundary: reopen `MetaStore` and rebuild the registry on the same temporary directory, the way existing metastore tests do; integration tests stop and start a `TestServer` on the same path. +- **Response loss**: the test drops the command future after `AfterMetastoreCommit` and then replays or inspects. +- **Representative schemas**: synthetic multi-table schemas with indexes, triggers, views and an FTS5 table (FTS5 is compiled in). +- `TXN_TIMEOUT` is 100 ms in test builds; drain tests that hold a writer longer than that use their own `txn_timeout` or a hook, never a sleep. + +## 17. Acceptance tests map + +Planned test names; the table is updated as tests land. + +| # | Requirement | Planned tests | +|---|---|---| +| 1 | Concurrent acquisition by two operations: one owner, typed conflict for the loser | `fence::tests::acquire_race_single_owner`; `tests::fence::admin::concurrent_acquire_one_owner` | +| 2 | Active writer commits or is rolled back before freeze acknowledgement; nothing commits after | `fence::drain::tests::active_writer_commits_before_ack`, `forced_rollback_before_ack`, `no_commit_after_ack` | +| 3 | Autocommit, explicit transactions, queued writers, batches, DDL, schema jobs, old WebSockets, read-to-write upgrades cannot bypass | `connection_manager::tests::fence_rejects_queued_writer`, `fence_rejects_read_to_write_upgrade`, `fence_rejects_ddl_and_pragma`, `fence_rejects_raw_with_raw_write`; `tests::fence::protocol::old_ws_session_cannot_write`, `batch_denied_mid_batch`; `fence::tests::acquire_rejects_shared_schema` | +| 4 | Program that captured config before the fence is rejected at the WAL | `connection_core::tests::wal_gate_rejects_program_admitted_before_fence` | +| 5 | Pre-fence transactions cannot write after release or publication | `fence::tests::stale_generation_cannot_write_after_release`, `stale_generation_cannot_write_after_enable_writes` | +| 6 | Acquisition timeout returns `DRAINING`, admission stays closed | `fence::drain::tests::deadline_returns_draining_and_stays_closed` | +| 7 | Restart at every persistence boundary; indeterminate persistence keeps the gate closed until same-command reconciliation | `fence::tests::restart_at_each_boundary` (parameterised over hook points), `indeterminate_commit_keeps_gate_closed` | +| 8 | Evict and lazily reload a fenced namespace; identical admission | `tests::fence::lifecycle::evicted_namespace_reloads_same_gate` | +| 9 | Filesystem recovery, `destroy_on_error`, undecodable records, missing target quarantine, metastore backup rollback fail closed with provenance | `meta_store::tests::fs_recovery_with_marker_unavailable`, `destroy_on_error_keeps_fenced_unavailable`, `undecodable_row_unavailable`, `incomplete_target_unavailable`, `metastore_rollback_detected_by_marker` | +| 10 | Wrong owner, stale revision, invalid role/state, replay, command-id reuse; replay before revision check | `fence::transition::tests::*` (exhaustive over states × commands) | +| 11 | Target creation raced with SQL, dump, replication, lifecycle never observable as writable or readable | `fence::target::tests::create_race_never_observable` | +| 12 | Only the matching import capability writes a quarantined target; admin credentials and admin shell cannot | `fence::target::tests::import_requires_matching_capability`; `tests::fence::admin::admin_shell_cannot_write_quarantined` | +| 13 | Seal enters `TARGET_IMPORT_DRAINING`, waits, reaches `TARGET_VALIDATING`, cannot resume import; only a durable validation receipt permits idempotent publication | `fence::target::tests::seal_waits_for_import_writers`, `sealed_target_rejects_import`, `publish_requires_validation_receipt`, `publish_is_idempotent` | +| 14 | Enable writes idempotent, survives restart and response loss, irreversible | `fence::target::tests::enable_writes_idempotent_and_irreversible`, `enable_writes_survives_restart` | +| 15 | Lost `EnableTargetWrites` response resolved from receipt/state | `fence::target::tests::enable_writes_response_loss_resolved` | +| 16 | Read fence drains SQL, dump, `log_entries`, `snapshot`, including dead peers and forced termination | `fence::read::tests::*`; `tests::fence::protocol::read_fence_ends_dump`, `read_fence_ends_log_entries_typed`, `read_fence_ends_snapshot`, `read_fence_forced_termination` | +| 17 | Delete, reset, fork, restore, config, schema mutation rejected | `tests::fence::lifecycle::lifecycle_rejected_while_fenced` | +| 18 | Codes through HTTP, Hrana, RPC, dump, replication, replica write proxy; distinguishable from auth/timeout/not-found; old peers compatible; no retry loops | `tests::fence::protocol::{http_codes, hrana_http_codes, hrana_ws_codes, rpc_codes, dump_codes, replication_codes, replica_proxy_preserves_code, auth_and_not_found_distinct, denial_not_retried}`; `libsql-replication` `proxy_error_stable_code_is_additive` | +| 19 | Corrupt or unknown durable fence state fails closed | `fence::store::tests::corrupt_payload_fails_closed`, `unknown_format_version_fails_closed` | +| 20 | Metrics and audit logs | `tests::fence::observability::metrics_and_labels`; `fence::audit::tests::audit_event_fields` | +| 21 | Capability discovery and mixed-version protection | `tests::fence::admin::capabilities`; `fence::store::tests::legacy_mirror_and_fk_guard` (bounded, see section 18) | +| 22 | Adoption is two-person/audited, keeps admission closed, cannot reverse publication | `fence::tests::adopt_requires_key_and_two_approvers`, `adopt_keeps_gates_closed`, `adopt_cannot_touch_writable` | +| — | Import API usable by bulk import | `fence::target::tests::import_session_loads_dump_into_quarantined_target` | + +## 18. Limits + +What this design and its tests do not prove: + +- **Mixed-version behaviour** is tested at the data and wire level (legacy mirror, foreign-key guard, proto unknown-field handling, capability endpoint), not by running an older binary against the same metastore. +- **Representative data** is synthetic. Real production schemas are not part of the test suite. +- **Client retry policy** of SDKs was not audited; the server guarantees only that fence denials use codes that are not conventionally retried. +- **"Commit unknown"** is the caller's classification when neither replay nor inspection answers; the server's part is that replay and inspection always answer when the server is reachable. +- **Two-person adoption** is a separate secret plus a recorded two-approver request, not verified identities. +- **Delivered data** cannot be recalled: bytes already sent, frames held by embedded replicas, and a replica server partitioned from the primary are outside the server's reach. +- **Metastore rollback detection** relies on the namespace directory's marker. If both the metastore and the namespace directory are lost or restored from backup together, the server cannot detect that a newer fence existed; the caller's durable intent is authoritative then. +- **Release and pinning** of a server build that contains the fence are outside this change. + +## 19. Module layout and commit series (Design) + +```text +libsql-server/proto/namespace_fence.proto +libsql-server/src/generated/namespace_fence.rs +libsql-server/src/namespace/fence/ + mod.rs re-exports, FENCE_PROTOCOL_VERSION + state.rs Role, FenceState, OperationClass, permission matrix + outcome.rs FenceOutcome, FenceError, protocol mappings + command.rs FenceCommand, fingerprint + transition.rs pure apply() + record.rs NamespaceFenceRecord, CommandReceipt, encode/decode + store.rs metastore tables, fence CAS, marker file + registry.rs FenceRegistry + controller.rs FenceController, GateSnapshot, generations, leases + drain.rs write, read and import drains + target.rs target lifecycle, MigrationCapability, ImportSession, ValidationSession + audit.rs audit events and metrics + hooks.rs cfg(test) FenceTestHooks +libsql-server/src/http/admin/fence.rs +``` + +Planned commits, each leaving the crate building with its tests passing: + +1. `libsql-server: document namespace fence contract` (this document) +2. `libsql-server: add namespace fence types and transition logic` +3. `libsql-server: persist namespace fences in the metastore` +4. `libsql-server: fail closed on ambiguous metastore recovery` +5. `libsql-server: add fence registry and controller, install before first connection` +6. `libsql-server: gate write transactions at the WAL by admission generation` +7. `libsql-server: positive source write drain` +8. `libsql-server: source read fence with read and stream leases` +9. `libsql-server: quarantined target lifecycle and migration capabilities` +10. `libsql-server: namespace fence admin API and capability discovery` +11. `libsql-server: deny lifecycle operations on fenced namespaces` +12. `libsql-replication: add stable error code and replicated fence to protocols` +13. `libsql-server: typed fence outcomes across HTTP, Hrana, RPC, dump and replication` +14. `libsql-server: legacy-binary protection and restart/eviction tests` +15. `libsql-server: namespace fence adoption` +16. `libsql-server: namespace fence metrics and audit log` + +## 20. Positions on specific hazards + +| Hazard | Position | +|---|---| +| One WAL choke point for primary connections | The gate lives in `ManagedConnectionWalWrapper::begin_write_txn`; nothing else is authoritative. | +| Writers outside the wrapper (logger's raw connection, `checkpoint_db`, bottomless restore, fork copy) | Logger and `checkpoint_db` are maintenance and never change logical contents. Bottomless restore and fork copy are lifecycle operations and are denied at the lifecycle layer. | +| Fresh transactions are visible at `begin_read_txn` | Used to capture `txn_generation`. | +| `SQLITE_BUSY` is retried and the error path releases a slot it may not hold | Check before `acquire()`, return `SQLITE_AUTH`, carry the typed reason out of band. | +| Waking queued writers | Reuse the `sync_token` mass-wake shape. | +| Frozen boundary | Captured with no writer holding the manager slot, after `insert_frames` published it. | +| `VACUUM` | Not maintenance; skipped while writes are fenced. | +| ATTACH of a fenced namespace | Checked as a `NormalRead` of the attached namespace through a non-creating lookup. | +| Open-default `handle()`, including ATTACH authorisation and replica lazy creation | Non-creating lookups on read paths; creating paths refuse names with a record or marker. | +| `process()` publishing unpersisted config | Fixed for all config writes. | +| The scheduler's second metastore connection | Fence CAS and config writes use `BEGIN IMMEDIATE`; shared schema is excluded. | +| Replica servers inherit config | Legacy `block_*` mirror plus additive `ReplicatedFence`; typed terminal status installs a local read denial. | +| Proxy errors lose their type | Additive `stable_code = 4`. | +| Write proxy retries `UNAVAILABLE` forever | Fence denials are never `UNAVAILABLE`. | +| Admin API has no principal and may have no key | Fence mutators require a configured key; adoption requires a second key and two recorded approvers. | From 6a40d12e969959986f83876f8b2efbb9da87dcd3 Mon Sep 17 00:00:00 2001 From: River Date: Tue, 29 Sep 2026 14:11:22 +0000 Subject: [PATCH 04/33] libsql-server: add namespace fence types and transition logic Add the I/O-free core of the namespace fence described in docs/NAMESPACE_FENCE.md: - state.rs: roles, fence states (including the non-durable UNFENCED, ABSENT and derived UNKNOWN_UNAVAILABLE), operation classes and the permission matrix. - outcome.rs: stable outcome codes, their admin HTTP, user HTTP, Hrana, gRPC and proxy mappings (fence denials are never 5xx, 429 or gRPC UNAVAILABLE), bounded detail reasons and FenceError. - command.rs: fence commands and requests, and the canonical SHA-256 request fingerprint over everything except command_id. - record.rs: fence records, command receipts and the on-disk marker, with a strict protobuf encoding (proto/namespace_fence.proto) whose decoder rejects unknown versions, enum values, malformed ids and self-contradictory records instead of defaulting. - transition.rs: the pure apply() and complete_drain() functions. Replay and fingerprint conflicts are decided before owner, role, state and revision; TARGET_WRITABLE has no reverse transition; adoption changes only the owner. Nothing outside the module uses it yet; the metastore, controller and protocol layers follow. Co-authored-by: Tomasz Szymczyszyn --- docs/NAMESPACE_FENCE.md | 14 +- libsql-server/proto/namespace_fence.proto | 205 ++ .../src/generated/namespace_fence.rs | 507 +++++ libsql-server/src/namespace/fence/command.rs | 420 ++++ libsql-server/src/namespace/fence/mod.rs | 28 + libsql-server/src/namespace/fence/outcome.rs | 409 ++++ libsql-server/src/namespace/fence/record.rs | 908 +++++++++ libsql-server/src/namespace/fence/state.rs | 403 ++++ .../src/namespace/fence/transition.rs | 1794 +++++++++++++++++ libsql-server/src/namespace/mod.rs | 1 + libsql-server/tests/bootstrap.rs | 2 +- 11 files changed, 4684 insertions(+), 7 deletions(-) create mode 100644 libsql-server/proto/namespace_fence.proto create mode 100644 libsql-server/src/generated/namespace_fence.rs create mode 100644 libsql-server/src/namespace/fence/command.rs create mode 100644 libsql-server/src/namespace/fence/mod.rs create mode 100644 libsql-server/src/namespace/fence/outcome.rs create mode 100644 libsql-server/src/namespace/fence/record.rs create mode 100644 libsql-server/src/namespace/fence/state.rs create mode 100644 libsql-server/src/namespace/fence/transition.rs diff --git a/docs/NAMESPACE_FENCE.md b/docs/NAMESPACE_FENCE.md index c2964dfbf2..965e44de52 100644 --- a/docs/NAMESPACE_FENCE.md +++ b/docs/NAMESPACE_FENCE.md @@ -298,15 +298,17 @@ CREATE TABLE IF NOT EXISTS namespace_fence_receipts ( For every mutating command, under the per-namespace transition lock: 1. Authenticate (admin auth) and check the deployment flag. -2. Compute the fingerprint: SHA-256 over the deterministic protobuf encoding of the command *including* namespace, `operation_id`, command kind and every argument, *excluding* `command_id`. +2. Compute the fingerprint: SHA-256 over the deterministic protobuf encoding of the command *including* namespace, `operation_id`, command kind, `expected_state`, `expected_revision` and every argument, *excluding* `command_id`. 3. Look up `(namespace, operation_id, command_id)`: - found, same fingerprint, final outcome: return the stored result with `"replayed": true`. This holds even though the revision has since advanced. - found, same fingerprint, in-progress (`DRAINING`, or an indeterminate commit being reconciled): resume that same command (section 8.4). - found, different fingerprint: `FENCE_COMMAND_CONFLICT`. Nothing changes. -4. Only a non-replay proceeds: owner check (`FENCE_OWNED_BY_ANOTHER_OPERATION`), role and transition check (`INVALID_FENCE_TRANSITION`, with `detail: role_mismatch` where the role is wrong), `expected_state` and `expected_revision` (`FENCE_REVISION_MISMATCH`), then command-specific preconditions (`FENCE_PRECONDITION_FAILED`). -5. A command from the owner that asks for the state the record is already in (for example `EnableTargetWrites` when already `TARGET_WRITABLE`) returns `ALREADY_APPLIED`, records a receipt so its own replay is stable, and does not change the revision. +4. Only a non-replay proceeds. A namespace whose control state cannot be established refuses everything with `FENCE_STATE_UNAVAILABLE`, except the two commands that reconcile it: a replay of the `CreateTargetQuarantined` that left the marker (section 10.1), and `AdoptFence` after a metastore rollback (section 12). +5. Owner check (`FENCE_OWNED_BY_ANOTHER_OPERATION`) against an unfinished record. +6. A command from the owner that asks for the state the record is already in (for example `EnableTargetWrites` when already `TARGET_WRITABLE`) returns `ALREADY_APPLIED`, records a receipt so its own replay is stable, and does not change the revision. This is checked before the revision, because the caller's expectation is typically the state before a response it never received. Likewise, a new command from the owner that asks for the drain the record is already in (for example `AcquireSourceWriteFence` in `SOURCE_DRAINING`, after an adoption or a lost `command_id`) records a `DRAINING` receipt without changing the revision and joins that drain. +7. Role and transition check (`INVALID_FENCE_TRANSITION`, with `detail: role_mismatch` where the role is wrong, or `operation_finished` when the owner has already finished with the namespace), `expected_state` and `expected_revision` (`FENCE_REVISION_MISMATCH`), then command-specific preconditions (`FENCE_PRECONDITION_FAILED`). -The pure part of this — steps 3 to 5 as `apply(record, receipts, command) -> (next record, receipt, outcome)` — is a function with no I/O and is unit-tested exhaustively. +The pure part of this — steps 3 to 7 as `apply(current, stored receipt, request, env) -> decision`, and the completion of a drain as `complete_drain(record, draining receipt, evidence) -> (next record, final receipt)` — is a function with no I/O (`namespace/fence/transition.rs`) and is unit-tested exhaustively over every state and command. ### 5.4 Transaction domain @@ -330,7 +332,7 @@ Receipts of the operation that currently owns a record are never pruned. Receipt ### 5.6 On-disk marker -Each fenced namespace directory holds a small file `dbs//.fence` containing `format_version`, `operation_id`, role, state and revision. It is written (and fsynced) **after** the metastore commit of each transition, and **before** the metastore transaction for `CreateTargetQuarantined` (section 10.1). It exists so that recovery paths that lose or roll back the metastore can tell a fenced namespace from a legacy one: +Each fenced namespace directory holds a small file `dbs//.fence` containing `format_version` and a copy of the last committed record (so its `operation_id`, role, state, revision and the `command_id` that produced it). It is written (and fsynced) **after** the metastore commit of each transition, and **before** the metastore transaction for `CreateTargetQuarantined` (section 10.1). It exists so that recovery paths that lose or roll back the metastore can tell a fenced namespace from a legacy one: - metastore row present, marker absent or older: the metastore is authoritative; the marker is rewritten on load (a crash between commit and marker write); - marker present, metastore row absent or at a lower revision: the metastore was lost or rolled back; the namespace is `UNKNOWN_UNAVAILABLE`, with provenance `metastore_behind_marker`; @@ -361,7 +363,7 @@ Notes: - `FENCE_COMMAND_CONFLICT`: a `command_id` reused with a different request fingerprint. - `FENCE_COMMIT_INDETERMINATE`: the server could not establish whether its own commit landed (for example an I/O error on `COMMIT`). Admission stays closed; only a replay of the same `command_id` reconciles it; every other command on the namespace receives this code until then. -- `FENCE_PRECONDITION_FAILED` carries `detail`, one of: `admin_auth_required`, `fence_disabled`, `not_primary`, `shared_schema_unsupported`, `namespace_identity_mismatch`, `namespace_exists`, `validation_receipt_required`, `restore_not_allowed`, `adoption_not_authorised`. +- `FENCE_PRECONDITION_FAILED` carries `detail`, one of: `admin_auth_required`, `fence_disabled`, `not_primary`, `shared_schema_unsupported`, `namespace_identity_mismatch`, `namespace_exists`, `validation_receipt_required`, `restore_not_allowed`, `adoption_not_authorised`, `invalid_argument`. `INVALID_FENCE_TRANSITION` may carry `role_mismatch` or `operation_finished`. `FENCE_STATE_UNAVAILABLE` carries the reason the state cannot be established: `corrupt_record`, `unsupported_format_version`, `incomplete_target_creation`, `metastore_behind_marker` or `indeterminate_commit`. - **Data-plane denials are never `500`, `503`, `429` or gRPC `UNAVAILABLE`.** `423 Locked` is chosen because common HTTP clients do not retry it. The JSON error body of the user HTTP API gains an additive `"code"` field (`{"error": "...", "code": "MIGRATION_WRITE_FENCED"}`); the existing `Blocked` error (from `block_reads`/`block_writes`) keeps its current mapping. - gRPC statuses carry the code in the `x-libsql-fence-code` metadata entry and as the message prefix `": "`. - Authentication (`401`), missing namespace (`404`), timeouts and transport errors are distinct from all of the above. diff --git a/libsql-server/proto/namespace_fence.proto b/libsql-server/proto/namespace_fence.proto new file mode 100644 index 0000000000..efdee1ab81 --- /dev/null +++ b/libsql-server/proto/namespace_fence.proto @@ -0,0 +1,205 @@ +// Durable encoding of namespace fence records, command receipts and markers. +// +// See docs/NAMESPACE_FENCE.md. Every message here is stored, so fields are only ever added. +// Readers reject unknown enum values and missing required fields instead of guessing. +syntax = "proto3"; + +package namespace_fence; + +enum FenceRole { + FENCE_ROLE_UNSPECIFIED = 0; + FENCE_ROLE_SOURCE = 1; + FENCE_ROLE_TARGET = 2; +} + +// `UNFENCED` and `ABSENT` are never stored in a record; they appear as the `expected_state` of a +// request against a namespace that has no record. `UNKNOWN_UNAVAILABLE` is derived and never +// stored either. +enum FenceState { + FENCE_STATE_UNSPECIFIED = 0; + FENCE_STATE_UNFENCED = 1; + FENCE_STATE_ABSENT = 2; + FENCE_STATE_SOURCE_DRAINING = 3; + FENCE_STATE_SOURCE_WRITE_FENCED = 4; + FENCE_STATE_SOURCE_READ_DRAINING = 5; + FENCE_STATE_SOURCE_READ_FENCED = 6; + FENCE_STATE_RELEASED = 7; + FENCE_STATE_TARGET_QUARANTINED = 8; + FENCE_STATE_TARGET_IMPORT_DRAINING = 9; + FENCE_STATE_TARGET_VALIDATING = 10; + FENCE_STATE_TARGET_WRITE_FENCED = 11; + FENCE_STATE_TARGET_WRITABLE = 12; + FENCE_STATE_TARGET_ABORTED = 13; + FENCE_STATE_UNKNOWN_UNAVAILABLE = 14; +} + +enum CommandKind { + COMMAND_KIND_UNSPECIFIED = 0; + COMMAND_KIND_ACQUIRE_SOURCE_WRITE_FENCE = 1; + COMMAND_KIND_SET_SOURCE_READ_FENCE = 2; + COMMAND_KIND_CLEAR_SOURCE_READ_FENCE = 3; + COMMAND_KIND_RELEASE_SOURCE_WRITE_FENCE = 4; + COMMAND_KIND_CREATE_TARGET_QUARANTINED = 5; + COMMAND_KIND_SEAL_TARGET_IMPORT = 6; + COMMAND_KIND_RECORD_TARGET_VALIDATION = 7; + COMMAND_KIND_PUBLISH_TARGET_READABLE_WRITE_FENCED = 8; + COMMAND_KIND_ENABLE_TARGET_WRITES = 9; + COMMAND_KIND_ABORT_QUARANTINED_TARGET = 10; + COMMAND_KIND_ADOPT_FENCE = 11; +} + +// Only the outcomes a receipt can record. Errors are never stored. +enum ReceiptOutcome { + RECEIPT_OUTCOME_UNSPECIFIED = 0; + RECEIPT_OUTCOME_APPLIED = 1; + RECEIPT_OUTCOME_ALREADY_APPLIED = 2; + RECEIPT_OUTCOME_DRAINING = 3; +} + +enum OnDeadline { + ON_DEADLINE_UNSPECIFIED = 0; + ON_DEADLINE_FAIL = 1; + ON_DEADLINE_FORCE_ROLLBACK = 2; +} + +enum ValidationResult { + VALIDATION_RESULT_UNSPECIFIED = 0; + VALIDATION_RESULT_OK = 1; + VALIDATION_RESULT_FAILED = 2; +} + +message DrainPolicy { + uint64 deadline_ms = 1; + OnDeadline on_deadline = 2; +} + +message FrozenBoundary { + string log_id = 1; + uint64 frame_no = 2; +} + +message LegacyBlocks { + bool block_reads = 1; + bool block_writes = 2; + optional string block_reason = 3; +} + +message ServerIdentity { + string build = 1; + string instance_id = 2; +} + +message TargetConfig { + optional uint64 max_db_size = 1; + optional string jwt_key = 2; + optional uint64 txn_timeout_s = 3; + bool allow_attach = 4; + optional string durability_mode = 5; + optional string bottomless_db_id = 6; +} + +message ValidationSnapshot { + string log_id = 1; + uint64 frame_no = 2; + uint64 page_count = 3; +} + +message ValidationRecord { + string operation_id = 1; + string command_id = 2; + ValidationResult result = 3; + string summary = 4; + optional ValidationSnapshot snapshot = 5; + int64 recorded_at_ms = 6; +} + +message Adoption { + string previous_operation_id = 1; + string new_operation_id = 2; + string command_id = 3; + repeated string approvers = 4; + string incident_ref = 5; + string reason = 6; + int64 at_ms = 7; + uint64 revision = 8; +} + +message FenceRecord { + string namespace = 1; + FenceRole role = 2; + FenceState state = 3; + uint64 revision = 4; + string operation_id = 5; + optional string log_id = 6; + optional string target_incarnation_id = 7; + optional DrainPolicy drain_policy = 8; + optional int64 drain_started_at_ms = 9; + optional FrozenBoundary frozen_boundary = 10; + optional ValidationRecord validation = 11; + LegacyBlocks legacy_blocks = 12; + int64 created_at_ms = 13; + int64 last_transition_at_ms = 14; + string last_command_id = 15; + ServerIdentity written_by = 16; + repeated Adoption adoptions = 17; +} + +message CommandReceipt { + string namespace = 1; + string operation_id = 2; + string command_id = 3; + CommandKind command = 4; + bytes fingerprint = 5; + ReceiptOutcome outcome = 6; + uint64 revision_before = 7; + uint64 revision_after = 8; + FenceState state_after = 9; + int64 applied_at_ms = 10; + string instance_id = 11; + optional Adoption adoption = 12; +} + +// Contents of `dbs//.fence`: a copy of the last committed record (or, for +// `CreateTargetQuarantined`, of the record about to be committed). +message FenceMarker { + uint32 format_version = 1; + FenceRecord record = 2; +} + +// Canonical input of a command fingerprint: everything in the request except `command_id`. +message FingerprintInput { + string namespace = 1; + string operation_id = 2; + CommandKind kind = 3; + FenceState expected_state = 4; + uint64 expected_revision = 5; + oneof args { + AcquireSourceWriteFenceArgs acquire_source_write_fence = 10; + DrainArgs set_source_read_fence = 11; + DrainArgs seal_target_import = 12; + TargetConfig create_target_quarantined = 13; + RecordTargetValidationArgs record_target_validation = 14; + AdoptFenceArgs adopt_fence = 15; + } +} + +message AcquireSourceWriteFenceArgs { + string expected_log_id = 1; + optional DrainPolicy drain_policy = 2; +} + +message DrainArgs { + optional DrainPolicy drain_policy = 1; +} + +message RecordTargetValidationArgs { + ValidationResult result = 1; + string summary = 2; +} + +message AdoptFenceArgs { + string current_operation_id = 1; + repeated string approvers = 2; + string incident_ref = 3; + string reason = 4; +} diff --git a/libsql-server/src/generated/namespace_fence.rs b/libsql-server/src/generated/namespace_fence.rs new file mode 100644 index 0000000000..dcbbcafe96 --- /dev/null +++ b/libsql-server/src/generated/namespace_fence.rs @@ -0,0 +1,507 @@ +// This file is @generated by prost-build. +#[allow(clippy::derive_partial_eq_without_eq)] +#[derive(Clone, PartialEq, ::prost::Message)] +pub struct DrainPolicy { + #[prost(uint64, tag = "1")] + pub deadline_ms: u64, + #[prost(enumeration = "OnDeadline", tag = "2")] + pub on_deadline: i32, +} +#[allow(clippy::derive_partial_eq_without_eq)] +#[derive(Clone, PartialEq, ::prost::Message)] +pub struct FrozenBoundary { + #[prost(string, tag = "1")] + pub log_id: ::prost::alloc::string::String, + #[prost(uint64, tag = "2")] + pub frame_no: u64, +} +#[allow(clippy::derive_partial_eq_without_eq)] +#[derive(Clone, PartialEq, ::prost::Message)] +pub struct LegacyBlocks { + #[prost(bool, tag = "1")] + pub block_reads: bool, + #[prost(bool, tag = "2")] + pub block_writes: bool, + #[prost(string, optional, tag = "3")] + pub block_reason: ::core::option::Option<::prost::alloc::string::String>, +} +#[allow(clippy::derive_partial_eq_without_eq)] +#[derive(Clone, PartialEq, ::prost::Message)] +pub struct ServerIdentity { + #[prost(string, tag = "1")] + pub build: ::prost::alloc::string::String, + #[prost(string, tag = "2")] + pub instance_id: ::prost::alloc::string::String, +} +#[allow(clippy::derive_partial_eq_without_eq)] +#[derive(Clone, PartialEq, ::prost::Message)] +pub struct TargetConfig { + #[prost(uint64, optional, tag = "1")] + pub max_db_size: ::core::option::Option, + #[prost(string, optional, tag = "2")] + pub jwt_key: ::core::option::Option<::prost::alloc::string::String>, + #[prost(uint64, optional, tag = "3")] + pub txn_timeout_s: ::core::option::Option, + #[prost(bool, tag = "4")] + pub allow_attach: bool, + #[prost(string, optional, tag = "5")] + pub durability_mode: ::core::option::Option<::prost::alloc::string::String>, + #[prost(string, optional, tag = "6")] + pub bottomless_db_id: ::core::option::Option<::prost::alloc::string::String>, +} +#[allow(clippy::derive_partial_eq_without_eq)] +#[derive(Clone, PartialEq, ::prost::Message)] +pub struct ValidationSnapshot { + #[prost(string, tag = "1")] + pub log_id: ::prost::alloc::string::String, + #[prost(uint64, tag = "2")] + pub frame_no: u64, + #[prost(uint64, tag = "3")] + pub page_count: u64, +} +#[allow(clippy::derive_partial_eq_without_eq)] +#[derive(Clone, PartialEq, ::prost::Message)] +pub struct ValidationRecord { + #[prost(string, tag = "1")] + pub operation_id: ::prost::alloc::string::String, + #[prost(string, tag = "2")] + pub command_id: ::prost::alloc::string::String, + #[prost(enumeration = "ValidationResult", tag = "3")] + pub result: i32, + #[prost(string, tag = "4")] + pub summary: ::prost::alloc::string::String, + #[prost(message, optional, tag = "5")] + pub snapshot: ::core::option::Option, + #[prost(int64, tag = "6")] + pub recorded_at_ms: i64, +} +#[allow(clippy::derive_partial_eq_without_eq)] +#[derive(Clone, PartialEq, ::prost::Message)] +pub struct Adoption { + #[prost(string, tag = "1")] + pub previous_operation_id: ::prost::alloc::string::String, + #[prost(string, tag = "2")] + pub new_operation_id: ::prost::alloc::string::String, + #[prost(string, tag = "3")] + pub command_id: ::prost::alloc::string::String, + #[prost(string, repeated, tag = "4")] + pub approvers: ::prost::alloc::vec::Vec<::prost::alloc::string::String>, + #[prost(string, tag = "5")] + pub incident_ref: ::prost::alloc::string::String, + #[prost(string, tag = "6")] + pub reason: ::prost::alloc::string::String, + #[prost(int64, tag = "7")] + pub at_ms: i64, + #[prost(uint64, tag = "8")] + pub revision: u64, +} +#[allow(clippy::derive_partial_eq_without_eq)] +#[derive(Clone, PartialEq, ::prost::Message)] +pub struct FenceRecord { + #[prost(string, tag = "1")] + pub namespace: ::prost::alloc::string::String, + #[prost(enumeration = "FenceRole", tag = "2")] + pub role: i32, + #[prost(enumeration = "FenceState", tag = "3")] + pub state: i32, + #[prost(uint64, tag = "4")] + pub revision: u64, + #[prost(string, tag = "5")] + pub operation_id: ::prost::alloc::string::String, + #[prost(string, optional, tag = "6")] + pub log_id: ::core::option::Option<::prost::alloc::string::String>, + #[prost(string, optional, tag = "7")] + pub target_incarnation_id: ::core::option::Option<::prost::alloc::string::String>, + #[prost(message, optional, tag = "8")] + pub drain_policy: ::core::option::Option, + #[prost(int64, optional, tag = "9")] + pub drain_started_at_ms: ::core::option::Option, + #[prost(message, optional, tag = "10")] + pub frozen_boundary: ::core::option::Option, + #[prost(message, optional, tag = "11")] + pub validation: ::core::option::Option, + #[prost(message, optional, tag = "12")] + pub legacy_blocks: ::core::option::Option, + #[prost(int64, tag = "13")] + pub created_at_ms: i64, + #[prost(int64, tag = "14")] + pub last_transition_at_ms: i64, + #[prost(string, tag = "15")] + pub last_command_id: ::prost::alloc::string::String, + #[prost(message, optional, tag = "16")] + pub written_by: ::core::option::Option, + #[prost(message, repeated, tag = "17")] + pub adoptions: ::prost::alloc::vec::Vec, +} +#[allow(clippy::derive_partial_eq_without_eq)] +#[derive(Clone, PartialEq, ::prost::Message)] +pub struct CommandReceipt { + #[prost(string, tag = "1")] + pub namespace: ::prost::alloc::string::String, + #[prost(string, tag = "2")] + pub operation_id: ::prost::alloc::string::String, + #[prost(string, tag = "3")] + pub command_id: ::prost::alloc::string::String, + #[prost(enumeration = "CommandKind", tag = "4")] + pub command: i32, + #[prost(bytes = "vec", tag = "5")] + pub fingerprint: ::prost::alloc::vec::Vec, + #[prost(enumeration = "ReceiptOutcome", tag = "6")] + pub outcome: i32, + #[prost(uint64, tag = "7")] + pub revision_before: u64, + #[prost(uint64, tag = "8")] + pub revision_after: u64, + #[prost(enumeration = "FenceState", tag = "9")] + pub state_after: i32, + #[prost(int64, tag = "10")] + pub applied_at_ms: i64, + #[prost(string, tag = "11")] + pub instance_id: ::prost::alloc::string::String, + #[prost(message, optional, tag = "12")] + pub adoption: ::core::option::Option, +} +/// Contents of `dbs//.fence`: a copy of the last committed record (or, for +/// `CreateTargetQuarantined`, of the record about to be committed). +#[allow(clippy::derive_partial_eq_without_eq)] +#[derive(Clone, PartialEq, ::prost::Message)] +pub struct FenceMarker { + #[prost(uint32, tag = "1")] + pub format_version: u32, + #[prost(message, optional, tag = "2")] + pub record: ::core::option::Option, +} +/// Canonical input of a command fingerprint: everything in the request except `command_id`. +#[allow(clippy::derive_partial_eq_without_eq)] +#[derive(Clone, PartialEq, ::prost::Message)] +pub struct FingerprintInput { + #[prost(string, tag = "1")] + pub namespace: ::prost::alloc::string::String, + #[prost(string, tag = "2")] + pub operation_id: ::prost::alloc::string::String, + #[prost(enumeration = "CommandKind", tag = "3")] + pub kind: i32, + #[prost(enumeration = "FenceState", tag = "4")] + pub expected_state: i32, + #[prost(uint64, tag = "5")] + pub expected_revision: u64, + #[prost(oneof = "fingerprint_input::Args", tags = "10, 11, 12, 13, 14, 15")] + pub args: ::core::option::Option, +} +/// Nested message and enum types in `FingerprintInput`. +pub mod fingerprint_input { + #[allow(clippy::derive_partial_eq_without_eq)] + #[derive(Clone, PartialEq, ::prost::Oneof)] + pub enum Args { + #[prost(message, tag = "10")] + AcquireSourceWriteFence(super::AcquireSourceWriteFenceArgs), + #[prost(message, tag = "11")] + SetSourceReadFence(super::DrainArgs), + #[prost(message, tag = "12")] + SealTargetImport(super::DrainArgs), + #[prost(message, tag = "13")] + CreateTargetQuarantined(super::TargetConfig), + #[prost(message, tag = "14")] + RecordTargetValidation(super::RecordTargetValidationArgs), + #[prost(message, tag = "15")] + AdoptFence(super::AdoptFenceArgs), + } +} +#[allow(clippy::derive_partial_eq_without_eq)] +#[derive(Clone, PartialEq, ::prost::Message)] +pub struct AcquireSourceWriteFenceArgs { + #[prost(string, tag = "1")] + pub expected_log_id: ::prost::alloc::string::String, + #[prost(message, optional, tag = "2")] + pub drain_policy: ::core::option::Option, +} +#[allow(clippy::derive_partial_eq_without_eq)] +#[derive(Clone, PartialEq, ::prost::Message)] +pub struct DrainArgs { + #[prost(message, optional, tag = "1")] + pub drain_policy: ::core::option::Option, +} +#[allow(clippy::derive_partial_eq_without_eq)] +#[derive(Clone, PartialEq, ::prost::Message)] +pub struct RecordTargetValidationArgs { + #[prost(enumeration = "ValidationResult", tag = "1")] + pub result: i32, + #[prost(string, tag = "2")] + pub summary: ::prost::alloc::string::String, +} +#[allow(clippy::derive_partial_eq_without_eq)] +#[derive(Clone, PartialEq, ::prost::Message)] +pub struct AdoptFenceArgs { + #[prost(string, tag = "1")] + pub current_operation_id: ::prost::alloc::string::String, + #[prost(string, repeated, tag = "2")] + pub approvers: ::prost::alloc::vec::Vec<::prost::alloc::string::String>, + #[prost(string, tag = "3")] + pub incident_ref: ::prost::alloc::string::String, + #[prost(string, tag = "4")] + pub reason: ::prost::alloc::string::String, +} +#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash, PartialOrd, Ord, ::prost::Enumeration)] +#[repr(i32)] +pub enum FenceRole { + Unspecified = 0, + Source = 1, + Target = 2, +} +impl FenceRole { + /// String value of the enum field names used in the ProtoBuf definition. + /// + /// The values are not transformed in any way and thus are considered stable + /// (if the ProtoBuf definition does not change) and safe for programmatic use. + pub fn as_str_name(&self) -> &'static str { + match self { + FenceRole::Unspecified => "FENCE_ROLE_UNSPECIFIED", + FenceRole::Source => "FENCE_ROLE_SOURCE", + FenceRole::Target => "FENCE_ROLE_TARGET", + } + } + /// Creates an enum from field names used in the ProtoBuf definition. + pub fn from_str_name(value: &str) -> ::core::option::Option { + match value { + "FENCE_ROLE_UNSPECIFIED" => Some(Self::Unspecified), + "FENCE_ROLE_SOURCE" => Some(Self::Source), + "FENCE_ROLE_TARGET" => Some(Self::Target), + _ => None, + } + } +} +/// `UNFENCED` and `ABSENT` are never stored in a record; they appear as the `expected_state` of a +/// request against a namespace that has no record. `UNKNOWN_UNAVAILABLE` is derived and never +/// stored either. +#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash, PartialOrd, Ord, ::prost::Enumeration)] +#[repr(i32)] +pub enum FenceState { + Unspecified = 0, + Unfenced = 1, + Absent = 2, + SourceDraining = 3, + SourceWriteFenced = 4, + SourceReadDraining = 5, + SourceReadFenced = 6, + Released = 7, + TargetQuarantined = 8, + TargetImportDraining = 9, + TargetValidating = 10, + TargetWriteFenced = 11, + TargetWritable = 12, + TargetAborted = 13, + UnknownUnavailable = 14, +} +impl FenceState { + /// String value of the enum field names used in the ProtoBuf definition. + /// + /// The values are not transformed in any way and thus are considered stable + /// (if the ProtoBuf definition does not change) and safe for programmatic use. + pub fn as_str_name(&self) -> &'static str { + match self { + FenceState::Unspecified => "FENCE_STATE_UNSPECIFIED", + FenceState::Unfenced => "FENCE_STATE_UNFENCED", + FenceState::Absent => "FENCE_STATE_ABSENT", + FenceState::SourceDraining => "FENCE_STATE_SOURCE_DRAINING", + FenceState::SourceWriteFenced => "FENCE_STATE_SOURCE_WRITE_FENCED", + FenceState::SourceReadDraining => "FENCE_STATE_SOURCE_READ_DRAINING", + FenceState::SourceReadFenced => "FENCE_STATE_SOURCE_READ_FENCED", + FenceState::Released => "FENCE_STATE_RELEASED", + FenceState::TargetQuarantined => "FENCE_STATE_TARGET_QUARANTINED", + FenceState::TargetImportDraining => "FENCE_STATE_TARGET_IMPORT_DRAINING", + FenceState::TargetValidating => "FENCE_STATE_TARGET_VALIDATING", + FenceState::TargetWriteFenced => "FENCE_STATE_TARGET_WRITE_FENCED", + FenceState::TargetWritable => "FENCE_STATE_TARGET_WRITABLE", + FenceState::TargetAborted => "FENCE_STATE_TARGET_ABORTED", + FenceState::UnknownUnavailable => "FENCE_STATE_UNKNOWN_UNAVAILABLE", + } + } + /// Creates an enum from field names used in the ProtoBuf definition. + pub fn from_str_name(value: &str) -> ::core::option::Option { + match value { + "FENCE_STATE_UNSPECIFIED" => Some(Self::Unspecified), + "FENCE_STATE_UNFENCED" => Some(Self::Unfenced), + "FENCE_STATE_ABSENT" => Some(Self::Absent), + "FENCE_STATE_SOURCE_DRAINING" => Some(Self::SourceDraining), + "FENCE_STATE_SOURCE_WRITE_FENCED" => Some(Self::SourceWriteFenced), + "FENCE_STATE_SOURCE_READ_DRAINING" => Some(Self::SourceReadDraining), + "FENCE_STATE_SOURCE_READ_FENCED" => Some(Self::SourceReadFenced), + "FENCE_STATE_RELEASED" => Some(Self::Released), + "FENCE_STATE_TARGET_QUARANTINED" => Some(Self::TargetQuarantined), + "FENCE_STATE_TARGET_IMPORT_DRAINING" => Some(Self::TargetImportDraining), + "FENCE_STATE_TARGET_VALIDATING" => Some(Self::TargetValidating), + "FENCE_STATE_TARGET_WRITE_FENCED" => Some(Self::TargetWriteFenced), + "FENCE_STATE_TARGET_WRITABLE" => Some(Self::TargetWritable), + "FENCE_STATE_TARGET_ABORTED" => Some(Self::TargetAborted), + "FENCE_STATE_UNKNOWN_UNAVAILABLE" => Some(Self::UnknownUnavailable), + _ => None, + } + } +} +#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash, PartialOrd, Ord, ::prost::Enumeration)] +#[repr(i32)] +pub enum CommandKind { + Unspecified = 0, + AcquireSourceWriteFence = 1, + SetSourceReadFence = 2, + ClearSourceReadFence = 3, + ReleaseSourceWriteFence = 4, + CreateTargetQuarantined = 5, + SealTargetImport = 6, + RecordTargetValidation = 7, + PublishTargetReadableWriteFenced = 8, + EnableTargetWrites = 9, + AbortQuarantinedTarget = 10, + AdoptFence = 11, +} +impl CommandKind { + /// String value of the enum field names used in the ProtoBuf definition. + /// + /// The values are not transformed in any way and thus are considered stable + /// (if the ProtoBuf definition does not change) and safe for programmatic use. + pub fn as_str_name(&self) -> &'static str { + match self { + CommandKind::Unspecified => "COMMAND_KIND_UNSPECIFIED", + CommandKind::AcquireSourceWriteFence => { + "COMMAND_KIND_ACQUIRE_SOURCE_WRITE_FENCE" + } + CommandKind::SetSourceReadFence => "COMMAND_KIND_SET_SOURCE_READ_FENCE", + CommandKind::ClearSourceReadFence => "COMMAND_KIND_CLEAR_SOURCE_READ_FENCE", + CommandKind::ReleaseSourceWriteFence => { + "COMMAND_KIND_RELEASE_SOURCE_WRITE_FENCE" + } + CommandKind::CreateTargetQuarantined => { + "COMMAND_KIND_CREATE_TARGET_QUARANTINED" + } + CommandKind::SealTargetImport => "COMMAND_KIND_SEAL_TARGET_IMPORT", + CommandKind::RecordTargetValidation => { + "COMMAND_KIND_RECORD_TARGET_VALIDATION" + } + CommandKind::PublishTargetReadableWriteFenced => { + "COMMAND_KIND_PUBLISH_TARGET_READABLE_WRITE_FENCED" + } + CommandKind::EnableTargetWrites => "COMMAND_KIND_ENABLE_TARGET_WRITES", + CommandKind::AbortQuarantinedTarget => { + "COMMAND_KIND_ABORT_QUARANTINED_TARGET" + } + CommandKind::AdoptFence => "COMMAND_KIND_ADOPT_FENCE", + } + } + /// Creates an enum from field names used in the ProtoBuf definition. + pub fn from_str_name(value: &str) -> ::core::option::Option { + match value { + "COMMAND_KIND_UNSPECIFIED" => Some(Self::Unspecified), + "COMMAND_KIND_ACQUIRE_SOURCE_WRITE_FENCE" => { + Some(Self::AcquireSourceWriteFence) + } + "COMMAND_KIND_SET_SOURCE_READ_FENCE" => Some(Self::SetSourceReadFence), + "COMMAND_KIND_CLEAR_SOURCE_READ_FENCE" => Some(Self::ClearSourceReadFence), + "COMMAND_KIND_RELEASE_SOURCE_WRITE_FENCE" => { + Some(Self::ReleaseSourceWriteFence) + } + "COMMAND_KIND_CREATE_TARGET_QUARANTINED" => { + Some(Self::CreateTargetQuarantined) + } + "COMMAND_KIND_SEAL_TARGET_IMPORT" => Some(Self::SealTargetImport), + "COMMAND_KIND_RECORD_TARGET_VALIDATION" => Some(Self::RecordTargetValidation), + "COMMAND_KIND_PUBLISH_TARGET_READABLE_WRITE_FENCED" => { + Some(Self::PublishTargetReadableWriteFenced) + } + "COMMAND_KIND_ENABLE_TARGET_WRITES" => Some(Self::EnableTargetWrites), + "COMMAND_KIND_ABORT_QUARANTINED_TARGET" => Some(Self::AbortQuarantinedTarget), + "COMMAND_KIND_ADOPT_FENCE" => Some(Self::AdoptFence), + _ => None, + } + } +} +/// Only the outcomes a receipt can record. Errors are never stored. +#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash, PartialOrd, Ord, ::prost::Enumeration)] +#[repr(i32)] +pub enum ReceiptOutcome { + Unspecified = 0, + Applied = 1, + AlreadyApplied = 2, + Draining = 3, +} +impl ReceiptOutcome { + /// String value of the enum field names used in the ProtoBuf definition. + /// + /// The values are not transformed in any way and thus are considered stable + /// (if the ProtoBuf definition does not change) and safe for programmatic use. + pub fn as_str_name(&self) -> &'static str { + match self { + ReceiptOutcome::Unspecified => "RECEIPT_OUTCOME_UNSPECIFIED", + ReceiptOutcome::Applied => "RECEIPT_OUTCOME_APPLIED", + ReceiptOutcome::AlreadyApplied => "RECEIPT_OUTCOME_ALREADY_APPLIED", + ReceiptOutcome::Draining => "RECEIPT_OUTCOME_DRAINING", + } + } + /// Creates an enum from field names used in the ProtoBuf definition. + pub fn from_str_name(value: &str) -> ::core::option::Option { + match value { + "RECEIPT_OUTCOME_UNSPECIFIED" => Some(Self::Unspecified), + "RECEIPT_OUTCOME_APPLIED" => Some(Self::Applied), + "RECEIPT_OUTCOME_ALREADY_APPLIED" => Some(Self::AlreadyApplied), + "RECEIPT_OUTCOME_DRAINING" => Some(Self::Draining), + _ => None, + } + } +} +#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash, PartialOrd, Ord, ::prost::Enumeration)] +#[repr(i32)] +pub enum OnDeadline { + Unspecified = 0, + Fail = 1, + ForceRollback = 2, +} +impl OnDeadline { + /// String value of the enum field names used in the ProtoBuf definition. + /// + /// The values are not transformed in any way and thus are considered stable + /// (if the ProtoBuf definition does not change) and safe for programmatic use. + pub fn as_str_name(&self) -> &'static str { + match self { + OnDeadline::Unspecified => "ON_DEADLINE_UNSPECIFIED", + OnDeadline::Fail => "ON_DEADLINE_FAIL", + OnDeadline::ForceRollback => "ON_DEADLINE_FORCE_ROLLBACK", + } + } + /// Creates an enum from field names used in the ProtoBuf definition. + pub fn from_str_name(value: &str) -> ::core::option::Option { + match value { + "ON_DEADLINE_UNSPECIFIED" => Some(Self::Unspecified), + "ON_DEADLINE_FAIL" => Some(Self::Fail), + "ON_DEADLINE_FORCE_ROLLBACK" => Some(Self::ForceRollback), + _ => None, + } + } +} +#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash, PartialOrd, Ord, ::prost::Enumeration)] +#[repr(i32)] +pub enum ValidationResult { + Unspecified = 0, + Ok = 1, + Failed = 2, +} +impl ValidationResult { + /// String value of the enum field names used in the ProtoBuf definition. + /// + /// The values are not transformed in any way and thus are considered stable + /// (if the ProtoBuf definition does not change) and safe for programmatic use. + pub fn as_str_name(&self) -> &'static str { + match self { + ValidationResult::Unspecified => "VALIDATION_RESULT_UNSPECIFIED", + ValidationResult::Ok => "VALIDATION_RESULT_OK", + ValidationResult::Failed => "VALIDATION_RESULT_FAILED", + } + } + /// Creates an enum from field names used in the ProtoBuf definition. + pub fn from_str_name(value: &str) -> ::core::option::Option { + match value { + "VALIDATION_RESULT_UNSPECIFIED" => Some(Self::Unspecified), + "VALIDATION_RESULT_OK" => Some(Self::Ok), + "VALIDATION_RESULT_FAILED" => Some(Self::Failed), + _ => None, + } + } +} diff --git a/libsql-server/src/namespace/fence/command.rs b/libsql-server/src/namespace/fence/command.rs new file mode 100644 index 0000000000..718fc78d83 --- /dev/null +++ b/libsql-server/src/namespace/fence/command.rs @@ -0,0 +1,420 @@ +//! Fence commands, requests and their canonical fingerprint (`docs/NAMESPACE_FENCE.md` +//! sections 4.2, 4.4 and 5.3). + +use std::fmt; + +use prost::Message as _; +use sha2::{Digest as _, Sha256}; +use uuid::Uuid; + +use crate::namespace::NamespaceName; + +use super::proto; +use super::record::codec; +use super::state::FenceState; + +/// Largest accepted `RecordTargetValidation` summary, in bytes. +pub const MAX_VALIDATION_SUMMARY_BYTES: usize = 4096; + +/// What happens when a drain deadline passes. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] +pub enum OnDeadline { + /// Answer `DRAINING`; the durable state stays draining and admission stays closed. + Fail, + /// Roll back (or cancel, for reads) the work still holding the drain open, then keep + /// waiting for it to actually end. + ForceRollback, +} + +impl OnDeadline { + pub const fn as_str(self) -> &'static str { + match self { + OnDeadline::Fail => "fail", + OnDeadline::ForceRollback => "force_rollback", + } + } +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] +pub struct DrainPolicy { + pub deadline_ms: u64, + pub on_deadline: OnDeadline, +} + +/// The subset of namespace configuration `CreateTargetQuarantined` accepts. Restore options +/// and dump URLs are not part of it: import goes through the migration capability. +#[derive(Debug, Clone, Default, PartialEq, Eq, Hash)] +pub struct TargetConfig { + pub max_db_size: Option, + pub jwt_key: Option, + pub txn_timeout_s: Option, + pub allow_attach: bool, + pub durability_mode: Option, + pub bottomless_db_id: Option, +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] +pub enum ValidationResult { + Ok, + Failed, +} + +impl ValidationResult { + pub const fn as_str(self) -> &'static str { + match self { + ValidationResult::Ok => "ok", + ValidationResult::Failed => "failed", + } + } +} + +#[derive(Debug, Clone, PartialEq, Eq, Hash)] +pub struct AdoptArgs { + /// The operation that currently owns the record, as the adopter believes it. + pub current_operation_id: Uuid, + /// Two distinct, non-empty identities. Recorded, not verified (section 12). + pub approvers: Vec, + pub incident_ref: String, + pub reason: String, +} + +/// A mutating fence command and its command-specific arguments. +#[derive(Debug, Clone, PartialEq, Eq, Hash)] +pub enum FenceCommand { + AcquireSourceWriteFence { + /// The replication log id the caller observed on the source. + expected_log_id: Uuid, + drain_policy: Option, + }, + SetSourceReadFence { + drain_policy: Option, + }, + ClearSourceReadFence, + ReleaseSourceWriteFence, + CreateTargetQuarantined { + config: TargetConfig, + }, + SealTargetImport { + drain_policy: Option, + }, + RecordTargetValidation { + result: ValidationResult, + summary: String, + }, + PublishTargetReadableWriteFenced, + EnableTargetWrites, + AbortQuarantinedTarget, + AdoptFence(AdoptArgs), +} + +/// The kind of a command, without its arguments. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] +pub enum CommandKind { + AcquireSourceWriteFence, + SetSourceReadFence, + ClearSourceReadFence, + ReleaseSourceWriteFence, + CreateTargetQuarantined, + SealTargetImport, + RecordTargetValidation, + PublishTargetReadableWriteFenced, + EnableTargetWrites, + AbortQuarantinedTarget, + AdoptFence, +} + +impl CommandKind { + pub const ALL: [CommandKind; 11] = [ + CommandKind::AcquireSourceWriteFence, + CommandKind::SetSourceReadFence, + CommandKind::ClearSourceReadFence, + CommandKind::ReleaseSourceWriteFence, + CommandKind::CreateTargetQuarantined, + CommandKind::SealTargetImport, + CommandKind::RecordTargetValidation, + CommandKind::PublishTargetReadableWriteFenced, + CommandKind::EnableTargetWrites, + CommandKind::AbortQuarantinedTarget, + CommandKind::AdoptFence, + ]; + + pub const fn as_str(self) -> &'static str { + match self { + CommandKind::AcquireSourceWriteFence => "AcquireSourceWriteFence", + CommandKind::SetSourceReadFence => "SetSourceReadFence", + CommandKind::ClearSourceReadFence => "ClearSourceReadFence", + CommandKind::ReleaseSourceWriteFence => "ReleaseSourceWriteFence", + CommandKind::CreateTargetQuarantined => "CreateTargetQuarantined", + CommandKind::SealTargetImport => "SealTargetImport", + CommandKind::RecordTargetValidation => "RecordTargetValidation", + CommandKind::PublishTargetReadableWriteFenced => "PublishTargetReadableWriteFenced", + CommandKind::EnableTargetWrites => "EnableTargetWrites", + CommandKind::AbortQuarantinedTarget => "AbortQuarantinedTarget", + CommandKind::AdoptFence => "AdoptFence", + } + } + + /// The state a successful command finally leaves the record in, where that is a single + /// state. A command from the owner whose goal state the record is already in is + /// `ALREADY_APPLIED`. `RecordTargetValidation` and `AdoptFence` do not move the state and + /// have none. + pub const fn goal_state(self) -> Option { + match self { + CommandKind::AcquireSourceWriteFence => Some(FenceState::SourceWriteFenced), + CommandKind::SetSourceReadFence => Some(FenceState::SourceReadFenced), + CommandKind::ClearSourceReadFence => Some(FenceState::SourceWriteFenced), + CommandKind::ReleaseSourceWriteFence => Some(FenceState::Released), + CommandKind::CreateTargetQuarantined => Some(FenceState::TargetQuarantined), + CommandKind::SealTargetImport => Some(FenceState::TargetValidating), + CommandKind::PublishTargetReadableWriteFenced => Some(FenceState::TargetWriteFenced), + CommandKind::EnableTargetWrites => Some(FenceState::TargetWritable), + CommandKind::AbortQuarantinedTarget => Some(FenceState::TargetAborted), + CommandKind::RecordTargetValidation | CommandKind::AdoptFence => None, + } + } +} + +impl fmt::Display for CommandKind { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + f.write_str(self.as_str()) + } +} + +impl FenceCommand { + pub fn kind(&self) -> CommandKind { + match self { + FenceCommand::AcquireSourceWriteFence { .. } => CommandKind::AcquireSourceWriteFence, + FenceCommand::SetSourceReadFence { .. } => CommandKind::SetSourceReadFence, + FenceCommand::ClearSourceReadFence => CommandKind::ClearSourceReadFence, + FenceCommand::ReleaseSourceWriteFence => CommandKind::ReleaseSourceWriteFence, + FenceCommand::CreateTargetQuarantined { .. } => CommandKind::CreateTargetQuarantined, + FenceCommand::SealTargetImport { .. } => CommandKind::SealTargetImport, + FenceCommand::RecordTargetValidation { .. } => CommandKind::RecordTargetValidation, + FenceCommand::PublishTargetReadableWriteFenced => { + CommandKind::PublishTargetReadableWriteFenced + } + FenceCommand::EnableTargetWrites => CommandKind::EnableTargetWrites, + FenceCommand::AbortQuarantinedTarget => CommandKind::AbortQuarantinedTarget, + FenceCommand::AdoptFence(_) => CommandKind::AdoptFence, + } + } +} + +/// One mutating request: the common fields of section 4.2 and the command. +#[derive(Debug, Clone, PartialEq, Eq, Hash)] +pub struct FenceRequest { + pub namespace: NamespaceName, + pub operation_id: Uuid, + pub command_id: Uuid, + /// `Unfenced` or `Absent` for a namespace with no record. + pub expected_state: FenceState, + /// `0` for a namespace with no record. + pub expected_revision: u64, + pub command: FenceCommand, +} + +impl FenceRequest { + /// SHA-256 of the deterministic protobuf encoding of everything in the request except + /// `command_id` (section 5.3). Two requests with the same `(operation_id, command_id)` and + /// different fingerprints are a `FENCE_COMMAND_CONFLICT`. + pub fn fingerprint(&self) -> Fingerprint { + let input = proto::FingerprintInput { + namespace: self.namespace.as_str().to_string(), + operation_id: self.operation_id.to_string(), + kind: codec::command_kind_to_proto(self.command.kind()) as i32, + expected_state: codec::state_to_proto(self.expected_state) as i32, + expected_revision: self.expected_revision, + args: fingerprint_args(&self.command), + }; + Fingerprint(Sha256::digest(input.encode_to_vec()).into()) + } +} + +fn fingerprint_args(command: &FenceCommand) -> Option { + use proto::fingerprint_input::Args; + + let drain = |p: &Option| proto::DrainArgs { + drain_policy: p.as_ref().map(codec::drain_policy_to_proto), + }; + + match command { + FenceCommand::AcquireSourceWriteFence { + expected_log_id, + drain_policy, + } => Some(Args::AcquireSourceWriteFence( + proto::AcquireSourceWriteFenceArgs { + expected_log_id: expected_log_id.to_string(), + drain_policy: drain_policy.as_ref().map(codec::drain_policy_to_proto), + }, + )), + FenceCommand::SetSourceReadFence { drain_policy } => { + Some(Args::SetSourceReadFence(drain(drain_policy))) + } + FenceCommand::SealTargetImport { drain_policy } => { + Some(Args::SealTargetImport(drain(drain_policy))) + } + FenceCommand::CreateTargetQuarantined { config } => Some(Args::CreateTargetQuarantined( + codec::target_config_to_proto(config), + )), + FenceCommand::RecordTargetValidation { result, summary } => Some( + Args::RecordTargetValidation(proto::RecordTargetValidationArgs { + result: codec::validation_result_to_proto(*result) as i32, + summary: summary.clone(), + }), + ), + FenceCommand::AdoptFence(args) => Some(Args::AdoptFence(proto::AdoptFenceArgs { + current_operation_id: args.current_operation_id.to_string(), + approvers: args.approvers.clone(), + incident_ref: args.incident_ref.clone(), + reason: args.reason.clone(), + })), + FenceCommand::ClearSourceReadFence + | FenceCommand::ReleaseSourceWriteFence + | FenceCommand::PublishTargetReadableWriteFenced + | FenceCommand::EnableTargetWrites + | FenceCommand::AbortQuarantinedTarget => None, + } +} + +/// A canonical request fingerprint. +#[derive(Clone, Copy, PartialEq, Eq, Hash)] +pub struct Fingerprint(pub [u8; 32]); + +impl Fingerprint { + pub fn as_bytes(&self) -> &[u8; 32] { + &self.0 + } +} + +impl fmt::Display for Fingerprint { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + f.write_str("sha256:")?; + for b in self.0 { + write!(f, "{b:02x}")?; + } + Ok(()) + } +} + +impl fmt::Debug for Fingerprint { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + fmt::Display::fmt(self, f) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn request(command: FenceCommand) -> FenceRequest { + FenceRequest { + namespace: NamespaceName::from("db1"), + operation_id: Uuid::from_u128(1), + command_id: Uuid::from_u128(2), + expected_state: FenceState::Unfenced, + expected_revision: 0, + command, + } + } + + fn acquire() -> FenceCommand { + FenceCommand::AcquireSourceWriteFence { + expected_log_id: Uuid::from_u128(3), + drain_policy: Some(DrainPolicy { + deadline_ms: 1000, + on_deadline: OnDeadline::Fail, + }), + } + } + + #[test] + fn fingerprint_ignores_command_id() { + let a = request(acquire()); + let mut b = a.clone(); + b.command_id = Uuid::from_u128(99); + assert_eq!(a.fingerprint(), b.fingerprint()); + } + + #[test] + fn fingerprint_covers_every_other_field() { + let base = request(acquire()); + let fp = base.fingerprint(); + + let mut changed = Vec::new(); + let mut r = base.clone(); + r.namespace = NamespaceName::from("db2"); + changed.push(r); + let mut r = base.clone(); + r.operation_id = Uuid::from_u128(7); + changed.push(r); + let mut r = base.clone(); + r.expected_state = FenceState::Released; + changed.push(r); + let mut r = base.clone(); + r.expected_revision = 1; + changed.push(r); + let mut r = base.clone(); + r.command = FenceCommand::AcquireSourceWriteFence { + expected_log_id: Uuid::from_u128(4), + drain_policy: Some(DrainPolicy { + deadline_ms: 1000, + on_deadline: OnDeadline::Fail, + }), + }; + changed.push(r); + let mut r = base.clone(); + r.command = FenceCommand::AcquireSourceWriteFence { + expected_log_id: Uuid::from_u128(3), + drain_policy: Some(DrainPolicy { + deadline_ms: 1000, + on_deadline: OnDeadline::ForceRollback, + }), + }; + changed.push(r); + let mut r = base.clone(); + r.command = FenceCommand::AcquireSourceWriteFence { + expected_log_id: Uuid::from_u128(3), + drain_policy: None, + }; + changed.push(r); + + for r in changed { + assert_ne!(r.fingerprint(), fp, "{r:?}"); + } + } + + #[test] + fn fingerprint_distinguishes_argumentless_commands() { + let kinds = [ + FenceCommand::ClearSourceReadFence, + FenceCommand::ReleaseSourceWriteFence, + FenceCommand::PublishTargetReadableWriteFenced, + FenceCommand::EnableTargetWrites, + FenceCommand::AbortQuarantinedTarget, + FenceCommand::SetSourceReadFence { drain_policy: None }, + FenceCommand::SealTargetImport { drain_policy: None }, + ]; + let fps: std::collections::HashSet<_> = kinds + .into_iter() + .map(|c| request(c).fingerprint()) + .collect(); + assert_eq!(fps.len(), 7); + } + + /// The fingerprint is stored in receipts and compared on every replay, so its encoding must + /// never change: a change would turn every stored receipt into a command conflict. + #[test] + fn fingerprint_is_stable() { + assert_eq!( + request(acquire()).fingerprint().to_string(), + "sha256:c585d3570b5eb5a3ef9b8efb89eca26e2336c7e19a561fa3db667c4cde905515" + ); + } + + #[test] + fn every_kind_has_a_name() { + let names: std::collections::HashSet<_> = + CommandKind::ALL.iter().map(|k| k.as_str()).collect(); + assert_eq!(names.len(), CommandKind::ALL.len()); + } +} diff --git a/libsql-server/src/namespace/fence/mod.rs b/libsql-server/src/namespace/fence/mod.rs new file mode 100644 index 0000000000..3ffe2746f8 --- /dev/null +++ b/libsql-server/src/namespace/fence/mod.rs @@ -0,0 +1,28 @@ +//! Namespace fence: a durable, operation-owned control record that an external operation (for +//! example, moving a database between servers) uses as the data-plane authority boundary for +//! one namespace. +//! +//! `docs/NAMESPACE_FENCE.md` is the contract and the design. This module holds the parts with +//! no I/O: the states and permission matrix ([`state`]), the stable outcome codes and their +//! protocol mappings ([`outcome`]), commands and their canonical fingerprint ([`command`]), +//! records, receipts and markers with their strict durable encoding ([`record`]), and the pure +//! transition function ([`transition`]). + +// The persistence, controller and protocol layers that consume these types land in the +// following commits of this series; until then most of the module is unused by the rest of +// the crate. This attribute is removed once they are wired. +#![allow(dead_code)] + +pub mod command; +pub mod outcome; +pub mod record; +pub mod state; +pub mod transition; + +#[allow(clippy::all)] +pub(crate) mod proto { + include!("../../generated/namespace_fence.rs"); +} + +/// Version of the fence admin protocol reported by capability discovery. +pub const FENCE_PROTOCOL_VERSION: u32 = 1; diff --git a/libsql-server/src/namespace/fence/outcome.rs b/libsql-server/src/namespace/fence/outcome.rs new file mode 100644 index 0000000000..6a9b6a8b57 --- /dev/null +++ b/libsql-server/src/namespace/fence/outcome.rs @@ -0,0 +1,409 @@ +//! Stable outcome codes and their protocol mappings (`docs/NAMESPACE_FENCE.md` section 6). + +use std::fmt; +use std::str::FromStr; + +use hyper::StatusCode; + +/// gRPC metadata key carrying the stable code of a fence denial. +pub const GRPC_FENCE_CODE_METADATA: &str = "x-libsql-fence-code"; + +/// Every machine-readable outcome a fence command or a fenced data-plane operation can report. +/// Clients match on [`FenceOutcome::as_str`], never on a message. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] +pub enum FenceOutcome { + Applied, + AlreadyApplied, + Draining, + MigrationWriteFenced, + MigrationReadFenced, + MigrationTargetQuarantined, + FenceStateUnavailable, + OperationCapabilityRequired, + FenceOwnedByAnotherOperation, + FenceRevisionMismatch, + InvalidFenceTransition, + FenceCommandConflict, + FenceCommitIndeterminate, + FencePreconditionFailed, +} + +/// What kind of answer an outcome is. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum OutcomeKind { + Success, + InProgress, + /// A denial of ordinary data-plane or lifecycle work because of the fence. + DataPlane, + /// A refusal of a fence command (or of capability work). + Control, +} + +impl FenceOutcome { + pub const ALL: [FenceOutcome; 14] = [ + FenceOutcome::Applied, + FenceOutcome::AlreadyApplied, + FenceOutcome::Draining, + FenceOutcome::MigrationWriteFenced, + FenceOutcome::MigrationReadFenced, + FenceOutcome::MigrationTargetQuarantined, + FenceOutcome::FenceStateUnavailable, + FenceOutcome::OperationCapabilityRequired, + FenceOutcome::FenceOwnedByAnotherOperation, + FenceOutcome::FenceRevisionMismatch, + FenceOutcome::InvalidFenceTransition, + FenceOutcome::FenceCommandConflict, + FenceOutcome::FenceCommitIndeterminate, + FenceOutcome::FencePreconditionFailed, + ]; + + pub const fn as_str(self) -> &'static str { + match self { + FenceOutcome::Applied => "APPLIED", + FenceOutcome::AlreadyApplied => "ALREADY_APPLIED", + FenceOutcome::Draining => "DRAINING", + FenceOutcome::MigrationWriteFenced => "MIGRATION_WRITE_FENCED", + FenceOutcome::MigrationReadFenced => "MIGRATION_READ_FENCED", + FenceOutcome::MigrationTargetQuarantined => "MIGRATION_TARGET_QUARANTINED", + FenceOutcome::FenceStateUnavailable => "FENCE_STATE_UNAVAILABLE", + FenceOutcome::OperationCapabilityRequired => "OPERATION_CAPABILITY_REQUIRED", + FenceOutcome::FenceOwnedByAnotherOperation => "FENCE_OWNED_BY_ANOTHER_OPERATION", + FenceOutcome::FenceRevisionMismatch => "FENCE_REVISION_MISMATCH", + FenceOutcome::InvalidFenceTransition => "INVALID_FENCE_TRANSITION", + FenceOutcome::FenceCommandConflict => "FENCE_COMMAND_CONFLICT", + FenceOutcome::FenceCommitIndeterminate => "FENCE_COMMIT_INDETERMINATE", + FenceOutcome::FencePreconditionFailed => "FENCE_PRECONDITION_FAILED", + } + } + + pub const fn kind(self) -> OutcomeKind { + match self { + FenceOutcome::Applied | FenceOutcome::AlreadyApplied => OutcomeKind::Success, + FenceOutcome::Draining => OutcomeKind::InProgress, + FenceOutcome::MigrationWriteFenced + | FenceOutcome::MigrationReadFenced + | FenceOutcome::MigrationTargetQuarantined + | FenceOutcome::FenceStateUnavailable => OutcomeKind::DataPlane, + FenceOutcome::OperationCapabilityRequired + | FenceOutcome::FenceOwnedByAnotherOperation + | FenceOutcome::FenceRevisionMismatch + | FenceOutcome::InvalidFenceTransition + | FenceOutcome::FenceCommandConflict + | FenceOutcome::FenceCommitIndeterminate + | FenceOutcome::FencePreconditionFailed => OutcomeKind::Control, + } + } + + pub const fn is_error(self) -> bool { + matches!(self.kind(), OutcomeKind::DataPlane | OutcomeKind::Control) + } + + /// Status code on the admin API. + pub fn admin_http_status(self) -> StatusCode { + match self { + FenceOutcome::Applied | FenceOutcome::AlreadyApplied => StatusCode::OK, + FenceOutcome::Draining => StatusCode::ACCEPTED, + FenceOutcome::MigrationWriteFenced + | FenceOutcome::MigrationReadFenced + | FenceOutcome::MigrationTargetQuarantined + | FenceOutcome::FenceStateUnavailable => StatusCode::LOCKED, + FenceOutcome::OperationCapabilityRequired => StatusCode::FORBIDDEN, + FenceOutcome::FenceOwnedByAnotherOperation + | FenceOutcome::FenceRevisionMismatch + | FenceOutcome::InvalidFenceTransition + | FenceOutcome::FenceCommandConflict + | FenceOutcome::FenceCommitIndeterminate => StatusCode::CONFLICT, + FenceOutcome::FencePreconditionFailed => StatusCode::PRECONDITION_FAILED, + } + } + + /// Status code on the user HTTP API (`/`, `/v1`, `/v2`, `/v3`, `/dump`). Only data-plane + /// denials reach it. + pub fn user_http_status(self) -> Option { + match self.kind() { + OutcomeKind::DataPlane => Some(StatusCode::LOCKED), + _ => None, + } + } + + /// The Hrana error `code`. Only data-plane denials reach Hrana. + pub fn hrana_code(self) -> Option<&'static str> { + match self.kind() { + OutcomeKind::DataPlane => Some(self.as_str()), + _ => None, + } + } + + /// The gRPC status code on RPC, proxy connection and replication services. Never + /// `UNAVAILABLE`, which the write proxy retries without bound. + pub fn grpc_code(self) -> Option { + match self { + FenceOutcome::MigrationWriteFenced + | FenceOutcome::MigrationReadFenced + | FenceOutcome::MigrationTargetQuarantined + | FenceOutcome::FenceStateUnavailable + | FenceOutcome::OperationCapabilityRequired => Some(tonic::Code::FailedPrecondition), + _ => None, + } + } + + /// The value of the proxy protocol's `Error.stable_code` field. + pub fn proxy_stable_code(self) -> Option<&'static str> { + match self.kind() { + OutcomeKind::DataPlane => Some(self.as_str()), + _ => None, + } + } +} + +impl fmt::Display for FenceOutcome { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + f.write_str(self.as_str()) + } +} + +#[derive(Debug, Clone, PartialEq, Eq, thiserror::Error)] +#[error("unknown fence outcome `{0}`")] +pub struct UnknownFenceOutcome(pub String); + +impl FromStr for FenceOutcome { + type Err = UnknownFenceOutcome; + + fn from_str(s: &str) -> Result { + FenceOutcome::ALL + .iter() + .copied() + .find(|o| o.as_str() == s) + .ok_or_else(|| UnknownFenceOutcome(s.to_string())) + } +} + +/// The bounded `detail` reason of an error outcome. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] +pub enum FenceDetail { + // FENCE_PRECONDITION_FAILED + AdminAuthRequired, + FenceDisabled, + NotPrimary, + SharedSchemaUnsupported, + NamespaceIdentityMismatch, + NamespaceExists, + ValidationReceiptRequired, + RestoreNotAllowed, + AdoptionNotAuthorised, + InvalidArgument, + // INVALID_FENCE_TRANSITION + RoleMismatch, + OperationFinished, + // FENCE_STATE_UNAVAILABLE + CorruptRecord, + UnsupportedFormatVersion, + IncompleteTargetCreation, + MetastoreBehindMarker, + IndeterminateCommit, +} + +impl FenceDetail { + pub const fn as_str(self) -> &'static str { + match self { + FenceDetail::AdminAuthRequired => "admin_auth_required", + FenceDetail::FenceDisabled => "fence_disabled", + FenceDetail::NotPrimary => "not_primary", + FenceDetail::SharedSchemaUnsupported => "shared_schema_unsupported", + FenceDetail::NamespaceIdentityMismatch => "namespace_identity_mismatch", + FenceDetail::NamespaceExists => "namespace_exists", + FenceDetail::ValidationReceiptRequired => "validation_receipt_required", + FenceDetail::RestoreNotAllowed => "restore_not_allowed", + FenceDetail::AdoptionNotAuthorised => "adoption_not_authorised", + FenceDetail::InvalidArgument => "invalid_argument", + FenceDetail::RoleMismatch => "role_mismatch", + FenceDetail::OperationFinished => "operation_finished", + FenceDetail::CorruptRecord => "corrupt_record", + FenceDetail::UnsupportedFormatVersion => "unsupported_format_version", + FenceDetail::IncompleteTargetCreation => "incomplete_target_creation", + FenceDetail::MetastoreBehindMarker => "metastore_behind_marker", + FenceDetail::IndeterminateCommit => "indeterminate_commit", + } + } +} + +impl fmt::Display for FenceDetail { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + f.write_str(self.as_str()) + } +} + +/// An error outcome with its bounded detail and a human message. +#[derive(Debug, Clone, PartialEq, Eq, thiserror::Error)] +#[error("{outcome}: {message}")] +pub struct FenceError { + outcome: FenceOutcome, + detail: Option, + message: String, +} + +impl FenceError { + /// # Panics + /// + /// If `outcome` is not an error outcome. That is a programming error, never input. + pub fn new(outcome: FenceOutcome, message: impl Into) -> Self { + assert!(outcome.is_error(), "{outcome} is not an error outcome"); + Self { + outcome, + detail: None, + message: message.into(), + } + } + + pub fn with_detail(mut self, detail: FenceDetail) -> Self { + self.detail = Some(detail); + self + } + + pub fn outcome(&self) -> FenceOutcome { + self.outcome + } + + pub fn detail(&self) -> Option { + self.detail + } + + pub fn message(&self) -> &str { + &self.message + } + + /// A gRPC status for this error, if the outcome has a gRPC mapping. The code is in the + /// [`GRPC_FENCE_CODE_METADATA`] entry and prefixes the message. + pub fn to_grpc_status(&self) -> Option { + let code = self.outcome.grpc_code()?; + let mut status = tonic::Status::new(code, format!("{}: {}", self.outcome, self.message)); + status.metadata_mut().insert( + GRPC_FENCE_CODE_METADATA, + tonic::metadata::MetadataValue::from_static(self.outcome.as_str()), + ); + Some(status) + } + + /// The stable code carried by a gRPC status produced by [`FenceError::to_grpc_status`]. + pub fn outcome_from_grpc_status(status: &tonic::Status) -> Option { + status + .metadata() + .get(GRPC_FENCE_CODE_METADATA)? + .to_str() + .ok()? + .parse() + .ok() + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn codes_round_trip() { + for outcome in FenceOutcome::ALL { + assert_eq!(outcome.as_str().parse::().unwrap(), outcome); + } + assert!("applied".parse::().is_err()); + } + + /// Section 6: data-plane denials are never 500, 503, 429 or gRPC UNAVAILABLE. + #[test] + fn denials_are_never_retryable_statuses() { + let retryable = [ + StatusCode::INTERNAL_SERVER_ERROR, + StatusCode::SERVICE_UNAVAILABLE, + StatusCode::TOO_MANY_REQUESTS, + StatusCode::BAD_GATEWAY, + StatusCode::GATEWAY_TIMEOUT, + ]; + for outcome in FenceOutcome::ALL { + assert!( + !retryable.contains(&outcome.admin_http_status()), + "{outcome}" + ); + if let Some(status) = outcome.user_http_status() { + assert!(!retryable.contains(&status), "{outcome}"); + } + assert_ne!(outcome.grpc_code(), Some(tonic::Code::Unavailable)); + } + } + + #[test] + fn protocol_table() { + use FenceOutcome as O; + let rows = [ + (O::Applied, 200, None, None), + (O::AlreadyApplied, 200, None, None), + (O::Draining, 202, None, None), + ( + O::MigrationWriteFenced, + 423, + Some(423), + Some("MIGRATION_WRITE_FENCED"), + ), + ( + O::MigrationReadFenced, + 423, + Some(423), + Some("MIGRATION_READ_FENCED"), + ), + ( + O::MigrationTargetQuarantined, + 423, + Some(423), + Some("MIGRATION_TARGET_QUARANTINED"), + ), + ( + O::FenceStateUnavailable, + 423, + Some(423), + Some("FENCE_STATE_UNAVAILABLE"), + ), + (O::OperationCapabilityRequired, 403, None, None), + (O::FenceOwnedByAnotherOperation, 409, None, None), + (O::FenceRevisionMismatch, 409, None, None), + (O::InvalidFenceTransition, 409, None, None), + (O::FenceCommandConflict, 409, None, None), + (O::FenceCommitIndeterminate, 409, None, None), + (O::FencePreconditionFailed, 412, None, None), + ]; + assert_eq!(rows.len(), FenceOutcome::ALL.len()); + for (outcome, admin, user, hrana) in rows { + assert_eq!(outcome.admin_http_status().as_u16(), admin, "{outcome}"); + assert_eq!( + outcome.user_http_status().map(|s| s.as_u16()), + user, + "{outcome}" + ); + assert_eq!(outcome.hrana_code(), hrana, "{outcome}"); + assert_eq!(outcome.proxy_stable_code(), hrana, "{outcome}"); + } + } + + #[test] + fn grpc_status_carries_code() { + let err = FenceError::new(FenceOutcome::MigrationReadFenced, "reads are fenced"); + let status = err.to_grpc_status().unwrap(); + assert_eq!(status.code(), tonic::Code::FailedPrecondition); + assert!(status.message().starts_with("MIGRATION_READ_FENCED: ")); + assert_eq!( + FenceError::outcome_from_grpc_status(&status), + Some(FenceOutcome::MigrationReadFenced) + ); + assert_eq!( + FenceError::outcome_from_grpc_status(&tonic::Status::unavailable("x")), + None + ); + + let control = FenceError::new(FenceOutcome::FenceRevisionMismatch, "stale"); + assert!(control.to_grpc_status().is_none()); + } + + #[test] + #[should_panic] + fn success_is_not_an_error() { + FenceError::new(FenceOutcome::Applied, "nope"); + } +} diff --git a/libsql-server/src/namespace/fence/record.rs b/libsql-server/src/namespace/fence/record.rs new file mode 100644 index 0000000000..9599f2a849 --- /dev/null +++ b/libsql-server/src/namespace/fence/record.rs @@ -0,0 +1,908 @@ +//! Fence records, command receipts, the on-disk marker, and their durable encoding +//! (`docs/NAMESPACE_FENCE.md` sections 5.1, 5.2 and 5.6). +//! +//! Decoding is strict: an unknown format version, an unknown enum value, a missing required +//! field, a malformed id or a record that contradicts itself is an error, which the store turns +//! into `UNKNOWN_UNAVAILABLE`. Nothing is defaulted. + +use prost::Message as _; +use uuid::Uuid; + +use crate::namespace::NamespaceName; + +use super::command::{CommandKind, DrainPolicy, Fingerprint, TargetConfig, ValidationResult}; +use super::outcome::FenceOutcome; +use super::proto; +use super::state::{Admission, FenceState, Role}; + +/// The only `format_version` this server writes and reads. +pub const FENCE_FORMAT_VERSION: u32 = 1; + +/// Identity of the namespace copy a record is about. +#[derive(Debug, Clone, Default, PartialEq, Eq)] +pub struct NamespaceIdentity { + /// Replication log id: for a source, captured at acquisition and checked against the + /// caller's expectation; for a target, known once the namespace exists. + pub log_id: Option, + /// Server-generated id of a target created by `CreateTargetQuarantined`. + pub target_incarnation_id: Option, +} + +/// The source's replication position once no writer can commit. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct FrozenBoundary { + pub log_id: Uuid, + pub frame_no: u64, +} + +/// The pre-fence values of the legacy `block_*` configuration fields, restored when the +/// operation finishes (section 13.2). +#[derive(Debug, Clone, Default, PartialEq, Eq)] +pub struct LegacyBlocks { + pub block_reads: bool, + pub block_writes: bool, + pub block_reason: Option, +} + +/// The server process that wrote something. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct ServerIdentity { + pub build: String, + pub instance_id: Uuid, +} + +/// What the server observed of a target when a validation result was recorded. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct ValidationSnapshot { + pub log_id: Uuid, + pub frame_no: u64, + pub page_count: u64, +} + +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct ValidationRecord { + pub operation_id: Uuid, + pub command_id: Uuid, + pub result: ValidationResult, + pub summary: String, + pub snapshot: Option, + pub recorded_at_ms: i64, +} + +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct Adoption { + pub previous_operation_id: Uuid, + pub new_operation_id: Uuid, + pub command_id: Uuid, + pub approvers: Vec, + pub incident_ref: String, + pub reason: String, + pub at_ms: i64, + /// The revision the adoption produced. + pub revision: u64, +} + +/// The durable control record of one fenced namespace. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct NamespaceFenceRecord { + pub namespace: NamespaceName, + pub role: Role, + /// Always a durable state whose role is `role`. + pub state: FenceState, + /// Starts at 1 and increases by one on every applied transition. Survives restart. + pub revision: u64, + pub operation_id: Uuid, + pub identity: NamespaceIdentity, + pub drain_policy: Option, + /// When the current drain was requested, if the record is draining. + pub drain_started_at_ms: Option, + pub frozen_boundary: Option, + /// The most recent validation result of the owning operation (target). + pub validation: Option, + pub legacy_blocks: LegacyBlocks, + pub created_at_ms: i64, + pub last_transition_at_ms: i64, + /// The command that produced the current revision. + pub last_command_id: Uuid, + pub written_by: ServerIdentity, + pub adoptions: Vec, +} + +impl NamespaceFenceRecord { + pub fn write_admission(&self) -> Admission { + self.state.write_admission() + } + + pub fn read_admission(&self) -> Admission { + self.state.read_admission() + } + + /// Values of the legacy `block_*` configuration fields while this record is in force: the + /// fence state mirrored for an older binary, or the pre-fence values once the operation + /// has released the namespace. + pub fn legacy_mirror(&self) -> LegacyBlocks { + match self.state { + FenceState::Released | FenceState::TargetWritable => self.legacy_blocks.clone(), + state => LegacyBlocks { + block_reads: !state.read_admission().is_open(), + block_writes: !state.write_admission().is_open(), + block_reason: Some(format!( + "namespace fence: {state} (operation {})", + self.operation_id + )), + }, + } + } + + pub fn encode(&self) -> Vec { + codec::record_to_proto(self).encode_to_vec() + } + + /// Decode a stored record. `format_version` and `revision` are the columns stored beside + /// the payload; both must agree with it. + pub fn decode( + format_version: u32, + revision: u64, + bytes: &[u8], + ) -> Result { + check_format_version(format_version)?; + let msg = proto::FenceRecord::decode(bytes)?; + let record = codec::record_from_proto(msg)?; + if record.revision != revision { + return Err(FenceDecodeError::Invalid( + "revision column disagrees with payload", + )); + } + Ok(record) + } +} + +/// The durable result of one applied command, keyed by `(namespace, operation_id, command_id)`. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct CommandReceipt { + pub namespace: NamespaceName, + pub operation_id: Uuid, + pub command_id: Uuid, + pub command: CommandKind, + pub fingerprint: Fingerprint, + /// `APPLIED`, `ALREADY_APPLIED` or `DRAINING`. Errors are never stored. + pub outcome: FenceOutcome, + pub revision_before: u64, + pub revision_after: u64, + pub state_after: FenceState, + pub applied_at_ms: i64, + pub instance_id: Uuid, + pub adoption: Option, +} + +impl CommandReceipt { + /// Whether the receipt holds the command's final answer, as opposed to a drain that is + /// still to be completed. + pub fn is_final(&self) -> bool { + self.outcome != FenceOutcome::Draining + } + + pub fn encode(&self) -> Vec { + codec::receipt_to_proto(self).encode_to_vec() + } + + pub fn decode(format_version: u32, bytes: &[u8]) -> Result { + check_format_version(format_version)?; + let msg = proto::CommandReceipt::decode(bytes)?; + codec::receipt_from_proto(msg) + } +} + +/// Contents of the per-namespace marker file: a copy of the last committed record. Written +/// after each metastore commit, and before the metastore transaction of +/// `CreateTargetQuarantined`, so recovery can tell a fenced namespace from a legacy one and +/// detect a metastore that went backwards (section 5.6). +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct FenceMarker { + pub record: NamespaceFenceRecord, +} + +impl FenceMarker { + pub fn for_record(record: &NamespaceFenceRecord) -> Self { + Self { + record: record.clone(), + } + } + + pub fn encode(&self) -> Vec { + proto::FenceMarker { + format_version: FENCE_FORMAT_VERSION, + record: Some(codec::record_to_proto(&self.record)), + } + .encode_to_vec() + } + + pub fn decode(bytes: &[u8]) -> Result { + let msg = proto::FenceMarker::decode(bytes)?; + check_format_version(msg.format_version)?; + let record = msg + .record + .ok_or(FenceDecodeError::Invalid("marker without record"))?; + Ok(Self { + record: codec::record_from_proto(record)?, + }) + } +} + +/// Why stored fence state could not be read. Every variant means the namespace is +/// `UNKNOWN_UNAVAILABLE`. +#[derive(Debug, Clone, PartialEq, Eq, thiserror::Error)] +pub enum FenceDecodeError { + #[error("unsupported fence format version {0}")] + UnsupportedFormatVersion(u32), + #[error("undecodable fence payload: {0}")] + Undecodable(String), + #[error("invalid fence payload: {0}")] + Invalid(&'static str), +} + +impl From for FenceDecodeError { + fn from(e: prost::DecodeError) -> Self { + FenceDecodeError::Undecodable(e.to_string()) + } +} + +fn check_format_version(v: u32) -> Result<(), FenceDecodeError> { + if v != FENCE_FORMAT_VERSION { + return Err(FenceDecodeError::UnsupportedFormatVersion(v)); + } + Ok(()) +} + +/// Conversions between the domain types and their protobuf encoding. +pub(super) mod codec { + use super::*; + use crate::namespace::fence::command::OnDeadline; + + type R = Result; + + pub fn parse_uuid(s: &str) -> R { + // Only the canonical hyphenated form is ever written. + let id = Uuid::parse_str(s).map_err(|_| FenceDecodeError::Invalid("malformed id"))?; + if id.hyphenated().to_string() != s { + return Err(FenceDecodeError::Invalid("non-canonical id")); + } + Ok(id) + } + + fn parse_opt_uuid(s: Option<&str>) -> R> { + s.map(parse_uuid).transpose() + } + + fn namespace(s: String) -> R { + NamespaceName::from_string(s).map_err(|_| FenceDecodeError::Invalid("invalid namespace")) + } + + pub fn role_to_proto(role: Role) -> proto::FenceRole { + match role { + Role::Source => proto::FenceRole::Source, + Role::Target => proto::FenceRole::Target, + } + } + + pub fn role_from_proto(v: i32) -> R { + match proto::FenceRole::try_from(v) { + Ok(proto::FenceRole::Source) => Ok(Role::Source), + Ok(proto::FenceRole::Target) => Ok(Role::Target), + _ => Err(FenceDecodeError::Invalid("unknown role")), + } + } + + pub fn state_to_proto(state: FenceState) -> proto::FenceState { + use proto::FenceState as P; + match state { + FenceState::Unfenced => P::Unfenced, + FenceState::Absent => P::Absent, + FenceState::SourceDraining => P::SourceDraining, + FenceState::SourceWriteFenced => P::SourceWriteFenced, + FenceState::SourceReadDraining => P::SourceReadDraining, + FenceState::SourceReadFenced => P::SourceReadFenced, + FenceState::Released => P::Released, + FenceState::TargetQuarantined => P::TargetQuarantined, + FenceState::TargetImportDraining => P::TargetImportDraining, + FenceState::TargetValidating => P::TargetValidating, + FenceState::TargetWriteFenced => P::TargetWriteFenced, + FenceState::TargetWritable => P::TargetWritable, + FenceState::TargetAborted => P::TargetAborted, + FenceState::UnknownUnavailable => P::UnknownUnavailable, + } + } + + pub fn state_from_proto(v: i32) -> R { + use proto::FenceState as P; + let p = P::try_from(v).map_err(|_| FenceDecodeError::Invalid("unknown state"))?; + Ok(match p { + P::Unspecified => return Err(FenceDecodeError::Invalid("unspecified state")), + P::Unfenced => FenceState::Unfenced, + P::Absent => FenceState::Absent, + P::SourceDraining => FenceState::SourceDraining, + P::SourceWriteFenced => FenceState::SourceWriteFenced, + P::SourceReadDraining => FenceState::SourceReadDraining, + P::SourceReadFenced => FenceState::SourceReadFenced, + P::Released => FenceState::Released, + P::TargetQuarantined => FenceState::TargetQuarantined, + P::TargetImportDraining => FenceState::TargetImportDraining, + P::TargetValidating => FenceState::TargetValidating, + P::TargetWriteFenced => FenceState::TargetWriteFenced, + P::TargetWritable => FenceState::TargetWritable, + P::TargetAborted => FenceState::TargetAborted, + P::UnknownUnavailable => FenceState::UnknownUnavailable, + }) + } + + /// A state stored in a record or marker: durable, and of the stated role. + pub fn durable_state_from_proto(v: i32, role: Role) -> R { + let state = state_from_proto(v)?; + match state.role() { + Some(r) if r == role => Ok(state), + Some(_) => Err(FenceDecodeError::Invalid("state does not match role")), + None => Err(FenceDecodeError::Invalid("state is not durable")), + } + } + + pub fn command_kind_to_proto(kind: CommandKind) -> proto::CommandKind { + use proto::CommandKind as P; + match kind { + CommandKind::AcquireSourceWriteFence => P::AcquireSourceWriteFence, + CommandKind::SetSourceReadFence => P::SetSourceReadFence, + CommandKind::ClearSourceReadFence => P::ClearSourceReadFence, + CommandKind::ReleaseSourceWriteFence => P::ReleaseSourceWriteFence, + CommandKind::CreateTargetQuarantined => P::CreateTargetQuarantined, + CommandKind::SealTargetImport => P::SealTargetImport, + CommandKind::RecordTargetValidation => P::RecordTargetValidation, + CommandKind::PublishTargetReadableWriteFenced => P::PublishTargetReadableWriteFenced, + CommandKind::EnableTargetWrites => P::EnableTargetWrites, + CommandKind::AbortQuarantinedTarget => P::AbortQuarantinedTarget, + CommandKind::AdoptFence => P::AdoptFence, + } + } + + fn command_kind_from_proto(v: i32) -> R { + use proto::CommandKind as P; + let p = P::try_from(v).map_err(|_| FenceDecodeError::Invalid("unknown command"))?; + Ok(match p { + P::Unspecified => return Err(FenceDecodeError::Invalid("unspecified command")), + P::AcquireSourceWriteFence => CommandKind::AcquireSourceWriteFence, + P::SetSourceReadFence => CommandKind::SetSourceReadFence, + P::ClearSourceReadFence => CommandKind::ClearSourceReadFence, + P::ReleaseSourceWriteFence => CommandKind::ReleaseSourceWriteFence, + P::CreateTargetQuarantined => CommandKind::CreateTargetQuarantined, + P::SealTargetImport => CommandKind::SealTargetImport, + P::RecordTargetValidation => CommandKind::RecordTargetValidation, + P::PublishTargetReadableWriteFenced => CommandKind::PublishTargetReadableWriteFenced, + P::EnableTargetWrites => CommandKind::EnableTargetWrites, + P::AbortQuarantinedTarget => CommandKind::AbortQuarantinedTarget, + P::AdoptFence => CommandKind::AdoptFence, + }) + } + + fn outcome_to_proto(outcome: FenceOutcome) -> proto::ReceiptOutcome { + match outcome { + FenceOutcome::Applied => proto::ReceiptOutcome::Applied, + FenceOutcome::AlreadyApplied => proto::ReceiptOutcome::AlreadyApplied, + FenceOutcome::Draining => proto::ReceiptOutcome::Draining, + other => unreachable!("receipts never store {other}"), + } + } + + fn outcome_from_proto(v: i32) -> R { + match proto::ReceiptOutcome::try_from(v) { + Ok(proto::ReceiptOutcome::Applied) => Ok(FenceOutcome::Applied), + Ok(proto::ReceiptOutcome::AlreadyApplied) => Ok(FenceOutcome::AlreadyApplied), + Ok(proto::ReceiptOutcome::Draining) => Ok(FenceOutcome::Draining), + _ => Err(FenceDecodeError::Invalid("unknown receipt outcome")), + } + } + + pub fn drain_policy_to_proto(p: &DrainPolicy) -> proto::DrainPolicy { + proto::DrainPolicy { + deadline_ms: p.deadline_ms, + on_deadline: match p.on_deadline { + OnDeadline::Fail => proto::OnDeadline::Fail, + OnDeadline::ForceRollback => proto::OnDeadline::ForceRollback, + } as i32, + } + } + + fn drain_policy_from_proto(p: proto::DrainPolicy) -> R { + let on_deadline = match proto::OnDeadline::try_from(p.on_deadline) { + Ok(proto::OnDeadline::Fail) => OnDeadline::Fail, + Ok(proto::OnDeadline::ForceRollback) => OnDeadline::ForceRollback, + _ => return Err(FenceDecodeError::Invalid("unknown drain deadline policy")), + }; + Ok(DrainPolicy { + deadline_ms: p.deadline_ms, + on_deadline, + }) + } + + pub fn validation_result_to_proto(r: ValidationResult) -> proto::ValidationResult { + match r { + ValidationResult::Ok => proto::ValidationResult::Ok, + ValidationResult::Failed => proto::ValidationResult::Failed, + } + } + + fn validation_result_from_proto(v: i32) -> R { + match proto::ValidationResult::try_from(v) { + Ok(proto::ValidationResult::Ok) => Ok(ValidationResult::Ok), + Ok(proto::ValidationResult::Failed) => Ok(ValidationResult::Failed), + _ => Err(FenceDecodeError::Invalid("unknown validation result")), + } + } + + pub fn target_config_to_proto(c: &TargetConfig) -> proto::TargetConfig { + proto::TargetConfig { + max_db_size: c.max_db_size, + jwt_key: c.jwt_key.clone(), + txn_timeout_s: c.txn_timeout_s, + allow_attach: c.allow_attach, + durability_mode: c.durability_mode.clone(), + bottomless_db_id: c.bottomless_db_id.clone(), + } + } + + fn validation_to_proto(v: &ValidationRecord) -> proto::ValidationRecord { + proto::ValidationRecord { + operation_id: v.operation_id.to_string(), + command_id: v.command_id.to_string(), + result: validation_result_to_proto(v.result) as i32, + summary: v.summary.clone(), + snapshot: v.snapshot.map(|s| proto::ValidationSnapshot { + log_id: s.log_id.to_string(), + frame_no: s.frame_no, + page_count: s.page_count, + }), + recorded_at_ms: v.recorded_at_ms, + } + } + + fn validation_from_proto(v: proto::ValidationRecord) -> R { + Ok(ValidationRecord { + operation_id: parse_uuid(&v.operation_id)?, + command_id: parse_uuid(&v.command_id)?, + result: validation_result_from_proto(v.result)?, + summary: v.summary, + snapshot: v + .snapshot + .map(|s| { + Ok::<_, FenceDecodeError>(ValidationSnapshot { + log_id: parse_uuid(&s.log_id)?, + frame_no: s.frame_no, + page_count: s.page_count, + }) + }) + .transpose()?, + recorded_at_ms: v.recorded_at_ms, + }) + } + + fn adoption_to_proto(a: &Adoption) -> proto::Adoption { + proto::Adoption { + previous_operation_id: a.previous_operation_id.to_string(), + new_operation_id: a.new_operation_id.to_string(), + command_id: a.command_id.to_string(), + approvers: a.approvers.clone(), + incident_ref: a.incident_ref.clone(), + reason: a.reason.clone(), + at_ms: a.at_ms, + revision: a.revision, + } + } + + fn adoption_from_proto(a: proto::Adoption) -> R { + Ok(Adoption { + previous_operation_id: parse_uuid(&a.previous_operation_id)?, + new_operation_id: parse_uuid(&a.new_operation_id)?, + command_id: parse_uuid(&a.command_id)?, + approvers: a.approvers, + incident_ref: a.incident_ref, + reason: a.reason, + at_ms: a.at_ms, + revision: a.revision, + }) + } + + pub fn record_to_proto(r: &NamespaceFenceRecord) -> proto::FenceRecord { + proto::FenceRecord { + namespace: r.namespace.as_str().to_string(), + role: role_to_proto(r.role) as i32, + state: state_to_proto(r.state) as i32, + revision: r.revision, + operation_id: r.operation_id.to_string(), + log_id: r.identity.log_id.map(|id| id.to_string()), + target_incarnation_id: r.identity.target_incarnation_id.map(|id| id.to_string()), + drain_policy: r.drain_policy.as_ref().map(drain_policy_to_proto), + drain_started_at_ms: r.drain_started_at_ms, + frozen_boundary: r.frozen_boundary.map(|b| proto::FrozenBoundary { + log_id: b.log_id.to_string(), + frame_no: b.frame_no, + }), + validation: r.validation.as_ref().map(validation_to_proto), + legacy_blocks: Some(proto::LegacyBlocks { + block_reads: r.legacy_blocks.block_reads, + block_writes: r.legacy_blocks.block_writes, + block_reason: r.legacy_blocks.block_reason.clone(), + }), + created_at_ms: r.created_at_ms, + last_transition_at_ms: r.last_transition_at_ms, + last_command_id: r.last_command_id.to_string(), + written_by: Some(proto::ServerIdentity { + build: r.written_by.build.clone(), + instance_id: r.written_by.instance_id.to_string(), + }), + adoptions: r.adoptions.iter().map(adoption_to_proto).collect(), + } + } + + pub fn record_from_proto(m: proto::FenceRecord) -> R { + let role = role_from_proto(m.role)?; + let state = durable_state_from_proto(m.state, role)?; + if m.revision == 0 { + return Err(FenceDecodeError::Invalid("record revision is zero")); + } + let legacy = m + .legacy_blocks + .ok_or(FenceDecodeError::Invalid("missing legacy blocks"))?; + let written_by = m + .written_by + .ok_or(FenceDecodeError::Invalid("missing server identity"))?; + let record = NamespaceFenceRecord { + namespace: namespace(m.namespace)?, + role, + state, + revision: m.revision, + operation_id: parse_uuid(&m.operation_id)?, + identity: NamespaceIdentity { + log_id: parse_opt_uuid(m.log_id.as_deref())?, + target_incarnation_id: parse_opt_uuid(m.target_incarnation_id.as_deref())?, + }, + drain_policy: m.drain_policy.map(drain_policy_from_proto).transpose()?, + drain_started_at_ms: m.drain_started_at_ms, + frozen_boundary: m + .frozen_boundary + .map(|b| { + Ok::<_, FenceDecodeError>(FrozenBoundary { + log_id: parse_uuid(&b.log_id)?, + frame_no: b.frame_no, + }) + }) + .transpose()?, + validation: m.validation.map(validation_from_proto).transpose()?, + legacy_blocks: LegacyBlocks { + block_reads: legacy.block_reads, + block_writes: legacy.block_writes, + block_reason: legacy.block_reason, + }, + created_at_ms: m.created_at_ms, + last_transition_at_ms: m.last_transition_at_ms, + last_command_id: parse_uuid(&m.last_command_id)?, + written_by: ServerIdentity { + build: written_by.build, + instance_id: parse_uuid(&written_by.instance_id)?, + }, + adoptions: m + .adoptions + .into_iter() + .map(adoption_from_proto) + .collect::>()?, + }; + check_record_invariants(&record)?; + Ok(record) + } + + /// Facts every record written by `apply` satisfies. A record that breaks one was not + /// written by this server and is not trusted. + fn check_record_invariants(r: &NamespaceFenceRecord) -> R<()> { + use FenceState::*; + let frozen_required = matches!( + r.state, + SourceWriteFenced | SourceReadDraining | SourceReadFenced + ); + if frozen_required && r.frozen_boundary.is_none() { + return Err(FenceDecodeError::Invalid( + "fenced source without frozen boundary", + )); + } + if r.role == Role::Source && r.identity.log_id.is_none() { + return Err(FenceDecodeError::Invalid( + "source without namespace identity", + )); + } + if r.role == Role::Target && r.identity.target_incarnation_id.is_none() { + return Err(FenceDecodeError::Invalid("target without incarnation id")); + } + if matches!(r.state, TargetWriteFenced | TargetWritable) + && !r + .validation + .as_ref() + .is_some_and(|v| v.result == ValidationResult::Ok) + { + return Err(FenceDecodeError::Invalid( + "published target without validation", + )); + } + Ok(()) + } + + pub fn receipt_to_proto(r: &CommandReceipt) -> proto::CommandReceipt { + proto::CommandReceipt { + namespace: r.namespace.as_str().to_string(), + operation_id: r.operation_id.to_string(), + command_id: r.command_id.to_string(), + command: command_kind_to_proto(r.command) as i32, + fingerprint: r.fingerprint.as_bytes().to_vec(), + outcome: outcome_to_proto(r.outcome) as i32, + revision_before: r.revision_before, + revision_after: r.revision_after, + state_after: state_to_proto(r.state_after) as i32, + applied_at_ms: r.applied_at_ms, + instance_id: r.instance_id.to_string(), + adoption: r.adoption.as_ref().map(adoption_to_proto), + } + } + + pub fn receipt_from_proto(m: proto::CommandReceipt) -> R { + let fingerprint: [u8; 32] = m + .fingerprint + .as_slice() + .try_into() + .map_err(|_| FenceDecodeError::Invalid("fingerprint is not 32 bytes"))?; + let state_after = state_from_proto(m.state_after)?; + if !state_after.is_durable() { + return Err(FenceDecodeError::Invalid("receipt state is not durable")); + } + if m.revision_after < m.revision_before { + return Err(FenceDecodeError::Invalid("receipt revision went backwards")); + } + Ok(CommandReceipt { + namespace: namespace(m.namespace)?, + operation_id: parse_uuid(&m.operation_id)?, + command_id: parse_uuid(&m.command_id)?, + command: command_kind_from_proto(m.command)?, + fingerprint: Fingerprint(fingerprint), + outcome: outcome_from_proto(m.outcome)?, + revision_before: m.revision_before, + revision_after: m.revision_after, + state_after, + applied_at_ms: m.applied_at_ms, + instance_id: parse_uuid(&m.instance_id)?, + adoption: m.adoption.map(adoption_from_proto).transpose()?, + }) + } +} + +#[cfg(test)] +pub(super) mod tests { + use super::*; + use crate::namespace::fence::command::OnDeadline; + + pub fn sample_record() -> NamespaceFenceRecord { + NamespaceFenceRecord { + namespace: NamespaceName::from("db1"), + role: Role::Source, + state: FenceState::SourceWriteFenced, + revision: 2, + operation_id: Uuid::from_u128(1), + identity: NamespaceIdentity { + log_id: Some(Uuid::from_u128(10)), + target_incarnation_id: None, + }, + drain_policy: Some(DrainPolicy { + deadline_ms: 5000, + on_deadline: OnDeadline::ForceRollback, + }), + drain_started_at_ms: Some(100), + frozen_boundary: Some(FrozenBoundary { + log_id: Uuid::from_u128(10), + frame_no: 1234, + }), + validation: None, + legacy_blocks: LegacyBlocks { + block_reads: false, + block_writes: true, + block_reason: Some("maintenance".into()), + }, + created_at_ms: 100, + last_transition_at_ms: 200, + last_command_id: Uuid::from_u128(2), + written_by: ServerIdentity { + build: "test".into(), + instance_id: Uuid::from_u128(99), + }, + adoptions: vec![Adoption { + previous_operation_id: Uuid::from_u128(5), + new_operation_id: Uuid::from_u128(1), + command_id: Uuid::from_u128(6), + approvers: vec!["a".into(), "b".into()], + incident_ref: "inc".into(), + reason: "lost control plane".into(), + at_ms: 150, + revision: 2, + }], + } + } + + fn sample_receipt() -> CommandReceipt { + CommandReceipt { + namespace: NamespaceName::from("db1"), + operation_id: Uuid::from_u128(1), + command_id: Uuid::from_u128(2), + command: CommandKind::AcquireSourceWriteFence, + fingerprint: Fingerprint([7; 32]), + outcome: FenceOutcome::Applied, + revision_before: 0, + revision_after: 2, + state_after: FenceState::SourceWriteFenced, + applied_at_ms: 200, + instance_id: Uuid::from_u128(99), + adoption: None, + } + } + + #[test] + fn record_round_trips() { + let record = sample_record(); + let bytes = record.encode(); + let decoded = NamespaceFenceRecord::decode(FENCE_FORMAT_VERSION, 2, &bytes).unwrap(); + assert_eq!(decoded, record); + } + + #[test] + fn receipt_round_trips() { + let receipt = sample_receipt(); + let decoded = CommandReceipt::decode(FENCE_FORMAT_VERSION, &receipt.encode()).unwrap(); + assert_eq!(decoded, receipt); + } + + #[test] + fn marker_round_trips() { + let marker = FenceMarker::for_record(&sample_record()); + assert_eq!(FenceMarker::decode(&marker.encode()).unwrap(), marker); + } + + #[test] + fn unknown_format_version_is_rejected() { + let bytes = sample_record().encode(); + assert_eq!( + NamespaceFenceRecord::decode(2, 2, &bytes), + Err(FenceDecodeError::UnsupportedFormatVersion(2)) + ); + assert!(matches!( + CommandReceipt::decode(0, &sample_receipt().encode()), + Err(FenceDecodeError::UnsupportedFormatVersion(0)) + )); + let mut marker = proto::FenceMarker::decode( + FenceMarker::for_record(&sample_record()) + .encode() + .as_slice(), + ) + .unwrap(); + marker.format_version = 7; + assert_eq!( + FenceMarker::decode(&marker.encode_to_vec()), + Err(FenceDecodeError::UnsupportedFormatVersion(7)) + ); + } + + #[test] + fn garbage_is_rejected() { + assert!(matches!( + NamespaceFenceRecord::decode(FENCE_FORMAT_VERSION, 2, &[0xff, 0xff, 0xff]), + Err(FenceDecodeError::Undecodable(_)) + )); + // An empty payload decodes as an all-default message, which is not a valid record. + assert!(NamespaceFenceRecord::decode(FENCE_FORMAT_VERSION, 0, &[]).is_err()); + assert!(CommandReceipt::decode(FENCE_FORMAT_VERSION, &[]).is_err()); + assert!(FenceMarker::decode(&[]).is_err()); + } + + #[test] + fn revision_column_must_match() { + let bytes = sample_record().encode(); + assert!(matches!( + NamespaceFenceRecord::decode(FENCE_FORMAT_VERSION, 3, &bytes), + Err(FenceDecodeError::Invalid(_)) + )); + } + + fn mutate( + f: impl FnOnce(&mut proto::FenceRecord), + ) -> Result { + let mut m = codec::record_to_proto(&sample_record()); + f(&mut m); + let revision = m.revision; + NamespaceFenceRecord::decode(FENCE_FORMAT_VERSION, revision, &m.encode_to_vec()) + } + + #[test] + fn invalid_records_are_rejected() { + assert!(mutate(|m| m.state = 999).is_err(), "unknown state"); + assert!( + mutate(|m| m.state = proto::FenceState::Unfenced as i32).is_err(), + "non-durable" + ); + assert!( + mutate(|m| m.state = proto::FenceState::UnknownUnavailable as i32).is_err(), + "derived state stored" + ); + assert!( + mutate(|m| m.state = proto::FenceState::TargetWritable as i32).is_err(), + "state/role mismatch" + ); + assert!(mutate(|m| m.role = 0).is_err(), "unspecified role"); + assert!(mutate(|m| m.operation_id = "not-a-uuid".into()).is_err()); + assert!( + mutate(|m| m.operation_id = m.operation_id.replace('-', "")).is_err(), + "non-canonical id" + ); + assert!(mutate(|m| m.namespace = String::new()).is_err()); + assert!(mutate(|m| m.legacy_blocks = None).is_err()); + assert!(mutate(|m| m.written_by = None).is_err()); + assert!( + mutate(|m| m.frozen_boundary = None).is_err(), + "fenced without boundary" + ); + assert!( + mutate(|m| m.log_id = None).is_err(), + "source without identity" + ); + assert!( + mutate(|m| { + m.revision = 0; + }) + .is_err(), + "revision zero" + ); + assert!( + mutate(|m| m.drain_policy.as_mut().unwrap().on_deadline = 0).is_err(), + "unspecified drain policy" + ); + } + + #[test] + fn invalid_receipts_are_rejected() { + let enc = |f: &dyn Fn(&mut proto::CommandReceipt)| { + let mut m = codec::receipt_to_proto(&sample_receipt()); + f(&mut m); + CommandReceipt::decode(FENCE_FORMAT_VERSION, &m.encode_to_vec()) + }; + assert!(enc(&|m| m.fingerprint = vec![1; 31]).is_err()); + assert!(enc(&|m| m.outcome = 0).is_err()); + assert!(enc(&|m| m.command = 0).is_err()); + assert!(enc(&|m| m.state_after = proto::FenceState::Unfenced as i32).is_err()); + assert!( + enc(&|m| m.revision_before = 5).is_err(), + "revision went backwards" + ); + } + + #[test] + fn legacy_mirror_follows_state() { + let mut record = sample_record(); + let m = record.legacy_mirror(); + assert!(!m.block_reads); + assert!(m.block_writes); + assert!(m.block_reason.unwrap().contains("SOURCE_WRITE_FENCED")); + + record.state = FenceState::SourceReadFenced; + let m = record.legacy_mirror(); + assert!(m.block_reads && m.block_writes); + + record.state = FenceState::Released; + assert_eq!(record.legacy_mirror(), record.legacy_blocks); + + record.role = Role::Target; + record.state = FenceState::TargetQuarantined; + let m = record.legacy_mirror(); + assert!(m.block_reads && m.block_writes); + record.state = FenceState::TargetWriteFenced; + let m = record.legacy_mirror(); + assert!(!m.block_reads && m.block_writes); + } +} diff --git a/libsql-server/src/namespace/fence/state.rs b/libsql-server/src/namespace/fence/state.rs new file mode 100644 index 0000000000..53a681775c --- /dev/null +++ b/libsql-server/src/namespace/fence/state.rs @@ -0,0 +1,403 @@ +//! Roles, states, operation classes and the permission matrix of `docs/NAMESPACE_FENCE.md` +//! sections 3 and 7.3. + +use std::fmt; +use std::str::FromStr; + +use super::outcome::FenceOutcome; + +/// Which side of a move a fence record belongs to. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] +pub enum Role { + Source, + Target, +} + +impl Role { + pub const fn as_str(self) -> &'static str { + match self { + Role::Source => "SOURCE", + Role::Target => "TARGET", + } + } +} + +impl fmt::Display for Role { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + f.write_str(self.as_str()) + } +} + +/// The state of a namespace as the fence sees it. +/// +/// `Unfenced` and `Absent` describe a namespace with no record (an ordinary namespace, or no +/// namespace at all). `UnknownUnavailable` is derived when the server cannot establish the +/// control state. None of those three is ever stored in a record. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] +pub enum FenceState { + Unfenced, + Absent, + SourceDraining, + SourceWriteFenced, + SourceReadDraining, + SourceReadFenced, + Released, + TargetQuarantined, + TargetImportDraining, + TargetValidating, + TargetWriteFenced, + TargetWritable, + TargetAborted, + UnknownUnavailable, +} + +impl FenceState { + pub const ALL: [FenceState; 14] = [ + FenceState::Unfenced, + FenceState::Absent, + FenceState::SourceDraining, + FenceState::SourceWriteFenced, + FenceState::SourceReadDraining, + FenceState::SourceReadFenced, + FenceState::Released, + FenceState::TargetQuarantined, + FenceState::TargetImportDraining, + FenceState::TargetValidating, + FenceState::TargetWriteFenced, + FenceState::TargetWritable, + FenceState::TargetAborted, + FenceState::UnknownUnavailable, + ]; + + pub const fn as_str(self) -> &'static str { + match self { + FenceState::Unfenced => "UNFENCED", + FenceState::Absent => "ABSENT", + FenceState::SourceDraining => "SOURCE_DRAINING", + FenceState::SourceWriteFenced => "SOURCE_WRITE_FENCED", + FenceState::SourceReadDraining => "SOURCE_READ_DRAINING", + FenceState::SourceReadFenced => "SOURCE_READ_FENCED", + FenceState::Released => "RELEASED", + FenceState::TargetQuarantined => "TARGET_QUARANTINED", + FenceState::TargetImportDraining => "TARGET_IMPORT_DRAINING", + FenceState::TargetValidating => "TARGET_VALIDATING", + FenceState::TargetWriteFenced => "TARGET_WRITE_FENCED", + FenceState::TargetWritable => "TARGET_WRITABLE", + FenceState::TargetAborted => "TARGET_ABORTED", + FenceState::UnknownUnavailable => "UNKNOWN_UNAVAILABLE", + } + } + + /// The role a record in this state has, if the state belongs to one. + pub const fn role(self) -> Option { + match self { + FenceState::SourceDraining + | FenceState::SourceWriteFenced + | FenceState::SourceReadDraining + | FenceState::SourceReadFenced + | FenceState::Released => Some(Role::Source), + FenceState::TargetQuarantined + | FenceState::TargetImportDraining + | FenceState::TargetValidating + | FenceState::TargetWriteFenced + | FenceState::TargetWritable + | FenceState::TargetAborted => Some(Role::Target), + FenceState::Unfenced | FenceState::Absent | FenceState::UnknownUnavailable => None, + } + } + + /// Whether this state can be stored in a fence record. + pub const fn is_durable(self) -> bool { + self.role().is_some() + } + + /// Whether the operation that owns a record in this state has finished with it. + pub const fn is_operation_finished(self) -> bool { + matches!( + self, + FenceState::Released | FenceState::TargetWritable | FenceState::TargetAborted + ) + } + + /// Whether a record in this state counts as an active fence (section 4.4, + /// `active_fences`): anything except an ordinary namespace, a released source or a + /// published target. + pub const fn is_active(self) -> bool { + !matches!( + self, + FenceState::Unfenced + | FenceState::Absent + | FenceState::Released + | FenceState::TargetWritable + ) + } + + /// A state in which the server is waiting for work admitted earlier to end. + pub const fn is_draining(self) -> bool { + matches!( + self, + FenceState::SourceDraining + | FenceState::SourceReadDraining + | FenceState::TargetImportDraining + ) + } + + /// The permission-matrix decision for work of `class` (section 3.3). + /// + /// For the capability classes this decides only whether the state admits capability work + /// at all; whether a particular capability is valid is the controller's decision. + pub fn permits(self, class: OperationClass) -> Result<(), FenceOutcome> { + use FenceState::*; + use OperationClass::*; + + match class { + Maintenance | Observability => return Ok(()), + _ => (), + } + + if self == UnknownUnavailable { + return Err(FenceOutcome::FenceStateUnavailable); + } + + match class { + NormalRead | Stream => match self { + Unfenced | Absent | Released | SourceDraining | SourceWriteFenced + | TargetWriteFenced | TargetWritable => Ok(()), + SourceReadDraining | SourceReadFenced => Err(FenceOutcome::MigrationReadFenced), + TargetQuarantined | TargetImportDraining | TargetValidating | TargetAborted => { + Err(FenceOutcome::MigrationTargetQuarantined) + } + UnknownUnavailable => unreachable!(), + }, + NormalWrite | Vacuum | Lifecycle => match self { + Unfenced | Absent | Released | TargetWritable => Ok(()), + SourceDraining | SourceWriteFenced | SourceReadDraining | SourceReadFenced + | TargetWriteFenced => Err(FenceOutcome::MigrationWriteFenced), + TargetQuarantined | TargetImportDraining | TargetValidating | TargetAborted => { + Err(FenceOutcome::MigrationTargetQuarantined) + } + UnknownUnavailable => unreachable!(), + }, + CapabilityImport => match self { + TargetQuarantined => Ok(()), + // Existing import writers may finish while draining, but no new one starts: + // that distinction is the controller's, which tracks issued capabilities. + TargetImportDraining => Ok(()), + _ => Err(FenceOutcome::OperationCapabilityRequired), + }, + CapabilityValidate => match self { + TargetValidating | TargetWriteFenced => Ok(()), + _ => Err(FenceOutcome::OperationCapabilityRequired), + }, + Maintenance | Observability => unreachable!(), + } + } + + /// Whether normal write admission is open in this state. + pub fn write_admission(self) -> Admission { + Admission::from(self.permits(OperationClass::NormalWrite)) + } + + /// Whether normal read admission is open in this state. + pub fn read_admission(self) -> Admission { + Admission::from(self.permits(OperationClass::NormalRead)) + } +} + +impl fmt::Display for FenceState { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + f.write_str(self.as_str()) + } +} + +#[derive(Debug, Clone, PartialEq, Eq, thiserror::Error)] +#[error("unknown fence state `{0}`")] +pub struct UnknownFenceState(pub String); + +impl FromStr for FenceState { + type Err = UnknownFenceState; + + fn from_str(s: &str) -> Result { + FenceState::ALL + .iter() + .copied() + .find(|state| state.as_str() == s) + .ok_or_else(|| UnknownFenceState(s.to_string())) + } +} + +/// Whether an admission path is open, and if it is closed, the code a denial carries. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum Admission { + Open, + Closed(FenceOutcome), +} + +impl Admission { + pub fn is_open(self) -> bool { + matches!(self, Admission::Open) + } + + pub fn as_str(self) -> &'static str { + match self { + Admission::Open => "open", + Admission::Closed(_) => "closed", + } + } +} + +impl From> for Admission { + fn from(value: Result<(), FenceOutcome>) -> Self { + match value { + Ok(()) => Admission::Open, + Err(code) => Admission::Closed(code), + } + } +} + +/// The class of a piece of work, which the permission matrix is keyed by (section 7.3). +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] +pub enum OperationClass { + /// Any logical write that is not operation-owned: SQL over every protocol, the admin + /// shell, schema migration, dump load outside an import capability. + NormalWrite, + /// Work that cannot change logical contents: `TRUNCATE` checkpoints, the storage monitor, + /// bottomless WAL upload, the replication logger's own connection. + Maintenance, + /// `VACUUM`. It takes a write transaction and produces replicated frames, so it is not + /// maintenance and is skipped wherever normal writes are denied. + Vacuum, + /// Generic lifecycle and configuration: config mutation, delete, reset, fork, create over + /// an existing record, restore, dump load, shared-schema linking. + Lifecycle, + /// Writes of an import session holding a `MigrationCapability`. + CapabilityImport, + /// Read-only validation through a `MigrationCapability`. + CapabilityValidate, + /// SQL programs, Hrana cursors, `/beta/listen`, ATTACH of the namespace. + NormalRead, + /// `/dump` and replication streams (`hello`, `log_entries`, `batch_log_entries`, + /// `snapshot`). + Stream, + /// Stats, jobs and metrics. Never a read lease. + Observability, +} + +impl OperationClass { + pub const ALL: [OperationClass; 9] = [ + OperationClass::NormalWrite, + OperationClass::Maintenance, + OperationClass::Vacuum, + OperationClass::Lifecycle, + OperationClass::CapabilityImport, + OperationClass::CapabilityValidate, + OperationClass::NormalRead, + OperationClass::Stream, + OperationClass::Observability, + ]; +} + +#[cfg(test)] +mod tests { + use super::*; + + use FenceOutcome as O; + use FenceState as S; + use OperationClass as C; + + #[test] + fn state_names_round_trip() { + for state in FenceState::ALL { + assert_eq!(state.as_str().parse::().unwrap(), state); + } + assert!("source_draining".parse::().is_err()); + assert!("".parse::().is_err()); + } + + #[test] + fn durable_states_have_a_role() { + for state in FenceState::ALL { + let expected = !matches!(state, S::Unfenced | S::Absent | S::UnknownUnavailable); + assert_eq!(state.is_durable(), expected, "{state}"); + } + } + + /// The whole permission matrix of section 3.3, row by row. + #[test] + fn permission_matrix() { + let allow = Ok(()); + let wf = Err(O::MigrationWriteFenced); + let rf = Err(O::MigrationReadFenced); + let tq = Err(O::MigrationTargetQuarantined); + let un = Err(O::FenceStateUnavailable); + let cap = Err(O::OperationCapabilityRequired); + + // state: (read, stream, write, lifecycle, import, validate) + let rows = [ + (S::Unfenced, [allow, allow, allow, allow, cap, cap]), + (S::Released, [allow, allow, allow, allow, cap, cap]), + (S::SourceDraining, [allow, allow, wf, wf, cap, cap]), + (S::SourceWriteFenced, [allow, allow, wf, wf, cap, cap]), + (S::SourceReadDraining, [rf, rf, wf, wf, cap, cap]), + (S::SourceReadFenced, [rf, rf, wf, wf, cap, cap]), + (S::TargetQuarantined, [tq, tq, tq, tq, allow, cap]), + (S::TargetImportDraining, [tq, tq, tq, tq, allow, cap]), + (S::TargetValidating, [tq, tq, tq, tq, cap, allow]), + (S::TargetWriteFenced, [allow, allow, wf, wf, cap, allow]), + (S::TargetWritable, [allow, allow, allow, allow, cap, cap]), + (S::TargetAborted, [tq, tq, tq, tq, cap, cap]), + (S::UnknownUnavailable, [un, un, un, un, un, un]), + ]; + + for (state, expected) in rows { + let classes = [ + C::NormalRead, + C::Stream, + C::NormalWrite, + C::Lifecycle, + C::CapabilityImport, + C::CapabilityValidate, + ]; + for (class, expected) in classes.into_iter().zip(expected) { + assert_eq!(state.permits(class), expected, "{state} {class:?}"); + } + // Vacuum follows the normal write column. + assert_eq!( + state.permits(C::Vacuum), + state.permits(C::NormalWrite), + "{state}" + ); + // Maintenance and observability continue in every state. + assert_eq!(state.permits(C::Maintenance), Ok(()), "{state}"); + assert_eq!(state.permits(C::Observability), Ok(()), "{state}"); + } + } + + #[test] + fn denials_are_data_plane_codes() { + for state in FenceState::ALL { + for class in OperationClass::ALL { + if let Err(code) = state.permits(class) { + assert!(code.is_error(), "{state} {class:?} {code}"); + } + } + } + } + + #[test] + fn active_and_finished() { + assert!(!S::Unfenced.is_active()); + assert!(!S::Released.is_active()); + assert!(!S::TargetWritable.is_active()); + assert!(S::TargetAborted.is_active()); + assert!(S::UnknownUnavailable.is_active()); + assert!(S::SourceDraining.is_active()); + + for state in FenceState::ALL { + assert_eq!( + state.is_operation_finished(), + matches!(state, S::Released | S::TargetWritable | S::TargetAborted) + ); + } + } +} diff --git a/libsql-server/src/namespace/fence/transition.rs b/libsql-server/src/namespace/fence/transition.rs new file mode 100644 index 0000000000..f1de43f747 --- /dev/null +++ b/libsql-server/src/namespace/fence/transition.rs @@ -0,0 +1,1794 @@ +//! The pure fence transition function (`docs/NAMESPACE_FENCE.md` sections 3.2 and 5.3). +//! +//! [`apply`] decides what a command does to a namespace's fence, given everything the store +//! read inside its transaction. It performs no I/O, takes no locks and reads no clock: the +//! store supplies the current record, the stored receipt for the request's +//! `(operation_id, command_id)` if there is one, and the facts in [`ApplyEnv`]. The store +//! persists whatever [`Decision::Apply`] returns, in one transaction, before anything is +//! published or answered. +//! +//! Checks run in this order, and the order is part of the contract: +//! +//! 1. **Replay.** A stored receipt with the same fingerprint is answered from the receipt +//! (`Replay`, or `Resume` for a drain still in progress), whatever has happened to the +//! record since. A stored receipt with a different fingerprint is `FENCE_COMMAND_CONFLICT`. +//! 2. **Unavailable state.** A record the server cannot establish refuses everything with +//! `FENCE_STATE_UNAVAILABLE`, except the two commands that can reconcile it: a replay of the +//! `CreateTargetQuarantined` that left the marker, and an adoption after a metastore +//! rollback. +//! 3. **Owner.** An unfinished record owned by another operation is +//! `FENCE_OWNED_BY_ANOTHER_OPERATION`. +//! 4. **Already applied.** A command from the owner whose goal state the record is already in +//! is `ALREADY_APPLIED`: a receipt is stored, the record and its revision do not change. +//! This is checked before the revision, because the caller's stated expectation is +//! typically the state before a response it never received. +//! 5. **Transition.** Role, then legality of the transition from the current state +//! (`INVALID_FENCE_TRANSITION`). +//! 6. **Expectation.** `expected_state` and `expected_revision` (`FENCE_REVISION_MISMATCH`). +//! 7. **Preconditions** of the command (`FENCE_PRECONDITION_FAILED`). +//! +//! A drain that starts in `apply` (`DRAINING`) is finished by [`complete_drain`], once the +//! controller has proven that the work admitted earlier has ended. + +use uuid::Uuid; + +use super::command::{ + AdoptArgs, CommandKind, FenceCommand, FenceRequest, ValidationResult, + MAX_VALIDATION_SUMMARY_BYTES, +}; +use super::outcome::{FenceDetail, FenceError, FenceOutcome}; +use super::record::{ + Adoption, CommandReceipt, FrozenBoundary, LegacyBlocks, NamespaceFenceRecord, + NamespaceIdentity, ServerIdentity, ValidationRecord, ValidationSnapshot, +}; +use super::state::{FenceState, Role}; + +/// What the store knows about a namespace's fence. +#[derive(Debug, Clone, Copy)] +pub enum CurrentFence<'a> { + /// No record. `namespace_exists` distinguishes `UNFENCED` from `ABSENT`. + None { + namespace_exists: bool, + }, + Record(&'a NamespaceFenceRecord), + /// The control state cannot be established. `marker` is the record the namespace's marker + /// file holds, when it has a readable one. + Unavailable { + detail: FenceDetail, + marker: Option<&'a NamespaceFenceRecord>, + }, +} + +impl CurrentFence<'_> { + pub fn state(&self) -> FenceState { + match self { + CurrentFence::None { + namespace_exists: true, + } => FenceState::Unfenced, + CurrentFence::None { + namespace_exists: false, + } => FenceState::Absent, + CurrentFence::Record(r) => r.state, + CurrentFence::Unavailable { .. } => FenceState::UnknownUnavailable, + } + } + + pub fn revision(&self) -> u64 { + match self { + CurrentFence::None { .. } => 0, + CurrentFence::Record(r) => r.revision, + CurrentFence::Unavailable { marker, .. } => marker.map_or(0, |m| m.revision), + } + } +} + +/// Facts `apply` needs that are not in the record. The store fills them from inside the same +/// transaction and the live namespace. +#[derive(Debug, Clone)] +pub struct ApplyEnv { + /// Wall-clock time, in milliseconds since the Unix epoch. Informational only: nothing in + /// the fence expires. + pub now_ms: i64, + pub server: ServerIdentity, + /// The namespace's current replication log id, if it exists and has one. + pub namespace_log_id: Option, + /// Whether the namespace is a shared schema or linked to one. + pub shared_schema: bool, + /// The namespace config's current `block_*` values, saved when a source is acquired and + /// restored when it is released. + pub legacy_blocks: LegacyBlocks, + /// A fresh id, used as `target_incarnation_id` by `CreateTargetQuarantined`. + pub new_incarnation_id: Uuid, + /// Whether the request carried the configured adoption key. + pub adoption_authorised: bool, + /// What the server observed of the target, for `RecordTargetValidation`. + pub validation_snapshot: Option, +} + +/// What a command does. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum Decision { + /// The command was applied before and its answer is final: return this receipt, with + /// `replayed: true`. Nothing is written. + Replay(CommandReceipt), + /// The same command started a drain that has not been completed: resume it. Nothing is + /// written. + Resume(CommandReceipt), + /// Persist `record` (when `Some`; `None` leaves the record as it is) and `receipt` in one + /// transaction, then answer `receipt.outcome`. A `DRAINING` receipt means the controller + /// must now run the drain and finish it with [`complete_drain`]. + Apply { + record: Option, + receipt: CommandReceipt, + }, +} + +/// Evidence, gathered by the controller, that a drain is complete. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum DrainCompletion { + /// No normal writer holds or can take the write slot; `boundary` was read with the slot + /// free, after the last commit published its frame. + SourceWrites { boundary: FrozenBoundary }, + /// Every SQL, dump and replication read lease has been released. + SourceReads, + /// Every import writer has finished and no capability writer holds the slot. + TargetImport, +} + +fn err(outcome: FenceOutcome, message: impl Into) -> FenceError { + FenceError::new(outcome, message) +} + +fn invalid(message: impl Into) -> FenceError { + err(FenceOutcome::InvalidFenceTransition, message) +} + +fn precondition(detail: FenceDetail, message: impl Into) -> FenceError { + err(FenceOutcome::FencePreconditionFailed, message).with_detail(detail) +} + +/// The role a command acts on. `None` for adoption, which acts on either. +fn command_role(kind: CommandKind) -> Option { + match kind { + CommandKind::AcquireSourceWriteFence + | CommandKind::SetSourceReadFence + | CommandKind::ClearSourceReadFence + | CommandKind::ReleaseSourceWriteFence => Some(Role::Source), + CommandKind::CreateTargetQuarantined + | CommandKind::SealTargetImport + | CommandKind::RecordTargetValidation + | CommandKind::PublishTargetReadableWriteFenced + | CommandKind::EnableTargetWrites + | CommandKind::AbortQuarantinedTarget => Some(Role::Target), + CommandKind::AdoptFence => None, + } +} + +/// The state a legal command moves a record from `from` to, and the receipt outcome. This is +/// the transition graph of section 3.2, minus the drain completions and adoption. +fn transition(kind: CommandKind, from: FenceState) -> Option<(FenceState, FenceOutcome)> { + use CommandKind as K; + use FenceOutcome::{Applied, Draining}; + use FenceState as S; + + Some(match (kind, from) { + (K::AcquireSourceWriteFence, S::Unfenced | S::Released | S::TargetWritable) => { + (S::SourceDraining, Draining) + } + (K::SetSourceReadFence, S::SourceWriteFenced) => (S::SourceReadDraining, Draining), + (K::ClearSourceReadFence, S::SourceReadDraining | S::SourceReadFenced) => { + (S::SourceWriteFenced, Applied) + } + (K::ReleaseSourceWriteFence, S::SourceDraining | S::SourceWriteFenced) => { + (S::Released, Applied) + } + (K::CreateTargetQuarantined, S::Absent) => (S::TargetQuarantined, Applied), + (K::SealTargetImport, S::TargetQuarantined) => (S::TargetImportDraining, Draining), + (K::RecordTargetValidation, S::TargetValidating) => (S::TargetValidating, Applied), + (K::PublishTargetReadableWriteFenced, S::TargetValidating) => { + (S::TargetWriteFenced, Applied) + } + (K::EnableTargetWrites, S::TargetWriteFenced) => (S::TargetWritable, Applied), + ( + K::AbortQuarantinedTarget, + S::TargetQuarantined + | S::TargetImportDraining + | S::TargetValidating + | S::TargetWriteFenced, + ) => (S::TargetAborted, Applied), + _ => return None, + }) +} + +/// The draining state a command's drain runs in, for commands that drain. +fn drain_state(kind: CommandKind) -> Option { + match kind { + CommandKind::AcquireSourceWriteFence => Some(FenceState::SourceDraining), + CommandKind::SetSourceReadFence => Some(FenceState::SourceReadDraining), + CommandKind::SealTargetImport => Some(FenceState::TargetImportDraining), + _ => None, + } +} + +fn receipt( + request: &FenceRequest, + env: &ApplyEnv, + outcome: FenceOutcome, + revision_before: u64, + revision_after: u64, + state_after: FenceState, +) -> CommandReceipt { + CommandReceipt { + namespace: request.namespace.clone(), + operation_id: request.operation_id, + command_id: request.command_id, + command: request.command.kind(), + fingerprint: request.fingerprint(), + outcome, + revision_before, + revision_after, + state_after, + applied_at_ms: env.now_ms, + instance_id: env.server.instance_id, + adoption: None, + } +} + +/// Decide what `request` does to the namespace's fence. See the module documentation for the +/// order of the checks. +/// +/// `existing` is the stored receipt for `(request.namespace, request.operation_id, +/// request.command_id)`, if any. +pub fn apply( + current: CurrentFence<'_>, + existing: Option<&CommandReceipt>, + request: &FenceRequest, + env: &ApplyEnv, +) -> Result { + let kind = request.command.kind(); + + // 1. Replay, before anything about the record is looked at. + if let Some(existing) = existing { + debug_assert_eq!(existing.operation_id, request.operation_id); + debug_assert_eq!(existing.command_id, request.command_id); + if existing.fingerprint != request.fingerprint() { + return Err(err( + FenceOutcome::FenceCommandConflict, + format!( + "command {} was already used for a different request", + request.command_id + ), + )); + } + return Ok(if existing.is_final() { + Decision::Replay(existing.clone()) + } else { + Decision::Resume(existing.clone()) + }); + } + + // 2. A state the server cannot establish. + let record = match current { + CurrentFence::None { .. } => None, + CurrentFence::Record(record) => Some(record), + CurrentFence::Unavailable { detail, marker } => { + return apply_unavailable(detail, marker, request, env); + } + }; + + if let FenceCommand::AdoptFence(args) = &request.command { + return apply_adopt(current, record, args, request, env); + } + + if let Some(record) = record { + // 3. Owner. + let finished = record.state.is_operation_finished(); + if !finished && record.operation_id != request.operation_id { + return Err(err( + FenceOutcome::FenceOwnedByAnotherOperation, + format!( + "namespace fence is owned by operation {}", + record.operation_id + ), + )); + } + + let own = record.operation_id == request.operation_id; + + // 4. Already applied. + if own && kind.goal_state() == Some(record.state) { + let receipt = receipt( + request, + env, + FenceOutcome::AlreadyApplied, + record.revision, + record.revision, + record.state, + ); + return Ok(Decision::Apply { + record: None, + receipt, + }); + } + + // The owner joining its own drain under a new command id (for example after + // adoption, or after losing the original command id): nothing changes but the receipt, + // and the controller resumes the drain. + if own && drain_state(kind) == Some(record.state) { + check_expectation(current, request)?; + let receipt = receipt( + request, + env, + FenceOutcome::Draining, + record.revision, + record.revision, + record.state, + ); + return Ok(Decision::Apply { + record: None, + receipt, + }); + } + + if own && finished { + return Err(invalid(format!( + "operation {} has finished with this namespace ({})", + record.operation_id, record.state + )) + .with_detail(FenceDetail::OperationFinished)); + } + } + + // `CreateTargetQuarantined` needs a name nobody uses, whatever the record says. + if kind == CommandKind::CreateTargetQuarantined && current.state() != FenceState::Absent { + return Err(precondition( + FenceDetail::NamespaceExists, + "the namespace already exists", + )); + } + + // 5. Role and transition. + let from = current.state(); + let role = command_role(kind).expect("adoption is handled above"); + let current_role = match from { + // A published target, or a released source, may be acquired as a source by a new + // operation. + FenceState::Released | FenceState::TargetWritable | FenceState::Unfenced => { + Some(Role::Source) + } + FenceState::Absent => Some(Role::Target), + other => other.role(), + }; + if current_role != Some(role) { + return Err( + invalid(format!("{kind} does not apply to a namespace in {from}")) + .with_detail(FenceDetail::RoleMismatch), + ); + } + let Some((to, outcome)) = transition(kind, from) else { + return Err(invalid(format!("{kind} is not a transition from {from}"))); + }; + + // 6. Expectation. + check_expectation(current, request)?; + + // 7. Preconditions, and the next record. + let revision_before = current.revision(); + let revision_after = revision_before + 1; + let mut next = match record { + Some(record) if !record.state.is_operation_finished() => record.clone(), + _ => fresh_record(request, env, role), + }; + + match &request.command { + FenceCommand::AcquireSourceWriteFence { + expected_log_id, + drain_policy, + } => { + if env.shared_schema { + return Err(precondition( + FenceDetail::SharedSchemaUnsupported, + "shared-schema namespaces cannot be fenced", + )); + } + if env.namespace_log_id != Some(*expected_log_id) { + return Err(precondition( + FenceDetail::NamespaceIdentityMismatch, + "the namespace's replication log id is not the one the caller observed", + )); + } + next.identity = NamespaceIdentity { + log_id: Some(*expected_log_id), + target_incarnation_id: None, + }; + next.legacy_blocks = env.legacy_blocks.clone(); + next.drain_policy = *drain_policy; + next.drain_started_at_ms = Some(env.now_ms); + } + FenceCommand::SetSourceReadFence { drain_policy } + | FenceCommand::SealTargetImport { drain_policy } => { + next.drain_policy = *drain_policy; + next.drain_started_at_ms = Some(env.now_ms); + } + FenceCommand::ClearSourceReadFence + | FenceCommand::ReleaseSourceWriteFence + | FenceCommand::EnableTargetWrites + | FenceCommand::AbortQuarantinedTarget => { + next.drain_policy = None; + next.drain_started_at_ms = None; + } + FenceCommand::CreateTargetQuarantined { .. } => { + next.identity = NamespaceIdentity { + log_id: None, + target_incarnation_id: Some(env.new_incarnation_id), + }; + next.legacy_blocks = LegacyBlocks::default(); + } + FenceCommand::RecordTargetValidation { result, summary } => { + if summary.len() > MAX_VALIDATION_SUMMARY_BYTES { + return Err(precondition( + FenceDetail::InvalidArgument, + format!( + "validation summary is longer than {MAX_VALIDATION_SUMMARY_BYTES} bytes" + ), + )); + } + next.validation = Some(ValidationRecord { + operation_id: request.operation_id, + command_id: request.command_id, + result: *result, + summary: summary.clone(), + snapshot: env.validation_snapshot, + recorded_at_ms: env.now_ms, + }); + } + FenceCommand::PublishTargetReadableWriteFenced => { + let validated = next.validation.as_ref().is_some_and(|v| { + v.result == ValidationResult::Ok && v.operation_id == next.operation_id + }); + if !validated { + return Err(precondition( + FenceDetail::ValidationReceiptRequired, + "publication requires a successful validation receipt from the owning operation", + )); + } + } + FenceCommand::AdoptFence(_) => unreachable!("handled above"), + } + + next.state = to; + next.revision = revision_after; + next.last_transition_at_ms = env.now_ms; + next.last_command_id = request.command_id; + next.written_by = env.server.clone(); + + let receipt = receipt(request, env, outcome, revision_before, revision_after, to); + Ok(Decision::Apply { + record: Some(next), + receipt, + }) +} + +/// A record for an operation that is starting on this namespace. +fn fresh_record(request: &FenceRequest, env: &ApplyEnv, role: Role) -> NamespaceFenceRecord { + NamespaceFenceRecord { + namespace: request.namespace.clone(), + role, + state: FenceState::Unfenced, + revision: 0, + operation_id: request.operation_id, + identity: NamespaceIdentity::default(), + drain_policy: None, + drain_started_at_ms: None, + frozen_boundary: None, + validation: None, + legacy_blocks: LegacyBlocks::default(), + created_at_ms: env.now_ms, + last_transition_at_ms: env.now_ms, + last_command_id: request.command_id, + written_by: env.server.clone(), + adoptions: Vec::new(), + } +} + +fn check_expectation(current: CurrentFence<'_>, request: &FenceRequest) -> Result<(), FenceError> { + let (state, revision) = (current.state(), current.revision()); + if request.expected_state != state || request.expected_revision != revision { + return Err(err( + FenceOutcome::FenceRevisionMismatch, + format!( + "expected {} at revision {}, found {} at revision {}", + request.expected_state, request.expected_revision, state, revision + ), + )); + } + Ok(()) +} + +fn apply_unavailable( + detail: FenceDetail, + marker: Option<&NamespaceFenceRecord>, + request: &FenceRequest, + env: &ApplyEnv, +) -> Result { + let unavailable = || { + err( + FenceOutcome::FenceStateUnavailable, + "the namespace's fence state cannot be established", + ) + .with_detail(detail) + }; + + match (&request.command, detail, marker) { + // A crash between writing the marker and committing the target's rows: only the same + // command completes the creation. + ( + FenceCommand::CreateTargetQuarantined { .. }, + FenceDetail::IncompleteTargetCreation, + Some(m), + ) if m.state == FenceState::TargetQuarantined + && m.revision == 1 + && m.operation_id == request.operation_id + && m.last_command_id == request.command_id => + { + let decision = apply( + CurrentFence::None { + namespace_exists: false, + }, + None, + request, + &ApplyEnv { + new_incarnation_id: m + .identity + .target_incarnation_id + .unwrap_or(env.new_incarnation_id), + ..env.clone() + }, + )?; + Ok(decision) + } + // The metastore went backwards: adoption re-establishes the record the marker last + // recorded (it is written only after a commit), under the adopting operation. + (FenceCommand::AdoptFence(args), FenceDetail::MetastoreBehindMarker, Some(m)) => { + apply_adopt(CurrentFence::Record(m), Some(m), args, request, env) + } + _ => Err(unavailable()), + } +} + +fn apply_adopt( + current: CurrentFence<'_>, + record: Option<&NamespaceFenceRecord>, + args: &AdoptArgs, + request: &FenceRequest, + env: &ApplyEnv, +) -> Result { + let Some(record) = record else { + return Err(invalid("there is no fence to adopt")); + }; + if record.state.is_operation_finished() { + return Err( + invalid(format!("a fence in {} cannot be adopted", record.state)) + .with_detail(FenceDetail::OperationFinished), + ); + } + if args.current_operation_id != record.operation_id { + return Err(err( + FenceOutcome::FenceOwnedByAnotherOperation, + format!( + "namespace fence is owned by operation {}", + record.operation_id + ), + )); + } + if request.operation_id == record.operation_id { + return Err(invalid("an operation cannot adopt its own fence")); + } + check_expectation(current, request)?; + if !env.adoption_authorised { + return Err(precondition( + FenceDetail::AdoptionNotAuthorised, + "adoption requires the configured adoption key", + )); + } + let approvers: Vec<&str> = args.approvers.iter().map(|a| a.trim()).collect(); + let two_distinct = approvers.len() == 2 + && approvers.iter().all(|a| !a.is_empty()) + && approvers[0] != approvers[1]; + if !two_distinct || args.incident_ref.trim().is_empty() || args.reason.trim().is_empty() { + return Err(precondition( + FenceDetail::AdoptionNotAuthorised, + "adoption requires two distinct approvers, an incident reference and a reason", + )); + } + + let revision_after = record.revision + 1; + let adoption = Adoption { + previous_operation_id: record.operation_id, + new_operation_id: request.operation_id, + command_id: request.command_id, + approvers: args.approvers.clone(), + incident_ref: args.incident_ref.clone(), + reason: args.reason.clone(), + at_ms: env.now_ms, + revision: revision_after, + }; + + // Ownership changes and nothing else: the state, and so every gate, stays as it was. + let mut next = record.clone(); + next.operation_id = request.operation_id; + next.revision = revision_after; + next.last_transition_at_ms = env.now_ms; + next.last_command_id = request.command_id; + next.written_by = env.server.clone(); + next.adoptions.push(adoption.clone()); + + let mut receipt = receipt( + request, + env, + FenceOutcome::Applied, + record.revision, + revision_after, + record.state, + ); + receipt.adoption = Some(adoption); + Ok(Decision::Apply { + record: Some(next), + receipt, + }) +} + +/// Finish a drain that `apply` started. `receipt` is the owning operation's `DRAINING` receipt +/// for it; the returned receipt replaces it (same key) with the final `APPLIED` answer. +pub fn complete_drain( + record: &NamespaceFenceRecord, + receipt: &CommandReceipt, + completion: DrainCompletion, + env: &ApplyEnv, +) -> Result<(NamespaceFenceRecord, CommandReceipt), FenceError> { + if receipt.outcome != FenceOutcome::Draining || receipt.operation_id != record.operation_id { + return Err(invalid( + "there is no drain of the owning operation to complete", + )); + } + + let (from, to) = match completion { + DrainCompletion::SourceWrites { .. } => { + (FenceState::SourceDraining, FenceState::SourceWriteFenced) + } + DrainCompletion::SourceReads => { + (FenceState::SourceReadDraining, FenceState::SourceReadFenced) + } + DrainCompletion::TargetImport => ( + FenceState::TargetImportDraining, + FenceState::TargetValidating, + ), + }; + if record.state != from || drain_state(receipt.command) != Some(from) { + return Err(invalid(format!( + "{} cannot complete a drain from {}", + receipt.command, record.state + ))); + } + + let mut next = record.clone(); + if let DrainCompletion::SourceWrites { boundary } = completion { + if record.identity.log_id != Some(boundary.log_id) { + return Err(precondition( + FenceDetail::NamespaceIdentityMismatch, + "the frozen boundary belongs to a different replication log", + )); + } + next.frozen_boundary = Some(boundary); + } + next.state = to; + next.revision = record.revision + 1; + next.drain_policy = None; + next.drain_started_at_ms = None; + next.last_transition_at_ms = env.now_ms; + next.last_command_id = receipt.command_id; + next.written_by = env.server.clone(); + + let mut final_receipt = receipt.clone(); + final_receipt.outcome = FenceOutcome::Applied; + final_receipt.revision_after = next.revision; + final_receipt.state_after = to; + final_receipt.applied_at_ms = env.now_ms; + final_receipt.instance_id = env.server.instance_id; + + Ok((next, final_receipt)) +} + +#[cfg(test)] +mod tests { + use super::*; + + use crate::namespace::fence::command::{DrainPolicy, OnDeadline, TargetConfig}; + use crate::namespace::fence::record::{FenceMarker, FENCE_FORMAT_VERSION}; + use crate::namespace::NamespaceName; + + use FenceOutcome as O; + use FenceState as S; + + const LOG: Uuid = Uuid::from_u128(0x10); + const INCARNATION: Uuid = Uuid::from_u128(0x20); + const OP: Uuid = Uuid::from_u128(0xa); + const OTHER_OP: Uuid = Uuid::from_u128(0xb); + + fn env() -> ApplyEnv { + ApplyEnv { + now_ms: 1_000, + server: ServerIdentity { + build: "test".into(), + instance_id: Uuid::from_u128(0x99), + }, + namespace_log_id: Some(LOG), + shared_schema: false, + legacy_blocks: LegacyBlocks::default(), + new_incarnation_id: INCARNATION, + adoption_authorised: false, + validation_snapshot: None, + } + } + + fn policy() -> Option { + Some(DrainPolicy { + deadline_ms: 1_000, + on_deadline: OnDeadline::Fail, + }) + } + + fn command(kind: CommandKind) -> FenceCommand { + match kind { + CommandKind::AcquireSourceWriteFence => FenceCommand::AcquireSourceWriteFence { + expected_log_id: LOG, + drain_policy: policy(), + }, + CommandKind::SetSourceReadFence => FenceCommand::SetSourceReadFence { + drain_policy: policy(), + }, + CommandKind::ClearSourceReadFence => FenceCommand::ClearSourceReadFence, + CommandKind::ReleaseSourceWriteFence => FenceCommand::ReleaseSourceWriteFence, + CommandKind::CreateTargetQuarantined => FenceCommand::CreateTargetQuarantined { + config: TargetConfig::default(), + }, + CommandKind::SealTargetImport => FenceCommand::SealTargetImport { + drain_policy: policy(), + }, + CommandKind::RecordTargetValidation => FenceCommand::RecordTargetValidation { + result: ValidationResult::Ok, + summary: "row counts match".into(), + }, + CommandKind::PublishTargetReadableWriteFenced => { + FenceCommand::PublishTargetReadableWriteFenced + } + CommandKind::EnableTargetWrites => FenceCommand::EnableTargetWrites, + CommandKind::AbortQuarantinedTarget => FenceCommand::AbortQuarantinedTarget, + CommandKind::AdoptFence => FenceCommand::AdoptFence(AdoptArgs { + current_operation_id: OP, + approvers: vec!["alice".into(), "bob".into()], + incident_ref: "INC-1".into(), + reason: "control plane lost".into(), + }), + } + } + + /// A little driver that plays the store: it keeps the record and receipts and applies + /// decisions the way the store will. + #[derive(Default)] + struct Harness { + namespace_exists: bool, + record: Option, + receipts: Vec, + next_command: u128, + } + + impl Harness { + fn source() -> Self { + Self { + namespace_exists: true, + ..Default::default() + } + } + + fn target() -> Self { + Self::default() + } + + fn current(&self) -> CurrentFence<'_> { + match &self.record { + Some(r) => CurrentFence::Record(r), + None => CurrentFence::None { + namespace_exists: self.namespace_exists, + }, + } + } + + fn request(&mut self, op: Uuid, command: FenceCommand) -> FenceRequest { + self.next_command += 1; + FenceRequest { + namespace: NamespaceName::from("db1"), + operation_id: op, + command_id: Uuid::from_u128(0x1000 + self.next_command), + expected_state: self.current().state(), + expected_revision: self.current().revision(), + command, + } + } + + fn lookup(&self, request: &FenceRequest) -> Option<&CommandReceipt> { + self.receipts.iter().find(|r| { + r.operation_id == request.operation_id && r.command_id == request.command_id + }) + } + + fn decide(&self, request: &FenceRequest, env: &ApplyEnv) -> Result { + apply(self.current(), self.lookup(request), request, env) + } + + fn persist(&mut self, decision: &Decision) { + if let Decision::Apply { record, receipt } = decision { + if let Some(record) = record { + // Everything apply writes must survive the durable encoding. + let decoded = NamespaceFenceRecord::decode( + FENCE_FORMAT_VERSION, + record.revision, + &record.encode(), + ) + .unwrap(); + assert_eq!(&decoded, record); + if let Some(old) = &self.record { + assert!(record.revision > old.revision, "revision must increase"); + } + self.record = Some(record.clone()); + } + self.store_receipt(receipt.clone()); + } + } + + fn store_receipt(&mut self, receipt: CommandReceipt) { + self.receipts.retain(|r| { + !(r.operation_id == receipt.operation_id && r.command_id == receipt.command_id) + }); + self.receipts.push(receipt); + } + + /// Send a fresh command with correct expectations and persist the result. + fn run(&mut self, op: Uuid, kind: CommandKind) -> Result { + let request = self.request(op, command(kind)); + self.run_request(&request, &env()) + } + + fn run_request( + &mut self, + request: &FenceRequest, + env: &ApplyEnv, + ) -> Result { + let decision = self.decide(request, env)?; + self.persist(&decision); + Ok(decision) + } + + fn complete(&mut self, completion: DrainCompletion) { + let record = self.record.clone().unwrap(); + let receipt = self + .receipts + .iter() + .find(|r| r.outcome == O::Draining && r.command_id == record.last_command_id) + .or_else(|| { + self.receipts + .iter() + .rev() + .find(|r| r.outcome == O::Draining) + }) + .unwrap() + .clone(); + let (next, receipt) = complete_drain(&record, &receipt, completion, &env()).unwrap(); + assert_eq!(next.revision, record.revision + 1); + self.record = Some(next); + self.store_receipt(receipt); + } + + fn state(&self) -> FenceState { + self.current().state() + } + + fn revision(&self) -> u64 { + self.current().revision() + } + + /// Drive the harness into `state` with operation `OP`. + fn in_state(state: FenceState) -> Self { + use CommandKind as K; + let boundary = DrainCompletion::SourceWrites { + boundary: FrozenBoundary { + log_id: LOG, + frame_no: 42, + }, + }; + let mut h = if state.role() == Some(Role::Target) || state == S::Absent { + Self::target() + } else { + Self::source() + }; + let steps: &[&dyn Fn(&mut Harness)] = match state { + S::Unfenced | S::Absent => &[], + S::SourceDraining => &[&|h| drop(h.run(OP, K::AcquireSourceWriteFence).unwrap())], + S::SourceWriteFenced => &[ + &|h| drop(h.run(OP, K::AcquireSourceWriteFence).unwrap()), + &|h| h.complete(boundary), + ], + S::SourceReadDraining => &[ + &|h| drop(h.run(OP, K::AcquireSourceWriteFence).unwrap()), + &|h| h.complete(boundary), + &|h| drop(h.run(OP, K::SetSourceReadFence).unwrap()), + ], + S::SourceReadFenced => &[ + &|h| drop(h.run(OP, K::AcquireSourceWriteFence).unwrap()), + &|h| h.complete(boundary), + &|h| drop(h.run(OP, K::SetSourceReadFence).unwrap()), + &|h| h.complete(DrainCompletion::SourceReads), + ], + S::Released => &[ + &|h| drop(h.run(OP, K::AcquireSourceWriteFence).unwrap()), + &|h| h.complete(boundary), + &|h| drop(h.run(OP, K::ReleaseSourceWriteFence).unwrap()), + ], + S::TargetQuarantined => { + &[&|h| drop(h.run(OP, K::CreateTargetQuarantined).unwrap())] + } + S::TargetImportDraining => &[ + &|h| drop(h.run(OP, K::CreateTargetQuarantined).unwrap()), + &|h| drop(h.run(OP, K::SealTargetImport).unwrap()), + ], + S::TargetValidating => &[ + &|h| drop(h.run(OP, K::CreateTargetQuarantined).unwrap()), + &|h| drop(h.run(OP, K::SealTargetImport).unwrap()), + &|h| h.complete(DrainCompletion::TargetImport), + ], + S::TargetWriteFenced => &[ + &|h| drop(h.run(OP, K::CreateTargetQuarantined).unwrap()), + &|h| drop(h.run(OP, K::SealTargetImport).unwrap()), + &|h| h.complete(DrainCompletion::TargetImport), + &|h| drop(h.run(OP, K::RecordTargetValidation).unwrap()), + &|h| drop(h.run(OP, K::PublishTargetReadableWriteFenced).unwrap()), + ], + S::TargetWritable => &[ + &|h| drop(h.run(OP, K::CreateTargetQuarantined).unwrap()), + &|h| drop(h.run(OP, K::SealTargetImport).unwrap()), + &|h| h.complete(DrainCompletion::TargetImport), + &|h| drop(h.run(OP, K::RecordTargetValidation).unwrap()), + &|h| drop(h.run(OP, K::PublishTargetReadableWriteFenced).unwrap()), + &|h| drop(h.run(OP, K::EnableTargetWrites).unwrap()), + ], + S::TargetAborted => &[ + &|h| drop(h.run(OP, K::CreateTargetQuarantined).unwrap()), + &|h| drop(h.run(OP, K::AbortQuarantinedTarget).unwrap()), + ], + S::UnknownUnavailable => panic!("not reachable by transitions"), + }; + for step in steps { + step(&mut h); + } + assert_eq!(h.state(), state); + h + } + } + + fn outcome_of(result: &Result) -> FenceOutcome { + match result { + Ok(Decision::Apply { receipt, .. }) => receipt.outcome, + Ok(Decision::Replay(r)) | Ok(Decision::Resume(r)) => r.outcome, + Err(e) => e.outcome(), + } + } + + fn state_after(result: &Result) -> Option { + match result { + Ok(Decision::Apply { receipt, .. }) => Some(receipt.state_after), + _ => None, + } + } + + const RECORD_STATES: [FenceState; 13] = [ + S::Unfenced, + S::Absent, + S::SourceDraining, + S::SourceWriteFenced, + S::SourceReadDraining, + S::SourceReadFenced, + S::Released, + S::TargetQuarantined, + S::TargetImportDraining, + S::TargetValidating, + S::TargetWriteFenced, + S::TargetWritable, + S::TargetAborted, + ]; + + /// Every (state, command) pair from the owning operation, with correct expectations, + /// against the full expected table: the legal transitions of section 3.2, the + /// `ALREADY_APPLIED` goal states, drain joins, and a refusal for everything else. + #[test] + fn exhaustive_owner_commands() { + use CommandKind as K; + + let expected = + |state: FenceState, kind: CommandKind| -> (FenceOutcome, Option) { + if kind == K::AdoptFence { + // Adopting one's own fence is never allowed; finished fences cannot be + // adopted; no record has nothing to adopt. + return (O::InvalidFenceTransition, None); + } + if kind == K::CreateTargetQuarantined { + return match state { + S::Absent => (O::Applied, Some(S::TargetQuarantined)), + S::TargetQuarantined => (O::AlreadyApplied, Some(S::TargetQuarantined)), + // The owner has finished with the namespace. + s if s.is_operation_finished() => (O::InvalidFenceTransition, None), + // The name is in use. + _ => (O::FencePreconditionFailed, None), + }; + } + if kind.goal_state() == Some(state) { + return (O::AlreadyApplied, Some(state)); + } + if drain_state(kind) == Some(state) { + return (O::Draining, Some(state)); + } + if state.is_operation_finished() && state != S::Unfenced { + return (O::InvalidFenceTransition, None); + } + match transition(kind, state) { + Some((to, outcome)) => { + if kind == K::PublishTargetReadableWriteFenced { + // No validation receipt has been recorded on this path. + (O::FencePreconditionFailed, None) + } else { + (outcome, Some(to)) + } + } + None => (O::InvalidFenceTransition, None), + } + }; + + for state in RECORD_STATES { + for kind in CommandKind::ALL { + // A target driven to TARGET_VALIDATING has no validation receipt yet. + let mut h = Harness::in_state(state); + let result = h.run(OP, kind); + let (outcome, to) = expected(state, kind); + assert_eq!(outcome_of(&result), outcome, "{state} {kind}: {result:?}"); + assert_eq!(state_after(&result), to, "{state} {kind}"); + } + } + } + + #[test] + fn source_happy_path() { + let mut h = Harness::source(); + let r = h.run(OP, CommandKind::AcquireSourceWriteFence).unwrap(); + let Decision::Apply { record, receipt } = r else { + panic!() + }; + let record = record.unwrap(); + assert_eq!(record.state, S::SourceDraining); + assert_eq!(record.revision, 1); + assert_eq!(receipt.outcome, O::Draining); + assert_eq!((receipt.revision_before, receipt.revision_after), (0, 1)); + assert_eq!(record.identity.log_id, Some(LOG)); + assert!(!record.write_admission().is_open()); + assert!(record.read_admission().is_open()); + + h.complete(DrainCompletion::SourceWrites { + boundary: FrozenBoundary { + log_id: LOG, + frame_no: 7, + }, + }); + assert_eq!(h.state(), S::SourceWriteFenced); + assert_eq!(h.revision(), 2); + assert_eq!( + h.record.as_ref().unwrap().frozen_boundary.unwrap().frame_no, + 7 + ); + let acquire_receipt = h + .receipts + .iter() + .find(|r| r.command == CommandKind::AcquireSourceWriteFence) + .unwrap(); + assert_eq!(acquire_receipt.outcome, O::Applied); + assert_eq!(acquire_receipt.revision_after, 2); + + h.run(OP, CommandKind::SetSourceReadFence).unwrap(); + assert_eq!((h.state(), h.revision()), (S::SourceReadDraining, 3)); + h.complete(DrainCompletion::SourceReads); + assert_eq!((h.state(), h.revision()), (S::SourceReadFenced, 4)); + h.run(OP, CommandKind::ClearSourceReadFence).unwrap(); + assert_eq!((h.state(), h.revision()), (S::SourceWriteFenced, 5)); + h.run(OP, CommandKind::ReleaseSourceWriteFence).unwrap(); + assert_eq!((h.state(), h.revision()), (S::Released, 6)); + assert!(h.record.as_ref().unwrap().write_admission().is_open()); + + // A new operation may acquire the released namespace; the revision keeps counting. + let r = h + .run(OTHER_OP, CommandKind::AcquireSourceWriteFence) + .unwrap(); + assert!(matches!(r, Decision::Apply { .. })); + let record = h.record.as_ref().unwrap(); + assert_eq!( + (record.state, record.revision, record.operation_id), + (S::SourceDraining, 7, OTHER_OP) + ); + assert!(record.frozen_boundary.is_none()); + } + + #[test] + fn target_happy_path() { + let mut h = Harness::target(); + h.run(OP, CommandKind::CreateTargetQuarantined).unwrap(); + let record = h.record.clone().unwrap(); + assert_eq!( + (record.role, record.state, record.revision), + (Role::Target, S::TargetQuarantined, 1) + ); + assert_eq!(record.identity.target_incarnation_id, Some(INCARNATION)); + assert!(!record.read_admission().is_open()); + + h.run(OP, CommandKind::SealTargetImport).unwrap(); + assert_eq!((h.state(), h.revision()), (S::TargetImportDraining, 2)); + h.complete(DrainCompletion::TargetImport); + assert_eq!((h.state(), h.revision()), (S::TargetValidating, 3)); + + // Publication needs a successful validation receipt first. + let err = h + .run(OP, CommandKind::PublishTargetReadableWriteFenced) + .unwrap_err(); + assert_eq!(err.outcome(), O::FencePreconditionFailed); + assert_eq!(err.detail(), Some(FenceDetail::ValidationReceiptRequired)); + + let failed = h.request( + OP, + FenceCommand::RecordTargetValidation { + result: ValidationResult::Failed, + summary: "mismatch".into(), + }, + ); + h.run_request(&failed, &env()).unwrap(); + assert_eq!((h.state(), h.revision()), (S::TargetValidating, 4)); + let err = h + .run(OP, CommandKind::PublishTargetReadableWriteFenced) + .unwrap_err(); + assert_eq!(err.detail(), Some(FenceDetail::ValidationReceiptRequired)); + + let snapshot = ValidationSnapshot { + log_id: LOG, + frame_no: 9, + page_count: 3, + }; + let ok = h.request(OP, command(CommandKind::RecordTargetValidation)); + h.run_request( + &ok, + &ApplyEnv { + validation_snapshot: Some(snapshot), + ..env() + }, + ) + .unwrap(); + assert_eq!(h.revision(), 5); + assert_eq!( + h.record + .as_ref() + .unwrap() + .validation + .as_ref() + .unwrap() + .snapshot, + Some(snapshot) + ); + + h.run(OP, CommandKind::PublishTargetReadableWriteFenced) + .unwrap(); + assert_eq!((h.state(), h.revision()), (S::TargetWriteFenced, 6)); + assert!(h.record.as_ref().unwrap().read_admission().is_open()); + assert!(!h.record.as_ref().unwrap().write_admission().is_open()); + + h.run(OP, CommandKind::EnableTargetWrites).unwrap(); + assert_eq!((h.state(), h.revision()), (S::TargetWritable, 7)); + assert!(h.record.as_ref().unwrap().write_admission().is_open()); + } + + #[test] + fn validation_summary_is_bounded() { + let mut h = Harness::in_state(S::TargetValidating); + let request = h.request( + OP, + FenceCommand::RecordTargetValidation { + result: ValidationResult::Ok, + summary: "x".repeat(MAX_VALIDATION_SUMMARY_BYTES + 1), + }, + ); + let err = h.run_request(&request, &env()).unwrap_err(); + assert_eq!(err.detail(), Some(FenceDetail::InvalidArgument)); + } + + /// Section 2.8: nothing moves a published target back, for the owning operation. + #[test] + fn target_writable_is_irreversible() { + for kind in CommandKind::ALL { + let mut h = Harness::in_state(S::TargetWritable); + let before = h.record.clone(); + let result = h.run(OP, kind); + match kind { + CommandKind::EnableTargetWrites => { + assert_eq!(outcome_of(&result), O::AlreadyApplied) + } + _ => assert_eq!(outcome_of(&result), O::InvalidFenceTransition, "{kind}"), + } + assert_eq!(h.record, before, "{kind} changed a published target"); + } + + // Another operation cannot move it to a frozen, aborted or absent target state + // either; it can only start a new move with the namespace as a source. + for kind in CommandKind::ALL { + let mut h = Harness::in_state(S::TargetWritable); + let result = h.run(OTHER_OP, kind); + if kind == CommandKind::AcquireSourceWriteFence { + assert_eq!(state_after(&result), Some(S::SourceDraining)); + } else { + assert!(outcome_of(&result).is_error(), "{kind}: {result:?}"); + assert_eq!(h.state(), S::TargetWritable); + } + } + } + + #[test] + fn exact_replay_after_revision_advanced() { + let mut h = Harness::source(); + let acquire = h.request(OP, command(CommandKind::AcquireSourceWriteFence)); + h.run_request(&acquire, &env()).unwrap(); + + // While draining, a replay resumes the same drain. + let d = h.decide(&acquire, &env()).unwrap(); + assert!(matches!(&d, Decision::Resume(r) if r.command_id == acquire.command_id)); + + h.complete(DrainCompletion::SourceWrites { + boundary: FrozenBoundary { + log_id: LOG, + frame_no: 1, + }, + }); + h.run(OP, CommandKind::SetSourceReadFence).unwrap(); + h.complete(DrainCompletion::SourceReads); + assert_eq!(h.revision(), 4); + + // The original acquire's expectations (UNFENCED, 0) are long stale, but a replay is + // answered from its receipt before any revision check. + let d = h.decide(&acquire, &env()).unwrap(); + let Decision::Replay(receipt) = d else { + panic!("{d:?}") + }; + assert_eq!(receipt.outcome, O::Applied); + assert_eq!(receipt.state_after, S::SourceWriteFenced); + assert_eq!(receipt.revision_after, 2); + } + + #[test] + fn command_id_reuse_with_different_fingerprint_conflicts() { + let mut h = Harness::source(); + let acquire = h.request(OP, command(CommandKind::AcquireSourceWriteFence)); + h.run_request(&acquire, &env()).unwrap(); + let before = (h.record.clone(), h.receipts.clone()); + + let mut reused = acquire.clone(); + reused.command = FenceCommand::ReleaseSourceWriteFence; + reused.expected_state = S::SourceDraining; + reused.expected_revision = 1; + let err = h.decide(&reused, &env()).unwrap_err(); + assert_eq!(err.outcome(), O::FenceCommandConflict); + + let mut reused = acquire.clone(); + reused.command = FenceCommand::AcquireSourceWriteFence { + expected_log_id: LOG, + drain_policy: None, + }; + assert_eq!( + h.decide(&reused, &env()).unwrap_err().outcome(), + O::FenceCommandConflict + ); + + assert_eq!((h.record.clone(), h.receipts.clone()), before); + } + + #[test] + fn wrong_owner_is_refused() { + for state in RECORD_STATES { + if !state.is_durable() || state.is_operation_finished() { + continue; + } + for kind in CommandKind::ALL { + if kind == CommandKind::AdoptFence { + continue; + } + let mut h = Harness::in_state(state); + let before = h.record.clone(); + let result = h.run(OTHER_OP, kind); + // The owner is checked before anything about the command. + assert_eq!( + outcome_of(&result), + O::FenceOwnedByAnotherOperation, + "{state} {kind}" + ); + assert_eq!(h.record, before); + } + } + } + + #[test] + fn stale_revision_is_refused() { + let mut h = Harness::in_state(S::SourceWriteFenced); + let mut request = h.request(OP, command(CommandKind::SetSourceReadFence)); + request.expected_revision -= 1; + assert_eq!( + h.decide(&request, &env()).unwrap_err().outcome(), + O::FenceRevisionMismatch + ); + + let mut request = h.request(OP, command(CommandKind::SetSourceReadFence)); + request.expected_state = S::SourceDraining; + assert_eq!( + h.decide(&request, &env()).unwrap_err().outcome(), + O::FenceRevisionMismatch + ); + + let mut request = h.request(OP, command(CommandKind::SetSourceReadFence)); + request.expected_revision += 1; + assert_eq!( + h.decide(&request, &env()).unwrap_err().outcome(), + O::FenceRevisionMismatch + ); + } + + #[test] + fn role_mismatch() { + let mut h = Harness::in_state(S::SourceWriteFenced); + let err = h.run(OP, CommandKind::SealTargetImport).unwrap_err(); + assert_eq!(err.outcome(), O::InvalidFenceTransition); + assert_eq!(err.detail(), Some(FenceDetail::RoleMismatch)); + + let mut h = Harness::in_state(S::TargetQuarantined); + let err = h.run(OP, CommandKind::SetSourceReadFence).unwrap_err(); + assert_eq!(err.detail(), Some(FenceDetail::RoleMismatch)); + + let mut h = Harness::source(); + let err = h.run(OP, CommandKind::EnableTargetWrites).unwrap_err(); + assert_eq!(err.detail(), Some(FenceDetail::RoleMismatch)); + + let mut h = Harness::target(); + let err = h.run(OP, CommandKind::AcquireSourceWriteFence).unwrap_err(); + assert_eq!(err.detail(), Some(FenceDetail::RoleMismatch)); + } + + /// The pure half of `acquire_race_single_owner`: two operations race to acquire the same + /// namespace; the store's serialised transactions mean the second decides against the + /// first's committed record and loses with a typed conflict. + #[test] + fn acquire_race_single_owner() { + let mut h = Harness::source(); + let a = h.request(OP, command(CommandKind::AcquireSourceWriteFence)); + let b = h.request(OTHER_OP, command(CommandKind::AcquireSourceWriteFence)); + h.run_request(&a, &env()).unwrap(); + let err = h.run_request(&b, &env()).unwrap_err(); + assert_eq!(err.outcome(), O::FenceOwnedByAnotherOperation); + assert_eq!(h.record.as_ref().unwrap().operation_id, OP); + } + + #[test] + fn acquire_preconditions() { + let mut h = Harness::source(); + let request = h.request(OP, command(CommandKind::AcquireSourceWriteFence)); + let err = h + .decide( + &request, + &ApplyEnv { + namespace_log_id: Some(Uuid::from_u128(0x11)), + ..env() + }, + ) + .unwrap_err(); + assert_eq!(err.detail(), Some(FenceDetail::NamespaceIdentityMismatch)); + + let err = h + .decide( + &request, + &ApplyEnv { + shared_schema: true, + ..env() + }, + ) + .unwrap_err(); + assert_eq!(err.detail(), Some(FenceDetail::SharedSchemaUnsupported)); + } + + #[test] + fn acquire_saves_and_release_restores_legacy_blocks() { + let saved = LegacyBlocks { + block_reads: false, + block_writes: true, + block_reason: Some("maintenance".into()), + }; + let mut h = Harness::source(); + let request = h.request(OP, command(CommandKind::AcquireSourceWriteFence)); + h.run_request( + &request, + &ApplyEnv { + legacy_blocks: saved.clone(), + ..env() + }, + ) + .unwrap(); + let mirror = h.record.as_ref().unwrap().legacy_mirror(); + assert!(mirror.block_writes); + assert!(mirror + .block_reason + .unwrap() + .starts_with("namespace fence: SOURCE_DRAINING")); + + h.run(OP, CommandKind::ReleaseSourceWriteFence).unwrap(); + assert_eq!(h.record.as_ref().unwrap().legacy_mirror(), saved); + } + + #[test] + fn release_from_draining_is_a_precommit_rollback() { + let mut h = Harness::in_state(S::SourceDraining); + h.run(OP, CommandKind::ReleaseSourceWriteFence).unwrap(); + assert_eq!((h.state(), h.revision()), (S::Released, 2)); + } + + #[test] + fn owner_joins_its_own_drain_with_a_new_command() { + let mut h = Harness::in_state(S::SourceDraining); + let d = h.run(OP, CommandKind::AcquireSourceWriteFence).unwrap(); + let Decision::Apply { record, receipt } = d else { + panic!() + }; + assert!(record.is_none()); + assert_eq!(receipt.outcome, O::Draining); + assert_eq!(h.revision(), 1); + + // The drain completes through the new command's receipt. + let record = h.record.clone().unwrap(); + let (next, final_receipt) = complete_drain( + &record, + &receipt, + DrainCompletion::SourceWrites { + boundary: FrozenBoundary { + log_id: LOG, + frame_no: 3, + }, + }, + &env(), + ) + .unwrap(); + assert_eq!(next.state, S::SourceWriteFenced); + assert_eq!(final_receipt.command_id, receipt.command_id); + assert_eq!(final_receipt.outcome, O::Applied); + } + + #[test] + fn complete_drain_checks() { + let h = Harness::in_state(S::SourceDraining); + let record = h.record.clone().unwrap(); + let receipt = h.receipts[0].clone(); + + // Wrong kind of completion for the state. + assert!(complete_drain(&record, &receipt, DrainCompletion::SourceReads, &env()).is_err()); + assert!(complete_drain(&record, &receipt, DrainCompletion::TargetImport, &env()).is_err()); + + // A boundary from another log. + let err = complete_drain( + &record, + &receipt, + DrainCompletion::SourceWrites { + boundary: FrozenBoundary { + log_id: Uuid::from_u128(0x77), + frame_no: 1, + }, + }, + &env(), + ) + .unwrap_err(); + assert_eq!(err.detail(), Some(FenceDetail::NamespaceIdentityMismatch)); + + // A final receipt, or another operation's. + let mut final_receipt = receipt.clone(); + final_receipt.outcome = O::Applied; + let boundary = DrainCompletion::SourceWrites { + boundary: FrozenBoundary { + log_id: LOG, + frame_no: 1, + }, + }; + assert!(complete_drain(&record, &final_receipt, boundary, &env()).is_err()); + let mut other = receipt.clone(); + other.operation_id = OTHER_OP; + assert!(complete_drain(&record, &other, boundary, &env()).is_err()); + + // Not draining any more. + let h = Harness::in_state(S::SourceWriteFenced); + let record = h.record.clone().unwrap(); + assert!(complete_drain(&record, &receipt, boundary, &env()).is_err()); + } + + #[test] + fn unavailable_refuses_everything_else() { + let marker = Harness::in_state(S::SourceWriteFenced).record.unwrap(); + for detail in [ + FenceDetail::CorruptRecord, + FenceDetail::UnsupportedFormatVersion, + FenceDetail::MetastoreBehindMarker, + FenceDetail::IncompleteTargetCreation, + FenceDetail::IndeterminateCommit, + ] { + for kind in CommandKind::ALL { + if kind == CommandKind::AdoptFence && detail == FenceDetail::MetastoreBehindMarker { + continue; + } + let current = CurrentFence::Unavailable { + detail, + marker: Some(&marker), + }; + let request = FenceRequest { + namespace: NamespaceName::from("db1"), + operation_id: OP, + command_id: Uuid::from_u128(0x5000), + expected_state: S::UnknownUnavailable, + expected_revision: marker.revision, + command: command(kind), + }; + let err = apply(current, None, &request, &env()).unwrap_err(); + assert_eq!(err.outcome(), O::FenceStateUnavailable, "{detail} {kind}"); + assert_eq!(err.detail(), Some(detail)); + } + } + } + + #[test] + fn incomplete_target_creation_is_completed_only_by_the_same_command() { + let mut h = Harness::target(); + let create = h.request(OP, command(CommandKind::CreateTargetQuarantined)); + let Decision::Apply { record, .. } = h.decide(&create, &env()).unwrap() else { + panic!() + }; + // The store writes this marker, then crashes before the metastore commit. + let marker = FenceMarker::for_record(record.as_ref().unwrap()); + let marker = FenceMarker::decode(&marker.encode()).unwrap().record; + let current = CurrentFence::Unavailable { + detail: FenceDetail::IncompleteTargetCreation, + marker: Some(&marker), + }; + + // Another command id, even from the same operation, cannot complete it. + let mut other = create.clone(); + other.command_id = Uuid::from_u128(0x6000); + let err = apply(current, None, &other, &env()).unwrap_err(); + assert_eq!(err.outcome(), O::FenceStateUnavailable); + + // The same command does, with the incarnation id the marker recorded. + let d = apply( + current, + None, + &create, + &ApplyEnv { + new_incarnation_id: Uuid::from_u128(0x7777), + ..env() + }, + ) + .unwrap(); + let Decision::Apply { + record: Some(record), + receipt, + } = d + else { + panic!() + }; + assert_eq!(record, marker); + assert_eq!(receipt.outcome, O::Applied); + } + + fn adopt_request(h: &mut Harness, args: AdoptArgs) -> FenceRequest { + h.request(OTHER_OP, FenceCommand::AdoptFence(args)) + } + + fn adopt_args() -> AdoptArgs { + match command(CommandKind::AdoptFence) { + FenceCommand::AdoptFence(args) => args, + _ => unreachable!(), + } + } + + fn authorised() -> ApplyEnv { + ApplyEnv { + adoption_authorised: true, + ..env() + } + } + + #[test] + fn adopt_requires_key_and_two_approvers() { + let mut h = Harness::in_state(S::SourceWriteFenced); + let request = adopt_request(&mut h, adopt_args()); + let err = h.decide(&request, &env()).unwrap_err(); + assert_eq!(err.detail(), Some(FenceDetail::AdoptionNotAuthorised)); + + let bad = [ + AdoptArgs { + approvers: vec!["alice".into()], + ..adopt_args() + }, + AdoptArgs { + approvers: vec!["alice".into(), " alice ".into()], + ..adopt_args() + }, + AdoptArgs { + approvers: vec!["alice".into(), "".into()], + ..adopt_args() + }, + AdoptArgs { + approvers: vec!["a".into(), "b".into(), "c".into()], + ..adopt_args() + }, + AdoptArgs { + incident_ref: " ".into(), + ..adopt_args() + }, + AdoptArgs { + reason: "".into(), + ..adopt_args() + }, + ]; + for args in bad { + let request = adopt_request(&mut h, args.clone()); + let err = h.decide(&request, &authorised()).unwrap_err(); + assert_eq!( + err.detail(), + Some(FenceDetail::AdoptionNotAuthorised), + "{args:?}" + ); + } + + let wrong_owner = AdoptArgs { + current_operation_id: Uuid::from_u128(0xdead), + ..adopt_args() + }; + let request = adopt_request(&mut h, wrong_owner); + assert_eq!( + h.decide(&request, &authorised()).unwrap_err().outcome(), + O::FenceOwnedByAnotherOperation + ); + } + + #[test] + fn adopt_keeps_gates_closed() { + for state in [ + S::SourceDraining, + S::SourceWriteFenced, + S::SourceReadDraining, + S::SourceReadFenced, + S::TargetQuarantined, + S::TargetImportDraining, + S::TargetValidating, + S::TargetWriteFenced, + ] { + let mut h = Harness::in_state(state); + let before = h.record.clone().unwrap(); + let request = adopt_request(&mut h, adopt_args()); + h.run_request(&request, &authorised()).unwrap(); + let after = h.record.clone().unwrap(); + assert_eq!(after.state, state); + assert_eq!(after.write_admission(), before.write_admission()); + assert_eq!(after.read_admission(), before.read_admission()); + assert_eq!(after.revision, before.revision + 1); + assert_eq!(after.operation_id, OTHER_OP); + assert_eq!(after.adoptions.len(), 1); + assert_eq!(after.adoptions[0].approvers, vec!["alice", "bob"]); + let receipt = h.receipts.last().unwrap(); + assert_eq!(receipt.adoption.as_ref().unwrap().previous_operation_id, OP); + + // The old owner is now locked out. + let result = h.run(OP, CommandKind::ReleaseSourceWriteFence); + assert_eq!(outcome_of(&result), O::FenceOwnedByAnotherOperation); + } + } + + #[test] + fn adopt_cannot_touch_finished_fences() { + for state in [S::TargetWritable, S::TargetAborted, S::Released] { + let mut h = Harness::in_state(state); + let request = adopt_request(&mut h, adopt_args()); + let err = h.decide(&request, &authorised()).unwrap_err(); + assert_eq!(err.outcome(), O::InvalidFenceTransition, "{state}"); + assert_eq!(err.detail(), Some(FenceDetail::OperationFinished)); + } + let mut h = Harness::source(); + let request = adopt_request(&mut h, adopt_args()); + assert_eq!( + h.decide(&request, &authorised()).unwrap_err().outcome(), + O::InvalidFenceTransition + ); + } + + #[test] + fn adopted_owner_can_finish_the_drain() { + let mut h = Harness::in_state(S::SourceDraining); + let request = adopt_request(&mut h, adopt_args()); + h.run_request(&request, &authorised()).unwrap(); + let d = h + .run(OTHER_OP, CommandKind::AcquireSourceWriteFence) + .unwrap(); + assert!( + matches!(d, Decision::Apply { record: None, ref receipt } if receipt.outcome == O::Draining) + ); + h.complete(DrainCompletion::SourceWrites { + boundary: FrozenBoundary { + log_id: LOG, + frame_no: 5, + }, + }); + assert_eq!(h.state(), S::SourceWriteFenced); + assert_eq!(h.record.as_ref().unwrap().operation_id, OTHER_OP); + } + + #[test] + fn adopt_after_metastore_rollback_restores_marker_record() { + let marker = Harness::in_state(S::SourceReadFenced).record.unwrap(); + let current = CurrentFence::Unavailable { + detail: FenceDetail::MetastoreBehindMarker, + marker: Some(&marker), + }; + let request = FenceRequest { + namespace: NamespaceName::from("db1"), + operation_id: OTHER_OP, + command_id: Uuid::from_u128(0x8000), + expected_state: S::SourceReadFenced, + expected_revision: marker.revision, + command: FenceCommand::AdoptFence(adopt_args()), + }; + let d = apply(current, None, &request, &authorised()).unwrap(); + let Decision::Apply { + record: Some(record), + .. + } = d + else { + panic!() + }; + assert_eq!(record.state, S::SourceReadFenced); + assert_eq!(record.revision, marker.revision + 1); + assert_eq!(record.operation_id, OTHER_OP); + assert_eq!(record.frozen_boundary, marker.frozen_boundary); + } + + #[test] + fn revision_increases_by_one_per_applied_transition() { + let h = Harness::in_state(S::TargetWritable); + let mut receipts = h.receipts.clone(); + receipts.sort_by_key(|r| r.revision_after); + let mut last = 0; + for r in receipts { + if r.outcome == O::AlreadyApplied { + continue; + } + assert_eq!( + r.revision_after, + last + if r.command == CommandKind::SealTargetImport { + 2 + } else { + 1 + }, + "{r:?}" + ); + last = r.revision_after; + } + assert_eq!(h.revision(), 6); + } +} diff --git a/libsql-server/src/namespace/mod.rs b/libsql-server/src/namespace/mod.rs index ec45b50445..cba4030090 100644 --- a/libsql-server/src/namespace/mod.rs +++ b/libsql-server/src/namespace/mod.rs @@ -20,6 +20,7 @@ pub use self::store::NamespaceStore; pub mod broadcasters; pub(crate) mod configurator; +pub mod fence; pub mod meta_store; mod name; pub mod replication_wal; diff --git a/libsql-server/tests/bootstrap.rs b/libsql-server/tests/bootstrap.rs index a464f53288..912015e1f9 100644 --- a/libsql-server/tests/bootstrap.rs +++ b/libsql-server/tests/bootstrap.rs @@ -3,7 +3,7 @@ use std::process::Command; #[test] fn bootstrap() { - let iface_files = &["proto/admin_shell.proto"]; + let iface_files = &["proto/admin_shell.proto", "proto/namespace_fence.proto"]; let dirs = &["proto"]; let out_dir = PathBuf::from(std::env!("CARGO_MANIFEST_DIR")) From 8f05ee6fdf431579020e236c140127e448ab278c Mon Sep 17 00:00:00 2001 From: River Date: Tue, 29 Sep 2026 14:32:58 +0000 Subject: [PATCH 05/33] libsql-server: persist namespace fences in the metastore Add the additive `namespace_fences` and `namespace_fence_receipts` tables (created with --enable-namespace-fence, loaded and enforced whenever they exist) and run every fence command as a compare-and-swap in one BEGIN IMMEDIATE metastore transaction: read the record, the command's receipt and the config row, decide with the pure transition function, then commit the record, the receipt and the legacy block_* mirror together. The revision is stored, so it survives restart. Stored state is read strictly: an unknown format version, an undecodable payload, a revision column that disagrees with its payload, or a per-namespace `.fence` marker that is ahead of the metastore makes the namespace UNKNOWN_UNAVAILABLE instead of guessing. The marker is written after each commit (before it, for target creation) and rewritten on load when it fell behind. Ordinary config writes and deletes now take BEGIN IMMEDIATE and refuse while the fence denies lifecycle operations, and a config write that failed to persist is no longer published to the in-memory config. Receipts of finished operations are pruned after a retention period. Co-authored-by: Tomasz Szymczyszyn --- docs/NAMESPACE_FENCE.md | 8 +- libsql-server/src/config.rs | 6 + libsql-server/src/error.rs | 3 + libsql-server/src/main.rs | 14 + libsql-server/src/namespace/fence/mod.rs | 6 +- libsql-server/src/namespace/fence/record.rs | 2 +- libsql-server/src/namespace/fence/store.rs | 951 +++++++++++++ libsql-server/src/namespace/meta_store.rs | 1381 ++++++++++++++++++- 8 files changed, 2348 insertions(+), 23 deletions(-) create mode 100644 libsql-server/src/namespace/fence/store.rs diff --git a/docs/NAMESPACE_FENCE.md b/docs/NAMESPACE_FENCE.md index 965e44de52..d26eefca59 100644 --- a/docs/NAMESPACE_FENCE.md +++ b/docs/NAMESPACE_FENCE.md @@ -324,11 +324,15 @@ All namespace-config writes already pass through the metastore's single `MetaSto Ordinary config writes (`try_process`) and `remove` read the fence row inside their own transaction and refuse when the state denies lifecycle operations. Because both take `BEGIN IMMEDIATE` on the same database, a config write cannot interleave with a fence transition. +While a record is in force, the stored config row carries the legacy mirror (section 13.2) and the in-memory config is the namespace's own configuration: a transition overlays the mirror on the config row *as read inside its transaction*, so a config change committed by another metastore connection is preserved underneath it, and loading the metastore puts the saved `block_*` values back into the in-memory config. A namespace whose fence cannot be established keeps its stored row, mirror included, in memory. + +A fence command that returns an error writes nothing. The store answers a replay or a resumed drain without writing; otherwise it commits the record (when it changed), the receipt and the mirror together, writes the marker, and only then returns. `CreateTargetQuarantined` returns the created namespace config without publishing it; the caller publishes it after installing the target's gate (section 10.1). + `try_process` today publishes the new config to the in-memory watch even when persisting failed. That is fixed for all config writes: the watch is updated only after commit, and the error is returned. ### 5.5 Receipt retention (Contract) -Receipts of the operation that currently owns a record are never pruned. Receipts of finished operations (`RELEASED`, `TARGET_WRITABLE`, or superseded by adoption) are kept for at least `--namespace-fence-receipt-retention` (default 30 days) and are pruned only inside a later transition on the same namespace. Delete of a namespace in `UNFENCED`, `RELEASED` or `TARGET_WRITABLE` removes its fence row and receipts in the same transaction and logs them. +Receipts of the operation that currently owns a record are never pruned. Receipts of finished operations (`RELEASED`, `TARGET_WRITABLE`, or superseded by adoption) are kept for at least `--namespace-fence-receipt-retention-s` (default 30 days) and are pruned only inside a later transition on the same namespace. Delete of a namespace in `UNFENCED`, `RELEASED` or `TARGET_WRITABLE` removes its fence row and receipts in the same transaction and logs them; its marker is removed before that transaction commits, so a crash in between leaves a record without a marker (repaired on load) rather than a marker without a record. ### 5.6 On-disk marker @@ -595,7 +599,7 @@ The server cannot verify who the approvers are: the admin API has one shared key - Off, but fence tables or markers exist (the flag was turned off after use): fences are still loaded and enforced; mutating routes are disabled. - On: tables are created, routes are served, and the fail-closed recovery rules of section 13.3 apply. -Related flags: `--namespace-fence-receipt-retention`, `--namespace-fence-adoption-key`, `--namespace-fence-keepalive-interval`, and default drain deadlines `--namespace-fence-default-write-drain-ms` and `--namespace-fence-default-read-drain-ms` (used when a request has no `drain_policy`). +Related flags: `--namespace-fence-receipt-retention-s`, `--namespace-fence-adoption-key`, `--namespace-fence-keepalive-interval`, and default drain deadlines `--namespace-fence-default-write-drain-ms` and `--namespace-fence-default-read-drain-ms` (used when a request has no `drain_policy`). Upgrade order: deploy a binary with capability discovery and proxy `stable_code` support on every primary and replica; confirm with `GET /v1/fence/capabilities`; then enable the flag; then use fences. Rollback to a binary without fence support is refused by deployment tooling while `active_fences > 0`. diff --git a/libsql-server/src/config.rs b/libsql-server/src/config.rs index 2c3c302a6d..9ac7add98b 100644 --- a/libsql-server/src/config.rs +++ b/libsql-server/src/config.rs @@ -187,6 +187,12 @@ pub struct MetaStoreConfig { pub allow_recover_from_fs: bool, /// Destroy the metastore if there is a restore error pub destroy_on_error: bool, + /// Allow namespace fences to be used: creates the fence tables. Fences that already exist + /// are loaded and enforced whether or not this is set. + pub namespace_fence: bool, + /// How long receipts of finished fence operations are kept. `None` is the default of + /// 30 days. + pub namespace_fence_receipt_retention: Option, } #[derive(Debug, Clone)] diff --git a/libsql-server/src/error.rs b/libsql-server/src/error.rs index bfe67f47c7..f0cb631769 100644 --- a/libsql-server/src/error.rs +++ b/libsql-server/src/error.rs @@ -128,6 +128,8 @@ pub enum Error { RuntimeTaskJoinError(#[from] tokio::task::JoinError), #[error("database is not a primary")] NotAPrimary, + #[error(transparent)] + NamespaceFence(#[from] crate::namespace::fence::outcome::FenceError), } impl AsRef for Error { @@ -224,6 +226,7 @@ impl IntoResponse for &Error { AttachInMigration => self.format_err(StatusCode::BAD_REQUEST), RuntimeTaskJoinError(_) => self.format_err(StatusCode::INTERNAL_SERVER_ERROR), NotAPrimary => self.format_err(StatusCode::BAD_REQUEST), + NamespaceFence(e) => self.format_err(e.outcome().admin_http_status()), } } } diff --git a/libsql-server/src/main.rs b/libsql-server/src/main.rs index 307d5482fe..5d738a1ed5 100644 --- a/libsql-server/src/main.rs +++ b/libsql-server/src/main.rs @@ -258,6 +258,16 @@ struct Cli { #[clap(long, env = "SQLD_ALLOW_METASTORE_RECOVERY")] allow_metastore_recovery: bool, + /// Allow namespace fences to be used (see `docs/NAMESPACE_FENCE.md`). Off by default. + /// Fences that already exist in the metastore are enforced either way. + #[clap(long, env = "SQLD_ENABLE_NAMESPACE_FENCE")] + enable_namespace_fence: bool, + + /// How long, in seconds, receipts of finished namespace-fence operations are kept. + /// Defaults to 30 days. + #[clap(long, env = "SQLD_NAMESPACE_FENCE_RECEIPT_RETENTION_S")] + namespace_fence_receipt_retention_s: Option, + /// Shutdown timeout duration in seconds, defaults to 30 seconds. #[clap(long, env = "SQLD_SHUTDOWN_TIMEOUT")] shutdown_timeout: Option, @@ -650,6 +660,10 @@ fn make_meta_store_config(config: &Cli) -> anyhow::Result { bottomless, allow_recover_from_fs: config.allow_metastore_recovery, destroy_on_error: config.meta_store_destroy_on_error, + namespace_fence: config.enable_namespace_fence, + namespace_fence_receipt_retention: config + .namespace_fence_receipt_retention_s + .map(Duration::from_secs), }) } diff --git a/libsql-server/src/namespace/fence/mod.rs b/libsql-server/src/namespace/fence/mod.rs index 3ffe2746f8..672b0565e8 100644 --- a/libsql-server/src/namespace/fence/mod.rs +++ b/libsql-server/src/namespace/fence/mod.rs @@ -5,8 +5,9 @@ //! `docs/NAMESPACE_FENCE.md` is the contract and the design. This module holds the parts with //! no I/O: the states and permission matrix ([`state`]), the stable outcome codes and their //! protocol mappings ([`outcome`]), commands and their canonical fingerprint ([`command`]), -//! records, receipts and markers with their strict durable encoding ([`record`]), and the pure -//! transition function ([`transition`]). +//! records, receipts and markers with their strict durable encoding ([`record`]), the pure +//! transition function ([`transition`]), and the metastore tables, compare-and-swap and marker +//! file that persist them ([`store`], driven by `MetaStore::apply_fence_command`). // The persistence, controller and protocol layers that consume these types land in the // following commits of this series; until then most of the module is unused by the rest of @@ -17,6 +18,7 @@ pub mod command; pub mod outcome; pub mod record; pub mod state; +pub mod store; pub mod transition; #[allow(clippy::all)] diff --git a/libsql-server/src/namespace/fence/record.rs b/libsql-server/src/namespace/fence/record.rs index 9599f2a849..a2d9f7f5dd 100644 --- a/libsql-server/src/namespace/fence/record.rs +++ b/libsql-server/src/namespace/fence/record.rs @@ -728,7 +728,7 @@ pub(super) mod tests { } } - fn sample_receipt() -> CommandReceipt { + pub fn sample_receipt() -> CommandReceipt { CommandReceipt { namespace: NamespaceName::from("db1"), operation_id: Uuid::from_u128(1), diff --git a/libsql-server/src/namespace/fence/store.rs b/libsql-server/src/namespace/fence/store.rs new file mode 100644 index 0000000000..2d6f9ca217 --- /dev/null +++ b/libsql-server/src/namespace/fence/store.rs @@ -0,0 +1,951 @@ +//! Durable fence state: the metastore tables, the fence compare-and-swap, and the +//! per-namespace marker file (`docs/NAMESPACE_FENCE.md` sections 5.1 and 5.4 to 5.6). +//! +//! Everything here runs on a metastore connection, inside a transaction the caller opened with +//! `BEGIN IMMEDIATE`, so a fence transition, an ordinary config write and a delete of the same +//! namespace are serialised by SQLite's write lock whichever connection they use. The functions +//! only read and write rows and files: what a command does is decided by +//! [`transition::apply`](super::transition::apply), and when the result is published is the +//! caller's decision, made only after the transaction has committed. +//! +//! Reading is strict. A row with an unknown format version, an undecodable payload, or a +//! revision column that disagrees with its payload, and a marker that says more than the +//! metastore does, all read as [`StoredFence::Unavailable`]: the namespace is +//! `UNKNOWN_UNAVAILABLE` and every gate that consults it stays closed. + +use std::fs::{self, File}; +use std::io::{self, Write as _}; +use std::path::{Path, PathBuf}; +use std::time::Duration; + +use prost::Message as _; +use rusqlite::{params, OptionalExtension}; +use uuid::Uuid; + +use crate::connection::config::{DatabaseConfig, DurabilityMode}; +use crate::namespace::NamespaceName; +use crate::LIBSQL_PAGE_SIZE; +use libsql_replication::rpc::metadata; + +use super::command::TargetConfig; +use super::outcome::{FenceDetail, FenceError, FenceOutcome}; +use super::record::{ + CommandReceipt, FenceDecodeError, FenceMarker, LegacyBlocks, NamespaceFenceRecord, + FENCE_FORMAT_VERSION, +}; +use super::state::{FenceState, OperationClass}; +use super::transition::CurrentFence; + +/// Name of the per-namespace marker file, inside the namespace's directory. +pub const MARKER_FILE_NAME: &str = ".fence"; + +/// How long receipts of finished operations are kept by default (section 5.5). +pub const DEFAULT_RECEIPT_RETENTION: Duration = Duration::from_secs(30 * 24 * 60 * 60); + +const CREATE_FENCES_TABLE: &str = " + CREATE TABLE IF NOT EXISTS namespace_fences ( + namespace TEXT NOT NULL PRIMARY KEY, + format_version INTEGER NOT NULL, + revision INTEGER NOT NULL, + record BLOB NOT NULL, + FOREIGN KEY (namespace) REFERENCES namespace_configs (namespace) + ON DELETE RESTRICT ON UPDATE RESTRICT + )"; + +const CREATE_RECEIPTS_TABLE: &str = " + CREATE TABLE IF NOT EXISTS namespace_fence_receipts ( + namespace TEXT NOT NULL, + operation_id TEXT NOT NULL, + command_id TEXT NOT NULL, + format_version INTEGER NOT NULL, + revision_after INTEGER NOT NULL, + applied_at INTEGER NOT NULL, + receipt BLOB NOT NULL, + PRIMARY KEY (namespace, operation_id, command_id) + )"; + +/// Errors of the persistence layer. A fence outcome is a result the caller answers with; the +/// others are faults of the metastore or the filesystem. +#[derive(Debug, thiserror::Error)] +pub enum FenceStoreError { + #[error(transparent)] + Fence(#[from] FenceError), + #[error("metastore error: {0}")] + Sqlite(#[from] rusqlite::Error), + #[error("fence marker I/O error: {0}")] + Io(#[from] io::Error), +} + +/// Create the fence tables. Called when the fence is enabled; the tables are additive, and +/// the metastore of a server that never enabled the fence does not have them. +pub fn create_tables(conn: &rusqlite::Connection) -> rusqlite::Result<()> { + conn.execute(CREATE_FENCES_TABLE, ())?; + conn.execute(CREATE_RECEIPTS_TABLE, ())?; + Ok(()) +} + +/// Whether this metastore has ever held fence state. Once it has, fences are loaded and +/// enforced whether or not the fence is enabled (section 13.1). +pub fn tables_exist(conn: &rusqlite::Connection) -> rusqlite::Result { + let count: i64 = conn.query_row( + "SELECT count(*) FROM sqlite_master WHERE type = 'table' + AND name IN ('namespace_fences', 'namespace_fence_receipts')", + (), + |row| row.get(0), + )?; + Ok(count > 0) +} + +/// What the store established about one namespace's fence. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum StoredFence { + /// No fence record and no marker. `namespace_exists` is whether the namespace has a config + /// row: `UNFENCED` when it does, `ABSENT` when it does not. + None { + namespace_exists: bool, + }, + Record(NamespaceFenceRecord), + /// The control state cannot be established: `UNKNOWN_UNAVAILABLE`. + Unavailable { + detail: FenceDetail, + /// What was wrong, for the operator log and `InspectFence`. + reason: String, + /// The record the marker file holds, when it has a readable one. + marker: Option, + }, +} + +impl StoredFence { + pub fn as_current(&self) -> CurrentFence<'_> { + match self { + StoredFence::None { namespace_exists } => CurrentFence::None { + namespace_exists: *namespace_exists, + }, + StoredFence::Record(r) => CurrentFence::Record(r), + StoredFence::Unavailable { detail, marker, .. } => CurrentFence::Unavailable { + detail: *detail, + marker: marker.as_ref(), + }, + } + } + + pub fn state(&self) -> FenceState { + self.as_current().state() + } + + pub fn revision(&self) -> u64 { + self.as_current().revision() + } + + pub fn record(&self) -> Option<&NamespaceFenceRecord> { + match self { + StoredFence::Record(r) => Some(r), + _ => None, + } + } + + /// The permission-matrix decision for work of `class`, as an error a caller can return. + pub fn permits(&self, class: OperationClass) -> Result<(), FenceError> { + self.state().permits(class).map_err(|outcome| { + let err = FenceError::new(outcome, self.denial_message(class)); + match self { + StoredFence::Unavailable { detail, .. } => err.with_detail(*detail), + _ => err, + } + }) + } + + fn denial_message(&self, class: OperationClass) -> String { + match self { + StoredFence::Record(r) => format!( + "{class:?} is not permitted while the namespace fence is {} (operation {}, revision {})", + r.state, r.operation_id, r.revision + ), + StoredFence::Unavailable { reason, .. } => { + format!("the namespace's fence state cannot be established: {reason}") + } + StoredFence::None { .. } => format!("{class:?} is not permitted"), + } + } +} + +/// Where the marker of `namespace` lives, under the server's `dbs` directory. +pub fn marker_path(dbs_path: &Path, namespace: &NamespaceName) -> PathBuf { + dbs_path.join(namespace.as_str()).join(MARKER_FILE_NAME) +} + +/// Read a namespace's marker. `Ok(None)` when there is none; `Ok(Some(Err(_)))` when there is +/// one that cannot be decoded. +pub fn read_marker( + dbs_path: &Path, + namespace: &NamespaceName, +) -> io::Result>> { + match fs::read(marker_path(dbs_path, namespace)) { + Ok(bytes) => Ok(Some(FenceMarker::decode(&bytes))), + Err(e) if e.kind() == io::ErrorKind::NotFound => Ok(None), + Err(e) => Err(e), + } +} + +/// Durably replace a namespace's marker with a copy of `record`: write a temporary file, fsync +/// it, rename it over the marker and fsync the directory. Creates the namespace directory if +/// it does not exist yet (a target that is being created). +pub fn write_marker(dbs_path: &Path, record: &NamespaceFenceRecord) -> io::Result<()> { + let path = marker_path(dbs_path, &record.namespace); + let dir = path.parent().expect("marker path has a parent"); + fs::create_dir_all(dir)?; + let tmp = dir.join(format!("{MARKER_FILE_NAME}.tmp")); + { + let mut file = File::create(&tmp)?; + file.write_all(&FenceMarker::for_record(record).encode())?; + file.sync_all()?; + } + fs::rename(&tmp, &path)?; + File::open(dir)?.sync_all()?; + Ok(()) +} + +/// Remove a namespace's marker, if it has one, and fsync its directory. +pub fn remove_marker(dbs_path: &Path, namespace: &NamespaceName) -> io::Result<()> { + let path = marker_path(dbs_path, namespace); + match fs::remove_file(&path) { + Ok(()) => File::open(path.parent().expect("marker path has a parent"))?.sync_all(), + Err(e) if e.kind() == io::ErrorKind::NotFound => Ok(()), + Err(e) => Err(e), + } +} + +/// Whether the marker agrees with what the metastore says. Returned by [`read_fence`] so a +/// loader can repair a marker that fell behind (a crash between commit and marker write). +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum MarkerStatus { + /// The marker holds the stored record, or there is neither. + Current, + /// The metastore has a record and the marker is missing or older: rewrite it. + Stale, + /// The marker is what makes the namespace unavailable, or cannot be read. + Conflicting, +} + +struct RawFenceRow { + format_version: i64, + revision: i64, + record: Vec, +} + +fn read_raw_row( + conn: &rusqlite::Connection, + namespace: &NamespaceName, +) -> rusqlite::Result> { + conn.query_row( + "SELECT format_version, revision, record FROM namespace_fences WHERE namespace = ?1", + [namespace.as_str()], + |row| { + Ok(RawFenceRow { + format_version: row.get(0)?, + revision: row.get(1)?, + record: row.get(2)?, + }) + }, + ) + .optional() +} + +/// The stored revision column of `namespace`'s fence row, whatever its payload says. +pub fn stored_revision( + conn: &rusqlite::Connection, + namespace: &NamespaceName, +) -> rusqlite::Result> { + conn.query_row( + "SELECT revision FROM namespace_fences WHERE namespace = ?1", + [namespace.as_str()], + |row| row.get(0), + ) + .optional() +} + +fn decode_row(row: &RawFenceRow) -> Result { + let format_version = u32::try_from(row.format_version) + .map_err(|_| FenceDecodeError::UnsupportedFormatVersion(u32::MAX))?; + let revision = u64::try_from(row.revision) + .map_err(|_| FenceDecodeError::Invalid("negative revision column"))?; + NamespaceFenceRecord::decode(format_version, revision, &row.record) +} + +fn decode_detail(e: &FenceDecodeError) -> FenceDetail { + match e { + FenceDecodeError::UnsupportedFormatVersion(_) => FenceDetail::UnsupportedFormatVersion, + FenceDecodeError::Undecodable(_) | FenceDecodeError::Invalid(_) => { + FenceDetail::CorruptRecord + } + } +} + +/// Read the config row of `namespace`, if it has one. +pub fn read_config_row( + conn: &rusqlite::Connection, + namespace: &NamespaceName, +) -> Result, FenceStoreError> { + let bytes: Option> = conn + .query_row( + "SELECT config FROM namespace_configs WHERE namespace = ?1", + [namespace.as_str()], + |row| row.get(0), + ) + .optional()?; + match bytes { + None => Ok(None), + Some(bytes) => match metadata::DatabaseConfig::decode(&bytes[..]) { + Ok(c) => Ok(Some(DatabaseConfig::from(&c))), + Err(e) => Err(FenceError::new( + FenceOutcome::FenceStateUnavailable, + format!("the namespace's config row cannot be decoded: {e}"), + ) + .with_detail(FenceDetail::CorruptRecord) + .into()), + }, + } +} + +fn config_row_exists( + conn: &rusqlite::Connection, + namespace: &NamespaceName, +) -> rusqlite::Result { + conn.query_row( + "SELECT count(*) FROM namespace_configs WHERE namespace = ?1", + [namespace.as_str()], + |row| row.get::<_, i64>(0), + ) + .map(|n| n > 0) +} + +/// Establish the fence of `namespace` from its row and its marker (section 5.6). +/// +/// Only a metastore that has the fence tables can hold a record; `dbs_path` is where the +/// namespace directories, and so the markers, are. +pub fn read_fence( + conn: &rusqlite::Connection, + dbs_path: &Path, + namespace: &NamespaceName, +) -> Result<(StoredFence, MarkerStatus), FenceStoreError> { + let row = read_raw_row(conn, namespace)?; + let marker = read_marker(dbs_path, namespace)?; + + let marker_record = match &marker { + Some(Ok(m)) => Some(m.record.clone()), + _ => None, + }; + + let Some(row) = row else { + return Ok(match marker { + None => ( + StoredFence::None { + namespace_exists: config_row_exists(conn, namespace)?, + }, + MarkerStatus::Current, + ), + Some(Err(e)) => ( + StoredFence::Unavailable { + detail: FenceDetail::CorruptRecord, + reason: format!("the fence marker cannot be decoded: {e}"), + marker: None, + }, + MarkerStatus::Conflicting, + ), + Some(Ok(m)) => { + let incomplete_creation = m.record.state == FenceState::TargetQuarantined + && m.record.revision == 1 + && !config_row_exists(conn, namespace)?; + let (detail, reason) = if incomplete_creation { + ( + FenceDetail::IncompleteTargetCreation, + "a target creation was interrupted before its metastore commit".to_string(), + ) + } else { + ( + FenceDetail::MetastoreBehindMarker, + format!( + "the marker records revision {} but the metastore has no fence record", + m.record.revision + ), + ) + }; + ( + StoredFence::Unavailable { + detail, + reason, + marker: Some(m.record), + }, + MarkerStatus::Conflicting, + ) + } + }); + }; + + let record = match decode_row(&row) { + Ok(record) => record, + Err(e) => { + return Ok(( + StoredFence::Unavailable { + detail: decode_detail(&e), + reason: format!("the fence record cannot be read: {e}"), + marker: marker_record, + }, + MarkerStatus::Conflicting, + )) + } + }; + + if record.namespace != *namespace { + return Ok(( + StoredFence::Unavailable { + detail: FenceDetail::CorruptRecord, + reason: format!("the fence record names namespace `{}`", record.namespace), + marker: marker_record, + }, + MarkerStatus::Conflicting, + )); + } + + Ok(match marker { + Some(Ok(m)) if m.record.revision > record.revision => ( + StoredFence::Unavailable { + detail: FenceDetail::MetastoreBehindMarker, + reason: format!( + "the marker records revision {} but the metastore has revision {}", + m.record.revision, record.revision + ), + marker: Some(m.record), + }, + MarkerStatus::Conflicting, + ), + Some(Ok(m)) if m.record.revision == record.revision && m.record != record => ( + StoredFence::Unavailable { + detail: FenceDetail::MetastoreBehindMarker, + reason: format!( + "the marker and the metastore disagree at revision {}", + record.revision + ), + marker: Some(m.record), + }, + MarkerStatus::Conflicting, + ), + Some(Ok(m)) if m.record == record => (StoredFence::Record(record), MarkerStatus::Current), + // Missing, older, or unreadable while the metastore has a well-formed record: the + // metastore is authoritative and the marker is rewritten. + _ => (StoredFence::Record(record), MarkerStatus::Stale), + }) +} + +/// Look up the receipt for `(namespace, operation_id, command_id)`. +pub fn read_receipt( + conn: &rusqlite::Connection, + namespace: &NamespaceName, + operation_id: Uuid, + command_id: Uuid, +) -> Result>, rusqlite::Error> { + let row: Option<(i64, Vec)> = conn + .query_row( + "SELECT format_version, receipt FROM namespace_fence_receipts + WHERE namespace = ?1 AND operation_id = ?2 AND command_id = ?3", + params![ + namespace.as_str(), + operation_id.to_string(), + command_id.to_string() + ], + |row| Ok((row.get(0)?, row.get(1)?)), + ) + .optional()?; + Ok(row.map(|(v, bytes)| decode_receipt(v, &bytes, namespace, operation_id, command_id))) +} + +fn decode_receipt( + format_version: i64, + bytes: &[u8], + namespace: &NamespaceName, + operation_id: Uuid, + command_id: Uuid, +) -> Result { + let format_version = u32::try_from(format_version) + .map_err(|_| FenceDecodeError::UnsupportedFormatVersion(u32::MAX))?; + let receipt = CommandReceipt::decode(format_version, bytes)?; + if receipt.namespace != *namespace + || receipt.operation_id != operation_id + || receipt.command_id != command_id + { + return Err(FenceDecodeError::Invalid("receipt disagrees with its key")); + } + Ok(receipt) +} + +/// One stored receipt, as read for inspection. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct StoredReceipt { + pub operation_id: String, + pub command_id: String, + pub revision_after: i64, + pub applied_at_ms: i64, + pub receipt: Result, +} + +/// Every receipt of `namespace`, oldest first. +pub fn read_receipts( + conn: &rusqlite::Connection, + namespace: &NamespaceName, +) -> rusqlite::Result> { + let mut stmt = conn.prepare( + "SELECT operation_id, command_id, format_version, revision_after, applied_at, receipt + FROM namespace_fence_receipts WHERE namespace = ?1 + ORDER BY applied_at, revision_after, operation_id, command_id", + )?; + let rows = stmt.query_map([namespace.as_str()], |row| { + Ok(( + row.get::<_, String>(0)?, + row.get::<_, String>(1)?, + row.get::<_, i64>(2)?, + row.get::<_, i64>(3)?, + row.get::<_, i64>(4)?, + row.get::<_, Vec>(5)?, + )) + })?; + let mut out = Vec::new(); + for row in rows { + let (op, cmd, version, revision_after, applied_at, bytes) = row?; + let receipt = match (Uuid::parse_str(&op), Uuid::parse_str(&cmd)) { + (Ok(op_id), Ok(cmd_id)) => decode_receipt(version, &bytes, namespace, op_id, cmd_id), + _ => Err(FenceDecodeError::Invalid("receipt key is not a uuid")), + }; + out.push(StoredReceipt { + operation_id: op, + command_id: cmd, + revision_after, + applied_at_ms: applied_at, + receipt, + }); + } + Ok(out) +} + +/// Compare-and-swap the fence row of `record.namespace`: it must currently have revision +/// `previous` (`None`: no row). The caller holds the write lock, so this only fails if the +/// caller read something other than what is stored, which is a bug; it is still checked. +pub fn write_record( + conn: &rusqlite::Connection, + record: &NamespaceFenceRecord, + previous: Option, +) -> Result<(), FenceStoreError> { + let revision = i64::try_from(record.revision).expect("revision fits in i64"); + let bytes = record.encode(); + let changed = match previous { + None => conn.execute( + "INSERT INTO namespace_fences (namespace, format_version, revision, record) + VALUES (?1, ?2, ?3, ?4)", + params![ + record.namespace.as_str(), + FENCE_FORMAT_VERSION, + revision, + bytes + ], + )?, + Some(previous) => conn.execute( + "UPDATE namespace_fences SET format_version = ?2, revision = ?3, record = ?4 + WHERE namespace = ?1 AND revision = ?5", + params![ + record.namespace.as_str(), + FENCE_FORMAT_VERSION, + revision, + bytes, + previous + ], + )?, + }; + if changed != 1 { + return Err(FenceError::new( + FenceOutcome::FenceRevisionMismatch, + "the stored fence record changed under the transition", + ) + .into()); + } + Ok(()) +} + +/// Store `receipt`, replacing any receipt with the same key (a finished drain replaces its +/// `DRAINING` receipt). +pub fn write_receipt( + conn: &rusqlite::Connection, + receipt: &CommandReceipt, +) -> rusqlite::Result<()> { + conn.execute( + "INSERT OR REPLACE INTO namespace_fence_receipts + (namespace, operation_id, command_id, format_version, revision_after, applied_at, receipt) + VALUES (?1, ?2, ?3, ?4, ?5, ?6, ?7)", + params![ + receipt.namespace.as_str(), + receipt.operation_id.to_string(), + receipt.command_id.to_string(), + FENCE_FORMAT_VERSION, + i64::try_from(receipt.revision_after).expect("revision fits in i64"), + receipt.applied_at_ms, + receipt.encode(), + ], + )?; + Ok(()) +} + +/// Prune receipts of operations other than `owner` that are older than `retention` at +/// `now_ms` (section 5.5). The owner's receipts are never pruned. +pub fn prune_receipts( + conn: &rusqlite::Connection, + namespace: &NamespaceName, + owner: Uuid, + now_ms: i64, + retention: Duration, +) -> rusqlite::Result { + let cutoff = now_ms.saturating_sub(i64::try_from(retention.as_millis()).unwrap_or(i64::MAX)); + conn.execute( + "DELETE FROM namespace_fence_receipts + WHERE namespace = ?1 AND operation_id != ?2 AND applied_at < ?3", + params![namespace.as_str(), owner.to_string(), cutoff], + ) +} + +/// Remove every trace of `namespace`'s fence, for a delete of a namespace whose fence permits +/// it. Returns the number of receipts removed. +pub fn delete_fence( + conn: &rusqlite::Connection, + namespace: &NamespaceName, +) -> rusqlite::Result { + conn.execute( + "DELETE FROM namespace_fences WHERE namespace = ?1", + [namespace.as_str()], + )?; + conn.execute( + "DELETE FROM namespace_fence_receipts WHERE namespace = ?1", + [namespace.as_str()], + ) +} + +/// The config as it is stored while `blocks` is the legacy mirror: `config` with its +/// `block_*` fields replaced (section 13.2). +pub fn with_legacy_blocks(config: &DatabaseConfig, blocks: &LegacyBlocks) -> DatabaseConfig { + DatabaseConfig { + block_reads: blocks.block_reads, + block_writes: blocks.block_writes, + block_reason: blocks.block_reason.clone(), + ..config.clone() + } +} + +/// The `block_*` fields of `config`. +pub fn legacy_blocks_of(config: &DatabaseConfig) -> LegacyBlocks { + LegacyBlocks { + block_reads: config.block_reads, + block_writes: config.block_writes, + block_reason: config.block_reason.clone(), + } +} + +/// Upsert the config row of `namespace`. +pub fn write_config_row( + conn: &rusqlite::Connection, + namespace: &NamespaceName, + config: &DatabaseConfig, +) -> rusqlite::Result<()> { + let encoded = metadata::DatabaseConfig::from(config).encode_to_vec(); + conn.execute( + "INSERT INTO namespace_configs (namespace, config) VALUES (?1, ?2) + ON CONFLICT(namespace) DO UPDATE SET config = excluded.config", + params![namespace.as_str(), encoded], + )?; + Ok(()) +} + +/// The namespace config a target is created with, before the legacy mirror is applied. +pub fn target_database_config(config: &TargetConfig) -> Result { + let invalid = |message: String| { + FenceError::new(FenceOutcome::FencePreconditionFailed, message) + .with_detail(FenceDetail::InvalidArgument) + }; + let mut out = DatabaseConfig::default(); + if let Some(bytes) = config.max_db_size { + out.max_db_pages = bytes / LIBSQL_PAGE_SIZE; + } + out.jwt_key = config.jwt_key.clone(); + if let Some(s) = config.txn_timeout_s { + out.txn_timeout = Some(Duration::from_secs(s)); + } + out.allow_attach = config.allow_attach; + if let Some(mode) = &config.durability_mode { + out.durability_mode = mode + .parse::() + .map_err(|()| invalid(format!("unknown durability mode `{mode}`")))?; + } + out.bottomless_db_id = config.bottomless_db_id.clone(); + Ok(out) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::namespace::fence::record::tests as samples; + + fn conn() -> rusqlite::Connection { + let conn = rusqlite::Connection::open_in_memory().unwrap(); + conn.execute("PRAGMA foreign_keys=ON", ()).unwrap(); + conn.execute( + "CREATE TABLE namespace_configs (namespace TEXT NOT NULL PRIMARY KEY, config BLOB NOT NULL)", + (), + ) + .unwrap(); + create_tables(&conn).unwrap(); + conn + } + + fn sample() -> NamespaceFenceRecord { + samples::sample_record() + } + + #[test] + fn tables_are_additive_and_detectable() { + let conn = rusqlite::Connection::open_in_memory().unwrap(); + assert!(!tables_exist(&conn).unwrap()); + create_tables(&conn).unwrap(); + assert!(tables_exist(&conn).unwrap()); + // Idempotent. + create_tables(&conn).unwrap(); + } + + #[test] + fn record_round_trips_and_cas_checks_revision() { + let dir = tempfile::tempdir().unwrap(); + let conn = conn(); + let record = sample(); + write_config_row(&conn, &record.namespace, &DatabaseConfig::default()).unwrap(); + write_record(&conn, &record, None).unwrap(); + // A second insert, or an update from the wrong revision, is refused. + assert!(write_record(&conn, &record, None).is_err()); + let mut next = record.clone(); + next.revision += 1; + let err = write_record(&conn, &next, Some(1)).unwrap_err(); + assert!( + matches!(err, FenceStoreError::Fence(e) if e.outcome() == FenceOutcome::FenceRevisionMismatch) + ); + write_record(&conn, &next, Some(2)).unwrap(); + + let (stored, marker) = read_fence(&conn, dir.path(), &record.namespace).unwrap(); + assert_eq!(stored, StoredFence::Record(next)); + assert_eq!(marker, MarkerStatus::Stale); + } + + #[test] + fn fence_row_needs_a_config_row() { + let conn = conn(); + let err = write_record(&conn, &sample(), None).unwrap_err(); + assert!(matches!(err, FenceStoreError::Sqlite(_)), "{err:?}"); + } + + #[test] + fn delete_of_config_row_is_restricted_by_the_fence_row() { + let conn = conn(); + let record = sample(); + write_config_row(&conn, &record.namespace, &DatabaseConfig::default()).unwrap(); + write_record(&conn, &record, None).unwrap(); + assert!(conn + .execute( + "DELETE FROM namespace_configs WHERE namespace = ?1", + [record.namespace.as_str()] + ) + .is_err()); + delete_fence(&conn, &record.namespace).unwrap(); + conn.execute( + "DELETE FROM namespace_configs WHERE namespace = ?1", + [record.namespace.as_str()], + ) + .unwrap(); + } + + #[test] + fn marker_round_trips_and_is_compared() { + let dir = tempfile::tempdir().unwrap(); + let conn = conn(); + let record = sample(); + write_config_row(&conn, &record.namespace, &DatabaseConfig::default()).unwrap(); + write_record(&conn, &record, None).unwrap(); + write_marker(dir.path(), &record).unwrap(); + assert!(!dir + .path() + .join(record.namespace.as_str()) + .join(".fence.tmp") + .exists()); + let (stored, status) = read_fence(&conn, dir.path(), &record.namespace).unwrap(); + assert_eq!(stored, StoredFence::Record(record.clone())); + assert_eq!(status, MarkerStatus::Current); + + // A marker ahead of the metastore: the metastore went backwards. + let mut ahead = record.clone(); + ahead.revision += 1; + write_marker(dir.path(), &ahead).unwrap(); + let (stored, status) = read_fence(&conn, dir.path(), &record.namespace).unwrap(); + assert!(matches!( + stored, + StoredFence::Unavailable { detail: FenceDetail::MetastoreBehindMarker, marker: Some(ref m), .. } if *m == ahead + )); + assert_eq!(status, MarkerStatus::Conflicting); + + // Same revision, different contents. + let mut forked = record.clone(); + forked.operation_id = Uuid::from_u128(77); + write_marker(dir.path(), &forked).unwrap(); + let (stored, _) = read_fence(&conn, dir.path(), &record.namespace).unwrap(); + assert_eq!(stored.state(), FenceState::UnknownUnavailable); + + // An undecodable marker beside a good record: the record wins and the marker is stale. + fs::write(marker_path(dir.path(), &record.namespace), b"garbage").unwrap(); + let (stored, status) = read_fence(&conn, dir.path(), &record.namespace).unwrap(); + assert_eq!(stored, StoredFence::Record(record.clone())); + assert_eq!(status, MarkerStatus::Stale); + + // Marker without a row. + delete_fence(&conn, &record.namespace).unwrap(); + write_marker(dir.path(), &record).unwrap(); + let (stored, _) = read_fence(&conn, dir.path(), &record.namespace).unwrap(); + assert!(matches!( + stored, + StoredFence::Unavailable { + detail: FenceDetail::MetastoreBehindMarker, + .. + } + )); + fs::write(marker_path(dir.path(), &record.namespace), b"garbage").unwrap(); + let (stored, _) = read_fence(&conn, dir.path(), &record.namespace).unwrap(); + assert!(matches!( + stored, + StoredFence::Unavailable { + detail: FenceDetail::CorruptRecord, + marker: None, + .. + } + )); + } + + #[test] + fn corrupt_rows_are_unavailable() { + let dir = tempfile::tempdir().unwrap(); + let conn = conn(); + let record = sample(); + let ns = record.namespace.clone(); + write_config_row(&conn, &ns, &DatabaseConfig::default()).unwrap(); + write_record(&conn, &record, None).unwrap(); + + let set = |sql: &str| { + conn.execute(sql, [ns.as_str()]).unwrap(); + }; + let detail = || match read_fence(&conn, dir.path(), &ns).unwrap().0 { + StoredFence::Unavailable { detail, .. } => detail, + other => panic!("expected unavailable, got {other:?}"), + }; + + set("UPDATE namespace_fences SET format_version = 2 WHERE namespace = ?1"); + assert_eq!(detail(), FenceDetail::UnsupportedFormatVersion); + set("UPDATE namespace_fences SET format_version = 1, revision = 3 WHERE namespace = ?1"); + assert_eq!(detail(), FenceDetail::CorruptRecord); + set("UPDATE namespace_fences SET revision = 2, record = x'ffff' WHERE namespace = ?1"); + assert_eq!(detail(), FenceDetail::CorruptRecord); + set("UPDATE namespace_fences SET revision = -1 WHERE namespace = ?1"); + assert_eq!(detail(), FenceDetail::CorruptRecord); + + let stored = read_fence(&conn, dir.path(), &ns).unwrap().0; + for class in OperationClass::ALL { + let r = stored.permits(class); + match class { + OperationClass::Maintenance | OperationClass::Observability => assert!(r.is_ok()), + _ => { + let e = r.unwrap_err(); + assert_eq!(e.outcome(), FenceOutcome::FenceStateUnavailable); + assert_eq!(e.detail(), Some(FenceDetail::CorruptRecord)); + } + } + } + } + + #[test] + fn receipts_round_trip_and_prune() { + let conn = conn(); + let record = sample(); + let ns = record.namespace.clone(); + let base = samples::sample_receipt(); + let owner = base.operation_id; + let other = Uuid::from_u128(0xbeef); + + let mut r_owner_old = base.clone(); + r_owner_old.applied_at_ms = 10; + let mut r_other_old = base.clone(); + r_other_old.operation_id = other; + r_other_old.applied_at_ms = 10; + let mut r_other_new = base.clone(); + r_other_new.operation_id = other; + r_other_new.command_id = Uuid::from_u128(0xc0de); + r_other_new.applied_at_ms = 5_000; + for r in [&r_owner_old, &r_other_old, &r_other_new] { + write_receipt(&conn, r).unwrap(); + } + assert_eq!( + read_receipt(&conn, &ns, owner, base.command_id) + .unwrap() + .unwrap() + .unwrap(), + r_owner_old + ); + assert!(read_receipt(&conn, &ns, owner, Uuid::from_u128(1234)) + .unwrap() + .is_none()); + + // Retention 1s at t=6s: only the other operation's old receipt goes. + let pruned = prune_receipts(&conn, &ns, owner, 6_000, Duration::from_secs(1)).unwrap(); + assert_eq!(pruned, 1); + let left: Vec<_> = read_receipts(&conn, &ns) + .unwrap() + .into_iter() + .map(|r| r.receipt.unwrap()) + .collect(); + assert_eq!(left, vec![r_owner_old.clone(), r_other_new]); + + // A receipt stored under the wrong key reads as corrupt. + conn.execute( + "UPDATE namespace_fence_receipts SET command_id = ?1 WHERE operation_id = ?2", + params![Uuid::from_u128(4321).to_string(), owner.to_string()], + ) + .unwrap(); + assert!(read_receipt(&conn, &ns, owner, Uuid::from_u128(4321)) + .unwrap() + .unwrap() + .is_err()); + } + + #[test] + fn target_config_conversion() { + let c = target_database_config(&TargetConfig { + max_db_size: Some(4096 * 10), + jwt_key: Some("k".into()), + txn_timeout_s: Some(7), + allow_attach: true, + durability_mode: Some("strong".into()), + bottomless_db_id: Some("b".into()), + }) + .unwrap(); + assert_eq!(c.max_db_pages, 10); + assert_eq!(c.jwt_key.as_deref(), Some("k")); + assert_eq!(c.txn_timeout, Some(Duration::from_secs(7))); + assert!(c.allow_attach); + assert_eq!(c.durability_mode, DurabilityMode::Strong); + assert_eq!(c.bottomless_db_id.as_deref(), Some("b")); + assert!(!c.block_reads && !c.block_writes); + + let e = target_database_config(&TargetConfig { + durability_mode: Some("nope".into()), + ..Default::default() + }) + .unwrap_err(); + assert_eq!(e.detail(), Some(FenceDetail::InvalidArgument)); + } +} diff --git a/libsql-server/src/namespace/meta_store.rs b/libsql-server/src/namespace/meta_store.rs index 70b419ebe9..098102e3f7 100644 --- a/libsql-server/src/namespace/meta_store.rs +++ b/libsql-server/src/namespace/meta_store.rs @@ -1,6 +1,7 @@ #![allow(clippy::mutable_key_type)] -use std::path::Path; +use std::path::{Path, PathBuf}; use std::sync::Arc; +use std::time::Duration; use std::{collections::HashMap, fs::read_dir}; use bottomless::bottomless_wal::BottomlessWalWrapper; @@ -14,11 +15,13 @@ use libsql_sys::wal::{ }; use parking_lot::Mutex; use prost::Message; +use rusqlite::TransactionBehavior; use tokio::sync::oneshot; use tokio::sync::{ mpsc, watch::{self, Receiver, Sender}, }; +use uuid::Uuid; use crate::config::BottomlessConfig; use crate::connection::config::DatabaseConfig; @@ -28,6 +31,16 @@ use crate::{ config::MetaStoreConfig, connection::legacy::open_conn_active_checkpoint, error::Error, Result, }; +use super::fence::command::{FenceCommand, FenceRequest}; +use super::fence::outcome::{FenceDetail, FenceError, FenceOutcome}; +use super::fence::record::{ + CommandReceipt, NamespaceFenceRecord, ServerIdentity, ValidationSnapshot, +}; +use super::fence::state::OperationClass; +use super::fence::store::{ + self as fence_store, FenceStoreError, MarkerStatus, StoredFence, StoredReceipt, +}; +use super::fence::transition::{self, ApplyEnv, Decision, DrainCompletion}; use super::NamespaceName; type ChangeMsg = ( @@ -75,6 +88,19 @@ struct MetaStoreInner { conn: tokio::sync::Mutex, wal_manager: MetaStoreWalManager, db_kind: DatabaseKind, + /// `/dbs`, where namespace directories and their fence markers are. + dbs_path: PathBuf, + fence: FenceSettings, +} + +/// How this metastore treats namespace fences (`docs/NAMESPACE_FENCE.md` section 13.1). +#[derive(Debug, Clone, Copy)] +struct FenceSettings { + /// The fence may be used: its tables exist and commands are accepted. + enabled: bool, + /// The fence tables exist, so fence state is loaded and enforced. + tables: bool, + receipt_retention: Duration, } fn setup_connection(conn: &rusqlite::Connection) -> Result<()> { @@ -187,12 +213,24 @@ impl MetaStoreInner { db_kind: DatabaseKind, ) -> Result { setup_connection(&conn)?; + if config.namespace_fence { + fence_store::create_tables(&conn)?; + } + let fence = FenceSettings { + enabled: config.namespace_fence, + tables: fence_store::tables_exist(&conn)?, + receipt_retention: config + .namespace_fence_receipt_retention + .unwrap_or(fence_store::DEFAULT_RECEIPT_RETENTION), + }; let mut this = MetaStoreInner { configs: Default::default(), conn: conn.into(), wal_manager, db_kind, + dbs_path: base_path.join("dbs"), + fence, }; if config.allow_recover_from_fs { @@ -200,6 +238,9 @@ impl MetaStoreInner { } this.restore()?; + if this.fence.tables { + this.restore_fences()?; + } Ok(this) } @@ -301,6 +342,53 @@ impl MetaStoreInner { Ok(()) } + + /// Load every namespace's fence after the configs (section 5.6). The stored config row of + /// a fenced namespace carries the legacy mirror of the fence in its `block_*` fields + /// (section 13.2); the in-memory config is the namespace's own configuration, so those + /// fields are put back to the values the record saved. A marker that fell behind its + /// record is rewritten. A namespace whose fence cannot be established is logged and keeps + /// its stored config, mirror included. + fn restore_fences(&mut self) -> Result<()> { + let namespaces: Vec = self.configs.get_mut().keys().cloned().collect(); + let conn = self.conn.get_mut(); + let mut fenced = 0usize; + for ns in namespaces { + let (stored, marker) = match fence_store::read_fence(conn, &self.dbs_path, &ns) { + Ok(r) => r, + Err(FenceStoreError::Sqlite(e)) => return Err(e.into()), + Err(e) => { + tracing::error!(namespace = %ns, "cannot establish namespace fence: {e}"); + continue; + } + }; + match &stored { + StoredFence::None { .. } => continue, + StoredFence::Record(record) => { + fenced += 1; + if marker == MarkerStatus::Stale { + if let Err(e) = fence_store::write_marker(&self.dbs_path, record) { + tracing::error!(namespace = %ns, "failed to rewrite fence marker: {e}"); + } + } + let sender = self.configs.get_mut().get_mut(&ns).expect("listed above"); + let config = sender.borrow().config.clone(); + let config = fence_store::with_legacy_blocks(&config, &record.legacy_blocks); + sender.send_modify(|c| c.config = Arc::new(config)); + } + StoredFence::Unavailable { detail, reason, .. } => { + fenced += 1; + tracing::error!( + namespace = %ns, + %detail, + "namespace fence state is UNKNOWN_UNAVAILABLE: {reason}" + ); + } + } + } + tracing::info!("loaded {fenced} namespace fence(s)"); + Ok(()) + } } /// Handles config change updates by inserting them into the database and in-memory @@ -313,6 +401,11 @@ fn process(msg: ChangeMsg, inner: Arc) { } else { Ok(()) }; + // A config that was not persisted is not published. + if ret.is_err() { + let _ = ret_chan.send(ret); + return; + } let mut configs = inner.configs.blocking_lock(); if let Some(config_watch) = configs.get_mut(&namespace) { let new_version = config_watch.borrow().version.wrapping_add(1); @@ -349,22 +442,22 @@ fn try_process( namespace: &NamespaceName, config: &DatabaseConfig, ) -> Result<()> { - let config_encoded = metadata::DatabaseConfig::from(&*config).encode_to_vec(); - let mut conn = inner.conn.blocking_lock(); + // `BEGIN IMMEDIATE`: the write lock is what serialises this write with fence transitions + // (docs/NAMESPACE_FENCE.md section 5.4), including those of other metastore connections. + let tx = conn.transaction_with_behavior(TransactionBehavior::Immediate)?; + if inner.fence.tables { + let (stored, _) = + fence_store::read_fence(&tx, &inner.dbs_path, namespace).map_err(fence_store_error)?; + stored.permits(OperationClass::Lifecycle)?; + } if let Some(schema) = config.shared_schema_name.as_ref() { - let tx = conn.transaction()?; if inner.db_kind.is_primary() { - if let Some(ref schema) = config.shared_schema_name { - if crate::schema::db::has_pending_migration_jobs(&tx, schema)? { - return Err(crate::Error::PendingMigrationOnSchema(schema.clone())); - } + if crate::schema::db::has_pending_migration_jobs(&tx, schema)? { + return Err(crate::Error::PendingMigrationOnSchema(schema.clone())); } } - tx.execute( - "INSERT INTO namespace_configs (namespace, config) VALUES (?1, ?2) ON CONFLICT(namespace) DO UPDATE SET config=excluded.config", - rusqlite::params![namespace.as_str(), config_encoded], - )?; + fence_store::write_config_row(&tx, namespace, config)?; tx.execute( "DELETE FROM shared_schema_links WHERE namespace = ?", rusqlite::params![namespace.as_str()], @@ -373,13 +466,10 @@ fn try_process( "INSERT OR REPLACE INTO shared_schema_links (shared_schema_name, namespace) VALUES (?1, ?2)", rusqlite::params![schema.as_str(), namespace.as_str()], )?; - tx.commit()?; } else { - conn.execute( - "INSERT INTO namespace_configs (namespace, config) VALUES (?1, ?2) ON CONFLICT(namespace) DO UPDATE SET config=excluded.config", - rusqlite::params![namespace.as_str(), config_encoded], - )?; + fence_store::write_config_row(&tx, namespace, config)?; } + tx.commit()?; if let Err(e) = checkpoint(&conn) { tracing::warn!("failed to checkpoint metastore: {e}"); @@ -388,11 +478,335 @@ fn try_process( Ok(()) } +fn fence_store_error(e: FenceStoreError) -> Error { + match e { + FenceStoreError::Fence(e) => Error::NamespaceFence(e), + FenceStoreError::Sqlite(e) => Error::RusqliteError(e), + FenceStoreError::Io(e) => Error::IOError(e), + } +} + fn checkpoint(conn: &rusqlite::Connection) -> Result<()> { conn.query_row("PRAGMA wal_checkpoint(TRUNCATE)", (), |_| Ok(()))?; Ok(()) } +/// Facts about the server and the live namespace that a fence command needs and the metastore +/// does not hold (see [`ApplyEnv`]). The store adds what it reads inside the transaction. +#[derive(Debug, Clone)] +pub struct FenceContext { + pub server: ServerIdentity, + /// Wall-clock time in milliseconds since the Unix epoch. + pub now_ms: i64, + /// The namespace's current replication log id, if it exists and has one. + pub namespace_log_id: Option, + /// A fresh id for `CreateTargetQuarantined`. + pub new_incarnation_id: Uuid, + /// Whether the request carried the configured adoption key. + pub adoption_authorised: bool, + /// What the server observed of a target, for `RecordTargetValidation`. + pub validation_snapshot: Option, +} + +impl FenceContext { + /// A context for `server` at the current time with a fresh incarnation id. + pub fn now(server: ServerIdentity, namespace_log_id: Option) -> Self { + let now_ms = std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .map_or(0, |d| i64::try_from(d.as_millis()).unwrap_or(i64::MAX)); + Self { + server, + now_ms, + namespace_log_id, + new_incarnation_id: Uuid::new_v4(), + adoption_authorised: false, + validation_snapshot: None, + } + } +} + +/// How a fence command was answered. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum FenceCommitKind { + /// The command had been applied before; nothing was written. + Replayed, + /// The command started a drain that is still to be completed; nothing was written and the + /// controller resumes the drain. + Resumed, + /// The command's receipt (and, where it changed, the record) was committed. + Committed, +} + +/// The committed result of a fence command. +#[derive(Debug, Clone)] +pub struct FenceCommit { + pub kind: FenceCommitKind, + pub receipt: CommandReceipt, + /// The namespace's fence record after the command (for a replay, as it is now). + pub record: Option, + /// For a `CreateTargetQuarantined` that was committed, the namespace config it created. + /// It is not in the in-memory config map yet: the caller installs the target's gate first + /// and then publishes it. + pub created_config: Option>, +} + +/// A namespace's fence as read by `inspect_fence`. +#[derive(Debug, Clone)] +pub struct FenceInspection { + pub fence: StoredFence, + pub receipts: Vec, +} + +fn fence_disabled() -> FenceError { + FenceError::new( + FenceOutcome::FencePreconditionFailed, + "namespace fences are not enabled on this server", + ) + .with_detail(FenceDetail::FenceDisabled) +} + +fn not_primary() -> FenceError { + FenceError::new( + FenceOutcome::FencePreconditionFailed, + "namespace fences are only changed on a primary", + ) + .with_detail(FenceDetail::NotPrimary) +} + +fn unavailable_receipt(e: impl std::fmt::Display) -> FenceError { + FenceError::new( + FenceOutcome::FenceStateUnavailable, + format!("the stored receipt for this command cannot be read: {e}"), + ) + .with_detail(FenceDetail::CorruptRecord) +} + +/// The `ApplyEnv` for a command: the caller's context plus what the transaction read. +fn apply_env( + ctx: &FenceContext, + stored: &StoredFence, + config: Option<&DatabaseConfig>, +) -> ApplyEnv { + ApplyEnv { + now_ms: ctx.now_ms, + server: ctx.server.clone(), + namespace_log_id: ctx.namespace_log_id, + shared_schema: config.is_some_and(|c| c.is_shared_schema || c.shared_schema_name.is_some()), + // The namespace's own values: a stored record's saved values while it is in force, + // otherwise the config row as stored. + legacy_blocks: match stored { + StoredFence::Record(r) if !r.state.is_operation_finished() => r.legacy_blocks.clone(), + _ => config + .map(fence_store::legacy_blocks_of) + .unwrap_or_default(), + }, + new_incarnation_id: ctx.new_incarnation_id, + adoption_authorised: ctx.adoption_authorised, + validation_snapshot: ctx.validation_snapshot, + } +} + +fn apply_fence_command( + inner: &MetaStoreInner, + request: &FenceRequest, + ctx: &FenceContext, +) -> std::result::Result { + if !inner.fence.enabled || !inner.fence.tables { + return Err(fence_disabled().into()); + } + if !inner.db_kind.is_primary() { + return Err(not_primary().into()); + } + let ns = &request.namespace; + let mut conn = inner.conn.blocking_lock(); + let tx = conn.transaction_with_behavior(TransactionBehavior::Immediate)?; + + let (stored, marker) = fence_store::read_fence(&tx, &inner.dbs_path, ns)?; + let existing = + match fence_store::read_receipt(&tx, ns, request.operation_id, request.command_id)? { + None => None, + Some(Ok(r)) => Some(r), + Some(Err(e)) => return Err(unavailable_receipt(e).into()), + }; + let config = fence_store::read_config_row(&tx, ns)?; + let env = apply_env(ctx, &stored, config.as_ref()); + + let decision = transition::apply(stored.as_current(), existing.as_ref(), request, &env)?; + let (record, receipt) = match decision { + Decision::Replay(receipt) | Decision::Resume(receipt) => { + let kind = if receipt.is_final() { + FenceCommitKind::Replayed + } else { + FenceCommitKind::Resumed + }; + return Ok(FenceCommit { + kind, + receipt, + record: stored.record().cloned(), + created_config: None, + }); + } + Decision::Apply { record, receipt } => (record, receipt), + }; + + let mut created_config = None; + if let Some(next) = &record { + let previous = fence_store::stored_revision(&tx, ns)?; + if let FenceCommand::CreateTargetQuarantined { config: target } = &request.command { + // Section 10.1: the marker first, then the config row, the record and the receipt + // in one transaction. A crash in between leaves a marker without rows, which only a + // replay of this command completes. + let logical = fence_store::target_database_config(target)?; + fence_store::write_marker(&inner.dbs_path, next)?; + fence_store::write_config_row( + &tx, + ns, + &fence_store::with_legacy_blocks(&logical, &next.legacy_mirror()), + )?; + created_config = Some(Arc::new(logical)); + } else { + let Some(config) = &config else { + return Err(FenceError::new( + FenceOutcome::FenceStateUnavailable, + "the fenced namespace has no config row", + ) + .with_detail(FenceDetail::CorruptRecord) + .into()); + }; + fence_store::write_config_row( + &tx, + ns, + &fence_store::with_legacy_blocks(config, &next.legacy_mirror()), + )?; + } + fence_store::write_record(&tx, next, previous)?; + } + fence_store::write_receipt(&tx, &receipt)?; + let owner = record + .as_ref() + .or(stored.record()) + .map_or(request.operation_id, |r| r.operation_id); + fence_store::prune_receipts(&tx, ns, owner, ctx.now_ms, inner.fence.receipt_retention)?; + tx.commit()?; + + let current = record.or_else(|| stored.record().cloned()); + after_fence_commit( + inner, + &conn, + current.as_ref(), + record_changed(¤t, &stored, marker), + ); + + Ok(FenceCommit { + kind: FenceCommitKind::Committed, + receipt, + record: current, + created_config, + }) +} + +/// Whether the marker has to be written after a commit: the record changed, or it had fallen +/// behind. +fn record_changed( + current: &Option, + stored: &StoredFence, + marker: MarkerStatus, +) -> bool { + current.as_ref() != stored.record() || marker == MarkerStatus::Stale +} + +fn after_fence_commit( + inner: &MetaStoreInner, + conn: &rusqlite::Connection, + record: Option<&NamespaceFenceRecord>, + write_marker: bool, +) { + if let (Some(record), true) = (record, write_marker) { + // The metastore is authoritative; a marker that is missing or behind is repaired on + // the next load (section 5.6). + if let Err(e) = fence_store::write_marker(&inner.dbs_path, record) { + tracing::error!(namespace = %record.namespace, "failed to write fence marker: {e}"); + } + } + if let Err(e) = checkpoint(conn) { + tracing::warn!("failed to checkpoint metastore: {e}"); + } +} + +fn complete_fence_drain( + inner: &MetaStoreInner, + ns: &NamespaceName, + operation_id: Uuid, + command_id: Uuid, + completion: DrainCompletion, + ctx: &FenceContext, +) -> std::result::Result { + if !inner.fence.tables { + return Err(fence_disabled().into()); + } + let mut conn = inner.conn.blocking_lock(); + let tx = conn.transaction_with_behavior(TransactionBehavior::Immediate)?; + + let (stored, _) = fence_store::read_fence(&tx, &inner.dbs_path, ns)?; + let record = match &stored { + StoredFence::Record(r) => r.clone(), + other => { + return Err(other + .permits(OperationClass::NormalWrite) + .err() + .unwrap_or_else(|| { + FenceError::new( + FenceOutcome::InvalidFenceTransition, + "the namespace has no fence record", + ) + }) + .into()) + } + }; + let receipt = match fence_store::read_receipt(&tx, ns, operation_id, command_id)? { + Some(Ok(r)) => r, + Some(Err(e)) => return Err(unavailable_receipt(e).into()), + None => { + return Err(FenceError::new( + FenceOutcome::InvalidFenceTransition, + format!("command {command_id} of operation {operation_id} has no receipt"), + ) + .into()) + } + }; + if receipt.is_final() { + return Ok(FenceCommit { + kind: FenceCommitKind::Replayed, + receipt, + record: Some(record), + created_config: None, + }); + } + + let config = fence_store::read_config_row(&tx, ns)?; + let env = apply_env(ctx, &stored, config.as_ref()); + let (next, final_receipt) = transition::complete_drain(&record, &receipt, completion, &env)?; + if let Some(config) = &config { + fence_store::write_config_row( + &tx, + ns, + &fence_store::with_legacy_blocks(config, &next.legacy_mirror()), + )?; + } + fence_store::write_record(&tx, &next, fence_store::stored_revision(&tx, ns)?)?; + fence_store::write_receipt(&tx, &final_receipt)?; + tx.commit()?; + + after_fence_commit(inner, &conn, Some(&next), true); + + Ok(FenceCommit { + kind: FenceCommitKind::Committed, + receipt: final_receipt, + record: Some(next), + created_config: None, + }) +} + impl MetaStore { #[tracing::instrument(skip(config, base_path, conn, wal_manager))] pub async fn new( @@ -523,7 +937,26 @@ impl MetaStore { let r = if let Some(sender) = configs.get(&namespace) { tracing::debug!("removed namespace `{}` from meta store", namespace); let config = sender.borrow().clone(); - let tx = conn.transaction()?; + let tx = conn.transaction_with_behavior(TransactionBehavior::Immediate)?; + if self.inner.fence.tables { + let (stored, _) = fence_store::read_fence(&tx, &self.inner.dbs_path, &namespace) + .map_err(fence_store_error)?; + stored.permits(OperationClass::Lifecycle)?; + if !matches!(stored, StoredFence::None { .. }) { + // The marker goes before the commit: a crash in between leaves a record + // without a marker, which is repaired on load, rather than a marker + // without a record, which would make the name unavailable. + fence_store::remove_marker(&self.inner.dbs_path, &namespace)?; + let receipts = fence_store::delete_fence(&tx, &namespace)?; + tracing::info!( + namespace = %namespace, + state = %stored.state(), + revision = stored.revision(), + receipts, + "removing namespace fence with its namespace" + ); + } + } if config.config.is_shared_schema { if crate::schema::db::schema_has_linked_dbs(&tx, &namespace)? { return Err(crate::Error::HasLinkedDbs(namespace.clone())); @@ -562,6 +995,115 @@ impl MetaStore { self.inner.configs.lock().await.contains_key(namespace) } + /// Whether namespace fences may be used on this server. + pub fn fence_enabled(&self) -> bool { + self.inner.fence.enabled + } + + /// Whether this metastore holds fence state, so fences are loaded and enforced. + pub fn fence_enforced(&self) -> bool { + self.inner.fence.tables + } + + /// Run one fence command as a compare-and-swap in a single metastore transaction + /// (`docs/NAMESPACE_FENCE.md` sections 5.3 and 5.4). Nothing is published here: the + /// caller publishes the result only after this returns, which is after the commit. + /// + /// A fence outcome that is an error (`FENCE_REVISION_MISMATCH`, …) is returned as + /// [`Error::NamespaceFence`] and nothing is written. + pub async fn apply_fence_command( + &self, + request: FenceRequest, + ctx: FenceContext, + ) -> Result { + let inner = self.inner.clone(); + tokio::task::spawn_blocking(move || apply_fence_command(&inner, &request, &ctx)) + .await? + .map_err(fence_store_error) + } + + /// Finish the drain that the owning operation's `DRAINING` receipt + /// `(operation_id, command_id)` started, once the controller has proven `completion`. The + /// final receipt replaces the `DRAINING` one. If the drain was already completed, the + /// final receipt is returned as a replay. + pub async fn complete_fence_drain( + &self, + namespace: NamespaceName, + operation_id: Uuid, + command_id: Uuid, + completion: DrainCompletion, + ctx: FenceContext, + ) -> Result { + let inner = self.inner.clone(); + tokio::task::spawn_blocking(move || { + complete_fence_drain( + &inner, + &namespace, + operation_id, + command_id, + completion, + &ctx, + ) + }) + .await? + .map_err(fence_store_error) + } + + /// Read a namespace's fence and all of its receipts (`InspectFence`). Never writes. + pub async fn inspect_fence(&self, namespace: NamespaceName) -> Result { + let inner = self.inner.clone(); + tokio::task::spawn_blocking(move || -> std::result::Result<_, FenceStoreError> { + let mut conn = inner.conn.blocking_lock(); + if !inner.fence.tables { + let tx = conn.transaction()?; + let exists = fence_store::read_config_row(&tx, &namespace)?.is_some(); + return Ok(FenceInspection { + fence: StoredFence::None { + namespace_exists: exists, + }, + receipts: Vec::new(), + }); + } + let tx = conn.transaction()?; + let (fence, _) = fence_store::read_fence(&tx, &inner.dbs_path, &namespace)?; + let receipts = fence_store::read_receipts(&tx, &namespace)?; + Ok(FenceInspection { fence, receipts }) + }) + .await? + .map_err(fence_store_error) + } + + /// Every namespace with fence state, and that state. Namespaces without a record or a + /// marker are left out. + pub async fn load_fences(&self) -> Result> { + let inner = self.inner.clone(); + tokio::task::spawn_blocking(move || -> std::result::Result<_, FenceStoreError> { + if !inner.fence.tables { + return Ok(Vec::new()); + } + let mut conn = inner.conn.blocking_lock(); + let tx = conn.transaction()?; + let names: Vec = { + let mut stmt = tx.prepare("SELECT namespace FROM namespace_configs")?; + let rows = stmt.query_map((), |r| r.get::<_, String>(0))?; + rows.collect::>()? + }; + let mut out = Vec::new(); + for name in names { + let Ok(ns) = NamespaceName::from_string(name) else { + continue; + }; + let (fence, _) = fence_store::read_fence(&tx, &inner.dbs_path, &ns)?; + if !matches!(fence, StoredFence::None { .. }) { + out.push((ns, fence)); + } + } + Ok(out) + }) + .await? + .map_err(fence_store_error) + } + pub(crate) async fn shutdown(&self) -> crate::Result<()> { let replicator = self.inner.wal_manager.wrapper().as_ref(); @@ -729,3 +1271,806 @@ impl MetaStoreHandle { &self.namespace } } + +#[cfg(test)] +mod fence_tests { + use std::path::Path; + + use tempfile::tempdir; + + use super::*; + use crate::namespace::fence::command::TargetConfig; + use crate::namespace::fence::record::{FenceMarker, FrozenBoundary}; + use crate::namespace::fence::state::FenceState; + + const LOG: Uuid = Uuid::from_u128(0x10); + const INCARNATION: Uuid = Uuid::from_u128(0x20); + const OP: Uuid = Uuid::from_u128(0xa); + const OTHER_OP: Uuid = Uuid::from_u128(0xb); + + async fn open_with(dir: &Path, config: MetaStoreConfig) -> MetaStore { + let (maker, manager) = metastore_connection_maker(None, dir).await.unwrap(); + let conn = maker().unwrap(); + MetaStore::new(config, dir, conn, manager, DatabaseKind::Primary) + .await + .unwrap() + } + + async fn open(dir: &Path, fence: bool) -> MetaStore { + open_with( + dir, + MetaStoreConfig { + namespace_fence: fence, + ..Default::default() + }, + ) + .await + } + + /// A second, independent connection to the same metastore database (like the schema + /// scheduler's). + async fn raw(dir: &Path) -> MetaStoreConnection { + let (maker, _) = metastore_connection_maker(None, dir).await.unwrap(); + maker().unwrap() + } + + fn raw_config(conn: &rusqlite::Connection, ns: &str) -> DatabaseConfig { + fence_store::read_config_row(conn, &NamespaceName::from(ns.to_string().leak() as &str)) + .unwrap() + .unwrap() + } + + fn ctx(now_ms: i64) -> FenceContext { + FenceContext { + server: ServerIdentity { + build: "test".into(), + instance_id: Uuid::from_u128(0x99), + }, + now_ms, + namespace_log_id: Some(LOG), + new_incarnation_id: INCARNATION, + adoption_authorised: false, + validation_snapshot: None, + } + } + + fn request( + ns: &'static str, + op: Uuid, + command_id: u128, + expected_state: FenceState, + expected_revision: u64, + command: FenceCommand, + ) -> FenceRequest { + FenceRequest { + namespace: ns.into(), + operation_id: op, + command_id: Uuid::from_u128(command_id), + expected_state, + expected_revision, + command, + } + } + + fn acquire(ns: &'static str, op: Uuid, command_id: u128) -> FenceRequest { + request( + ns, + op, + command_id, + FenceState::Unfenced, + 0, + FenceCommand::AcquireSourceWriteFence { + expected_log_id: LOG, + drain_policy: None, + }, + ) + } + + fn outcome_of(r: Result) -> FenceOutcome { + match r { + Ok(c) => c.receipt.outcome, + Err(Error::NamespaceFence(e)) => e.outcome(), + Err(e) => panic!("unexpected error: {e}"), + } + } + + fn fence_error(r: Result) -> FenceError { + match r { + Err(Error::NamespaceFence(e)) => e, + other => panic!("expected a fence error, got {other:?}"), + } + } + + async fn create_namespace(store: &MetaStore, ns: &'static str) -> MetaStoreHandle { + let handle = store.handle(ns.into()).await; + handle + .store(DatabaseConfig { + max_db_pages: 1234, + block_reason: Some("pre-fence".into()), + ..Default::default() + }) + .await + .unwrap(); + handle + } + + fn remove_blocking(store: &MetaStore, ns: &'static str) -> Result>> { + let store = store.clone(); + std::thread::spawn(move || store.remove(ns.into())) + .join() + .unwrap() + } + + #[tokio::test] + async fn fence_cas_persists_across_restart() { + let dir = tempdir().unwrap(); + let store = open(dir.path(), true).await; + let handle = create_namespace(&store, "db").await; + + let commit = store + .apply_fence_command(acquire("db", OP, 1), ctx(1_000)) + .await + .unwrap(); + assert_eq!(commit.kind, FenceCommitKind::Committed); + assert_eq!(commit.receipt.outcome, FenceOutcome::Draining); + let record = commit.record.unwrap(); + assert_eq!( + (record.state, record.revision), + (FenceState::SourceDraining, 1) + ); + + // The stored row carries the legacy mirror; the in-memory config is untouched. + let conn = raw(dir.path()).await; + let row = raw_config(&conn, "db"); + assert!(row.block_writes && !row.block_reads); + assert!(row + .block_reason + .unwrap() + .starts_with("namespace fence: SOURCE_DRAINING")); + assert!(!handle.get().block_writes); + + let boundary = FrozenBoundary { + log_id: LOG, + frame_no: 42, + }; + let commit = store + .complete_fence_drain( + "db".into(), + OP, + Uuid::from_u128(1), + DrainCompletion::SourceWrites { boundary }, + ctx(2_000), + ) + .await + .unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + let record = commit.record.unwrap(); + assert_eq!( + (record.state, record.revision), + (FenceState::SourceWriteFenced, 2) + ); + // Completing again is a replay of the final answer. + let again = store + .complete_fence_drain( + "db".into(), + OP, + Uuid::from_u128(1), + DrainCompletion::SourceWrites { boundary }, + ctx(2_500), + ) + .await + .unwrap(); + assert_eq!(again.kind, FenceCommitKind::Replayed); + + let marker = fence_store::read_marker(&dir.path().join("dbs"), &"db".into()) + .unwrap() + .unwrap() + .unwrap(); + assert_eq!(marker, FenceMarker::for_record(&record)); + + drop(handle); + drop(store); + + // Restart. + let store = open(dir.path(), true).await; + let inspection = store.inspect_fence("db".into()).await.unwrap(); + assert_eq!(inspection.fence, StoredFence::Record(record.clone())); + assert_eq!(inspection.fence.revision(), 2); + assert_eq!(record.frozen_boundary, Some(boundary)); + assert_eq!(inspection.receipts.len(), 1); + let receipt = inspection.receipts[0].receipt.clone().unwrap(); + assert_eq!( + (receipt.outcome, receipt.revision_after), + (FenceOutcome::Applied, 2) + ); + + // The in-memory config is the namespace's own, not the mirror. + let handle = store.handle("db".into()).await; + let config = handle.get(); + assert!(!config.block_writes && !config.block_reads); + assert_eq!(config.block_reason.as_deref(), Some("pre-fence")); + assert_eq!(config.max_db_pages, 1234); + + // The lost response of the first command is answered from its receipt, even though + // the revision has advanced. + let replay = store + .apply_fence_command(acquire("db", OP, 1), ctx(3_000)) + .await + .unwrap(); + assert_eq!(replay.kind, FenceCommitKind::Replayed); + assert_eq!(replay.receipt, receipt); + + let fences = store.load_fences().await.unwrap(); + assert_eq!( + fences, + vec![("db".into(), StoredFence::Record(record.clone()))] + ); + + // Release restores the legacy fields as they were before the fence. + let release = request( + "db", + OP, + 2, + FenceState::SourceWriteFenced, + 2, + FenceCommand::ReleaseSourceWriteFence, + ); + let commit = store + .apply_fence_command(release, ctx(4_000)) + .await + .unwrap(); + assert_eq!(commit.record.unwrap().revision, 3); + let row = raw_config(&conn, "db"); + assert!(!row.block_writes && !row.block_reads); + assert_eq!(row.block_reason.as_deref(), Some("pre-fence")); + assert_eq!(row.max_db_pages, 1234); + } + + #[tokio::test] + async fn flag_off_still_enforces_existing_fences() { + let dir = tempdir().unwrap(); + let store = open(dir.path(), true).await; + let handle = create_namespace(&store, "db").await; + store + .apply_fence_command(acquire("db", OP, 1), ctx(1_000)) + .await + .unwrap(); + drop(handle); + drop(store); + + let store = open(dir.path(), false).await; + assert!(!store.fence_enabled()); + assert!(store.fence_enforced()); + let e = fence_error( + store + .apply_fence_command(acquire("db", OTHER_OP, 2), ctx(2_000)) + .await, + ); + assert_eq!(e.detail(), Some(FenceDetail::FenceDisabled)); + let handle = store.handle("db".into()).await; + let e = fence_error(handle.store(DatabaseConfig::default()).await); + assert_eq!(e.outcome(), FenceOutcome::MigrationWriteFenced); + assert_eq!( + store + .inspect_fence("db".into()) + .await + .unwrap() + .fence + .state(), + FenceState::SourceDraining + ); + } + + #[tokio::test] + async fn disabled_fence_creates_nothing() { + let dir = tempdir().unwrap(); + let store = open(dir.path(), false).await; + let handle = create_namespace(&store, "db").await; + assert!(!store.fence_enforced()); + let e = fence_error( + store + .apply_fence_command(acquire("db", OP, 1), ctx(1_000)) + .await, + ); + assert_eq!(e.outcome(), FenceOutcome::FencePreconditionFailed); + assert_eq!(e.detail(), Some(FenceDetail::FenceDisabled)); + let conn = raw(dir.path()).await; + assert!(!fence_store::tables_exist(&conn).unwrap()); + handle + .store(DatabaseConfig { + max_db_pages: 7, + ..Default::default() + }) + .await + .unwrap(); + assert_eq!(raw_config(&conn, "db").max_db_pages, 7); + assert_eq!( + store.inspect_fence("db".into()).await.unwrap().fence, + StoredFence::None { + namespace_exists: true + } + ); + assert!(store.load_fences().await.unwrap().is_empty()); + } + + #[tokio::test] + async fn concurrent_cas_has_exactly_one_winner() { + let dir = tempdir().unwrap(); + let store = open(dir.path(), true).await; + let _handle = create_namespace(&store, "db").await; + + let attempts = (0..16u128).map(|i| { + let store = store.clone(); + async move { + store + .apply_fence_command(acquire("db", Uuid::from_u128(0x100 + i), 1), ctx(1_000)) + .await + } + }); + let outcomes: Vec<_> = futures::future::join_all(attempts) + .await + .into_iter() + .map(outcome_of) + .collect(); + assert_eq!( + outcomes + .iter() + .filter(|o| **o == FenceOutcome::Draining) + .count(), + 1, + "{outcomes:?}" + ); + assert!(outcomes.iter().all(|o| matches!( + o, + FenceOutcome::Draining | FenceOutcome::FenceOwnedByAnotherOperation + ))); + let inspection = store.inspect_fence("db".into()).await.unwrap(); + assert_eq!(inspection.fence.revision(), 1); + assert_eq!(inspection.receipts.len(), 1); + } + + #[tokio::test] + async fn fence_cas_and_config_writes_serialise() { + let dir = tempdir().unwrap(); + let store = open(dir.path(), true).await; + let handle = create_namespace(&store, "db").await; + + // Another metastore connection holds the write lock with an uncommitted config + // change. The fence transition cannot interleave with it: it gives up on the lock + // and writes nothing. + let mut other = raw(dir.path()).await; + let tx = other + .transaction_with_behavior(TransactionBehavior::Immediate) + .unwrap(); + let mut changed = raw_config(&tx, "db"); + changed.max_db_pages = 42; + fence_store::write_config_row(&tx, &"db".into(), &changed).unwrap(); + let r = store + .apply_fence_command(acquire("db", OP, 1), ctx(1_000)) + .await; + assert!( + matches!(r, Err(Error::RusqliteError(_))), + "expected the write lock to be busy, got {r:?}" + ); + tx.commit().unwrap(); + assert_eq!( + store + .inspect_fence("db".into()) + .await + .unwrap() + .fence + .state(), + FenceState::Unfenced + ); + + // After the commit the transition reads the row as committed, so the other writer's + // change survives underneath the mirror, although the in-memory config never saw it. + assert_eq!(handle.get().max_db_pages, 1234); + store + .apply_fence_command(acquire("db", OP, 1), ctx(1_000)) + .await + .unwrap(); + let conn = raw(dir.path()).await; + let row = raw_config(&conn, "db"); + assert_eq!(row.max_db_pages, 42); + assert!(row.block_writes); + + // An ordinary config write is refused inside its transaction while the fence denies + // lifecycle operations, and a refused write is not published. + let e = fence_error( + handle + .store(DatabaseConfig { + max_db_pages: 9, + ..Default::default() + }) + .await, + ); + assert_eq!(e.outcome(), FenceOutcome::MigrationWriteFenced); + assert_eq!(handle.get().max_db_pages, 1234); + assert_eq!(raw_config(&conn, "db").max_db_pages, 42); + let e = fence_error(handle.flush().await); + assert_eq!(e.outcome(), FenceOutcome::MigrationWriteFenced); + + // So is a delete, and the fence row would stop an older binary's delete too. + let e = fence_error(remove_blocking(&store, "db")); + assert_eq!(e.outcome(), FenceOutcome::MigrationWriteFenced); + assert!(conn + .execute("DELETE FROM namespace_configs WHERE namespace = 'db'", ()) + .is_err()); + + // Once released, config writes and delete work again; delete takes the fence with it. + let release = request( + "db", + OP, + 2, + FenceState::SourceDraining, + 1, + FenceCommand::ReleaseSourceWriteFence, + ); + store + .apply_fence_command(release, ctx(2_000)) + .await + .unwrap(); + handle + .store(DatabaseConfig { + max_db_pages: 9, + ..Default::default() + }) + .await + .unwrap(); + assert_eq!(handle.get().max_db_pages, 9); + drop(handle); + assert!(remove_blocking(&store, "db").unwrap().is_some()); + let inspection = store.inspect_fence("db".into()).await.unwrap(); + assert_eq!( + inspection.fence, + StoredFence::None { + namespace_exists: false + } + ); + assert!(inspection.receipts.is_empty()); + } + + #[tokio::test] + async fn corrupt_fence_row_fails_closed() { + let dir = tempdir().unwrap(); + let store = open(dir.path(), true).await; + let handle = create_namespace(&store, "db").await; + store + .apply_fence_command(acquire("db", OP, 1), ctx(1_000)) + .await + .unwrap(); + drop(handle); + drop(store); + + let conn = raw(dir.path()).await; + conn.execute( + "UPDATE namespace_fences SET record = x'00ff00' WHERE namespace = 'db'", + (), + ) + .unwrap(); + + // Startup does not fail, and does not guess. + let store = open(dir.path(), true).await; + let fence = store.inspect_fence("db".into()).await.unwrap().fence; + assert!(matches!( + fence, + StoredFence::Unavailable { + detail: FenceDetail::CorruptRecord, + .. + } + )); + // The config keeps the stored mirror, so statement-level checks stay closed too. + let handle = store.handle("db".into()).await; + assert!(handle.get().block_writes); + + let release = request( + "db", + OP, + 2, + FenceState::SourceDraining, + 1, + FenceCommand::ReleaseSourceWriteFence, + ); + let e = fence_error(store.apply_fence_command(release, ctx(2_000)).await); + assert_eq!(e.outcome(), FenceOutcome::FenceStateUnavailable); + assert_eq!(e.detail(), Some(FenceDetail::CorruptRecord)); + let e = fence_error(handle.store(DatabaseConfig::default()).await); + assert_eq!(e.outcome(), FenceOutcome::FenceStateUnavailable); + let e = fence_error( + store + .complete_fence_drain( + "db".into(), + OP, + Uuid::from_u128(1), + DrainCompletion::SourceWrites { + boundary: FrozenBoundary { + log_id: LOG, + frame_no: 1, + }, + }, + ctx(2_000), + ) + .await, + ); + assert_eq!(e.outcome(), FenceOutcome::FenceStateUnavailable); + + // An unknown format version is its own reason. + conn.execute( + "UPDATE namespace_fences SET format_version = 99 WHERE namespace = 'db'", + (), + ) + .unwrap(); + let fence = store.inspect_fence("db".into()).await.unwrap().fence; + assert!(matches!( + fence, + StoredFence::Unavailable { + detail: FenceDetail::UnsupportedFormatVersion, + .. + } + )); + } + + #[tokio::test] + async fn marker_tracks_the_metastore() { + let dir = tempdir().unwrap(); + let dbs = dir.path().join("dbs"); + let store = open(dir.path(), true).await; + let handle = create_namespace(&store, "db").await; + store + .apply_fence_command(acquire("db", OP, 1), ctx(1_000)) + .await + .unwrap(); + let conn = raw(dir.path()).await; + let (v1, r1, b1): (i64, i64, Vec) = conn + .query_row( + "SELECT format_version, revision, record FROM namespace_fences", + (), + |r| Ok((r.get(0)?, r.get(1)?, r.get(2)?)), + ) + .unwrap(); + let commit = store + .complete_fence_drain( + "db".into(), + OP, + Uuid::from_u128(1), + DrainCompletion::SourceWrites { + boundary: FrozenBoundary { + log_id: LOG, + frame_no: 7, + }, + }, + ctx(2_000), + ) + .await + .unwrap(); + let record = commit.record.unwrap(); + drop(handle); + drop(store); + + // A marker lost after the commit is rewritten from the metastore on load. + std::fs::remove_file(fence_store::marker_path(&dbs, &"db".into())).unwrap(); + let store = open(dir.path(), true).await; + let marker = fence_store::read_marker(&dbs, &"db".into()) + .unwrap() + .unwrap() + .unwrap(); + assert_eq!(marker.record, record); + drop(store); + + // A metastore that went backwards (restored to revision 1) is not trusted over the + // newer marker, and loading it does not overwrite the marker. + conn.execute( + "UPDATE namespace_fences SET format_version = ?1, revision = ?2, record = ?3", + rusqlite::params![v1, r1, b1], + ) + .unwrap(); + let store = open(dir.path(), true).await; + let fence = store.inspect_fence("db".into()).await.unwrap().fence; + match fence { + StoredFence::Unavailable { + detail: FenceDetail::MetastoreBehindMarker, + marker: Some(m), + .. + } => assert_eq!(m, record), + other => panic!("expected metastore_behind_marker, got {other:?}"), + } + let marker = fence_store::read_marker(&dbs, &"db".into()) + .unwrap() + .unwrap() + .unwrap(); + assert_eq!(marker.record, record); + let e = fence_error( + store + .apply_fence_command( + request( + "db", + OP, + 3, + FenceState::SourceWriteFenced, + 2, + FenceCommand::ReleaseSourceWriteFence, + ), + ctx(3_000), + ) + .await, + ); + assert_eq!(e.detail(), Some(FenceDetail::MetastoreBehindMarker)); + } + + fn create_target(ns: &'static str, command_id: u128) -> FenceRequest { + request( + ns, + OP, + command_id, + FenceState::Absent, + 0, + FenceCommand::CreateTargetQuarantined { + config: TargetConfig { + max_db_size: Some(4096 * 100), + ..Default::default() + }, + }, + ) + } + + #[tokio::test] + async fn target_creation_is_atomic_and_replayable() { + let dir = tempdir().unwrap(); + let dbs = dir.path().join("dbs"); + let store = open(dir.path(), true).await; + + let commit = store + .apply_fence_command(create_target("tgt", 1), ctx(1_000)) + .await + .unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + let record = commit.record.clone().unwrap(); + assert_eq!( + (record.state, record.revision), + (FenceState::TargetQuarantined, 1) + ); + assert_eq!(record.identity.target_incarnation_id, Some(INCARNATION)); + let created = commit.created_config.unwrap(); + assert_eq!(created.max_db_pages, 100); + assert!(!created.block_reads && !created.block_writes); + // Not published: the caller installs the gate first. + assert!(!store.exists(&"tgt".into()).await); + // Stored with the legacy mirror, and with its marker. + let conn = raw(dir.path()).await; + let row = raw_config(&conn, "tgt"); + assert!(row.block_reads && row.block_writes); + assert_eq!( + fence_store::read_marker(&dbs, &"tgt".into()) + .unwrap() + .unwrap() + .unwrap() + .record, + record + ); + let replay = store + .apply_fence_command(create_target("tgt", 1), ctx(2_000)) + .await + .unwrap(); + assert_eq!(replay.kind, FenceCommitKind::Replayed); + // A new command of the owner asking for the same thing is ALREADY_APPLIED; another + // operation cannot take the name. + let again = store + .apply_fence_command(create_target("tgt", 2), ctx(2_000)) + .await + .unwrap(); + assert_eq!(again.receipt.outcome, FenceOutcome::AlreadyApplied); + assert_eq!(again.record.unwrap().revision, 1); + let mut other = create_target("tgt", 5); + other.operation_id = OTHER_OP; + let e = fence_error(store.apply_fence_command(other, ctx(2_000)).await); + assert_eq!(e.outcome(), FenceOutcome::FenceOwnedByAnotherOperation); + + // A crash between the marker and the commit: the marker is all that is left. + store + .apply_fence_command(create_target("tgt2", 3), ctx(3_000)) + .await + .unwrap(); + let marker = fence_store::read_marker(&dbs, &"tgt2".into()) + .unwrap() + .unwrap() + .unwrap(); + for sql in [ + "DELETE FROM namespace_fence_receipts WHERE namespace = 'tgt2'", + "DELETE FROM namespace_fences WHERE namespace = 'tgt2'", + "DELETE FROM namespace_configs WHERE namespace = 'tgt2'", + ] { + conn.execute(sql, ()).unwrap(); + } + let fence = store.inspect_fence("tgt2".into()).await.unwrap().fence; + assert!(matches!( + fence, + StoredFence::Unavailable { + detail: FenceDetail::IncompleteTargetCreation, + .. + } + )); + // Only the same command completes it, keeping the incarnation id it announced. + let e = fence_error( + store + .apply_fence_command(create_target("tgt2", 4), ctx(4_000)) + .await, + ); + assert_eq!(e.outcome(), FenceOutcome::FenceStateUnavailable); + let mut later = ctx(5_000); + later.new_incarnation_id = Uuid::from_u128(0x21); + let commit = store + .apply_fence_command(create_target("tgt2", 3), later) + .await + .unwrap(); + assert_eq!(commit.kind, FenceCommitKind::Committed); + let record = commit.record.unwrap(); + assert_eq!( + record.identity.target_incarnation_id, + marker.record.identity.target_incarnation_id + ); + assert_eq!( + store.inspect_fence("tgt2".into()).await.unwrap().fence, + StoredFence::Record(record) + ); + } + + async fn run_source_operation(store: &MetaStore, op: Uuid, base: u128, rev: u64, now: i64) { + let expected = if rev == 0 { + FenceState::Unfenced + } else { + FenceState::Released + }; + let mut acq = acquire("db", op, base); + acq.expected_state = expected; + acq.expected_revision = rev; + store.apply_fence_command(acq, ctx(now)).await.unwrap(); + let release = request( + "db", + op, + base + 1, + FenceState::SourceDraining, + rev + 1, + FenceCommand::ReleaseSourceWriteFence, + ); + store + .apply_fence_command(release, ctx(now + 1)) + .await + .unwrap(); + } + + #[tokio::test] + async fn receipts_of_finished_operations_are_pruned_after_retention() { + let dir = tempdir().unwrap(); + let store = open_with( + dir.path(), + MetaStoreConfig { + namespace_fence: true, + namespace_fence_receipt_retention: Some(Duration::from_secs(1)), + ..Default::default() + }, + ) + .await; + let _handle = create_namespace(&store, "db").await; + + run_source_operation(&store, OP, 1, 0, 1_000).await; + // Within the retention period, the finished operation's receipts are kept. + run_source_operation(&store, OTHER_OP, 10, 2, 1_500).await; + let ops = |receipts: &[StoredReceipt]| { + receipts + .iter() + .map(|r| r.receipt.clone().unwrap().operation_id) + .collect::>() + }; + let inspection = store.inspect_fence("db".into()).await.unwrap(); + assert_eq!(ops(&inspection.receipts), vec![OP, OP, OTHER_OP, OTHER_OP]); + // Later, a transition prunes other operations' old receipts, never the owner's. + run_source_operation(&store, Uuid::from_u128(0xc), 20, 4, 10_000).await; + let inspection = store.inspect_fence("db".into()).await.unwrap(); + assert_eq!( + ops(&inspection.receipts), + vec![Uuid::from_u128(0xc), Uuid::from_u128(0xc)] + ); + assert_eq!(inspection.fence.revision(), 6); + } +} From 0cf1515fba09fb2f5213840707e718f349b4edca Mon Sep 17 00:00:00 2001 From: River Date: Tue, 29 Sep 2026 14:51:12 +0000 Subject: [PATCH 06/33] libsql-server: fail closed on ambiguous metastore recovery When namespace fences are in use (the flag is on, the fence tables exist, or a namespace directory holds a fence marker), a namespace whose state startup cannot establish is registered UNKNOWN_UNAVAILABLE instead of being skipped or given a default config: - an undecodable config row, fence row or marker; - a directory with a marker that the metastore has no trustworthy record for: filesystem recovery, a metastore rebuilt by destroy_on_error, a metastore restored from an older backup or without fence tables, and a target whose creation was interrupted. Such a namespace is refused by lookups, config writes and deletes with FENCE_STATE_UNAVAILABLE, is reported by InspectFence, and is settled only by a fence command that commits for it. MetaStore::handle() no longer default-creates on read paths: the new non-creating MetaStore::lookup serves NamespaceStore::with, fork's source and ATTACH authorisation. handle(), used only by creating paths, refuses unavailable names and names without a config whose directory holds a marker. Filesystem recovery skips marked directories, destroy_on_error moves the broken metastore aside instead of deleting it, and a fence row for a namespace name that cannot be decoded stops startup. Servers that never used fences recover as before. Co-authored-by: Tomasz Szymczyszyn --- docs/NAMESPACE_FENCE.md | 19 +- libsql-server/src/connection/program.rs | 8 +- libsql-server/src/namespace/fence/store.rs | 39 + libsql-server/src/namespace/meta_store.rs | 802 +++++++++++++++++++-- libsql-server/src/namespace/store.rs | 27 +- libsql-server/src/schema/db.rs | 3 + 6 files changed, 836 insertions(+), 62 deletions(-) diff --git a/docs/NAMESPACE_FENCE.md b/docs/NAMESPACE_FENCE.md index d26eefca59..0a007efc83 100644 --- a/docs/NAMESPACE_FENCE.md +++ b/docs/NAMESPACE_FENCE.md @@ -614,14 +614,17 @@ This is best effort. An older binary applies `block_*` at statement level only, ### 13.3 Fail-closed metastore recovery -With the flag on, or whenever fence tables or markers exist: +Recovery fails closed when the flag is on, when the fence tables exist, or when any namespace directory holds a marker. Otherwise (a server that never used fences) recovery behaves as before. Startup scans `dbs/` for markers first; a marker in a directory whose name is not a valid namespace name stops startup with an operator error. -- `MetaStore::handle()` no longer default-creates entries on read paths. `NamespaceStore::with`, `checkpoint`, ATTACH authorisation and `check_program_auth` use a non-creating lookup; only create, fork destination and replica lazy creation create, and they refuse names that have a fence record or marker. -- `restore()`: an undecodable namespace name, config or fence row marks that namespace `UNKNOWN_UNAVAILABLE` (when the name is decodable) or fails startup with an operator error (when it is not), instead of skipping the row. -- `maybe_recover_from_fs`: a directory with a marker is registered `UNKNOWN_UNAVAILABLE`; directories without markers keep today's behaviour (they are legacy, unfenced namespaces). -- `destroy_on_error`: the broken metastore is renamed aside rather than deleted, and directories with markers are registered `UNKNOWN_UNAVAILABLE` after the rebuild. -- Metastore restore from backup: the provenance (`restored_from_backup`, backup generation) is surfaced in the capability endpoint, in `InspectFence`, as a metric and in a startup log line; the marker comparison (section 5.6) makes any namespace whose record went backwards `UNKNOWN_UNAVAILABLE`. A restored record is never trusted over a newer marker. -- Replica-kind servers: lazy creation of a name refused by the primary with a fence code does not create a local default namespace. +A namespace that startup cannot recover is registered `UNKNOWN_UNAVAILABLE` in memory. It is never given a default config, never default-created, and every lookup, config write and delete of it is refused with `FENCE_STATE_UNAVAILABLE` and a detail. `InspectFence` and the fence list report it. Nothing is written for it: the registration is recomputed from the metastore and the markers at every start, and a fence command that commits for the name (the replay that completes an interrupted target creation, later an adoption) settles it. + +- `MetaStore::handle()` no longer default-creates entries on read paths. `NamespaceStore::with` (and so every SQL, Hrana, dump and replication entry point), fork's source and ATTACH authorisation (`check_program_auth`) use the non-creating `MetaStore::lookup`, which returns the existing handle, nothing, or the fence error. Only create, fork destination, reset, the default namespace and lazy creation call `handle()`, which refuses a registered name, and a name without a config whose directory holds a marker (a target being created, or a namespace the metastore lost). `NamespaceStore::checkpoint` does not touch the metastore and needs no lookup. +- `restore()`: an undecodable config row marks its namespace `UNKNOWN_UNAVAILABLE` (`corrupt_record`) and the row is left as it is. An undecodable namespace name cannot be addressed by any request, so a legacy row is still skipped; if the metastore holds a fence row for such a name, startup fails with an operator error. An undecodable, unknown-version or inconsistent fence row, or an unreadable marker, marks the namespace `UNKNOWN_UNAVAILABLE` (section 5.6). +- `maybe_recover_from_fs`: a directory with a marker is not recovered from `config.json` (or a default); it is registered `UNKNOWN_UNAVAILABLE` (`metastore_behind_marker`). Directories without markers keep today's behaviour: they are legacy, unfenced namespaces. +- `destroy_on_error`: the broken metastore is renamed to `metastore.broken-` rather than deleted (if the rename fails, startup fails instead), and directories with markers are registered `UNKNOWN_UNAVAILABLE` after the rebuild. Without fences it still deletes, as before. +- A metastore without fence tables next to a directory that holds a marker (rebuilt, recovered, or restored from a backup taken before the tables existed) makes that namespace `UNKNOWN_UNAVAILABLE`, even when its config row is present. +- Metastore restore from backup: the marker comparison (section 5.6) makes any namespace whose record went backwards or disappeared `UNKNOWN_UNAVAILABLE`. A restored record is never trusted over a newer marker. Surfacing the provenance (`restored_from_backup`, backup generation) in the capability endpoint, `InspectFence`, a metric and a startup line is part of the admin API (section 4). +- Replica-kind servers: lazy creation of a name refused by the primary with a fence code does not create a local default namespace. This needs the proxy's stable code (section 6.1) and lands with the protocol mappings. ### 13.4 Shared schema @@ -691,7 +694,7 @@ Planned test names; the table is updated as tests land. | 6 | Acquisition timeout returns `DRAINING`, admission stays closed | `fence::drain::tests::deadline_returns_draining_and_stays_closed` | | 7 | Restart at every persistence boundary; indeterminate persistence keeps the gate closed until same-command reconciliation | `fence::tests::restart_at_each_boundary` (parameterised over hook points), `indeterminate_commit_keeps_gate_closed` | | 8 | Evict and lazily reload a fenced namespace; identical admission | `tests::fence::lifecycle::evicted_namespace_reloads_same_gate` | -| 9 | Filesystem recovery, `destroy_on_error`, undecodable records, missing target quarantine, metastore backup rollback fail closed with provenance | `meta_store::tests::fs_recovery_with_marker_unavailable`, `destroy_on_error_keeps_fenced_unavailable`, `undecodable_row_unavailable`, `incomplete_target_unavailable`, `metastore_rollback_detected_by_marker` | +| 9 | Filesystem recovery, `destroy_on_error`, undecodable records, missing target quarantine, metastore backup rollback fail closed with provenance | `meta_store::fence_tests::recovery::{fs_recovery_with_marker_unavailable, destroy_on_error_keeps_fenced_unavailable, undecodable_row_unavailable, incomplete_target_unavailable, metastore_rollback_detected_by_marker, lookup_never_creates, undecodable_name_with_fence_fails_startup, marker_in_invalid_directory_fails_startup}`; legacy behaviour kept: `destroy_on_error_without_fences_is_unchanged`, `undecodable_row_without_fences_is_skipped_as_before`; `meta_store::fence_tests::corrupt_fence_row_fails_closed` | | 10 | Wrong owner, stale revision, invalid role/state, replay, command-id reuse; replay before revision check | `fence::transition::tests::*` (exhaustive over states × commands) | | 11 | Target creation raced with SQL, dump, replication, lifecycle never observable as writable or readable | `fence::target::tests::create_race_never_observable` | | 12 | Only the matching import capability writes a quarantined target; admin credentials and admin shell cannot | `fence::target::tests::import_requires_matching_capability`; `tests::fence::admin::admin_shell_cannot_write_quarantined` | diff --git a/libsql-server/src/connection/program.rs b/libsql-server/src/connection/program.rs index 08dd9526f3..4d5ada51ff 100644 --- a/libsql-server/src/connection/program.rs +++ b/libsql-server/src/connection/program.rs @@ -370,7 +370,13 @@ pub async fn check_program_auth( } StmtKind::Attach(ref ns) => { ctx.auth.has_right(ns, Permission::AttachRead)?; - if !ctx.meta_store.handle(ns.clone()).await.get().allow_attach { + // A non-creating lookup: a missing namespace does not allow attach, and one + // whose fence state is not established is refused with its fence error. + let allow_attach = match ctx.meta_store.lookup(ns).await? { + Some(handle) => handle.get().allow_attach, + None => false, + }; + if !allow_attach { return Err(Error::Forbidden(format!( "Namespace `{ns}` doesn't allow attach" ))); diff --git a/libsql-server/src/namespace/fence/store.rs b/libsql-server/src/namespace/fence/store.rs index 2d6f9ca217..bb7a92a20f 100644 --- a/libsql-server/src/namespace/fence/store.rs +++ b/libsql-server/src/namespace/fence/store.rs @@ -215,6 +215,32 @@ pub fn remove_marker(dbs_path: &Path, namespace: &NamespaceName) -> io::Result<( } } +/// The namespace directories under `dbs_path` that hold a marker, in no particular order. A +/// directory whose name is not a valid namespace name is returned as its raw name, so the +/// caller can refuse to start rather than ignore it. +pub fn scan_markers(dbs_path: &Path) -> io::Result>> { + let entries = match fs::read_dir(dbs_path) { + Ok(entries) => entries, + Err(e) if e.kind() == io::ErrorKind::NotFound => return Ok(Vec::new()), + Err(e) => return Err(e), + }; + let mut out = Vec::new(); + for entry in entries { + let entry = entry?; + if !entry.file_type()?.is_dir() || !entry.path().join(MARKER_FILE_NAME).try_exists()? { + continue; + } + let raw = entry.file_name(); + out.push(match raw.to_str() { + Some(name) => { + NamespaceName::from_string(name.to_string()).map_err(|_| name.to_string()) + } + None => Err(raw.to_string_lossy().into_owned()), + }); + } + Ok(out) +} + /// Whether the marker agrees with what the metastore says. Returned by [`read_fence`] so a /// loader can repair a marker that fell behind (a crash between commit and marker write). #[derive(Debug, Clone, Copy, PartialEq, Eq)] @@ -264,6 +290,19 @@ pub fn stored_revision( .optional() } +/// Like [`stored_revision`], for a namespace name that is not a valid [`NamespaceName`]. +pub fn stored_revision_raw( + conn: &rusqlite::Connection, + namespace: &str, +) -> rusqlite::Result> { + conn.query_row( + "SELECT revision FROM namespace_fences WHERE namespace = ?1", + [namespace], + |row| row.get(0), + ) + .optional() +} + fn decode_row(row: &RawFenceRow) -> Result { let format_version = u32::try_from(row.format_version) .map_err(|_| FenceDecodeError::UnsupportedFormatVersion(u32::MAX))?; diff --git a/libsql-server/src/namespace/meta_store.rs b/libsql-server/src/namespace/meta_store.rs index 098102e3f7..175fb24b55 100644 --- a/libsql-server/src/namespace/meta_store.rs +++ b/libsql-server/src/namespace/meta_store.rs @@ -91,6 +91,12 @@ struct MetaStoreInner { /// `/dbs`, where namespace directories and their fence markers are. dbs_path: PathBuf, fence: FenceSettings, + /// Namespaces whose state could not be recovered at startup and that are therefore + /// `UNKNOWN_UNAVAILABLE` (`docs/NAMESPACE_FENCE.md` section 13.3): an undecodable config + /// row, a fence that cannot be established, or a marker the metastore has no trustworthy + /// record for. They are refused by lookups and by every config or lifecycle change, and + /// never default-created. A fence command that commits for the name takes it out. + recovered: Mutex>, } /// How this metastore treats namespace fences (`docs/NAMESPACE_FENCE.md` section 13.1). @@ -100,6 +106,9 @@ struct FenceSettings { enabled: bool, /// The fence tables exist, so fence state is loaded and enforced. tables: bool, + /// Recovery fails closed (section 13.3): the flag is on, the fence tables exist, or a + /// namespace directory holds a marker. + fail_closed: bool, receipt_retention: Duration, } @@ -216,9 +225,13 @@ impl MetaStoreInner { if config.namespace_fence { fence_store::create_tables(&conn)?; } + let tables = fence_store::tables_exist(&conn)?; + let dbs_path = base_path.join("dbs"); + let marked = marked_namespaces(&dbs_path)?; let fence = FenceSettings { enabled: config.namespace_fence, - tables: fence_store::tables_exist(&conn)?, + tables, + fail_closed: config.namespace_fence || tables || !marked.is_empty(), receipt_retention: config .namespace_fence_receipt_retention .unwrap_or(fence_store::DEFAULT_RECEIPT_RETENTION), @@ -229,8 +242,9 @@ impl MetaStoreInner { conn: conn.into(), wal_manager, db_kind, - dbs_path: base_path.join("dbs"), + dbs_path, fence, + recovered: Default::default(), }; if config.allow_recover_from_fs { @@ -241,10 +255,72 @@ impl MetaStoreInner { if this.fence.tables { this.restore_fences()?; } + this.register_marked(&marked)?; Ok(this) } + /// Register every namespace directory with a marker that the metastore has no trustworthy + /// fence record for as `UNKNOWN_UNAVAILABLE` (section 13.3): a metastore that was rebuilt + /// (`destroy_on_error`), recovered from the filesystem, restored from an older backup, or + /// that lost its fence tables, and a target whose creation was interrupted. + fn register_marked(&mut self, marked: &[NamespaceName]) -> Result<()> { + for ns in marked { + if self.recovered.get_mut().contains_key(ns) { + continue; + } + let known = self.configs.get_mut().contains_key(ns); + if self.fence.tables { + if known { + // `restore_fences` compared the marker with the record. + continue; + } + let stored = match fence_store::read_fence(self.conn.get_mut(), &self.dbs_path, ns) + { + Ok((stored, _)) => stored, + Err(FenceStoreError::Sqlite(e)) => return Err(e.into()), + Err(e) => StoredFence::Unavailable { + detail: FenceDetail::CorruptRecord, + reason: format!("the fence marker cannot be read: {e}"), + marker: None, + }, + }; + let (detail, reason, marker) = match stored { + StoredFence::Unavailable { + detail, + reason, + marker, + } => (detail, reason, marker), + other => ( + FenceDetail::CorruptRecord, + format!( + "the namespace has fence state {} but no usable config row", + other.state() + ), + other.record().cloned(), + ), + }; + mark_unavailable(self.recovered.get_mut(), ns.clone(), detail, reason, marker); + } else { + let (detail, marker) = match fence_store::read_marker(&self.dbs_path, ns)? { + None => continue, + Some(Ok(m)) => (FenceDetail::MetastoreBehindMarker, Some(m.record)), + Some(Err(_)) => (FenceDetail::CorruptRecord, None), + }; + let reason = "the namespace directory holds a fence marker but the metastore has \ + no fence tables (it was rebuilt, recovered or restored without them)" + .to_string(); + mark_unavailable(self.recovered.get_mut(), ns.clone(), detail, reason, marker); + } + } + Ok(()) + } + + /// The fence error for a namespace registered as unavailable at startup. + fn recovery_denial(&self, namespace: &NamespaceName) -> Option { + self.recovered.lock().get(namespace).map(unavailable_error) + } + fn maybe_recover_from_fs(&mut self, base_path: &Path) -> Result<()> { let count = self.conn @@ -267,6 +343,16 @@ impl MetaStoreInner { let config_path = entry.path().join("config.json"); let name = NamespaceName::from_string(entry.file_name().to_str().unwrap().to_string())?; + if entry + .path() + .join(fence_store::MARKER_FILE_NAME) + .try_exists()? + { + // A fenced namespace is never recovered with a guessed config; it is + // registered as unavailable below (section 13.3). + tracing::warn!("not recovering fenced namespace `{name}` from the filesystem"); + continue; + } let config = if config_path.try_exists()? { let config_bytes = std::fs::read(&config_path)?; serde_json::from_slice(&config_bytes)? @@ -291,10 +377,10 @@ impl MetaStoreInner { fn restore(&mut self) -> Result<()> { tracing::info!("restoring meta store"); - let mut stmt = self - .conn - .get_mut() - .prepare("SELECT namespace, config FROM namespace_configs")?; + let fence = self.fence; + let conn: &rusqlite::Connection = self.conn.get_mut(); + let mut unavailable = Vec::new(); + let mut stmt = conn.prepare("SELECT namespace, config FROM namespace_configs")?; let rows = stmt.query(())?.mapped(|r| { let ns = r.get::<_, String>(0)?; @@ -306,9 +392,19 @@ impl MetaStoreInner { for row in rows { match row { Ok((k, v)) => { - let ns = match NamespaceName::from_string(k) { + let ns = match NamespaceName::from_string(k.clone()) { Ok(ns) => ns, Err(e) => { + // A name nothing can address cannot be served or default-created, + // so a legacy row is skipped as before. A fenced one is an operator + // problem: its fence could not be enforced or inspected. + if fence.tables && fence_store::stored_revision_raw(conn, &k)?.is_some() + { + return Err(Error::Internal(format!( + "the metastore holds a namespace fence for `{k}`, which is not \ + a valid namespace name; refusing to start" + ))); + } tracing::warn!("unable to convert namespace name: {}", e); continue; } @@ -316,6 +412,11 @@ impl MetaStoreInner { let config = match metadata::DatabaseConfig::decode(&v[..]) { Ok(c) => Arc::new(DatabaseConfig::from(&c)), + Err(e) if fence.fail_closed => { + unavailable + .push((ns, format!("the config row cannot be decoded: {e}"))); + continue; + } Err(e) => { tracing::warn!("unable to convert config: {}", e); continue; @@ -338,6 +439,17 @@ impl MetaStoreInner { } } + drop(stmt); + for (ns, reason) in unavailable { + mark_unavailable( + self.recovered.get_mut(), + ns, + FenceDetail::CorruptRecord, + reason, + None, + ); + } + tracing::info!("meta store restore completed"); Ok(()) @@ -358,7 +470,14 @@ impl MetaStoreInner { Ok(r) => r, Err(FenceStoreError::Sqlite(e)) => return Err(e.into()), Err(e) => { - tracing::error!(namespace = %ns, "cannot establish namespace fence: {e}"); + fenced += 1; + mark_unavailable( + self.recovered.get_mut(), + ns, + FenceDetail::CorruptRecord, + format!("the namespace fence cannot be established: {e}"), + None, + ); continue; } }; @@ -376,12 +495,18 @@ impl MetaStoreInner { let config = fence_store::with_legacy_blocks(&config, &record.legacy_blocks); sender.send_modify(|c| c.config = Arc::new(config)); } - StoredFence::Unavailable { detail, reason, .. } => { + StoredFence::Unavailable { + detail, + reason, + marker, + } => { fenced += 1; - tracing::error!( - namespace = %ns, - %detail, - "namespace fence state is UNKNOWN_UNAVAILABLE: {reason}" + mark_unavailable( + self.recovered.get_mut(), + ns, + *detail, + reason.clone(), + marker.clone(), ); } } @@ -391,6 +516,109 @@ impl MetaStoreInner { } } +fn mark_unavailable( + recovered: &mut HashMap, + namespace: NamespaceName, + detail: FenceDetail, + reason: String, + marker: Option, +) { + tracing::error!( + namespace = %namespace, + %detail, + "namespace is UNKNOWN_UNAVAILABLE: {reason}" + ); + recovered.insert( + namespace, + StoredFence::Unavailable { + detail, + reason, + marker, + }, + ); +} + +/// The namespaces under `dbs_path` whose directory holds a fence marker. A marker in a +/// directory that is not a valid namespace name stops startup: the fence it records could be +/// neither enforced nor inspected. +fn marked_namespaces(dbs_path: &Path) -> Result> { + fence_store::scan_markers(dbs_path)? + .into_iter() + .map(|m| { + m.map_err(|raw| { + Error::Internal(format!( + "namespace directory `{raw}` holds a fence marker but is not a valid \ + namespace name; refusing to start" + )) + }) + }) + .collect() +} + +/// A namespace's fence as the metastore holds it now, for a metastore with the fence tables: +/// the live record and marker, unless they read as established while startup could not +/// recover the namespace (an undecodable config row, for instance), which stays unavailable. +fn established_fence( + inner: &MetaStoreInner, + conn: &rusqlite::Connection, + namespace: &NamespaceName, +) -> std::result::Result { + let (live, _) = fence_store::read_fence(conn, &inner.dbs_path, namespace)?; + if matches!(live, StoredFence::Unavailable { .. }) { + return Ok(live); + } + Ok(inner + .recovered + .lock() + .get(namespace) + .cloned() + .unwrap_or(live)) +} + +/// The error returned for a namespace whose fence state is `UNKNOWN_UNAVAILABLE`. +fn unavailable_error(stored: &StoredFence) -> FenceError { + stored + .permits(OperationClass::NormalRead) + .err() + .unwrap_or_else(|| { + FenceError::new( + FenceOutcome::FenceStateUnavailable, + "the namespace's fence state cannot be established", + ) + }) +} + +/// Why a name that has no config must not be created: its directory holds a marker, so it is a +/// target being created or a namespace the metastore lost (section 13.3). +fn marker_denial(dbs_path: &Path, namespace: &NamespaceName) -> Result> { + Ok(match fence_store::read_marker(dbs_path, namespace)? { + None => None, + Some(Ok(m)) => { + let revision = m.record.revision; + Some( + StoredFence::Record(m.record) + .permits(OperationClass::Lifecycle) + .err() + .unwrap_or_else(|| { + unavailable_error(&StoredFence::Unavailable { + detail: FenceDetail::MetastoreBehindMarker, + reason: format!( + "the namespace directory holds a fence marker (revision \ + {revision}) but the metastore has no config for it" + ), + marker: None, + }) + }), + ) + } + Some(Err(e)) => Some(unavailable_error(&StoredFence::Unavailable { + detail: FenceDetail::CorruptRecord, + reason: format!("the fence marker cannot be decoded: {e}"), + marker: None, + })), + }) +} + /// Handles config change updates by inserting them into the database and in-memory /// cache of configs. fn process(msg: ChangeMsg, inner: Arc) { @@ -442,6 +670,9 @@ fn try_process( namespace: &NamespaceName, config: &DatabaseConfig, ) -> Result<()> { + if let Some(e) = inner.recovery_denial(namespace) { + return Err(e.into()); + } let mut conn = inner.conn.blocking_lock(); // `BEGIN IMMEDIATE`: the write lock is what serialises this write with fence transitions // (docs/NAMESPACE_FENCE.md section 5.4), including those of other metastore connections. @@ -688,6 +919,9 @@ fn apply_fence_command( .map_or(request.operation_id, |r| r.operation_id); fence_store::prune_receipts(&tx, ns, owner, ctx.now_ms, inner.fence.receipt_retention)?; tx.commit()?; + // The command established the fence from the durable state; whatever startup could not + // recover about this name is settled. + inner.recovered.lock().remove(ns); let current = record.or_else(|| stored.record().cloned()); after_fence_commit( @@ -796,6 +1030,7 @@ fn complete_fence_drain( fence_store::write_record(&tx, &next, fence_store::stored_revision(&tx, ns)?)?; fence_store::write_receipt(&tx, &final_receipt)?; tx.commit()?; + inner.recovered.lock().remove(ns); after_fence_commit(inner, &conn, Some(&next), true); @@ -834,12 +1069,36 @@ impl MetaStore { if destroy_on_error { let db_path = base_path.join("metastore"); - tracing::info!( - "meta store set to destroy on restore error, removing metastore db path folder ({:?})", db_path - ); + // With fences in use the broken metastore may hold the only record of a + // fence, so it is kept aside for the operator rather than deleted, and the + // rebuilt metastore registers every marked namespace as unavailable + // (section 13.3). + let keep = config.namespace_fence + || marked_namespaces(&base_path.join("dbs")) + .map_or(true, |marked| !marked.is_empty()); + if keep { + let millis = std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .map_or(0, |d| d.as_millis()); + let aside = base_path.join(format!("metastore.broken-{millis}")); + tracing::error!( + "meta store failed to restore ({e}); moving it aside to {aside:?} \ + and rebuilding it" + ); + if let Err(rename) = std::fs::rename(&db_path, &aside) { + tracing::error!( + "failed to move the metastore aside ({rename}); not destroying it" + ); + return Err(e); + } + } else { + tracing::info!( + "meta store set to destroy on restore error, removing metastore db path folder ({:?})", db_path + ); - if let Err(e) = std::fs::remove_dir_all(&db_path) { - tracing::error!("failed to remove base path({:?}): {}", &db_path, e); + if let Err(e) = std::fs::remove_dir_all(&db_path) { + tracing::error!("failed to remove base path({:?}): {}", &db_path, e); + } } if let Err(e) = std::fs::create_dir_all(&db_path) { @@ -900,11 +1159,38 @@ impl MetaStore { Ok(Self { changes_tx, inner }) } - pub async fn handle(&self, namespace: NamespaceName) -> MetaStoreHandle { + /// The handle of an existing namespace, without creating one (section 13.3). `Ok(None)` + /// when the namespace does not exist; a fence error when its state could not be recovered. + /// Every path that only reads or serves a namespace uses this. + pub async fn lookup(&self, namespace: &NamespaceName) -> Result> { + if let Some(e) = self.inner.recovery_denial(namespace) { + return Err(e.into()); + } + let configs = self.inner.configs.lock().await; + Ok(configs.get(namespace).map(|sender| MetaStoreHandle { + namespace: namespace.clone(), + inner: HandleState::External(self.changes_tx.clone(), sender.subscribe()), + })) + } + + /// The handle of `namespace`, creating an empty in-memory entry when it does not exist. + /// Only paths that create a namespace (create, fork destination, reset, lazy creation) use + /// this. It refuses a namespace whose state could not be recovered, and a name without a + /// config whose directory holds a fence marker (a target being created, or a namespace the + /// metastore lost): creating either would publish a default config where a fence belongs. + pub async fn handle(&self, namespace: NamespaceName) -> Result { tracing::debug!("getting meta store handle"); + if let Some(e) = self.inner.recovery_denial(&namespace) { + return Err(e.into()); + } let change_tx = self.changes_tx.clone(); let mut configs = self.inner.configs.lock().await; + if !configs.contains_key(&namespace) { + if let Some(e) = marker_denial(&self.inner.dbs_path, &namespace)? { + return Err(e.into()); + } + } let sender = configs.entry(namespace.clone()).or_insert_with(|| { // TODO(lucio): if no entry exists we need to ensure we send the update to // the bg channel. @@ -916,14 +1202,17 @@ impl MetaStore { tracing::debug!("meta handle subscribed"); - MetaStoreHandle { + Ok(MetaStoreHandle { namespace, inner: HandleState::External(change_tx, rx), - } + }) } pub fn remove(&self, namespace: NamespaceName) -> Result>> { tracing::debug!("removing namespace `{}` from meta store", namespace); + if let Some(e) = self.inner.recovery_denial(&namespace) { + return Err(e.into()); + } // "configs" lock can be used in both async and sync contexts while "conn" lock always used // in blocking context @@ -1055,17 +1344,24 @@ impl MetaStore { tokio::task::spawn_blocking(move || -> std::result::Result<_, FenceStoreError> { let mut conn = inner.conn.blocking_lock(); if !inner.fence.tables { - let tx = conn.transaction()?; - let exists = fence_store::read_config_row(&tx, &namespace)?.is_some(); + let recovered = inner.recovered.lock().get(&namespace).cloned(); + let fence = match recovered { + Some(fence) => fence, + None => { + let tx = conn.transaction()?; + StoredFence::None { + namespace_exists: fence_store::read_config_row(&tx, &namespace)? + .is_some(), + } + } + }; return Ok(FenceInspection { - fence: StoredFence::None { - namespace_exists: exists, - }, + fence, receipts: Vec::new(), }); } let tx = conn.transaction()?; - let (fence, _) = fence_store::read_fence(&tx, &inner.dbs_path, &namespace)?; + let fence = established_fence(&inner, &tx, &namespace)?; let receipts = fence_store::read_receipts(&tx, &namespace)?; Ok(FenceInspection { fence, receipts }) }) @@ -1078,22 +1374,33 @@ impl MetaStore { pub async fn load_fences(&self) -> Result> { let inner = self.inner.clone(); tokio::task::spawn_blocking(move || -> std::result::Result<_, FenceStoreError> { + let recovered: Vec<(NamespaceName, StoredFence)> = inner + .recovered + .lock() + .iter() + .map(|(ns, fence)| (ns.clone(), fence.clone())) + .collect(); if !inner.fence.tables { - return Ok(Vec::new()); + return Ok(recovered); } let mut conn = inner.conn.blocking_lock(); let tx = conn.transaction()?; - let names: Vec = { + let mut names: Vec = { let mut stmt = tx.prepare("SELECT namespace FROM namespace_configs")?; let rows = stmt.query_map((), |r| r.get::<_, String>(0))?; - rows.collect::>()? + rows.collect::>>()? + .into_iter() + .filter_map(|name| NamespaceName::from_string(name).ok()) + .collect() }; + for (ns, _) in recovered { + if !names.contains(&ns) { + names.push(ns); + } + } let mut out = Vec::new(); - for name in names { - let Ok(ns) = NamespaceName::from_string(name) else { - continue; - }; - let (fence, _) = fence_store::read_fence(&tx, &inner.dbs_path, &ns)?; + for ns in names { + let fence = established_fence(&inner, &tx, &ns)?; if !matches!(fence, StoredFence::None { .. }) { out.push((ns, fence)); } @@ -1382,7 +1689,7 @@ mod fence_tests { } async fn create_namespace(store: &MetaStore, ns: &'static str) -> MetaStoreHandle { - let handle = store.handle(ns.into()).await; + let handle = store.handle(ns.into()).await.unwrap(); handle .store(DatabaseConfig { max_db_pages: 1234, @@ -1485,7 +1792,7 @@ mod fence_tests { ); // The in-memory config is the namespace's own, not the mirror. - let handle = store.handle("db".into()).await; + let handle = store.handle("db".into()).await.unwrap(); let config = handle.get(); assert!(!config.block_writes && !config.block_reads); assert_eq!(config.block_reason.as_deref(), Some("pre-fence")); @@ -1547,7 +1854,7 @@ mod fence_tests { .await, ); assert_eq!(e.detail(), Some(FenceDetail::FenceDisabled)); - let handle = store.handle("db".into()).await; + let handle = store.handle("db".into()).await.unwrap(); let e = fence_error(handle.store(DatabaseConfig::default()).await); assert_eq!(e.outcome(), FenceOutcome::MigrationWriteFenced); assert_eq!( @@ -1760,9 +2067,29 @@ mod fence_tests { .. } )); - // The config keeps the stored mirror, so statement-level checks stay closed too. - let handle = store.handle("db".into()).await; - assert!(handle.get().block_writes); + // The namespace is not served and cannot be recreated. Its in-memory config keeps the + // stored mirror, so statement-level checks would stay closed too. + assert_eq!( + fence_error(store.lookup(&"db".into()).await).detail(), + Some(FenceDetail::CorruptRecord) + ); + assert_eq!( + fence_error(store.handle("db".into()).await).detail(), + Some(FenceDetail::CorruptRecord) + ); + let config = store.inner.configs.lock().await[&NamespaceName::from("db")] + .borrow() + .config + .clone(); + assert!(config.block_writes); + // A handle taken before the fence became unavailable cannot write the config either. + let handle = MetaStoreHandle { + namespace: "db".into(), + inner: HandleState::External( + store.changes_tx.clone(), + store.inner.configs.lock().await[&NamespaceName::from("db")].subscribe(), + ), + }; let release = request( "db", @@ -2073,4 +2400,397 @@ mod fence_tests { ); assert_eq!(inspection.fence.revision(), 6); } + + /// Fail-closed metastore recovery (`docs/NAMESPACE_FENCE.md` section 13.3). + mod recovery { + use super::*; + + fn unavailable_detail(r: Result) -> FenceDetail { + let e = fence_error(r); + assert_eq!(e.outcome(), FenceOutcome::FenceStateUnavailable, "{e}"); + e.detail().expect("unavailable carries a detail") + } + + async fn open_err(dir: &Path, config: MetaStoreConfig) -> Error { + let (maker, manager) = metastore_connection_maker(None, dir).await.unwrap(); + let conn = maker().unwrap(); + match MetaStore::new(config, dir, conn, manager, DatabaseKind::Primary).await { + Ok(_) => panic!("the metastore opened"), + Err(e) => e, + } + } + + fn recover_from_fs(fence: bool) -> MetaStoreConfig { + MetaStoreConfig { + allow_recover_from_fs: true, + namespace_fence: fence, + ..Default::default() + } + } + + /// A fenced namespace `db` (SOURCE_DRAINING, revision 1, with its marker). + async fn fenced_db(dir: &Path) -> NamespaceFenceRecord { + let store = open(dir, true).await; + let _handle = create_namespace(&store, "db").await; + store + .apply_fence_command(acquire("db", OP, 1), ctx(1_000)) + .await + .unwrap() + .record + .unwrap() + } + + fn assert_marker_unavailable(fence: &StoredFence, record: &NamespaceFenceRecord) { + match fence { + StoredFence::Unavailable { + detail: FenceDetail::MetastoreBehindMarker, + marker: Some(m), + .. + } => assert_eq!(m, record), + other => panic!("expected metastore_behind_marker, got {other:?}"), + } + } + + #[tokio::test] + async fn lookup_never_creates() { + let dir = tempdir().unwrap(); + let store = open(dir.path(), true).await; + assert!(store.lookup(&"missing".into()).await.unwrap().is_none()); + assert!(!store.exists(&"missing".into()).await); + + let _handle = create_namespace(&store, "db").await; + let found = store.lookup(&"db".into()).await.unwrap().unwrap(); + assert_eq!(found.get().max_db_pages, 1234); + // Only the creating path adds an entry. + let created = store.handle("new".into()).await.unwrap(); + assert_eq!( + created.get().max_db_pages, + DatabaseConfig::default().max_db_pages + ); + assert!(store.exists(&"new".into()).await); + } + + #[tokio::test] + async fn fs_recovery_with_marker_unavailable() { + for fence in [true, false] { + let dir = tempdir().unwrap(); + let record = fenced_db(dir.path()).await; + // A legacy namespace directory without a marker. + std::fs::create_dir_all(dir.path().join("dbs").join("legacy")).unwrap(); + std::fs::remove_dir_all(dir.path().join("metastore")).unwrap(); + + let store = open_with(dir.path(), recover_from_fs(fence)).await; + // The legacy directory is recovered as before. + let legacy = store.lookup(&"legacy".into()).await.unwrap().unwrap(); + assert_eq!( + legacy.get().max_db_pages, + DatabaseConfig::default().max_db_pages + ); + // The fenced one is not recovered with a guessed config, and is unavailable. + assert_eq!( + unavailable_detail(store.lookup(&"db".into()).await), + FenceDetail::MetastoreBehindMarker, + "fence flag {fence}" + ); + assert_eq!( + unavailable_detail(store.handle("db".into()).await), + FenceDetail::MetastoreBehindMarker + ); + assert_eq!( + unavailable_detail(remove_blocking(&store, "db")), + FenceDetail::MetastoreBehindMarker + ); + let inspection = store.inspect_fence("db".into()).await.unwrap(); + assert_marker_unavailable(&inspection.fence, &record); + let fences = store.load_fences().await.unwrap(); + assert_eq!(fences.len(), 1); + assert_marker_unavailable(&fences[0].1, &record); + // Nothing was written for it, and the marker is untouched. + let conn = raw(dir.path()).await; + assert!(fence_store::read_config_row(&conn, &"db".into()) + .unwrap() + .is_none()); + let marker = fence_store::read_marker(&dir.path().join("dbs"), &"db".into()) + .unwrap() + .unwrap() + .unwrap(); + assert_eq!(marker.record, record); + } + } + + #[tokio::test] + async fn destroy_on_error_keeps_fenced_unavailable() { + let dir = tempdir().unwrap(); + let record = fenced_db(dir.path()).await; + { + // Break the metastore so that restoring it fails. + let conn = raw(dir.path()).await; + conn.execute( + "ALTER TABLE namespace_configs RENAME COLUMN config TO broken", + (), + ) + .unwrap(); + } + let store = open_with( + dir.path(), + MetaStoreConfig { + destroy_on_error: true, + namespace_fence: true, + ..Default::default() + }, + ) + .await; + // The broken metastore is kept aside, not deleted. + let aside: Vec<_> = std::fs::read_dir(dir.path()) + .unwrap() + .map(|e| e.unwrap().file_name().into_string().unwrap()) + .filter(|n| n.starts_with("metastore.broken-")) + .collect(); + assert_eq!(aside.len(), 1, "{aside:?}"); + assert!(dir.path().join(&aside[0]).join("data").exists()); + // The rebuilt metastore knows nothing of `db`, so its marker makes it unavailable. + assert_eq!( + unavailable_detail(store.lookup(&"db".into()).await), + FenceDetail::MetastoreBehindMarker + ); + assert_eq!( + unavailable_detail(store.handle("db".into()).await), + FenceDetail::MetastoreBehindMarker + ); + assert_marker_unavailable( + &store.inspect_fence("db".into()).await.unwrap().fence, + &record, + ); + } + + #[tokio::test] + async fn destroy_on_error_without_fences_is_unchanged() { + let dir = tempdir().unwrap(); + { + let store = open(dir.path(), false).await; + let _handle = create_namespace(&store, "db").await; + } + { + let conn = raw(dir.path()).await; + conn.execute( + "ALTER TABLE namespace_configs RENAME COLUMN config TO broken", + (), + ) + .unwrap(); + } + let store = open_with( + dir.path(), + MetaStoreConfig { + destroy_on_error: true, + ..Default::default() + }, + ) + .await; + assert!(store.lookup(&"db".into()).await.unwrap().is_none()); + assert!(!std::fs::read_dir(dir.path()).unwrap().any(|e| e + .unwrap() + .file_name() + .to_string_lossy() + .starts_with("metastore."))); + } + + #[tokio::test] + async fn undecodable_row_unavailable() { + let dir = tempdir().unwrap(); + { + let store = open(dir.path(), true).await; + let _handle = create_namespace(&store, "good").await; + } + let conn = raw(dir.path()).await; + conn.execute( + "INSERT INTO namespace_configs VALUES ('bad', X'FFFFFFFF')", + (), + ) + .unwrap(); + let store = open(dir.path(), true).await; + assert!(store.lookup(&"good".into()).await.unwrap().is_some()); + assert_eq!( + unavailable_detail(store.lookup(&"bad".into()).await), + FenceDetail::CorruptRecord + ); + // Never replaced by a default config, nor deleted. + assert_eq!( + unavailable_detail(store.handle("bad".into()).await), + FenceDetail::CorruptRecord + ); + assert_eq!( + unavailable_detail(remove_blocking(&store, "bad")), + FenceDetail::CorruptRecord + ); + assert!(matches!( + store.inspect_fence("bad".into()).await.unwrap().fence, + StoredFence::Unavailable { + detail: FenceDetail::CorruptRecord, + .. + } + )); + let bytes: Vec = conn + .query_row( + "SELECT config FROM namespace_configs WHERE namespace = 'bad'", + (), + |r| r.get(0), + ) + .unwrap(); + assert_eq!(bytes, vec![0xff; 4]); + } + + #[tokio::test] + async fn undecodable_row_without_fences_is_skipped_as_before() { + let dir = tempdir().unwrap(); + drop(open(dir.path(), false).await); + let conn = raw(dir.path()).await; + conn.execute( + "INSERT INTO namespace_configs VALUES ('bad', X'FFFFFFFF')", + (), + ) + .unwrap(); + let store = open(dir.path(), false).await; + assert!(store.lookup(&"bad".into()).await.unwrap().is_none()); + } + + #[tokio::test] + async fn undecodable_name_with_fence_fails_startup() { + let dir = tempdir().unwrap(); + drop(open(dir.path(), true).await); + let conn = raw(dir.path()).await; + let config = metadata::DatabaseConfig::from(&DatabaseConfig::default()).encode_to_vec(); + conn.execute("INSERT INTO namespace_configs VALUES ('', ?1)", [&config]) + .unwrap(); + // Without a fence it is skipped, as before. + let store = open(dir.path(), true).await; + assert!(store.load_fences().await.unwrap().is_empty()); + drop(store); + conn.execute("INSERT INTO namespace_fences VALUES ('', 1, 1, X'00')", ()) + .unwrap(); + let e = open_err(dir.path(), MetaStoreConfig::default()).await; + assert!(matches!(e, Error::Internal(_)), "{e}"); + } + + #[tokio::test] + async fn marker_in_invalid_directory_fails_startup() { + use std::os::unix::ffi::OsStrExt; + let dir = tempdir().unwrap(); + let bad = dir + .path() + .join("dbs") + .join(std::ffi::OsStr::from_bytes(b"\xff")); + std::fs::create_dir_all(&bad).unwrap(); + std::fs::write(bad.join(fence_store::MARKER_FILE_NAME), b"x").unwrap(); + let e = open_err(dir.path(), MetaStoreConfig::default()).await; + assert!(matches!(e, Error::Internal(_)), "{e}"); + } + + #[tokio::test] + async fn incomplete_target_unavailable() { + let dir = tempdir().unwrap(); + let store = open(dir.path(), true).await; + let commit = store + .apply_fence_command(create_target("tgt", 3), ctx(1_000)) + .await + .unwrap(); + let record = commit.record.unwrap(); + // Committed but not yet published: nothing can create a default namespace over it. + let e = fence_error(store.handle("tgt".into()).await); + assert_eq!(e.outcome(), FenceOutcome::MigrationTargetQuarantined); + assert!(store.lookup(&"tgt".into()).await.unwrap().is_none()); + drop(store); + + // A crash between the marker and the commit: the marker is all that is left. + let conn = raw(dir.path()).await; + for sql in [ + "DELETE FROM namespace_fence_receipts WHERE namespace = 'tgt'", + "DELETE FROM namespace_fences WHERE namespace = 'tgt'", + "DELETE FROM namespace_configs WHERE namespace = 'tgt'", + ] { + conn.execute(sql, ()).unwrap(); + } + let store = open(dir.path(), true).await; + assert_eq!( + unavailable_detail(store.lookup(&"tgt".into()).await), + FenceDetail::IncompleteTargetCreation + ); + assert_eq!( + unavailable_detail(store.handle("tgt".into()).await), + FenceDetail::IncompleteTargetCreation + ); + let fences = store.load_fences().await.unwrap(); + assert!(matches!( + fences.as_slice(), + [( + _, + StoredFence::Unavailable { + detail: FenceDetail::IncompleteTargetCreation, + .. + } + )] + )); + // The same command completes the creation, which settles the name. + let commit = store + .apply_fence_command(create_target("tgt", 3), ctx(2_000)) + .await + .unwrap(); + assert_eq!(commit.kind, FenceCommitKind::Committed); + assert_eq!(commit.record.as_ref().unwrap().identity, record.identity); + assert!(store.lookup(&"tgt".into()).await.unwrap().is_none()); + let e = fence_error(store.handle("tgt".into()).await); + assert_eq!(e.outcome(), FenceOutcome::MigrationTargetQuarantined); + assert_eq!( + store.inspect_fence("tgt".into()).await.unwrap().fence, + StoredFence::Record(commit.record.unwrap()) + ); + } + + #[tokio::test] + async fn metastore_rollback_detected_by_marker() { + let dir = tempdir().unwrap(); + let conn = raw(dir.path()).await; + let record = { + let store = open(dir.path(), true).await; + let _handle = create_namespace(&store, "db").await; + let record = store + .apply_fence_command(acquire("db", OP, 1), ctx(1_000)) + .await + .unwrap() + .record + .unwrap(); + drop(store); + // A metastore restored from a backup taken before the fence: the namespace is + // unfenced there, and only its marker remembers the fence. + conn.execute("DELETE FROM namespace_fence_receipts", ()) + .unwrap(); + conn.execute("DELETE FROM namespace_fences", ()).unwrap(); + record + }; + let store = open(dir.path(), true).await; + assert_eq!( + unavailable_detail(store.lookup(&"db".into()).await), + FenceDetail::MetastoreBehindMarker + ); + // Neither served, nor deleted, nor given a new config. + assert_eq!( + unavailable_detail(store.handle("db".into()).await), + FenceDetail::MetastoreBehindMarker + ); + assert_eq!( + unavailable_detail(remove_blocking(&store, "db")), + FenceDetail::MetastoreBehindMarker + ); + assert_marker_unavailable( + &store.inspect_fence("db".into()).await.unwrap().fence, + &record, + ); + // A new operation cannot acquire over the lost fence either. + let e = fence_error( + store + .apply_fence_command(acquire("db", OTHER_OP, 2), ctx(2_000)) + .await, + ); + assert_eq!(e.detail(), Some(FenceDetail::MetastoreBehindMarker)); + } + } } diff --git a/libsql-server/src/namespace/store.rs b/libsql-server/src/namespace/store.rs index 86e9438ccd..1813ef5187 100644 --- a/libsql-server/src/namespace/store.rs +++ b/libsql-server/src/namespace/store.rs @@ -173,7 +173,7 @@ impl NamespaceStore { ns.destroy().await?; } - let db_config = self.inner.metadata.handle(namespace.clone()).await; + let db_config = self.inner.metadata.handle(namespace.clone()).await?; // destroy on-disk database self.cleanup( &namespace, @@ -240,7 +240,9 @@ impl NamespaceStore { return Err(crate::Error::NamespaceDoesntExist(from.to_string())); } - let from_config = self.inner.metadata.handle(from.clone()).await; + let Some(from_config) = self.inner.metadata.lookup(&from).await? else { + return Err(crate::Error::NamespaceDoesntExist(from.to_string())); + }; let from_entry = self .load_namespace(&from, from_config.clone(), RestoreOption::Latest) .await?; @@ -275,7 +277,7 @@ impl NamespaceStore { should_delete: true, }; - let handle = self.inner.metadata.handle(to.clone()).await; + let handle = self.inner.metadata.handle(to.clone()).await?; handle .store_and_maybe_flush(Some(to_config.into()), false) .await?; @@ -322,13 +324,6 @@ impl NamespaceStore { where Fun: FnOnce(&Namespace) -> R, { - if namespace != NamespaceName::default() - && !self.inner.metadata.exists(&namespace).await - && !self.inner.allow_lazy_creation - { - return Err(Error::NamespaceDoesntExist(namespace.to_string())); - } - let f = { let name = namespace.clone(); move |ns: NamespaceEntry| async move { @@ -341,7 +336,15 @@ impl NamespaceStore { } }; - let handle = self.inner.metadata.handle(namespace.to_owned()).await; + // A lookup that cannot create: only the default namespace and lazy creation create a + // namespace here, and those refuse a name whose fence state is not established. + let handle = match self.inner.metadata.lookup(&namespace).await? { + Some(handle) => handle, + None if namespace == NamespaceName::default() || self.inner.allow_lazy_creation => { + self.inner.metadata.handle(namespace.clone()).await? + } + None => return Err(Error::NamespaceDoesntExist(namespace.to_string())), + }; f(self .load_namespace(&namespace, handle, RestoreOption::Latest) .await?) @@ -440,7 +443,7 @@ impl NamespaceStore { } let db_config = Arc::new(db_config); - let handle = self.inner.metadata.handle(namespace.clone()).await; + let handle = self.inner.metadata.handle(namespace.clone()).await?; tracing::debug!("storing db config"); handle.store(db_config).await?; tracing::debug!("completed storing db config, loading namespace"); diff --git a/libsql-server/src/schema/db.rs b/libsql-server/src/schema/db.rs index ec8dcad840..d0bce10128 100644 --- a/libsql-server/src/schema/db.rs +++ b/libsql-server/src/schema/db.rs @@ -486,6 +486,7 @@ mod test { meta_store .handle(schema.into()) .await + .unwrap() .store(DatabaseConfig { is_shared_schema: true, ..Default::default() @@ -502,6 +503,7 @@ mod test { meta_store .handle(name.into()) .await + .unwrap() .store(DatabaseConfig { shared_schema_name: Some(schema.into()), ..Default::default() @@ -579,6 +581,7 @@ mod test { assert!(meta_store .handle("ns1".into()) .await + .unwrap() .store(DatabaseConfig { shared_schema_name: Some("schema1".into()), ..Default::default() From d01a72d96da2bd794785bbf0c525fb0dc50f184c Mon Sep 17 00:00:00 2001 From: River Date: Tue, 29 Sep 2026 15:24:54 +0000 Subject: [PATCH 07/33] libsql-server: add fence registry and controller, install before first connection Add the in-memory authority for namespace fences: - `FenceRegistry`, held by `NamespaceStore` outside the namespace cache and seeded from `MetaStore::load_fences()` before anything is served, so an evicted and reloaded namespace gets the controller it had. Namespaces without fence state get an UNFENCED controller on first load; deleting a namespace drops its controller. - `FenceController`: a per-namespace transition lock and a `watch` gate (`GateSnapshot`: the durable fence, a write generation and an indeterminate flag). Commands commit in the metastore, are published to the gate and only then answered, on their own task so a lost response does not lose the publication. An error before COMMIT leaves the gate unchanged; a failed COMMIT closes the gate and refuses other commands with FENCE_COMMIT_INDETERMINATE until the same command is replayed. - `FenceConnState`, bound to the controller for every connection a `MakeLegacyConnection` opens, starting with its held connection, and shared with the connection's WAL wrapper. The checks that use it land in the next commit. - `cfg(test)` `FenceTestHooks` with the named hook points of the design. `NamespaceStore::with` and `make_namespace` refuse a namespace whose fence state is unavailable before any setup. Co-authored-by: Tomasz Szymczyszyn --- docs/NAMESPACE_FENCE.md | 23 +- .../src/connection/connection_core.rs | 6 + .../src/connection/connection_manager.rs | 11 +- libsql-server/src/connection/legacy.rs | 23 +- .../src/namespace/configurator/helpers.rs | 3 + .../src/namespace/configurator/mod.rs | 2 + .../src/namespace/configurator/primary.rs | 6 + .../src/namespace/configurator/replica.rs | 5 + .../src/namespace/configurator/schema.rs | 4 + .../src/namespace/fence/controller.rs | 865 ++++++++++++++++++ libsql-server/src/namespace/fence/hooks.rs | 141 +++ libsql-server/src/namespace/fence/mod.rs | 17 +- libsql-server/src/namespace/fence/registry.rs | 221 +++++ libsql-server/src/namespace/meta_store.rs | 15 +- libsql-server/src/namespace/mod.rs | 12 + libsql-server/src/namespace/store.rs | 229 +++++ 16 files changed, 1561 insertions(+), 22 deletions(-) create mode 100644 libsql-server/src/namespace/fence/controller.rs create mode 100644 libsql-server/src/namespace/fence/hooks.rs create mode 100644 libsql-server/src/namespace/fence/registry.rs diff --git a/docs/NAMESPACE_FENCE.md b/docs/NAMESPACE_FENCE.md index 0a007efc83..4226d20629 100644 --- a/docs/NAMESPACE_FENCE.md +++ b/docs/NAMESPACE_FENCE.md @@ -397,19 +397,22 @@ This is additive in proto3: older peers skip the unknown field; a newer replica ### 7.1 Registry -`FenceRegistry` (in `NamespaceStore`, outside the moka cache) maps `NamespaceName -> Arc`. It is loaded from the metastore (and markers) at startup, before any namespace is served, and changes only after a durable commit. Namespaces without a record get an `UNFENCED` controller lazily. Because the registry is not the cache value, cache eviction and lazy reload reinstall the same controller (section 8.5). +`FenceRegistry` (in `NamespaceStore`, outside the moka cache) maps `NamespaceName -> Arc`. It is seeded from `MetaStore::load_fences()` (fence rows, markers, and the namespaces startup could not recover) in `NamespaceStore::new`, before any namespace is served. Namespaces without a record get an `UNFENCED` controller lazily, on first load. Because the registry is not the cache value, cache eviction and lazy reload reinstall the same controller (section 8.5). Deleting a namespace, which deletes its fence state in the same metastore transaction, removes its controller. ### 7.2 Controller state Per namespace: -- `transition_lock`: a `tokio::sync::Mutex` serialising commands on this namespace. -- `gate`: a `tokio::sync::watch` of `GateSnapshot { state, revision, write: Open | Closed(code), read: Open | Closed(code), write_generation, capabilities }`. The WAL wrapper, `CoreConnection`, dump, replication and lifecycle code read it without locks. -- `write_generation: u64`, stored in the snapshot and **incremented on every transition that closes or opens write admission** (acquire, release, create, seal, publish, enable, abort, adopt). +- `transition_lock`: a `tokio::sync::Mutex` serialising commands on this namespace. A command holds it from its first check to its response (`FenceController::begin_transition` returns a `Transition` that owns the guard). +- `gate`: a `tokio::sync::watch` of `GateSnapshot { fence, write_generation, indeterminate }`, where `fence` is the durable fence as last published (a record, no record, or `UNKNOWN_UNAVAILABLE` with its detail). State, revision, owning operation and every admission (`permits(class)`, `write()`, `read()`) are derived from it through the permission matrix. The WAL wrapper, `CoreConnection`, dump, replication and lifecycle code read it without locks. Phase 5 adds the live capability set. +- `write_generation: u64`, in the snapshot, **incremented on every publication that changes the fence state, the owning operation, or the indeterminate flag**. That covers every transition that closes or opens write admission (acquire, release, create, seal, publish, enable, abort, adopt), and is conservative for the others. A replay that publishes the same durable state does not move it. +- `indeterminate: Option<(operation_id, command_id)>`: set when a command's `COMMIT` failed (or the task running it died) so that whether it applied is unknown. While set, every class except `Maintenance` and `Observability` is denied with `FENCE_STATE_UNAVAILABLE` / `indeterminate_commit`, and every other command is refused with `FENCE_COMMIT_INDETERMINATE`. A replay of the same command is answered by the metastore from the durable row (replayed if it had committed, applied if it had not) and clears it (section 8.4). - writer tracking, provided by the namespace's `ManagedConnectionWalManager` (section 8.2). - `read_leases`: counters and cancel handles per lease class (`sql`, `dump`, `replication`), with a `Notify` on every release. - `capabilities`: the live `MigrationCapability` set and an import-writer counter. -- in `cfg(test)` builds only, an optional `FenceTestHooks` (section 16). +- in `cfg(test)` builds only, a `FenceTestHooks` (section 16). + +`FenceController::apply_command` runs the command on its own task: a caller that goes away after the commit (a lost response) does not prevent the publication. The metastore maps a failed `COMMIT` to `FENCE_COMMIT_INDETERMINATE`; any other error (a refusal by the transition function, a busy metastore, a failure before `COMMIT`) proves nothing was written and leaves the gate exactly as it was. ### 7.3 Operation classes @@ -473,8 +476,8 @@ This makes the WAL gate independent of statement classification: DDL, misclassif ### 8.5 Restart and eviction -- At startup the registry is built from the metastore and markers before `NamespaceStore` serves anything. `make_namespace` takes the controller from the registry and passes it into the configurator's `setup()`, down to `MakeLegacyConnection::new` and every `LegacyConnection`, **before** the first connection (the maker's held `_db` connection) is created. The replication logger, dump and replication services get the same controller. -- `NamespaceStore::with` checks the registry before `handle()` or `load_namespace`: `UNKNOWN_UNAVAILABLE` is refused before any setup work. +- At startup the registry is built from the metastore and markers before `NamespaceStore` serves anything. `make_namespace` takes the controller from the registry and passes it into the configurator's `setup()` (primary, schema and replica), down to `MakeLegacyConnection::new`, which binds a `FenceConnState` to it for every `LegacyConnection` it opens, **starting with** the maker's held `_db` connection. The `Namespace` keeps the same controller (`Namespace::fence()`), which is how dump, replication and lifecycle code, all of which reach a namespace through `NamespaceStore::with`, read its gate. +- `NamespaceStore::with` and `make_namespace` check the registry before `lookup()`, `handle()` or any setup: `UNKNOWN_UNAVAILABLE` is refused before any setup work. - Idle or capacity eviction shuts the namespace down but leaves the controller in the registry; a lazy reload reinstalls the identical gate, revision and generation. A drain waiter that holds the evicted manager sees its connections close and is notified. - Namespaces in `SOURCE_DRAINING` or `TARGET_IMPORT_DRAINING` after a restart stay closed until the same command is replayed. Nothing advances in the background. @@ -674,7 +677,7 @@ Every transition emits one structured log event (target `libsql_server::fence::a ## 16. Test strategy (Design) -- **No new dependency.** The crate has no failpoint library. Race tests use `#[cfg(test)]` hooks: `FenceTestHooks` holds named points (`AfterInstallingGate`, `BeforeMetastoreCommit`, `AfterMetastoreCommit`, `BeforeGatePublish`, `InBeginWriteTxnAfterCheck`, `AfterManagerRelease`, `BeforeBoundaryCapture`, `AfterTargetRowsCommitted`, `BeforeResponse`), each able to park the task on a `tokio::sync::Barrier` or `Notify` or to inject an error or an indeterminate commit. Hooks compile only in the library's own test build, so they cost nothing in release builds; integration tests under `tests/` cover protocol behaviour and do not rely on hooks. +- **No new dependency.** The crate has no failpoint library. Race tests use `#[cfg(test)]` hooks: `FenceTestHooks` holds named points (`AfterInstallingGate`, `BeforeMetastoreCommit`, `AfterMetastoreCommit`, `BeforeGatePublish`, `InBeginWriteTxnAfterCheck`, `AfterManagerRelease`, `BeforeBoundaryCapture`, `AfterTargetRowsCommitted`, `BeforeResponse`), each able to park the task on a pair of `Notify`s (`pause_at` returns handles to wait until the task arrives and to release it) or to inject an error or an indeterminate commit. An armed point fires once. `BeforeMetastoreCommit` is reached immediately before the metastore transaction is started (an injected error there is a failure before commit); an injected indeterminate outcome at `AfterMetastoreCommit` is a commit that happened but was not acknowledged. Hooks compile only in the library's own test build, so they cost nothing in release builds; integration tests under `tests/` cover protocol behaviour and do not rely on hooks. - **Restart** at a boundary: reopen `MetaStore` and rebuild the registry on the same temporary directory, the way existing metastore tests do; integration tests stop and start a `TestServer` on the same path. - **Response loss**: the test drops the command future after `AfterMetastoreCommit` and then replays or inspects. - **Representative schemas**: synthetic multi-table schemas with indexes, triggers, views and an FTS5 table (FTS5 is compiled in). @@ -692,8 +695,8 @@ Planned test names; the table is updated as tests land. | 4 | Program that captured config before the fence is rejected at the WAL | `connection_core::tests::wal_gate_rejects_program_admitted_before_fence` | | 5 | Pre-fence transactions cannot write after release or publication | `fence::tests::stale_generation_cannot_write_after_release`, `stale_generation_cannot_write_after_enable_writes` | | 6 | Acquisition timeout returns `DRAINING`, admission stays closed | `fence::drain::tests::deadline_returns_draining_and_stays_closed` | -| 7 | Restart at every persistence boundary; indeterminate persistence keeps the gate closed until same-command reconciliation | `fence::tests::restart_at_each_boundary` (parameterised over hook points), `indeterminate_commit_keeps_gate_closed` | -| 8 | Evict and lazily reload a fenced namespace; identical admission | `tests::fence::lifecycle::evicted_namespace_reloads_same_gate` | +| 7 | Restart at every persistence boundary; indeterminate persistence keeps the gate closed until same-command reconciliation | `fence::tests::restart_at_each_boundary` (parameterised over hook points), `indeterminate_commit_keeps_gate_closed`; landed: `fence::controller::tests::{indeterminate_commit_keeps_writes_closed_until_replayed, indeterminate_commit_that_did_not_apply_is_retried_by_replay, failed_before_commit_leaves_gate_unchanged, publication_happens_before_the_response, committed_command_is_published_when_the_caller_goes_away}`, `namespace::store::fence_tests::restart_installs_the_durable_gate_before_serving` | +| 8 | Evict and lazily reload a fenced namespace; identical admission | `tests::fence::lifecycle::evicted_namespace_reloads_same_gate`; landed at unit level: `namespace::store::fence_tests::evicted_namespace_reloads_with_the_same_controller`, `fence::registry::tests::seeded_from_load_fences_including_recovered_names` | | 9 | Filesystem recovery, `destroy_on_error`, undecodable records, missing target quarantine, metastore backup rollback fail closed with provenance | `meta_store::fence_tests::recovery::{fs_recovery_with_marker_unavailable, destroy_on_error_keeps_fenced_unavailable, undecodable_row_unavailable, incomplete_target_unavailable, metastore_rollback_detected_by_marker, lookup_never_creates, undecodable_name_with_fence_fails_startup, marker_in_invalid_directory_fails_startup}`; legacy behaviour kept: `destroy_on_error_without_fences_is_unchanged`, `undecodable_row_without_fences_is_skipped_as_before`; `meta_store::fence_tests::corrupt_fence_row_fails_closed` | | 10 | Wrong owner, stale revision, invalid role/state, replay, command-id reuse; replay before revision check | `fence::transition::tests::*` (exhaustive over states × commands) | | 11 | Target creation raced with SQL, dump, replication, lifecycle never observable as writable or readable | `fence::target::tests::create_race_never_observable` | diff --git a/libsql-server/src/connection/connection_core.rs b/libsql-server/src/connection/connection_core.rs index 17a6f52961..9025914c62 100644 --- a/libsql-server/src/connection/connection_core.rs +++ b/libsql-server/src/connection/connection_core.rs @@ -394,6 +394,7 @@ mod test { use crate::auth::Authenticated; use crate::connection::legacy::MakeLegacyConnection; use crate::connection::{Connection as _, RequestContext, TXN_TIMEOUT}; + use crate::namespace::fence::controller::FenceController; use crate::namespace::meta_store::{metastore_connection_maker, MetaStore}; use crate::namespace::NamespaceName; use crate::query_result_builder::test::{test_driver, TestBuilder}; @@ -454,6 +455,7 @@ mod test { Default::default(), Arc::new(|_| unreachable!()), Arc::new(|| Sqlite3WalManager::default()), + FenceController::unfenced(Default::default()), ) .await .unwrap(); @@ -500,6 +502,7 @@ mod test { Default::default(), Arc::new(|_| unreachable!()), Arc::new(|| Sqlite3WalManager::default()), + FenceController::unfenced(Default::default()), ) .await .unwrap(); @@ -551,6 +554,7 @@ mod test { Default::default(), Arc::new(|_| unreachable!()), Arc::new(|| Sqlite3WalManager::default()), + FenceController::unfenced(Default::default()), ) .await .unwrap(); @@ -634,6 +638,7 @@ mod test { Default::default(), Arc::new(|_| unreachable!()), Arc::new(|| Sqlite3WalManager::default()), + FenceController::unfenced(Default::default()), ) .await .unwrap(); @@ -727,6 +732,7 @@ mod test { Default::default(), Arc::new(|_| unreachable!()), Arc::new(|| Sqlite3WalManager::default()), + FenceController::unfenced(Default::default()), ) .await .unwrap(); diff --git a/libsql-server/src/connection/connection_manager.rs b/libsql-server/src/connection/connection_manager.rs index 9fa22fbb8e..4baaa0ddc0 100644 --- a/libsql-server/src/connection/connection_manager.rs +++ b/libsql-server/src/connection/connection_manager.rs @@ -14,6 +14,7 @@ use rusqlite::ErrorCode; use super::connection_core::CoreConnection; use super::TXN_TIMEOUT; +use crate::namespace::fence::controller::FenceConnState; pub type ConnId = u64; pub type InnerWalManager = Sqlite3WalManager; @@ -117,12 +118,18 @@ impl Default for ConnectionManagerInner { pub struct ManagedConnectionWalWrapper { id: ConnId, manager: ConnectionManager, + /// The connection's fence state, which `begin_write_txn` checks against the namespace's + /// gate (`docs/NAMESPACE_FENCE.md` section 8.1). + // Installed here so that no connection exists without it; the check itself lands in the + // next commit of this series. + #[allow(dead_code)] + fence: Arc, } impl ManagedConnectionWalWrapper { - pub(crate) fn new(manager: ConnectionManager) -> Self { + pub(crate) fn new(manager: ConnectionManager, fence: Arc) -> Self { let id = manager.inner.next_conn_id.fetch_add(1, Ordering::SeqCst); - Self { id, manager } + Self { id, manager, fence } } pub fn id(&self) -> ConnId { diff --git a/libsql-server/src/connection/legacy.rs b/libsql-server/src/connection/legacy.rs index ae5addd70d..29676d237a 100644 --- a/libsql-server/src/connection/legacy.rs +++ b/libsql-server/src/connection/legacy.rs @@ -14,6 +14,8 @@ use tokio::time::Duration; use crate::error::Error; use crate::metrics::DESCRIBE_COUNT; use crate::namespace::broadcasters::BroadcasterHandle; +use crate::namespace::fence::controller::{FenceConnState, FenceController}; +use crate::namespace::fence::state::OperationClass; use crate::namespace::meta_store::MetaStoreHandle; use crate::namespace::ResolveNamespacePathFn; use crate::query_result_builder::{QueryBuilderConfig, QueryResultBuilder}; @@ -47,6 +49,9 @@ pub struct MakeLegacyConnection { block_writes: Arc, resolve_attach_path: ResolveNamespacePathFn, make_wal_manager: Arc InnerWalManager + Sync + Send + 'static>, + /// The namespace's fence controller. Every connection this maker opens, starting with the + /// held `_db` connection, carries a fence state bound to it. + fence: Arc, } impl MakeLegacyConnection @@ -69,6 +74,7 @@ where block_writes: Arc, resolve_attach_path: ResolveNamespacePathFn, make_wal_manager: Arc InnerWalManager + Sync + Send + 'static>, + fence: Arc, ) -> Result { let txn_timeout = config_store.get().txn_timeout.unwrap_or(TXN_TIMEOUT); @@ -89,6 +95,7 @@ where resolve_attach_path, connection_manager: ConnectionManager::new(txn_timeout), make_wal_manager, + fence, }; let db = this.try_create_db().await?; @@ -146,6 +153,7 @@ where self.resolve_attach_path.clone(), self.connection_manager.clone(), self.make_wal_manager.clone(), + FenceConnState::new(self.fence.clone(), OperationClass::NormalWrite), ) .await } @@ -165,6 +173,10 @@ where pub struct LegacyConnection { pub(super) inner: Arc>>>, + /// Shared with the connection's WAL wrapper. + // Read by the WAL gate and the program admission check in the next commit of this series. + #[allow(dead_code)] + pub(super) fence: Arc, } #[cfg(test)] @@ -185,6 +197,10 @@ impl LegacyConnection { Arc::new(|_| unreachable!()), ConnectionManager::new(TXN_TIMEOUT), Arc::new(|| Sqlite3WalManager::default()), + FenceConnState::new( + FenceController::unfenced(Default::default()), + OperationClass::NormalWrite, + ), ) .await .unwrap() @@ -195,6 +211,7 @@ impl Clone for LegacyConnection { fn clone(&self) -> Self { Self { inner: self.inner.clone(), + fence: self.fence.clone(), } } } @@ -321,11 +338,13 @@ where resolve_attach_path: ResolveNamespacePathFn, connection_manager: ConnectionManager, make_wal: Arc InnerWalManager + Sync + Send + 'static>, + fence: Arc, ) -> crate::Result { let (conn, id) = tokio::task::spawn_blocking({ let connection_manager = connection_manager.clone(); + let fence = fence.clone(); move || -> crate::Result<_> { - let manager = ManagedConnectionWalWrapper::new(connection_manager); + let manager = ManagedConnectionWalWrapper::new(connection_manager, fence); let id = manager.id(); let wal = make_wal().wrap(manager).wrap(wal_wrapper); @@ -366,7 +385,7 @@ where connection_manager.register_connection(&inner, id); - Ok(Self { inner }) + Ok(Self { inner, fence }) } pub async fn execute( diff --git a/libsql-server/src/namespace/configurator/helpers.rs b/libsql-server/src/namespace/configurator/helpers.rs index 599320783d..1f2524cade 100644 --- a/libsql-server/src/namespace/configurator/helpers.rs +++ b/libsql-server/src/namespace/configurator/helpers.rs @@ -24,6 +24,7 @@ use crate::connection::{Connection as _, MakeConnection, MakeThrottledConnection use crate::database::{PrimaryConnection, PrimaryConnectionMaker}; use crate::error::LoadDumpError; use crate::namespace::broadcasters::BroadcasterHandle; +use crate::namespace::fence::controller::FenceController; use crate::namespace::meta_store::MetaStoreHandle; use crate::namespace::replication_wal::{make_replication_wal_wrapper, ReplicationWalWrapper}; use crate::namespace::{ @@ -50,6 +51,7 @@ pub(super) async fn make_primary_connection_maker( broadcaster: BroadcasterHandle, make_wal_manager: Arc InnerWalManager + Sync + Send + 'static>, encryption_config: Option, + fence: Arc, ) -> crate::Result<( Arc, ReplicationWalWrapper, @@ -174,6 +176,7 @@ pub(super) async fn make_primary_connection_maker( block_writes, resolve_attach_path, make_wal_manager.clone(), + fence, ) .await? .throttled( diff --git a/libsql-server/src/namespace/configurator/mod.rs b/libsql-server/src/namespace/configurator/mod.rs index 517b21ca5a..029ab0b3ce 100644 --- a/libsql-server/src/namespace/configurator/mod.rs +++ b/libsql-server/src/namespace/configurator/mod.rs @@ -13,6 +13,7 @@ use crate::replication::script_backup_manager::ScriptBackupManager; use crate::StatsSender; use super::broadcasters::BroadcasterHandle; +use super::fence::controller::FenceController; use super::meta_store::MetaStoreHandle; use super::{ Namespace, NamespaceBottomlessDbIdInit, NamespaceName, NamespaceStore, ResetCb, @@ -119,6 +120,7 @@ pub trait ConfigureNamespace { resolve_attach_path: ResolveNamespacePathFn, store: NamespaceStore, broadcaster: BroadcasterHandle, + fence: Arc, ) -> Pin> + Send + 'a>>; fn cleanup<'a>( diff --git a/libsql-server/src/namespace/configurator/primary.rs b/libsql-server/src/namespace/configurator/primary.rs index f68405fad6..7b63671ed1 100644 --- a/libsql-server/src/namespace/configurator/primary.rs +++ b/libsql-server/src/namespace/configurator/primary.rs @@ -13,6 +13,7 @@ use crate::connection::{Connection as _, MakeConnection}; use crate::database::{Database, PrimaryDatabase}; use crate::namespace::broadcasters::BroadcasterHandle; use crate::namespace::configurator::helpers::make_primary_connection_maker; +use crate::namespace::fence::controller::FenceController; use crate::namespace::meta_store::MetaStoreHandle; use crate::namespace::{ Namespace, NamespaceBottomlessDbIdInit, NamespaceName, NamespaceStore, ResetCb, @@ -53,6 +54,7 @@ impl PrimaryConfigurator { db_path: Arc, broadcaster: BroadcasterHandle, encryption_config: Option, + fence: Arc, ) -> crate::Result { let mut join_set = JoinSet::new(); @@ -72,6 +74,7 @@ impl PrimaryConfigurator { broadcaster, self.make_wal_manager.clone(), encryption_config, + fence.clone(), ) .await?; @@ -112,6 +115,7 @@ impl PrimaryConfigurator { stats, db_config_store: meta_store_handle, path: db_path.into(), + fence, }) } } @@ -126,6 +130,7 @@ impl ConfigureNamespace for PrimaryConfigurator { resolve_attach_path: ResolveNamespacePathFn, _store: NamespaceStore, broadcaster: BroadcasterHandle, + fence: Arc, ) -> Pin> + Send + 'a>> { Box::pin(async move { let db_path: Arc = self.base.base_path.join("dbs").join(name.as_str()).into(); @@ -140,6 +145,7 @@ impl ConfigureNamespace for PrimaryConfigurator { db_path.clone(), broadcaster, self.base.encryption_config.clone(), + fence, ) .await { diff --git a/libsql-server/src/namespace/configurator/replica.rs b/libsql-server/src/namespace/configurator/replica.rs index b1a108af73..adea0fd406 100644 --- a/libsql-server/src/namespace/configurator/replica.rs +++ b/libsql-server/src/namespace/configurator/replica.rs @@ -18,6 +18,7 @@ use crate::connection::MakeConnection; use crate::database::{Database, ReplicaDatabase}; use crate::namespace::broadcasters::BroadcasterHandle; use crate::namespace::configurator::helpers::{make_stats, run_storage_monitor}; +use crate::namespace::fence::controller::FenceController; use crate::namespace::meta_store::MetaStoreHandle; use crate::namespace::{Namespace, NamespaceBottomlessDbIdInit, RestoreOption}; use crate::namespace::{NamespaceName, NamespaceStore, ResetCb, ResetOp, ResolveNamespacePathFn}; @@ -60,6 +61,7 @@ impl ConfigureNamespace for ReplicaConfigurator { resolve_attach_path: ResolveNamespacePathFn, store: NamespaceStore, broadcaster: BroadcasterHandle, + fence: Arc, ) -> Pin> + Send + 'a>> { Box::pin(async move { tracing::debug!("creating replica namespace"); @@ -104,6 +106,7 @@ impl ConfigureNamespace for ReplicaConfigurator { resolve_attach_path, store, broadcaster, + fence, ) .await; } @@ -220,6 +223,7 @@ impl ConfigureNamespace for ReplicaConfigurator { Arc::new(AtomicBool::new(false)), // this is always false for write proxy resolve_attach_path, self.make_wal_manager.clone(), + fence.clone(), ) .await?; @@ -274,6 +278,7 @@ impl ConfigureNamespace for ReplicaConfigurator { stats, db_config_store: meta_store_handle, path: db_path.into(), + fence, }) }) } diff --git a/libsql-server/src/namespace/configurator/schema.rs b/libsql-server/src/namespace/configurator/schema.rs index 275fd71e93..411ec91271 100644 --- a/libsql-server/src/namespace/configurator/schema.rs +++ b/libsql-server/src/namespace/configurator/schema.rs @@ -7,6 +7,7 @@ use crate::connection::config::DatabaseConfig; use crate::connection::connection_manager::InnerWalManager; use crate::database::{Database, SchemaDatabase}; use crate::namespace::broadcasters::BroadcasterHandle; +use crate::namespace::fence::controller::FenceController; use crate::namespace::meta_store::MetaStoreHandle; use crate::namespace::{ Namespace, NamespaceName, NamespaceStore, ResetCb, ResolveNamespacePathFn, RestoreOption, @@ -49,6 +50,7 @@ impl ConfigureNamespace for SchemaConfigurator { resolve_attach_path: ResolveNamespacePathFn, _store: NamespaceStore, broadcaster: BroadcasterHandle, + fence: Arc, ) -> std::pin::Pin> + Send + 'a>> { Box::pin(async move { let mut join_set = JoinSet::new(); @@ -69,6 +71,7 @@ impl ConfigureNamespace for SchemaConfigurator { broadcaster, self.make_wal_manager.clone(), self.base.encryption_config.clone(), + fence.clone(), ) .await?; @@ -90,6 +93,7 @@ impl ConfigureNamespace for SchemaConfigurator { stats, db_config_store: db_config.clone(), path: db_path.into(), + fence, }) }) } diff --git a/libsql-server/src/namespace/fence/controller.rs b/libsql-server/src/namespace/fence/controller.rs new file mode 100644 index 0000000000..265e04e40f --- /dev/null +++ b/libsql-server/src/namespace/fence/controller.rs @@ -0,0 +1,865 @@ +//! The per-namespace fence controller (`docs/NAMESPACE_FENCE.md` sections 7 and 8.4). +//! +//! A [`FenceController`] is the in-memory authority for one namespace's fence. It owns the +//! transition lock that serialises fence commands on the namespace, and the gate that every +//! admission path reads: a `watch` of [`GateSnapshot`]. The gate changes only after the +//! metastore has committed (commit → publish → respond), except that a commit whose outcome is +//! unknown closes it until the same command is replayed. +//! +//! Controllers live in the [`FenceRegistry`](super::registry::FenceRegistry), not in the +//! namespace cache, so evicting and reloading a namespace hands the reloaded namespace the +//! same controller. + +use std::sync::atomic::{AtomicU64, Ordering}; +use std::sync::Arc; + +use parking_lot::Mutex; +use tokio::sync::{watch, OwnedMutexGuard}; +use uuid::Uuid; + +use crate::error::Error; +use crate::namespace::meta_store::{FenceCommit, FenceContext, MetaStore}; +use crate::namespace::NamespaceName; + +use super::command::FenceRequest; +#[cfg(test)] +use super::hooks::FenceTestHooks; +use super::hooks::{HookOutcome, HookPoint}; +use super::outcome::{FenceDetail, FenceError, FenceOutcome}; +use super::state::{Admission, FenceState, OperationClass}; +use super::store::StoredFence; +use super::transition::DrainCompletion; + +/// `(operation_id, command_id)` of a fence command. +pub type CommandKey = (Uuid, Uuid); + +/// What every admission path reads: the fence as last published, the write-admission +/// generation, and whether a commit is indeterminate. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct GateSnapshot { + /// The durable fence as of the last publication (or as loaded at startup). + pub fence: StoredFence, + /// Incremented on every published change of the fence state, of its owning operation, or + /// of the indeterminate flag. A write transaction may only start when the generation its + /// program and its read transaction were admitted under equals this one (section 8.1). + pub write_generation: u64, + /// A command whose commit outcome is unknown. While set, every class except maintenance + /// and observability is denied, and every other command is refused. + pub indeterminate: Option, +} + +impl GateSnapshot { + fn new(fence: StoredFence) -> Self { + Self { + fence, + write_generation: 0, + indeterminate: None, + } + } + + pub fn state(&self) -> FenceState { + self.fence.state() + } + + pub fn revision(&self) -> u64 { + self.fence.revision() + } + + pub fn operation_id(&self) -> Option { + self.fence.record().map(|r| r.operation_id) + } + + pub fn is_unavailable(&self) -> bool { + matches!(self.fence, StoredFence::Unavailable { .. }) + } + + /// The gate's decision for work of `class`. + pub fn permits(&self, class: OperationClass) -> Result<(), FenceError> { + if let Some((operation_id, command_id)) = self.indeterminate { + if !matches!( + class, + OperationClass::Maintenance | OperationClass::Observability + ) { + return Err(FenceError::new( + FenceOutcome::FenceStateUnavailable, + format!( + "the outcome of fence command {command_id} of operation {operation_id} \ + is not known yet" + ), + ) + .with_detail(FenceDetail::IndeterminateCommit)); + } + } + self.fence.permits(class) + } + + /// Normal write admission. + pub fn write(&self) -> Admission { + Admission::from( + self.permits(OperationClass::NormalWrite) + .map_err(|e| e.outcome()), + ) + } + + /// Normal read admission. + pub fn read(&self) -> Admission { + Admission::from( + self.permits(OperationClass::NormalRead) + .map_err(|e| e.outcome()), + ) + } +} + +/// The fence controller of one namespace. +pub struct FenceController { + namespace: NamespaceName, + transition_lock: Arc>, + gate: watch::Sender, + #[cfg(test)] + hooks: FenceTestHooks, +} + +impl std::fmt::Debug for FenceController { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("FenceController") + .field("namespace", &self.namespace) + .field("gate", &*self.gate.borrow()) + .finish_non_exhaustive() + } +} + +impl FenceController { + /// A controller whose gate starts from `fence`, as established by the metastore. + pub fn new(namespace: NamespaceName, fence: StoredFence) -> Arc { + let (gate, _) = watch::channel(GateSnapshot::new(fence)); + Arc::new(Self { + namespace, + transition_lock: Default::default(), + gate, + #[cfg(test)] + hooks: FenceTestHooks::default(), + }) + } + + /// A controller for a namespace without fence state. + pub fn unfenced(namespace: NamespaceName) -> Arc { + Self::new( + namespace, + StoredFence::None { + namespace_exists: true, + }, + ) + } + + pub fn namespace(&self) -> &NamespaceName { + &self.namespace + } + + /// A copy of the current gate. + pub fn gate(&self) -> GateSnapshot { + self.gate.borrow().clone() + } + + /// A receiver that observes every publication. + pub fn subscribe(&self) -> watch::Receiver { + self.gate.subscribe() + } + + pub fn write_generation(&self) -> u64 { + self.gate.borrow().write_generation + } + + /// The live gate's decision for work of `class`. + pub fn permits(&self, class: OperationClass) -> Result<(), FenceError> { + self.gate.borrow().permits(class) + } + + /// Take the namespace's transition lock. Every fence command on the namespace runs while + /// holding it, from its first check to its response. + pub async fn begin_transition(self: &Arc) -> Transition { + let guard = self.transition_lock.clone().lock_owned().await; + Transition { + controller: self.clone(), + _guard: guard, + } + } + + /// Run one fence command to completion: take the transition lock, commit it in the + /// metastore, publish the result to the gate and return it. + /// + /// The work runs on its own task, so a caller that goes away (a lost response) does not + /// stop the publication of a command that committed. + pub async fn apply_command( + self: &Arc, + meta: &MetaStore, + request: FenceRequest, + ctx: FenceContext, + ) -> crate::Result { + let this = self.clone(); + let meta = meta.clone(); + tokio::spawn(async move { + let mut transition = this.begin_transition().await; + transition.apply(&meta, request, ctx).await + }) + .await? + } + + #[cfg(test)] + pub fn hooks(&self) -> &FenceTestHooks { + &self.hooks + } + + /// Reach a test hook point. Outside the library's own test build this does nothing. + #[cfg(test)] + pub(crate) async fn hook(&self, point: HookPoint) -> HookOutcome { + self.hooks.hit(point).await + } + + #[cfg(not(test))] + #[inline(always)] + pub(crate) async fn hook(&self, _point: HookPoint) -> HookOutcome { + HookOutcome::Continue + } + + /// Publish a new gate. `fence: None` keeps the published fence. The write generation moves + /// whenever the state, the owning operation or the indeterminate flag changes. + fn publish(&self, fence: Option, indeterminate: Option) { + self.gate.send_modify(|gate| { + let fence = fence.unwrap_or_else(|| gate.fence.clone()); + let changed = fence.state() != gate.fence.state() + || fence.record().map(|r| r.operation_id) + != gate.fence.record().map(|r| r.operation_id) + || indeterminate != gate.indeterminate; + gate.fence = fence; + gate.indeterminate = indeterminate; + if changed { + gate.write_generation += 1; + } + }); + let gate = self.gate.borrow(); + tracing::debug!( + namespace = %self.namespace, + state = %gate.state(), + revision = gate.revision(), + write_generation = gate.write_generation, + indeterminate = gate.indeterminate.is_some(), + "published namespace fence gate" + ); + } +} + +/// A fence command in progress on one namespace. Holds the namespace's transition lock until +/// dropped. +pub struct Transition { + controller: Arc, + _guard: OwnedMutexGuard<()>, +} + +impl Transition { + pub fn controller(&self) -> &Arc { + &self.controller + } + + /// Commit `request` in the metastore and publish the result. + pub async fn apply( + &mut self, + meta: &MetaStore, + request: FenceRequest, + ctx: FenceContext, + ) -> crate::Result { + debug_assert_eq!(&request.namespace, self.controller.namespace()); + let key = (request.operation_id, request.command_id); + self.commit( + key, + async move { meta.apply_fence_command(request, ctx).await }, + ) + .await + } + + /// Complete the drain that the `DRAINING` receipt `key` started, once the caller has + /// proven `completion`, and publish the result. + pub async fn complete_drain( + &mut self, + meta: &MetaStore, + key: CommandKey, + completion: DrainCompletion, + ctx: FenceContext, + ) -> crate::Result { + let namespace = self.controller.namespace().clone(); + self.commit(key, async move { + meta.complete_fence_drain(namespace, key.0, key.1, completion, ctx) + .await + }) + .await + } + + /// The commit → publish → respond sequence of section 8.4. + /// + /// - An error that proves nothing was committed leaves the gate exactly as it was. + /// - A commit whose outcome is unknown closes the gate (every class but maintenance and + /// observability) and marks `key` indeterminate: other commands get + /// `FENCE_COMMIT_INDETERMINATE` until `key` is replayed, and the replay, which the + /// metastore answers from the durable row, reopens it to whatever is durable. + /// - A commit is published before it is answered. + async fn commit(&mut self, key: CommandKey, run: F) -> crate::Result + where + F: std::future::Future>, + { + let controller = self.controller.clone(); + if let Some(pending) = controller.gate.borrow().indeterminate { + if pending != key { + return Err(pending_indeterminate(pending).into()); + } + } + + if let HookOutcome::Fail(e) = controller.hook(HookPoint::BeforeMetastoreCommit).await { + return Err(e.into()); + } + + let result = match run.await { + Ok(commit) => match controller.hook(HookPoint::AfterMetastoreCommit).await { + HookOutcome::Continue => Ok(commit), + HookOutcome::Indeterminate | HookOutcome::Fail(_) => Err(indeterminate( + key, + "the commit was not acknowledged (test hook)", + )), + }, + Err(e) if is_indeterminate(&e) => Err(indeterminate(key, &e.to_string())), + Err(e) => return Err(e), + }; + + match result { + Ok(commit) => { + let _ = controller.hook(HookPoint::BeforeGatePublish).await; + controller.publish(commit.record.clone().map(StoredFence::Record), None); + let _ = controller.hook(HookPoint::BeforeResponse).await; + Ok(commit) + } + Err(e) => { + tracing::error!( + namespace = %controller.namespace, + operation_id = %key.0, + command_id = %key.1, + "fence commit outcome unknown; the namespace stays closed until the command \ + is replayed: {e}" + ); + controller.publish(None, Some(key)); + Err(e.into()) + } + } + } +} + +/// Whether a metastore error leaves the commit's outcome unknown: the commit itself failed, or +/// the task running it died. +fn is_indeterminate(e: &Error) -> bool { + match e { + Error::NamespaceFence(f) => f.outcome() == FenceOutcome::FenceCommitIndeterminate, + Error::RuntimeTaskJoinError(_) => true, + _ => false, + } +} + +fn indeterminate(key: CommandKey, why: &str) -> FenceError { + FenceError::new( + FenceOutcome::FenceCommitIndeterminate, + format!( + "whether fence command {} of operation {} was committed is unknown ({why}); replay \ + the same command to reconcile", + key.1, key.0 + ), + ) + .with_detail(FenceDetail::IndeterminateCommit) +} + +fn pending_indeterminate((operation_id, command_id): CommandKey) -> FenceError { + FenceError::new( + FenceOutcome::FenceCommitIndeterminate, + format!( + "fence command {command_id} of operation {operation_id} has an unknown outcome; \ + only a replay of that command is accepted until it is reconciled" + ), + ) + .with_detail(FenceDetail::IndeterminateCommit) +} + +/// The fence state of one connection, shared by its WAL wrapper and its `CoreConnection` +/// (section 7.4). The WAL gate reads and writes it; this commit only installs it. +#[derive(Debug)] +pub struct FenceConnState { + controller: Arc, + class: OperationClass, + /// The write generation the current program was admitted under. + program_generation: AtomicU64, + /// The write generation the current read transaction was opened under. + txn_generation: AtomicU64, + /// The typed outcome of the last refusal at the WAL. + denial: Mutex>, +} + +impl FenceConnState { + pub fn new(controller: Arc, class: OperationClass) -> Arc { + let generation = controller.write_generation(); + Arc::new(Self { + controller, + class, + program_generation: AtomicU64::new(generation), + txn_generation: AtomicU64::new(generation), + denial: Mutex::new(None), + }) + } + + pub fn controller(&self) -> &Arc { + &self.controller + } + + pub fn class(&self) -> OperationClass { + self.class + } + + pub fn program_generation(&self) -> u64 { + self.program_generation.load(Ordering::Acquire) + } + + pub fn txn_generation(&self) -> u64 { + self.txn_generation.load(Ordering::Acquire) + } + + pub fn take_denial(&self) -> Option { + self.denial.lock().take() + } +} + +#[cfg(test)] +mod tests { + use std::path::Path; + + use tempfile::tempdir; + + use super::*; + use crate::config::MetaStoreConfig; + use crate::connection::config::DatabaseConfig; + use crate::database::DatabaseKind; + use crate::namespace::fence::command::FenceCommand; + use crate::namespace::fence::record::{FrozenBoundary, ServerIdentity}; + use crate::namespace::meta_store::{metastore_connection_maker, FenceCommitKind}; + + const LOG: Uuid = Uuid::from_u128(0x10); + const OP: Uuid = Uuid::from_u128(0xa); + const OTHER_OP: Uuid = Uuid::from_u128(0xb); + + pub(crate) async fn open_metastore(dir: &Path) -> MetaStore { + let (maker, manager) = metastore_connection_maker(None, dir).await.unwrap(); + let conn = maker().unwrap(); + MetaStore::new( + MetaStoreConfig { + namespace_fence: true, + ..Default::default() + }, + dir, + conn, + manager, + DatabaseKind::Primary, + ) + .await + .unwrap() + } + + pub(crate) async fn create_namespace(meta: &MetaStore, ns: &'static str) { + meta.handle(ns.into()) + .await + .unwrap() + .store(DatabaseConfig::default()) + .await + .unwrap(); + } + + fn ctx() -> FenceContext { + FenceContext::now( + ServerIdentity { + build: "test".into(), + instance_id: Uuid::from_u128(0x99), + }, + Some(LOG), + ) + } + + fn acquire(ns: &'static str, op: Uuid, command_id: u128) -> FenceRequest { + FenceRequest { + namespace: ns.into(), + operation_id: op, + command_id: Uuid::from_u128(command_id), + expected_state: FenceState::Unfenced, + expected_revision: 0, + command: FenceCommand::AcquireSourceWriteFence { + expected_log_id: LOG, + drain_policy: None, + }, + } + } + + fn release(ns: &'static str, op: Uuid, command_id: u128, revision: u64) -> FenceRequest { + FenceRequest { + namespace: ns.into(), + operation_id: op, + command_id: Uuid::from_u128(command_id), + expected_state: FenceState::SourceWriteFenced, + expected_revision: revision, + command: FenceCommand::ReleaseSourceWriteFence, + } + } + + fn outcome(r: &crate::Result) -> FenceOutcome { + match r { + Ok(c) => c.receipt.outcome, + Err(Error::NamespaceFence(e)) => e.outcome(), + Err(e) => panic!("unexpected error: {e}"), + } + } + + /// Acquire and complete the drain directly, as the write drain will: the controller ends in + /// `SOURCE_WRITE_FENCED`. + async fn fence_source(meta: &MetaStore, controller: &Arc, op: Uuid) { + let commit = controller + .apply_command(meta, acquire("ns", op, 1), ctx()) + .await + .unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Draining); + let mut t = controller.begin_transition().await; + let commit = t + .complete_drain( + meta, + (op, Uuid::from_u128(1)), + DrainCompletion::SourceWrites { + boundary: FrozenBoundary { + log_id: LOG, + frame_no: 0, + }, + }, + ctx(), + ) + .await + .unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + } + + #[tokio::test] + async fn committed_command_publishes_new_revision_and_generation() { + let tmp = tempdir().unwrap(); + let meta = open_metastore(tmp.path()).await; + create_namespace(&meta, "ns").await; + let controller = FenceController::unfenced("ns".into()); + let mut rx = controller.subscribe(); + assert!(controller.permits(OperationClass::NormalWrite).is_ok()); + assert_eq!(controller.write_generation(), 0); + + let commit = controller + .apply_command(&meta, acquire("ns", OP, 1), ctx()) + .await + .unwrap(); + assert_eq!(commit.kind, FenceCommitKind::Committed); + assert!(rx.has_changed().unwrap()); + let gate = rx.borrow_and_update().clone(); + assert_eq!(gate.state(), FenceState::SourceDraining); + assert_eq!(gate.revision(), 1); + assert_eq!(gate.operation_id(), Some(OP)); + assert_eq!(gate.write_generation, 1); + assert_eq!( + controller + .permits(OperationClass::NormalWrite) + .unwrap_err() + .outcome(), + FenceOutcome::MigrationWriteFenced + ); + assert!(controller.permits(OperationClass::NormalRead).is_ok()); + assert!(controller.permits(OperationClass::Maintenance).is_ok()); + + // The published gate is the durable one. + let inspected = meta.inspect_fence("ns".into()).await.unwrap(); + assert_eq!(inspected.fence, gate.fence); + } + + #[tokio::test] + async fn write_generation_bumps_on_every_write_admission_change() { + let tmp = tempdir().unwrap(); + let meta = open_metastore(tmp.path()).await; + create_namespace(&meta, "ns").await; + let controller = FenceController::unfenced("ns".into()); + + let mut generations = vec![controller.write_generation()]; + fence_source(&meta, &controller, OP).await; + // UNFENCED -> SOURCE_DRAINING -> SOURCE_WRITE_FENCED: two changes of state. + generations.push(controller.write_generation()); + let revision = controller.gate().revision(); + controller + .apply_command(&meta, release("ns", OP, 2, revision), ctx()) + .await + .unwrap(); + assert_eq!(controller.gate().state(), FenceState::Released); + assert!(controller.permits(OperationClass::NormalWrite).is_ok()); + generations.push(controller.write_generation()); + assert_eq!(generations, vec![0, 2, 3]); + + // A replay publishes the same durable state and does not move the generation. + let before = controller.gate(); + let replay = controller + .apply_command(&meta, release("ns", OP, 2, revision), ctx()) + .await + .unwrap(); + assert_eq!(replay.kind, FenceCommitKind::Replayed); + assert_eq!(controller.gate(), before); + } + + #[tokio::test] + async fn failed_before_commit_leaves_gate_unchanged() { + let tmp = tempdir().unwrap(); + let meta = open_metastore(tmp.path()).await; + create_namespace(&meta, "ns").await; + let controller = FenceController::unfenced("ns".into()); + let before = controller.gate(); + + // A refusal by the transition function. + let mut wrong = acquire("ns", OP, 1); + wrong.expected_revision = 7; + let r = controller.apply_command(&meta, wrong, ctx()).await; + assert_eq!(outcome(&r), FenceOutcome::FenceRevisionMismatch); + assert_eq!(controller.gate(), before); + + // An injected failure before the metastore transaction. + controller.hooks().fail_at( + HookPoint::BeforeMetastoreCommit, + FenceError::new(FenceOutcome::FencePreconditionFailed, "injected"), + ); + let r = controller + .apply_command(&meta, acquire("ns", OP, 1), ctx()) + .await; + assert_eq!(outcome(&r), FenceOutcome::FencePreconditionFailed); + assert_eq!(controller.gate(), before); + + // The metastore is busy (another connection holds its write lock): nothing is written + // and the gate does not move. + let (maker, _) = metastore_connection_maker(None, tmp.path()).await.unwrap(); + let mut other = maker().unwrap(); + let lock = other + .transaction_with_behavior(rusqlite::TransactionBehavior::Immediate) + .unwrap(); + let r = controller + .apply_command(&meta, acquire("ns", OP, 1), ctx()) + .await; + assert!( + matches!(r, Err(Error::RusqliteError(_))), + "expected a busy metastore, got {r:?}" + ); + assert_eq!(controller.gate(), before); + drop(lock); + let inspected = meta.inspect_fence("ns".into()).await.unwrap(); + assert!(matches!(inspected.fence, StoredFence::None { .. })); + assert!(inspected.receipts.is_empty()); + } + + #[tokio::test] + async fn indeterminate_commit_keeps_writes_closed_until_replayed() { + let tmp = tempdir().unwrap(); + let meta = open_metastore(tmp.path()).await; + create_namespace(&meta, "ns").await; + let controller = FenceController::unfenced("ns".into()); + fence_source(&meta, &controller, OP).await; + let fenced = controller.gate(); + let revision = fenced.revision(); + + // Release commits, but its acknowledgement is lost. + controller.hooks().arm( + HookPoint::AfterMetastoreCommit, + super::super::hooks::HookAction::Indeterminate, + ); + let r = controller + .apply_command(&meta, release("ns", OP, 2, revision), ctx()) + .await; + assert_eq!(outcome(&r), FenceOutcome::FenceCommitIndeterminate); + let gate = controller.gate(); + assert_eq!(gate.indeterminate, Some((OP, Uuid::from_u128(2)))); + assert!(gate.write_generation > fenced.write_generation); + // The durable state says released, but nothing is admitted until it is reconciled. + for class in [ + OperationClass::NormalWrite, + OperationClass::NormalRead, + OperationClass::Stream, + OperationClass::Lifecycle, + ] { + let e = controller.permits(class).unwrap_err(); + assert_eq!(e.outcome(), FenceOutcome::FenceStateUnavailable); + assert_eq!(e.detail(), Some(FenceDetail::IndeterminateCommit)); + } + assert!(controller.permits(OperationClass::Maintenance).is_ok()); + + // Any other command is refused with the indeterminate code. + let r = controller + .apply_command(&meta, release("ns", OTHER_OP, 3, revision), ctx()) + .await; + assert_eq!(outcome(&r), FenceOutcome::FenceCommitIndeterminate); + + // The replay reconciles from the durable row and publishes it. + let r = controller + .apply_command(&meta, release("ns", OP, 2, revision), ctx()) + .await + .unwrap(); + assert_eq!(r.kind, FenceCommitKind::Replayed); + let gate = controller.gate(); + assert_eq!(gate.indeterminate, None); + assert_eq!(gate.state(), FenceState::Released); + assert!(controller.permits(OperationClass::NormalWrite).is_ok()); + } + + #[tokio::test] + async fn indeterminate_commit_that_did_not_apply_is_retried_by_replay() { + let tmp = tempdir().unwrap(); + let meta = open_metastore(tmp.path()).await; + create_namespace(&meta, "ns").await; + let controller = FenceController::unfenced("ns".into()); + + // Simulate an indeterminate outcome for a command that never reached the metastore: + // the replay applies it. + controller.publish(None, Some((OP, Uuid::from_u128(1)))); + assert!(controller.permits(OperationClass::NormalWrite).is_err()); + let r = controller + .apply_command(&meta, acquire("ns", OP, 1), ctx()) + .await + .unwrap(); + assert_eq!(r.kind, FenceCommitKind::Committed); + let gate = controller.gate(); + assert_eq!(gate.indeterminate, None); + assert_eq!(gate.state(), FenceState::SourceDraining); + } + + #[tokio::test] + async fn publication_happens_before_the_response() { + let tmp = tempdir().unwrap(); + let meta = open_metastore(tmp.path()).await; + create_namespace(&meta, "ns").await; + let controller = FenceController::unfenced("ns".into()); + let before_publish = controller.hooks().pause_at(HookPoint::BeforeGatePublish); + + let task = tokio::spawn({ + let controller = controller.clone(); + let meta = meta.clone(); + async move { + controller + .apply_command(&meta, acquire("ns", OP, 1), ctx()) + .await + } + }); + before_publish.reached().await; + // Committed, not yet published: the gate still shows the old state. + assert_eq!(controller.gate().state(), FenceState::Unfenced); + assert!(matches!( + meta.inspect_fence("ns".into()).await.unwrap().fence, + StoredFence::Record(_) + )); + let before_response = controller.hooks().pause_at(HookPoint::BeforeResponse); + before_publish.resume(); + before_response.reached().await; + assert_eq!(controller.gate().state(), FenceState::SourceDraining); + assert!(!task.is_finished()); + before_response.resume(); + assert_eq!(outcome(&task.await.unwrap()), FenceOutcome::Draining); + } + + #[tokio::test] + async fn committed_command_is_published_when_the_caller_goes_away() { + let tmp = tempdir().unwrap(); + let meta = open_metastore(tmp.path()).await; + create_namespace(&meta, "ns").await; + let controller = FenceController::unfenced("ns".into()); + let after_commit = controller.hooks().pause_at(HookPoint::AfterMetastoreCommit); + + let caller = tokio::spawn({ + let controller = controller.clone(); + let meta = meta.clone(); + async move { + controller + .apply_command(&meta, acquire("ns", OP, 1), ctx()) + .await + } + }); + after_commit.reached().await; + // The response is lost: the caller is cancelled after the commit. + caller.abort(); + assert!(caller.await.unwrap_err().is_cancelled()); + let published = controller.hooks().pause_at(HookPoint::BeforeResponse); + after_commit.resume(); + published.reached().await; + assert_eq!(controller.gate().state(), FenceState::SourceDraining); + published.resume(); + + // The transition lock is free again and a replay answers from the receipt. + let replay = controller + .apply_command(&meta, acquire("ns", OP, 1), ctx()) + .await + .unwrap(); + assert_eq!(replay.receipt.outcome, FenceOutcome::Draining); + } + + #[tokio::test] + async fn transition_lock_serialises_commands() { + let tmp = tempdir().unwrap(); + let meta = open_metastore(tmp.path()).await; + create_namespace(&meta, "ns").await; + let controller = FenceController::unfenced("ns".into()); + + let held = controller.begin_transition().await; + let second = tokio::spawn({ + let controller = controller.clone(); + let meta = meta.clone(); + async move { + controller + .apply_command(&meta, acquire("ns", OP, 1), ctx()) + .await + } + }); + tokio::task::yield_now().await; + assert!(!second.is_finished()); + assert!(controller.transition_lock.try_lock().is_err()); + drop(held); + assert_eq!(outcome(&second.await.unwrap()), FenceOutcome::Draining); + } + + #[test] + fn unavailable_gate_denies_every_class_but_maintenance() { + let controller = FenceController::new( + "ns".into(), + StoredFence::Unavailable { + detail: FenceDetail::CorruptRecord, + reason: "test".into(), + marker: None, + }, + ); + let gate = controller.gate(); + assert!(gate.is_unavailable()); + for class in OperationClass::ALL { + let r = gate.permits(class); + match class { + OperationClass::Maintenance | OperationClass::Observability => { + assert!(r.is_ok()) + } + _ => { + let e = r.unwrap_err(); + assert_eq!(e.outcome(), FenceOutcome::FenceStateUnavailable); + assert_eq!(e.detail(), Some(FenceDetail::CorruptRecord)); + } + } + } + } + + #[test] + fn conn_state_starts_at_the_current_generation() { + let controller = FenceController::unfenced("ns".into()); + controller.publish(None, Some((OP, OP))); + controller.publish(None, None); + let state = FenceConnState::new(controller.clone(), OperationClass::NormalWrite); + assert_eq!(state.program_generation(), 2); + assert_eq!(state.txn_generation(), 2); + assert!(Arc::ptr_eq(state.controller(), &controller)); + assert_eq!(state.class(), OperationClass::NormalWrite); + assert!(state.take_denial().is_none()); + } +} diff --git a/libsql-server/src/namespace/fence/hooks.rs b/libsql-server/src/namespace/fence/hooks.rs new file mode 100644 index 0000000000..90ef0d1685 --- /dev/null +++ b/libsql-server/src/namespace/fence/hooks.rs @@ -0,0 +1,141 @@ +//! Test hooks for fence race tests (`docs/NAMESPACE_FENCE.md` section 16). +//! +//! The crate has no failpoint library. Instead, a [`FenceController`] built by the library's +//! own test build carries a `FenceTestHooks`: named points on the transition path where a +//! test can park the task until it releases it, or inject a failure. Race tests are written +//! against these points and never against elapsed time. Outside `cfg(test)` only the point +//! names exist, and reaching a point does nothing. +//! +//! [`FenceController`]: super::controller::FenceController + +#[cfg(test)] +use std::collections::HashMap; +#[cfg(test)] +use std::sync::Arc; + +#[cfg(test)] +use parking_lot::Mutex; +#[cfg(test)] +use tokio::sync::Notify; + +use super::outcome::FenceError; + +/// A named point on a fence transition or a gated path. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] +pub enum HookPoint { + /// The in-memory `INSTALLING` gate of a closing transition has been published. + AfterInstallingGate, + /// Under the transition lock, immediately before the metastore transaction runs. + BeforeMetastoreCommit, + /// The metastore transaction returned a committed result. + AfterMetastoreCommit, + /// The committed result is about to be published to the gate. + BeforeGatePublish, + /// In `begin_write_txn`, after the gate check admitted the transaction. + InBeginWriteTxnAfterCheck, + /// The connection manager released the write slot. + AfterManagerRelease, + /// A drain is about to read the frozen boundary. + BeforeBoundaryCapture, + /// The rows of a quarantined target were committed; the target is not published yet. + AfterTargetRowsCommitted, + /// The gate is published; the answer is about to be returned. + BeforeResponse, +} + +#[cfg(test)] +/// What happens when a task reaches an armed point. Every action fires once: reaching the +/// point disarms it. +#[derive(Debug, Clone)] +pub enum HookAction { + /// Signal `reached`, then wait until `resume` is notified. + Pause { + reached: Arc, + resume: Arc, + }, + /// Fail at this point with `error`, as if the step had failed before it took effect. + Fail(FenceError), + /// At `AfterMetastoreCommit`: report the commit as indeterminate even though it happened, + /// which is what a lost commit acknowledgement looks like to the controller. + Indeterminate, +} + +/// What the task that reached a point has to do next. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum HookOutcome { + Continue, + Fail(FenceError), + Indeterminate, +} + +#[cfg(test)] +/// The armed points of one controller. +#[derive(Debug, Default)] +pub struct FenceTestHooks { + armed: Mutex>, +} + +#[cfg(test)] +/// The handles a test uses to follow a paused task. +#[derive(Debug, Clone)] +pub struct Paused { + pub reached: Arc, + pub resume: Arc, +} + +#[cfg(test)] +impl Paused { + /// Wait until the task reaches the point. + pub async fn reached(&self) { + self.reached.notified().await + } + + /// Let the task continue. + pub fn resume(&self) { + self.resume.notify_one() + } +} + +#[cfg(test)] +impl FenceTestHooks { + pub fn arm(&self, point: HookPoint, action: HookAction) { + self.armed.lock().insert(point, action); + } + + /// Arm `point` to pause, returning the handles to wait for it and to release it. + pub fn pause_at(&self, point: HookPoint) -> Paused { + let paused = Paused { + reached: Arc::new(Notify::new()), + resume: Arc::new(Notify::new()), + }; + self.arm( + point, + HookAction::Pause { + reached: paused.reached.clone(), + resume: paused.resume.clone(), + }, + ); + paused + } + + pub fn fail_at(&self, point: HookPoint, error: FenceError) { + self.arm(point, HookAction::Fail(error)); + } + + /// Called by the code under test when it reaches `point`. + pub async fn hit(&self, point: HookPoint) -> HookOutcome { + let action = self.armed.lock().remove(&point); + match action { + None => HookOutcome::Continue, + Some(HookAction::Pause { reached, resume }) => { + // `notify_one` stores a permit, so a test that starts waiting after the task + // got here still sees it. + reached.notify_one(); + resume.notified().await; + HookOutcome::Continue + } + Some(HookAction::Fail(e)) => HookOutcome::Fail(e), + Some(HookAction::Indeterminate) => HookOutcome::Indeterminate, + } + } +} diff --git a/libsql-server/src/namespace/fence/mod.rs b/libsql-server/src/namespace/fence/mod.rs index 672b0565e8..d5ac933e2e 100644 --- a/libsql-server/src/namespace/fence/mod.rs +++ b/libsql-server/src/namespace/fence/mod.rs @@ -2,12 +2,14 @@ //! example, moving a database between servers) uses as the data-plane authority boundary for //! one namespace. //! -//! `docs/NAMESPACE_FENCE.md` is the contract and the design. This module holds the parts with -//! no I/O: the states and permission matrix ([`state`]), the stable outcome codes and their -//! protocol mappings ([`outcome`]), commands and their canonical fingerprint ([`command`]), -//! records, receipts and markers with their strict durable encoding ([`record`]), the pure -//! transition function ([`transition`]), and the metastore tables, compare-and-swap and marker -//! file that persist them ([`store`], driven by `MetaStore::apply_fence_command`). +//! `docs/NAMESPACE_FENCE.md` is the contract and the design. This module holds the states and +//! permission matrix ([`state`]), the stable outcome codes and their protocol mappings +//! ([`outcome`]), commands and their canonical fingerprint ([`command`]), records, receipts and +//! markers with their strict durable encoding ([`record`]), the pure transition function +//! ([`transition`]), the metastore tables, compare-and-swap and marker file that persist them +//! ([`store`], driven by `MetaStore::apply_fence_command`), and the in-memory authority built +//! on them: the per-namespace [`controller`] with its gate, the [`registry`] that holds the +//! controllers outside the namespace cache, and the test [`hooks`] on their paths. // The persistence, controller and protocol layers that consume these types land in the // following commits of this series; until then most of the module is unused by the rest of @@ -15,8 +17,11 @@ #![allow(dead_code)] pub mod command; +pub mod controller; +pub mod hooks; pub mod outcome; pub mod record; +pub mod registry; pub mod state; pub mod store; pub mod transition; diff --git a/libsql-server/src/namespace/fence/registry.rs b/libsql-server/src/namespace/fence/registry.rs new file mode 100644 index 0000000000..610189677b --- /dev/null +++ b/libsql-server/src/namespace/fence/registry.rs @@ -0,0 +1,221 @@ +//! The fence registry (`docs/NAMESPACE_FENCE.md` section 7.1). +//! +//! One [`FenceController`] per namespace, held outside the namespace cache so that eviction and +//! lazy reload hand a reloaded namespace the controller it had, with the same gate, revision +//! and write generation. The registry is seeded from the metastore before the namespace store +//! serves anything, and namespaces without fence state get an `UNFENCED` controller on first +//! use. + +use std::collections::HashMap; +use std::sync::Arc; + +use parking_lot::Mutex; + +use crate::namespace::NamespaceName; + +use super::controller::FenceController; +use super::outcome::FenceError; +use super::state::OperationClass; +use super::store::StoredFence; + +#[derive(Debug, Default)] +pub struct FenceRegistry { + controllers: Mutex>>, +} + +impl FenceRegistry { + /// A registry holding a controller for every namespace with fence state, as returned by + /// `MetaStore::load_fences` (which includes the namespaces startup could not recover). + pub fn seeded(fences: impl IntoIterator) -> Self { + let controllers = fences + .into_iter() + .map(|(ns, fence)| { + if let StoredFence::Unavailable { detail, reason, .. } = &fence { + tracing::error!( + namespace = %ns, + %detail, + "namespace fence state is unavailable; the namespace is not served: {reason}" + ); + } + let controller = FenceController::new(ns.clone(), fence); + (ns, controller) + }) + .collect(); + Self { + controllers: Mutex::new(controllers), + } + } + + /// The controller of `namespace`, creating an `UNFENCED` one if it has none yet. + pub fn controller(&self, namespace: &NamespaceName) -> Arc { + self.controllers + .lock() + .entry(namespace.clone()) + .or_insert_with(|| FenceController::unfenced(namespace.clone())) + .clone() + } + + /// The controller of `namespace`, if it has one. + pub fn get(&self, namespace: &NamespaceName) -> Option> { + self.controllers.lock().get(namespace).cloned() + } + + /// Forget `namespace`'s controller. Only for a namespace that was deleted together with its + /// fence state. + pub fn remove(&self, namespace: &NamespaceName) -> Option> { + self.controllers.lock().remove(namespace) + } + + /// Refuse a namespace whose fence state is `UNKNOWN_UNAVAILABLE`, before any work is done + /// to serve it. + pub fn check_available(&self, namespace: &NamespaceName) -> Result<(), FenceError> { + match self.get(namespace) { + Some(controller) => { + let gate = controller.gate(); + if gate.is_unavailable() { + gate.permits(OperationClass::NormalRead) + } else { + Ok(()) + } + } + None => Ok(()), + } + } + + pub fn len(&self) -> usize { + self.controllers.lock().len() + } +} + +#[cfg(test)] +mod tests { + use tempfile::tempdir; + use uuid::Uuid; + + use super::*; + use crate::config::MetaStoreConfig; + use crate::connection::config::DatabaseConfig; + use crate::database::DatabaseKind; + use crate::namespace::fence::command::{FenceCommand, FenceRequest}; + use crate::namespace::fence::outcome::{FenceDetail, FenceOutcome}; + use crate::namespace::fence::record::ServerIdentity; + use crate::namespace::fence::state::FenceState; + use crate::namespace::fence::store as fence_store; + use crate::namespace::meta_store::{metastore_connection_maker, FenceContext, MetaStore}; + + const LOG: Uuid = Uuid::from_u128(0x10); + const OP: Uuid = Uuid::from_u128(0xa); + + async fn open(dir: &std::path::Path) -> MetaStore { + let (maker, manager) = metastore_connection_maker(None, dir).await.unwrap(); + let conn = maker().unwrap(); + MetaStore::new( + MetaStoreConfig { + namespace_fence: true, + ..Default::default() + }, + dir, + conn, + manager, + DatabaseKind::Primary, + ) + .await + .unwrap() + } + + fn ctx() -> FenceContext { + FenceContext::now( + ServerIdentity { + build: "test".into(), + instance_id: Uuid::from_u128(0x99), + }, + Some(LOG), + ) + } + + #[tokio::test] + async fn seeded_from_load_fences_including_recovered_names() { + let tmp = tempdir().unwrap(); + { + let meta = open(tmp.path()).await; + for ns in ["fenced", "plain"] { + meta.handle(ns.into()) + .await + .unwrap() + .store(DatabaseConfig::default()) + .await + .unwrap(); + } + let controller = FenceController::unfenced("fenced".into()); + controller + .apply_command( + &meta, + FenceRequest { + namespace: "fenced".into(), + operation_id: OP, + command_id: Uuid::from_u128(1), + expected_state: FenceState::Unfenced, + expected_revision: 0, + command: FenceCommand::AcquireSourceWriteFence { + expected_log_id: LOG, + drain_policy: None, + }, + }, + ctx(), + ) + .await + .unwrap(); + meta.shutdown().await.unwrap(); + } + // A directory with an unreadable marker and no config: startup cannot recover it. + let lost = tmp.path().join("dbs").join("lost"); + std::fs::create_dir_all(&lost).unwrap(); + std::fs::write(lost.join(fence_store::MARKER_FILE_NAME), b"garbage").unwrap(); + + // Restart. + let meta = open(tmp.path()).await; + let registry = FenceRegistry::seeded(meta.load_fences().await.unwrap()); + assert_eq!(registry.len(), 2); + + let fenced = registry.get(&"fenced".into()).unwrap(); + let gate = fenced.gate(); + assert_eq!(gate.state(), FenceState::SourceDraining); + assert_eq!(gate.revision(), 1); + assert_eq!(gate.operation_id(), Some(OP)); + assert_eq!( + fenced + .permits(OperationClass::NormalWrite) + .unwrap_err() + .outcome(), + FenceOutcome::MigrationWriteFenced + ); + assert!(registry.check_available(&"fenced".into()).is_ok()); + + let lost = registry.get(&"lost".into()).unwrap(); + assert!(lost.gate().is_unavailable()); + let e = registry.check_available(&"lost".into()).unwrap_err(); + assert_eq!(e.outcome(), FenceOutcome::FenceStateUnavailable); + assert_eq!(e.detail(), Some(FenceDetail::CorruptRecord)); + for class in [OperationClass::NormalRead, OperationClass::NormalWrite] { + assert!(lost.permits(class).is_err()); + } + + // An ordinary namespace has no controller until it is used, then an UNFENCED one. + assert!(registry.get(&"plain".into()).is_none()); + let plain = registry.controller(&"plain".into()); + assert_eq!(plain.gate().state(), FenceState::Unfenced); + assert!(plain.permits(OperationClass::NormalWrite).is_ok()); + assert_eq!(registry.len(), 3); + } + + #[test] + fn controller_is_stable_until_removed() { + let registry = FenceRegistry::default(); + let a = registry.controller(&"ns".into()); + let b = registry.controller(&"ns".into()); + assert!(Arc::ptr_eq(&a, &b)); + assert!(Arc::ptr_eq(®istry.get(&"ns".into()).unwrap(), &a)); + assert!(registry.remove(&"ns".into()).is_some()); + assert!(!Arc::ptr_eq(®istry.controller(&"ns".into()), &a)); + } +} diff --git a/libsql-server/src/namespace/meta_store.rs b/libsql-server/src/namespace/meta_store.rs index 175fb24b55..6ed7b29fd5 100644 --- a/libsql-server/src/namespace/meta_store.rs +++ b/libsql-server/src/namespace/meta_store.rs @@ -804,6 +804,17 @@ fn not_primary() -> FenceError { .with_detail(FenceDetail::NotPrimary) } +/// A fence transaction whose `COMMIT` failed: whether it took effect is unknown, and the +/// controller keeps the namespace closed until the same command is replayed (section 8.4). +fn commit_indeterminate(e: rusqlite::Error) -> FenceStoreError { + FenceError::new( + FenceOutcome::FenceCommitIndeterminate, + format!("the metastore commit of a fence transition failed: {e}"), + ) + .with_detail(FenceDetail::IndeterminateCommit) + .into() +} + fn unavailable_receipt(e: impl std::fmt::Display) -> FenceError { FenceError::new( FenceOutcome::FenceStateUnavailable, @@ -918,7 +929,7 @@ fn apply_fence_command( .or(stored.record()) .map_or(request.operation_id, |r| r.operation_id); fence_store::prune_receipts(&tx, ns, owner, ctx.now_ms, inner.fence.receipt_retention)?; - tx.commit()?; + tx.commit().map_err(commit_indeterminate)?; // The command established the fence from the durable state; whatever startup could not // recover about this name is settled. inner.recovered.lock().remove(ns); @@ -1029,7 +1040,7 @@ fn complete_fence_drain( } fence_store::write_record(&tx, &next, fence_store::stored_revision(&tx, ns)?)?; fence_store::write_receipt(&tx, &final_receipt)?; - tx.commit()?; + tx.commit().map_err(commit_indeterminate)?; inner.recovered.lock().remove(ns); after_fence_commit(inner, &conn, Some(&next), true); diff --git a/libsql-server/src/namespace/mod.rs b/libsql-server/src/namespace/mod.rs index cba4030090..28ba60ea26 100644 --- a/libsql-server/src/namespace/mod.rs +++ b/libsql-server/src/namespace/mod.rs @@ -14,6 +14,7 @@ use crate::connection::Connection as _; use crate::database::Database; use crate::stats::Stats; +use self::fence::controller::FenceController; use self::meta_store::MetaStoreHandle; pub use self::name::NamespaceName; pub use self::store::NamespaceStore; @@ -66,6 +67,10 @@ pub struct Namespace { stats: Arc, db_config_store: MetaStoreHandle, path: Arc, + /// The namespace's fence controller, from the store's registry. Every connection, and the + /// dump and replication services that reach the namespace through the store, read its + /// gate. + fence: Arc, } impl Namespace { @@ -98,6 +103,13 @@ impl Namespace { Ok(()) } + // Read by the protocol layers that consult the gate outside a connection (dump, + // replication, lifecycle), which land later in this series. + #[allow(dead_code)] + pub(crate) fn fence(&self) -> &Arc { + &self.fence + } + pub fn config(&self) -> Arc { self.db_config_store.get() } diff --git a/libsql-server/src/namespace/store.rs b/libsql-server/src/namespace/store.rs index 1813ef5187..3a132cf10f 100644 --- a/libsql-server/src/namespace/store.rs +++ b/libsql-server/src/namespace/store.rs @@ -21,6 +21,7 @@ use crate::stats::Stats; use super::broadcasters::{BroadcasterHandle, BroadcasterRegistry}; use super::configurator::{DynConfigurator, NamespaceConfigurators}; +use super::fence::registry::FenceRegistry; use super::meta_store::{MetaStore, MetaStoreHandle}; use super::schema_lock::SchemaLocksRegistry; use super::{Namespace, ResetCb, ResetOp, ResolveNamespacePathFn, RestoreOption}; @@ -50,6 +51,9 @@ pub struct NamespaceStoreInner { broadcasters: BroadcasterRegistry, configurators: NamespaceConfigurators, db_kind: DatabaseKind, + /// Fence controllers, outside the cache: a namespace that is evicted and reloaded gets the + /// controller it had. + fences: FenceRegistry, } impl NamespaceStore { @@ -84,6 +88,13 @@ impl NamespaceStore { .time_to_idle(Duration::from_secs(86400)) .build(); + // Every namespace with fence state gets its controller before anything is served + // (section 8.5). + let fences = FenceRegistry::seeded(metadata.load_fences().await?); + if fences.len() > 0 { + tracing::info!("loaded {} namespace fence controllers", fences.len()); + } + Ok(Self { inner: Arc::new(NamespaceStoreInner { store, @@ -95,6 +106,7 @@ impl NamespaceStore { broadcasters: Default::default(), configurators, db_kind, + fences, }), }) } @@ -120,6 +132,8 @@ impl NamespaceStore { } }) .await??; + // The namespace's fence state went with it. + self.inner.fences.remove(&namespace); let mut bottomless_db_id_init = NamespaceBottomlessDbIdInit::FetchFromConfig; if let Some(ns) = self.inner.store.remove(&namespace).await { @@ -336,6 +350,9 @@ impl NamespaceStore { } }; + // A namespace whose fence state is unavailable is refused before any setup work. + self.inner.fences.check_available(&namespace)?; + // A lookup that cannot create: only the default namespace and lazy creation create a // namespace here, and those refuse a name whose fence state is not established. let handle = match self.inner.metadata.lookup(&namespace).await? { @@ -371,6 +388,10 @@ impl NamespaceStore { config: MetaStoreHandle, restore_option: RestoreOption, ) -> crate::Result { + // The controller is handed to the namespace before its first connection exists, so no + // connection is ever opened without a gate (section 8.5). + self.inner.fences.check_available(namespace)?; + let fence = self.inner.fences.controller(namespace); let ns = self .get_configurator(&config.get()) .setup( @@ -381,6 +402,7 @@ impl NamespaceStore { self.resolve_attach_fn(), self.clone(), self.broadcaster(namespace.clone()), + fence, ) .await?; @@ -535,3 +557,210 @@ impl NamespaceStore { .await } } + +#[cfg(test)] +mod fence_tests { + use std::path::Path; + + use libsql_sys::wal::Sqlite3WalManager; + use tempfile::tempdir; + use tokio::sync::Semaphore; + use uuid::Uuid; + + use super::*; + use crate::config::MetaStoreConfig; + use crate::namespace::configurator::{BaseNamespaceConfig, PrimaryConfig, PrimaryConfigurator}; + use crate::namespace::fence::command::{FenceCommand, FenceRequest}; + use crate::namespace::fence::outcome::{FenceDetail, FenceOutcome}; + use crate::namespace::fence::record::ServerIdentity; + use crate::namespace::fence::state::{FenceState, OperationClass}; + use crate::namespace::fence::store as fence_store; + use crate::namespace::meta_store::{metastore_connection_maker, FenceContext}; + + const LOG: Uuid = Uuid::from_u128(0x10); + const OP: Uuid = Uuid::from_u128(0xa); + + async fn open_store(dir: &Path) -> NamespaceStore { + let (maker, manager) = metastore_connection_maker(None, dir).await.unwrap(); + let meta = MetaStore::new( + MetaStoreConfig { + namespace_fence: true, + ..Default::default() + }, + dir, + maker().unwrap(), + manager, + DatabaseKind::Primary, + ) + .await + .unwrap(); + let mut configurators = NamespaceConfigurators::empty(); + configurators.with_primary(PrimaryConfigurator::new( + BaseNamespaceConfig { + base_path: dir.to_path_buf().into(), + extensions: Arc::new([]), + stats_sender: tokio::sync::mpsc::channel(1).0, + max_response_size: 100_000_000, + max_total_response_size: 100_000_000, + max_concurrent_connections: Arc::new(Semaphore::new(10)), + max_concurrent_requests: 10_000, + encryption_config: None, + connection_creation_timeout: None, + disable_intelligent_throttling: false, + }, + PrimaryConfig { + max_log_size: 1_000_000_000, + max_log_duration: None, + bottomless_replication: None, + scripted_backup: None, + checkpoint_interval: None, + }, + Arc::new(|| Sqlite3WalManager::default()), + )); + NamespaceStore::new(false, false, 10, meta, configurators, DatabaseKind::Primary) + .await + .unwrap() + } + + fn acquire(ns: &'static str) -> FenceRequest { + FenceRequest { + namespace: ns.into(), + operation_id: OP, + command_id: Uuid::from_u128(1), + expected_state: FenceState::Unfenced, + expected_revision: 0, + command: FenceCommand::AcquireSourceWriteFence { + expected_log_id: LOG, + drain_policy: None, + }, + } + } + + fn ctx() -> FenceContext { + FenceContext::now( + ServerIdentity { + build: "test".into(), + instance_id: Uuid::from_u128(0x99), + }, + Some(LOG), + ) + } + + #[tokio::test] + async fn evicted_namespace_reloads_with_the_same_controller() { + let tmp = tempdir().unwrap(); + let store = open_store(tmp.path()).await; + store + .create("ns".into(), RestoreOption::Latest, Default::default()) + .await + .unwrap(); + let (fence, stats) = store + .with("ns".into(), |ns| (ns.fence().clone(), ns.stats())) + .await + .unwrap(); + assert!(Arc::ptr_eq( + &fence, + &store.inner.fences.get(&"ns".into()).unwrap() + )); + fence + .apply_command(store.meta_store(), acquire("ns"), ctx()) + .await + .unwrap(); + let gate = fence.gate(); + assert_eq!(gate.state(), FenceState::SourceDraining); + + // Evict the namespace, as idle or capacity eviction does, and let it shut down. + store + .inner + .store + .invalidate(&NamespaceName::from("ns")) + .await; + store.inner.store.run_pending_tasks().await; + assert!(store + .inner + .store + .get(&NamespaceName::from("ns")) + .await + .is_none()); + + let (reloaded, reloaded_stats) = store + .with("ns".into(), |ns| (ns.fence().clone(), ns.stats())) + .await + .unwrap(); + // A new namespace instance, the same controller and gate. + assert!(!Arc::ptr_eq(&stats, &reloaded_stats)); + assert!(Arc::ptr_eq(&fence, &reloaded)); + assert_eq!(reloaded.gate(), gate); + assert!(reloaded.permits(OperationClass::NormalWrite).is_err()); + } + + #[tokio::test] + async fn restart_installs_the_durable_gate_before_serving() { + let tmp = tempdir().unwrap(); + { + let store = open_store(tmp.path()).await; + store + .create("ns".into(), RestoreOption::Latest, Default::default()) + .await + .unwrap(); + let fence = store.inner.fences.controller(&"ns".into()); + fence + .apply_command(store.meta_store(), acquire("ns"), ctx()) + .await + .unwrap(); + store.shutdown().await.unwrap(); + } + + let store = open_store(tmp.path()).await; + // The registry holds the durable gate before the namespace is loaded. + let fence = store.inner.fences.get(&"ns".into()).unwrap(); + assert_eq!(fence.gate().state(), FenceState::SourceDraining); + assert_eq!(fence.gate().revision(), 1); + let loaded = store + .with("ns".into(), |ns| ns.fence().clone()) + .await + .unwrap(); + assert!(Arc::ptr_eq(&fence, &loaded)); + } + + #[tokio::test] + async fn unavailable_namespace_is_refused_before_setup() { + let tmp = tempdir().unwrap(); + // An unreadable marker in a directory with no config: the fence state cannot be + // established. + let dir = tmp.path().join("dbs").join("lost"); + std::fs::create_dir_all(&dir).unwrap(); + std::fs::write(dir.join(fence_store::MARKER_FILE_NAME), b"garbage").unwrap(); + + let store = open_store(tmp.path()).await; + let r = store.with("lost".into(), |_| ()).await; + match r { + Err(Error::NamespaceFence(e)) => { + assert_eq!(e.outcome(), FenceOutcome::FenceStateUnavailable); + assert_eq!(e.detail(), Some(FenceDetail::CorruptRecord)); + } + other => panic!("expected a fence error, got {other:?}"), + } + // Nothing was set up: no database file, and the namespace is not cached. + assert!(!dir.join("data").exists()); + assert!(store + .inner + .store + .get(&NamespaceName::from("lost")) + .await + .is_none()); + } + + #[tokio::test] + async fn destroy_forgets_the_controller() { + let tmp = tempdir().unwrap(); + let store = open_store(tmp.path()).await; + store + .create("ns".into(), RestoreOption::Latest, Default::default()) + .await + .unwrap(); + assert!(store.inner.fences.get(&"ns".into()).is_some()); + store.destroy("ns".into(), false).await.unwrap(); + assert!(store.inner.fences.get(&"ns".into()).is_none()); + } +} From 775068496127e9d871ad2d2b3a5466ea234149f2 Mon Sep 17 00:00:00 2001 From: River Date: Tue, 29 Sep 2026 15:46:13 +0000 Subject: [PATCH 08/33] libsql-server: gate write transactions at the WAL by admission generation Make ManagedConnectionWalWrapper::begin_write_txn the authoritative namespace fence check. Before queueing for the write slot it requires the live gate to admit the connection's operation class and the program and its read transaction to have been admitted under the gate's current write generation. A refusal returns SQLITE_AUTH (not BUSY, so SQLite does not retry it, and before acquire(), so no slot is released that was never held) and leaves the typed outcome in the connection's FenceConnState. - FenceConnState gains begin_program, begin_read_txn and admit_write. CoreConnection::run, every with_raw call and vacuum_if_needed start a program; the WAL wrapper records the generation of each new read transaction before its snapshot is taken. - The Vm refuses Write and DDL statements early against the live gate and reports a WAL refusal as Error::NamespaceFence instead of SQLITE_AUTH. A plain SQLITE_AUTH from an authorizer is left unchanged. - New detail stale_transaction on MIGRATION_WRITE_FENCED for a transaction or program that began under an earlier generation. - Namespaces without a fence stay at generation 0 and behave as before. Tests cover a program parked between admission and its write while the fence is acquired and released, read-to-write upgrades, DDL, a header pragma, BEGIN IMMEDIATE, VACUUM and raw writes, stale transactions after release, and unchanged behaviour of unfenced namespaces. Co-authored-by: Tomasz Szymczyszyn --- docs/NAMESPACE_FENCE.md | 18 +- .../src/connection/connection_core.rs | 21 +- .../src/connection/connection_manager.rs | 383 +++++++++++++++++- libsql-server/src/connection/legacy.rs | 10 +- libsql-server/src/connection/program.rs | 43 +- .../src/namespace/fence/controller.rs | 86 +++- libsql-server/src/namespace/fence/outcome.rs | 3 + 7 files changed, 539 insertions(+), 25 deletions(-) diff --git a/docs/NAMESPACE_FENCE.md b/docs/NAMESPACE_FENCE.md index 4226d20629..278f472c62 100644 --- a/docs/NAMESPACE_FENCE.md +++ b/docs/NAMESPACE_FENCE.md @@ -367,7 +367,7 @@ Notes: - `FENCE_COMMAND_CONFLICT`: a `command_id` reused with a different request fingerprint. - `FENCE_COMMIT_INDETERMINATE`: the server could not establish whether its own commit landed (for example an I/O error on `COMMIT`). Admission stays closed; only a replay of the same `command_id` reconciles it; every other command on the namespace receives this code until then. -- `FENCE_PRECONDITION_FAILED` carries `detail`, one of: `admin_auth_required`, `fence_disabled`, `not_primary`, `shared_schema_unsupported`, `namespace_identity_mismatch`, `namespace_exists`, `validation_receipt_required`, `restore_not_allowed`, `adoption_not_authorised`, `invalid_argument`. `INVALID_FENCE_TRANSITION` may carry `role_mismatch` or `operation_finished`. `FENCE_STATE_UNAVAILABLE` carries the reason the state cannot be established: `corrupt_record`, `unsupported_format_version`, `incomplete_target_creation`, `metastore_behind_marker` or `indeterminate_commit`. +- `FENCE_PRECONDITION_FAILED` carries `detail`, one of: `admin_auth_required`, `fence_disabled`, `not_primary`, `shared_schema_unsupported`, `namespace_identity_mismatch`, `namespace_exists`, `validation_receipt_required`, `restore_not_allowed`, `adoption_not_authorised`, `invalid_argument`. `INVALID_FENCE_TRANSITION` may carry `role_mismatch` or `operation_finished`. `FENCE_STATE_UNAVAILABLE` carries the reason the state cannot be established: `corrupt_record`, `unsupported_format_version`, `incomplete_target_creation`, `metastore_behind_marker` or `indeterminate_commit`. `MIGRATION_WRITE_FENCED` carries `stale_transaction` when the gate itself admits writes but the transaction (or the program) attempting one began under an earlier write generation (section 8.1): the client must roll back and begin a new transaction. - **Data-plane denials are never `500`, `503`, `429` or gRPC `UNAVAILABLE`.** `423 Locked` is chosen because common HTTP clients do not retry it. The JSON error body of the user HTTP API gains an additive `"code"` field (`{"error": "...", "code": "MIGRATION_WRITE_FENCED"}`); the existing `Blocked` error (from `block_reads`/`block_writes`) keeps its current mapping. - gRPC statuses carry the code in the `x-libsql-fence-code` metadata entry and as the message prefix `": "`. - Authentication (`401`), missing namespace (`404`), timeouts and transport errors are distinct from all of the above. @@ -431,14 +431,18 @@ Every write-transaction request at the WAL, every read lease and every lifecycle ### 7.4 Per-connection fence state -Every `LegacyConnection` receives a `FenceConnState` shared by its `ManagedConnectionWalWrapper` and its `CoreConnection`: the connection's class and capability (if any), `program_generation` (captured at program start), `txn_generation` (captured at `begin_read_txn` when a new read transaction starts), and a `denial` slot for the typed outcome of the last WAL refusal. +Every `LegacyConnection` receives a `FenceConnState` shared by its `ManagedConnectionWalWrapper` and its `CoreConnection`: the connection's class and capability (if any), `program_generation`, `txn_generation`, and a `denial` slot for the typed outcome of the last WAL refusal. + +- `begin_program()` records `program_generation` and clears the denial slot. It is called at the start of every `CoreConnection::run`, of every `with_raw` call (admin shell, schema migration, dump load, the configurators' own uses) and of `vacuum_if_needed`, so every way of running SQL on the connection is a program. +- `begin_read_txn()` records `txn_generation`. The WAL wrapper calls it from `begin_read_txn` **before** the snapshot is taken, so a transition racing with it leaves the transaction with the older generation, which can only refuse a later upgrade. +- `admit_write()` is the check of section 8.1 (2). A refusal is stored in the denial slot and returned. ## 8. Write admission and positive drain (Design) ### 8.1 The two checks -1. **Program and lifecycle admission (early).** `CoreConnection::run` reads the *live* gate at program start. If the program contains a statement that can write and the gate denies it, it fails at once with the typed error. It records `program_generation`. Lifecycle entry points check the gate the same way. -2. **WAL `begin_write_txn` (authoritative).** In `ManagedConnectionWalWrapper::begin_write_txn`, **before** `acquire()`, the wrapper requires: the gate permits the connection's class (and capability), and `program_generation == txn_generation == gate.write_generation`. On refusal it writes the typed outcome into the `denial` slot and returns `SQLITE_AUTH` — a non-`BUSY` code, so SQLite's busy handler does not retry it, and before `acquire()`, so no slot is released that was never held. `Vm::try_step` turns an `SQLITE_AUTH` with a filled `denial` slot into `Error::Fence(outcome)`; without a filled slot the SQLite error is returned unchanged. +1. **Program and lifecycle admission (early).** `CoreConnection::run` records `program_generation` at program start. Before each statement classified as `Write` or `DDL` runs, the `Vm` reads the *live* gate and, if it denies the connection's class, fails that step with `Error::NamespaceFence` (the step fails, as a `block_writes` denial does; read steps of the same batch are unaffected). Checking per statement rather than once per program also refuses the remaining writes of a batch the fence arrived in the middle of. Lifecycle entry points check the gate the same way. +2. **WAL `begin_write_txn` (authoritative).** In `ManagedConnectionWalWrapper::begin_write_txn`, **before** `acquire()`, the wrapper requires: the gate permits the connection's class (and capability), and `program_generation == txn_generation == gate.write_generation`. On refusal it writes the typed outcome into the `denial` slot and returns `SQLITE_AUTH` — a non-`BUSY` code, so SQLite's busy handler does not retry it, and before `acquire()`, so no slot is released that was never held. `Vm::try_step` turns an `SQLITE_AUTH` with a filled `denial` slot into `Error::NamespaceFence(outcome)`; without a filled slot (for example the `SQLITE_AUTH` of an authorizer) the SQLite error is returned unchanged. A `with_raw` caller sees the bare `SQLITE_AUTH` and can take the typed reason from the slot. When the gate admits writes but a generation is stale, the outcome is `MIGRATION_WRITE_FENCED` with detail `stale_transaction`. This makes the WAL gate independent of statement classification: DDL, misclassified PRAGMAs, `with_raw` users (admin shell, schema migrations, dump load), a program that snapshotted config before the fence, and read-to-write upgrades all converge on `begin_write_txn`. @@ -691,9 +695,9 @@ Planned test names; the table is updated as tests land. |---|---|---| | 1 | Concurrent acquisition by two operations: one owner, typed conflict for the loser | `fence::tests::acquire_race_single_owner`; `tests::fence::admin::concurrent_acquire_one_owner` | | 2 | Active writer commits or is rolled back before freeze acknowledgement; nothing commits after | `fence::drain::tests::active_writer_commits_before_ack`, `forced_rollback_before_ack`, `no_commit_after_ack` | -| 3 | Autocommit, explicit transactions, queued writers, batches, DDL, schema jobs, old WebSockets, read-to-write upgrades cannot bypass | `connection_manager::tests::fence_rejects_queued_writer`, `fence_rejects_read_to_write_upgrade`, `fence_rejects_ddl_and_pragma`, `fence_rejects_raw_with_raw_write`; `tests::fence::protocol::old_ws_session_cannot_write`, `batch_denied_mid_batch`; `fence::tests::acquire_rejects_shared_schema` | -| 4 | Program that captured config before the fence is rejected at the WAL | `connection_core::tests::wal_gate_rejects_program_admitted_before_fence` | -| 5 | Pre-fence transactions cannot write after release or publication | `fence::tests::stale_generation_cannot_write_after_release`, `stale_generation_cannot_write_after_enable_writes` | +| 3 | Autocommit, explicit transactions, queued writers, batches, DDL, schema jobs, old WebSockets, read-to-write upgrades cannot bypass | landed: `connection::connection_manager::fence_tests::{fence_rejects_read_to_write_upgrade, fence_rejects_ddl_and_pragma, fence_rejects_raw_with_raw_write}` (autocommit, explicit transactions, DDL, header-writing pragma, `BEGIN IMMEDIATE`, `VACUUM`, `with_raw` users); planned: `connection_manager::tests::fence_rejects_queued_writer`; `tests::fence::protocol::old_ws_session_cannot_write`, `batch_denied_mid_batch`; `fence::tests::acquire_rejects_shared_schema` | +| 4 | Program that captured config before the fence is rejected at the WAL | landed: `connection::connection_manager::fence_tests::wal_gate_rejects_program_admitted_before_fence` (a SQL function parks the program between admission and its write while the fence is acquired and released) | +| 5 | Pre-fence transactions cannot write after release or publication | landed: `connection::connection_manager::fence_tests::stale_generation_cannot_write_after_release`; unfenced behaviour unchanged: `unfenced_namespace_is_unchanged`; planned: `stale_generation_cannot_write_after_enable_writes` | | 6 | Acquisition timeout returns `DRAINING`, admission stays closed | `fence::drain::tests::deadline_returns_draining_and_stays_closed` | | 7 | Restart at every persistence boundary; indeterminate persistence keeps the gate closed until same-command reconciliation | `fence::tests::restart_at_each_boundary` (parameterised over hook points), `indeterminate_commit_keeps_gate_closed`; landed: `fence::controller::tests::{indeterminate_commit_keeps_writes_closed_until_replayed, indeterminate_commit_that_did_not_apply_is_retried_by_replay, failed_before_commit_leaves_gate_unchanged, publication_happens_before_the_response, committed_command_is_published_when_the_caller_goes_away}`, `namespace::store::fence_tests::restart_installs_the_durable_gate_before_serving` | | 8 | Evict and lazily reload a fenced namespace; identical admission | `tests::fence::lifecycle::evicted_namespace_reloads_same_gate`; landed at unit level: `namespace::store::fence_tests::evicted_namespace_reloads_with_the_same_controller`, `fence::registry::tests::seeded_from_load_fences_including_recovered_names` | diff --git a/libsql-server/src/connection/connection_core.rs b/libsql-server/src/connection/connection_core.rs index 9025914c62..ad2075166b 100644 --- a/libsql-server/src/connection/connection_core.rs +++ b/libsql-server/src/connection/connection_core.rs @@ -11,6 +11,7 @@ use crate::connection::legacy::open_conn_active_checkpoint; use crate::error::Error; use crate::metrics::{PROGRAM_EXEC_COUNT, QUERY_CANCELED, VACUUM_COUNT, WAL_CHECKPOINT_COUNT}; use crate::namespace::broadcasters::BroadcasterHandle; +use crate::namespace::fence::controller::FenceConnState; use crate::namespace::meta_store::MetaStoreHandle; use crate::namespace::ResolveNamespacePathFn; use crate::query_analysis::StmtKind; @@ -37,6 +38,8 @@ pub(super) struct CoreConnection { broadcaster: BroadcasterHandle, hooked: bool, canceled: Arc, + /// Shared with this connection's WAL wrapper (`docs/NAMESPACE_FENCE.md` section 7.4). + fence: Arc, } fn update_stats( @@ -68,6 +71,7 @@ impl CoreConnection { get_current_frame_no: GetCurrentFrameNo, block_writes: Arc, resolve_attach_path: ResolveNamespacePathFn, + fence: Arc, ) -> Result { let conn = open_conn_active_checkpoint( path, @@ -113,6 +117,7 @@ impl CoreConnection { hooked: false, canceled, get_current_frame_no, + fence, }; for ext in extensions.iter() { @@ -188,17 +193,21 @@ impl CoreConnection { pgm: Program, mut builder: B, ) -> Result { - let (config, stats, block_writes, resolve_attach_path) = { + let (config, stats, block_writes, resolve_attach_path, fence) = { let mut lock = this.lock(); let config = lock.config_store.get(); let stats = lock.stats.clone(); let block_writes = lock.block_writes.clone(); let resolve_attach_path = lock.resolve_attach_path.clone(); + let fence = lock.fence.clone(); lock.update_hooks(); - (config, stats, block_writes, resolve_attach_path) + (config, stats, block_writes, resolve_attach_path, fence) }; + // The program is admitted under the gate's current write generation; a write + // transaction it opens must start under the same one (section 8.1). + fence.begin_program(); builder.init(&this.lock().builder_config)?; let mut vm = Vm::new( @@ -229,7 +238,8 @@ impl CoreConnection { update_stats(&stats, sql, rows_read, rows_written, mem_used, elapsed) }, resolve_attach_path, - ); + ) + .with_fence(fence); let mut has_timeout = false; while !vm.finished() { @@ -296,6 +306,7 @@ impl CoreConnection { } pub(super) fn vacuum_if_needed(&self) -> Result<()> { + self.fence.begin_program(); let page_count = self .conn .query_row("PRAGMA page_count", (), |row| row.get::<_, i64>(0))?; @@ -416,6 +427,10 @@ mod test { hooked: false, canceled: Arc::new(false.into()), get_current_frame_no: Arc::new(|| None), + fence: FenceConnState::new( + FenceController::unfenced(Default::default()), + crate::namespace::fence::state::OperationClass::NormalWrite, + ), }; let conn = Arc::new(Mutex::new(conn)); diff --git a/libsql-server/src/connection/connection_manager.rs b/libsql-server/src/connection/connection_manager.rs index 4baaa0ddc0..4cf4675092 100644 --- a/libsql-server/src/connection/connection_manager.rs +++ b/libsql-server/src/connection/connection_manager.rs @@ -120,9 +120,6 @@ pub struct ManagedConnectionWalWrapper { manager: ConnectionManager, /// The connection's fence state, which `begin_write_txn` checks against the namespace's /// gate (`docs/NAMESPACE_FENCE.md` section 8.1). - // Installed here so that no connection exists without it; the check itself lands in the - // next commit of this series. - #[allow(dead_code)] fence: Arc, } @@ -417,6 +414,13 @@ impl WrapWal for ManagedConnectionWalWrapper { #[tracing::instrument(skip_all, fields(id = self.id))] fn begin_write_txn(&mut self, wrapped: &mut InnerWal) -> libsql_sys::wal::Result<()> { tracing::debug!("begin write"); + // The authoritative fence check. It runs before `acquire()`, so a refusal holds no slot + // and releases none, and it returns `SQLITE_AUTH` rather than `SQLITE_BUSY`, so SQLite's + // busy handler does not retry it. The typed reason is left in the connection's fence + // state for the program layer. + if self.fence.admit_write().is_err() { + return Err(rusqlite::ffi::Error::new(rusqlite::ffi::SQLITE_AUTH)); + } self.acquire()?; match wrapped.begin_write_txn() { Ok(_) => { @@ -498,6 +502,9 @@ impl WrapWal for ManagedConnectionWalWrapper { #[tracing::instrument(skip_all, fields(id = self.id))] fn begin_read_txn(&mut self, wrapped: &mut InnerWal) -> libsql_sys::wal::Result { tracing::debug!("begin read txn"); + // Recorded before the snapshot is taken: a transition racing with it leaves the + // transaction with the older generation, which can only refuse a later upgrade. + self.fence.begin_read_txn(); wrapped.begin_read_txn() } @@ -563,3 +570,373 @@ impl WrapWal for ManagedConnectionWalWrapper { ret } } + +/// The WAL write gate (`docs/NAMESPACE_FENCE.md` section 8.1): tests on real connections of a +/// namespace whose fence controller goes through committed metastore transitions. +#[cfg(test)] +mod fence_tests { + use std::path::Path; + use std::sync::Arc; + + use libsql_sys::wal::wrapper::PassthroughWalWrapper; + use libsql_sys::wal::Sqlite3WalManager; + use rusqlite::functions::FunctionFlags; + use rusqlite::ErrorCode; + use tempfile::tempdir; + + use crate::connection::connection_core::CoreConnection; + use crate::connection::legacy::{LegacyConnection, MakeLegacyConnection}; + use crate::connection::program::Program; + use crate::connection::Connection as _; + use crate::error::Error; + use crate::namespace::fence::controller::tests::{ + create_namespace, ctx, fence_source, open_metastore, release, OP, + }; + use crate::namespace::fence::controller::FenceController; + use crate::namespace::fence::outcome::{FenceDetail, FenceError, FenceOutcome}; + use crate::namespace::fence::state::FenceState; + use crate::namespace::meta_store::{MetaStore, MetaStoreHandle}; + use crate::query_result_builder::test::{StepResult, TestBuilder}; + use crate::query_result_builder::QueryResultBuilder as _; + use crate::DEFAULT_AUTO_CHECKPOINT; + + type Conn = LegacyConnection; + + struct Harness { + _dir: tempfile::TempDir, + meta: MetaStore, + controller: Arc, + maker: MakeLegacyConnection, + } + + impl Harness { + async fn new() -> Self { + let dir = tempdir().unwrap(); + let meta_dir = dir.path().join("meta"); + let db_dir = dir.path().join("db"); + std::fs::create_dir_all(&meta_dir).unwrap(); + std::fs::create_dir_all(&db_dir).unwrap(); + let meta = open_metastore(&meta_dir).await; + create_namespace(&meta, "ns").await; + let controller = FenceController::unfenced("ns".into()); + let maker = make_connections(&db_dir, controller.clone()).await; + let this = Self { + _dir: dir, + meta, + controller, + maker, + }; + let conn = this.conn().await; + assert_ok(&run(&conn, &["create table t (x)"]).await); + this + } + + async fn conn(&self) -> Conn { + self.maker.make_connection().await.unwrap() + } + + /// UNFENCED -> SOURCE_DRAINING -> SOURCE_WRITE_FENCED. + async fn fence(&self) { + fence_source(&self.meta, &self.controller, OP).await; + assert_eq!( + self.controller.gate().state(), + FenceState::SourceWriteFenced + ); + } + + /// SOURCE_WRITE_FENCED -> RELEASED: writes are admitted again, under a new generation. + async fn release(&self) { + let revision = self.controller.gate().revision(); + self.controller + .apply_command(&self.meta, release("ns", OP, 2, revision), ctx()) + .await + .unwrap(); + assert_eq!(self.controller.gate().state(), FenceState::Released); + } + } + + async fn make_connections( + path: &Path, + fence: Arc, + ) -> MakeLegacyConnection { + MakeLegacyConnection::new( + path.into(), + PassthroughWalWrapper, + Default::default(), + Default::default(), + MetaStoreHandle::load(path).unwrap(), + Arc::new([]), + 100000000, + 100000000, + DEFAULT_AUTO_CHECKPOINT, + Arc::new(|| None), + None, + Default::default(), + Arc::new(|_| unreachable!()), + Arc::new(|| Sqlite3WalManager::default()), + fence, + ) + .await + .unwrap() + } + + async fn run(conn: &Conn, stmts: &[&'static str]) -> Vec { + let inner = conn.inner.clone(); + let stmts = stmts.to_vec(); + tokio::task::spawn_blocking(move || { + CoreConnection::run(inner, Program::seq(&stmts), TestBuilder::default()) + .unwrap() + .into_ret() + }) + .await + .unwrap() + } + + fn assert_ok(steps: &[StepResult]) { + for (i, step) in steps.iter().enumerate() { + assert!(step.is_ok(), "step {i} failed: {step:?}"); + } + } + + fn fence_error(step: &StepResult) -> &FenceError { + match step { + Err(Error::NamespaceFence(e)) => e, + other => panic!("expected a fence denial, got {other:?}"), + } + } + + async fn count(conn: &Conn) -> i64 { + conn.with_raw(|c| c.query_row("select count(*) from t", (), |r| r.get(0))) + .unwrap() + } + + /// A program is admitted, the fence is acquired and released while it runs, and the write it + /// then attempts is refused at the WAL although the live gate is open again: the program was + /// admitted under a generation that is no longer current. The race is held open by a SQL + /// function that parks the program between admission and its write. + #[tokio::test(flavor = "multi_thread")] + async fn wal_gate_rejects_program_admitted_before_fence() { + let h = Harness::new().await; + let conn = h.conn().await; + + let (reached_tx, mut reached_rx) = tokio::sync::mpsc::unbounded_channel::<()>(); + let (resume_tx, resume_rx) = std::sync::mpsc::channel::<()>(); + let parked = std::panic::AssertUnwindSafe((reached_tx, std::sync::Mutex::new(resume_rx))); + conn.with_raw(move |c| { + c.create_scalar_function("park", 0, FunctionFlags::SQLITE_UTF8, move |_| { + let (reached, resume) = &*parked; + reached.send(()).unwrap(); + resume.lock().unwrap().recv().unwrap(); + Ok(1) + }) + }) + .unwrap(); + + let admitted_at = h.controller.write_generation(); + let program = tokio::spawn({ + let conn = conn.clone(); + async move { run(&conn, &["select park()", "insert into t values (1)"]).await } + }); + reached_rx.recv().await.unwrap(); + assert_eq!(conn.fence.program_generation(), admitted_at); + + h.fence().await; + h.release().await; + assert!(h.controller.write_generation() > admitted_at); + resume_tx.send(()).unwrap(); + + let steps = program.await.unwrap(); + assert!(steps[0].is_ok()); + let e = fence_error(&steps[1]); + assert_eq!(e.outcome(), FenceOutcome::MigrationWriteFenced); + assert_eq!(e.detail(), Some(FenceDetail::StaleTransaction)); + assert_eq!(count(&conn).await, 0); + + // The next program is admitted under the current generation and writes. + assert_ok(&run(&conn, &["insert into t values (2)"]).await); + assert_eq!(count(&conn).await, 1); + } + + /// A read transaction opened before a transition cannot be upgraded to a write transaction + /// after it, whether the gate is still closed or open again. + #[tokio::test(flavor = "multi_thread")] + async fn fence_rejects_read_to_write_upgrade() { + let h = Harness::new().await; + let conn = h.conn().await; + + // Gate closed: refused before the write runs. + assert_ok(&run(&conn, &["begin", "select * from t"]).await); + h.fence().await; + let steps = run(&conn, &["insert into t values (1)"]).await; + assert_eq!( + fence_error(&steps[0]).outcome(), + FenceOutcome::MigrationWriteFenced + ); + + // Gate open again: the transaction still belongs to the old generation, and only the + // WAL gate can tell. + h.release().await; + let steps = run(&conn, &["insert into t values (1)"]).await; + let e = fence_error(&steps[0]); + assert_eq!(e.outcome(), FenceOutcome::MigrationWriteFenced); + assert_eq!(e.detail(), Some(FenceDetail::StaleTransaction)); + assert_ok(&run(&conn, &["rollback"]).await); + + // A fresh transaction writes. + assert_ok(&run(&conn, &["begin", "insert into t values (1)", "commit"]).await); + assert_eq!(count(&conn).await, 1); + } + + /// DDL, a pragma that writes the header, and `BEGIN IMMEDIATE` are refused while writes are + /// fenced; reads keep working, and nothing was written. + #[tokio::test(flavor = "multi_thread")] + async fn fence_rejects_ddl_and_pragma() { + let h = Harness::new().await; + let conn = h.conn().await; + h.fence().await; + + for stmt in [ + "create table u (x)", + "create index i on t (x)", + "drop table t", + "pragma user_version = 7", + "begin immediate", + ] { + let steps = run(&conn, &[stmt]).await; + assert_eq!( + fence_error(&steps[0]).outcome(), + FenceOutcome::MigrationWriteFenced, + "{stmt}" + ); + assert!(conn.inner.lock().is_autocommit(), "{stmt}"); + } + assert_ok(&run(&conn, &["select * from t"]).await); + + h.release().await; + let version: i64 = conn + .with_raw(|c| c.query_row("pragma user_version", (), |r| r.get(0))) + .unwrap(); + assert_eq!(version, 0); + let tables: i64 = conn + .with_raw(|c| { + c.query_row( + "select count(*) from sqlite_schema where name in ('u', 'i')", + (), + |r| r.get(0), + ) + }) + .unwrap(); + assert_eq!(tables, 0); + } + + /// `with_raw` users (admin shell, schema migration, dump load) bypass statement + /// classification but not the WAL: the write fails with `SQLITE_AUTH` and the typed reason + /// is in the connection's denial slot. + #[tokio::test(flavor = "multi_thread")] + async fn fence_rejects_raw_with_raw_write() { + let h = Harness::new().await; + let conn = h.conn().await; + h.fence().await; + + for sql in ["insert into t values (1)", "begin immediate", "vacuum"] { + let err = conn.with_raw(|c| c.execute_batch(sql)).unwrap_err(); + match err { + rusqlite::Error::SqliteFailure(e, _) => { + assert_eq!(e.code, ErrorCode::AuthorizationForStatementDenied, "{sql}") + } + e => panic!("{sql}: unexpected error {e}"), + } + assert_eq!( + conn.fence.take_denial().unwrap().outcome(), + FenceOutcome::MigrationWriteFenced, + "{sql}" + ); + } + assert_eq!(count(&conn).await, 0); + + h.release().await; + conn.with_raw(|c| c.execute_batch("insert into t values (1)")) + .unwrap(); + assert_eq!(count(&conn).await, 1); + } + + /// The fence acquired and released between a transaction's first read and its write: the + /// write is refused although the gate is open, while a connection that was idle across the + /// transition, and the same connection after a rollback, write normally. + #[tokio::test(flavor = "multi_thread")] + async fn stale_generation_cannot_write_after_release() { + let h = Harness::new().await; + let in_txn = h.conn().await; + let idle = h.conn().await; + + assert_ok(&run(&in_txn, &["begin", "select count(*) from t"]).await); + h.fence().await; + h.release().await; + assert!(h + .controller + .permits(crate::namespace::fence::state::OperationClass::NormalWrite) + .is_ok()); + + let steps = run(&in_txn, &["insert into t values (1)", "commit"]).await; + assert_eq!( + fence_error(&steps[0]).detail(), + Some(FenceDetail::StaleTransaction) + ); + // The commit that follows ends the transaction without having written anything. + assert!(steps[1].is_ok()); + assert!(in_txn.inner.lock().is_autocommit()); + assert_ok(&run(&idle, &["insert into t values (2)"]).await); + assert_ok(&run(&in_txn, &["insert into t values (3)"]).await); + assert_eq!(count(&idle).await, 2); + } + + /// A namespace that never had a fence behaves as before: generation 0 everywhere, every + /// kind of write works, and a plain `SQLITE_AUTH` from an authorizer is reported as the + /// SQLite error it is, not as a fence denial. + #[tokio::test(flavor = "multi_thread")] + async fn unfenced_namespace_is_unchanged() { + let h = Harness::new().await; + let conn = h.conn().await; + + assert_ok( + &run( + &conn, + &[ + "insert into t values (1)", + "begin", + "select * from t", + "insert into t values (2)", + "commit", + "create table u (x)", + "pragma user_version = 3", + ], + ) + .await, + ); + conn.with_raw(|c| c.execute_batch("insert into t values (3)")) + .unwrap(); + assert_eq!(count(&conn).await, 3); + assert_eq!(h.controller.write_generation(), 0); + assert_eq!(conn.fence.program_generation(), 0); + assert_eq!(conn.fence.txn_generation(), 0); + + conn.with_raw(|c| { + c.authorizer(Some(|ctx: rusqlite::hooks::AuthContext<'_>| { + match ctx.action { + rusqlite::hooks::AuthAction::Insert { .. } => { + rusqlite::hooks::Authorization::Deny + } + _ => rusqlite::hooks::Authorization::Allow, + } + })) + }); + let steps = run(&conn, &["insert into t values (4)"]).await; + match &steps[0] { + Err(Error::RusqliteErrorExtended(rusqlite::Error::SqliteFailure(e, _), _)) => { + assert_eq!(e.code, ErrorCode::AuthorizationForStatementDenied) + } + other => panic!("expected the authorizer's error, got {other:?}"), + } + assert!(conn.fence.take_denial().is_none()); + } +} diff --git a/libsql-server/src/connection/legacy.rs b/libsql-server/src/connection/legacy.rs index 29676d237a..93ab86cb49 100644 --- a/libsql-server/src/connection/legacy.rs +++ b/libsql-server/src/connection/legacy.rs @@ -173,9 +173,7 @@ where pub struct LegacyConnection { pub(super) inner: Arc>>>, - /// Shared with the connection's WAL wrapper. - // Read by the WAL gate and the program admission check in the next commit of this series. - #[allow(dead_code)] + /// Shared with the connection's WAL wrapper and its `CoreConnection`. pub(super) fence: Arc, } @@ -344,7 +342,7 @@ where let connection_manager = connection_manager.clone(); let fence = fence.clone(); move || -> crate::Result<_> { - let manager = ManagedConnectionWalWrapper::new(connection_manager, fence); + let manager = ManagedConnectionWalWrapper::new(connection_manager, fence.clone()); let id = manager.id(); let wal = make_wal().wrap(manager).wrap(wal_wrapper); @@ -359,6 +357,7 @@ where current_frame_no_receiver, block_writes, resolve_attach_path, + fence, )?; let namespace = path @@ -464,6 +463,9 @@ where fn with_raw(&self, f: impl FnOnce(&mut rusqlite::Connection) -> R) -> R { let mut inner = self.inner.lock(); + // A raw use of the connection is a program like any other: the WAL gate admits a write + // transaction it opens only under the generation it started under. + self.fence.begin_program(); f(inner.raw_mut()) } } diff --git a/libsql-server/src/connection/program.rs b/libsql-server/src/connection/program.rs index 4d5ada51ff..8cafe681e7 100644 --- a/libsql-server/src/connection/program.rs +++ b/libsql-server/src/connection/program.rs @@ -7,6 +7,7 @@ use rusqlite::StatementStatus; use crate::auth::Permission; use crate::error::Error; use crate::metrics::{READ_QUERY_COUNT, WRITE_QUERY_COUNT}; +use crate::namespace::fence::controller::FenceConnState; use crate::namespace::{NamespaceName, ResolveNamespacePathFn}; use crate::query::Query; use crate::query_analysis::StmtKind; @@ -104,6 +105,7 @@ pub struct Vm<'a, B, F, S> { should_block: F, update_stats: S, resolve_attach_path: ResolveNamespacePathFn, + fence: Option>, } impl<'a, B, F, S> Vm<'a, B, F, S> @@ -127,9 +129,19 @@ where should_block, update_stats, resolve_attach_path, + fence: None, } } + /// Check the namespace fence on this connection: a statement that can write is refused + /// before it runs while the live gate denies the connection's class, and a refusal by the + /// WAL gate is reported as the fence error rather than as `SQLITE_AUTH` + /// (`docs/NAMESPACE_FENCE.md` section 8.1). + pub fn with_fence(mut self, fence: Arc) -> Self { + self.fence = Some(fence); + self + } + #[inline] fn current_step(&self) -> &Step { &self.program.steps()[self.current_step] @@ -164,7 +176,9 @@ where // builder error interrupt the execution of query. we should exit immediately. Err(e @ Error::BuilderError(_)) => return Err(e), Err(mut e) => { - if let Error::RusqliteError(err) = e { + if let Some(denial) = self.take_fence_denial(&e) { + e = Error::NamespaceFence(denial); + } else if let Error::RusqliteError(err) = e { let extended_code = unsafe { rusqlite::ffi::sqlite3_extended_errcode(conn.handle()) }; @@ -200,12 +214,39 @@ where Ok(query) } + /// The fence denial behind `e`, when `e` is the `SQLITE_AUTH` the WAL gate returns. A plain + /// `SQLITE_AUTH` (from an authorizer) is left alone, and so is an empty denial slot. + fn take_fence_denial(&self, e: &Error) -> Option { + let fence = self.fence.as_ref()?; + match e { + Error::RusqliteError(rusqlite::Error::SqliteFailure( + rusqlite::ffi::Error { + code: rusqlite::ErrorCode::AuthorizationForStatementDenied, + .. + }, + _, + )) => fence.take_denial(), + _ => None, + } + } + fn execute_query(&mut self, conn: &rusqlite::Connection) -> crate::Result<(u64, Option)> { tracing::debug!("executing query: {}", self.current_step().query.stmt.stmt); increment_counter!("libsql_server_libsql_query_execute"); let start = Instant::now(); + // Early fence admission for statements that can write. The WAL gate is authoritative + // and catches everything this misses (a misclassified statement, a read-to-write + // upgrade, a gate change while the statement runs). + if let Some(fence) = &self.fence { + if matches!( + self.current_step().query.stmt.kind, + StmtKind::Write | StmtKind::DDL + ) { + fence.controller().permits(fence.class())?; + } + } let (blocked, reason) = (self.should_block)(&self.current_step().query.stmt.kind); if blocked { return Err(Error::Blocked(reason)); diff --git a/libsql-server/src/namespace/fence/controller.rs b/libsql-server/src/namespace/fence/controller.rs index 265e04e40f..a9a31dc03a 100644 --- a/libsql-server/src/namespace/fence/controller.rs +++ b/libsql-server/src/namespace/fence/controller.rs @@ -384,7 +384,14 @@ fn pending_indeterminate((operation_id, command_id): CommandKey) -> FenceError { } /// The fence state of one connection, shared by its WAL wrapper and its `CoreConnection` -/// (section 7.4). The WAL gate reads and writes it; this commit only installs it. +/// (section 7.4). +/// +/// - A program (a `CoreConnection::run`, a `with_raw` call, a vacuum) starts with +/// [`begin_program`](Self::begin_program), which records the generation it was admitted under. +/// - The WAL wrapper calls [`begin_read_txn`](Self::begin_read_txn) whenever SQLite opens a read +/// transaction, and [`admit_write`](Self::admit_write) in `begin_write_txn` before it queues for +/// the write slot. A refusal leaves its typed outcome in the denial slot, which the program +/// layer takes to report the fence error instead of the bare `SQLITE_AUTH` the WAL returns. #[derive(Debug)] pub struct FenceConnState { controller: Arc, @@ -425,13 +432,69 @@ impl FenceConnState { self.txn_generation.load(Ordering::Acquire) } + /// Start a program on this connection: record the generation it is admitted under and + /// forget any denial a previous program left behind. Returns that generation. + pub fn begin_program(&self) -> u64 { + let generation = self.controller.write_generation(); + self.program_generation.store(generation, Ordering::Release); + *self.denial.lock() = None; + generation + } + + /// SQLite is opening a new read transaction on this connection: record the generation it + /// is opened under. A later upgrade of that transaction to a write transaction must happen + /// under the same generation. + pub fn begin_read_txn(&self) { + self.txn_generation + .store(self.controller.write_generation(), Ordering::Release); + } + + /// The authoritative write admission (section 8.1, check 2): the live gate permits this + /// connection's class, and the program and its read transaction were both admitted under + /// the gate's current write generation. On refusal the typed outcome is left in the denial + /// slot and returned. + pub fn admit_write(&self) -> Result<(), FenceError> { + let result = { + let gate = self.controller.gate.borrow(); + gate.permits(self.class).and_then(|()| { + let current = gate.write_generation; + let program = self.program_generation(); + let txn = self.txn_generation(); + if program == current && txn == current { + Ok(()) + } else { + Err(FenceError::new( + FenceOutcome::MigrationWriteFenced, + format!( + "the namespace fence changed after this transaction began \ + (program admitted at generation {program}, transaction opened at \ + generation {txn}, current generation {current}); roll back and \ + begin a new transaction" + ), + ) + .with_detail(FenceDetail::StaleTransaction)) + } + }) + }; + if let Err(e) = &result { + tracing::debug!( + namespace = %self.controller.namespace, + class = ?self.class, + "write transaction refused by the namespace fence: {e}" + ); + *self.denial.lock() = Some(e.clone()); + } + result + } + + /// Take the typed outcome of the last refusal at the WAL, if any. pub fn take_denial(&self) -> Option { self.denial.lock().take() } } #[cfg(test)] -mod tests { +pub(crate) mod tests { use std::path::Path; use tempfile::tempdir; @@ -445,7 +508,7 @@ mod tests { use crate::namespace::meta_store::{metastore_connection_maker, FenceCommitKind}; const LOG: Uuid = Uuid::from_u128(0x10); - const OP: Uuid = Uuid::from_u128(0xa); + pub(crate) const OP: Uuid = Uuid::from_u128(0xa); const OTHER_OP: Uuid = Uuid::from_u128(0xb); pub(crate) async fn open_metastore(dir: &Path) -> MetaStore { @@ -474,7 +537,7 @@ mod tests { .unwrap(); } - fn ctx() -> FenceContext { + pub(crate) fn ctx() -> FenceContext { FenceContext::now( ServerIdentity { build: "test".into(), @@ -484,7 +547,7 @@ mod tests { ) } - fn acquire(ns: &'static str, op: Uuid, command_id: u128) -> FenceRequest { + pub(crate) fn acquire(ns: &'static str, op: Uuid, command_id: u128) -> FenceRequest { FenceRequest { namespace: ns.into(), operation_id: op, @@ -498,7 +561,12 @@ mod tests { } } - fn release(ns: &'static str, op: Uuid, command_id: u128, revision: u64) -> FenceRequest { + pub(crate) fn release( + ns: &'static str, + op: Uuid, + command_id: u128, + revision: u64, + ) -> FenceRequest { FenceRequest { namespace: ns.into(), operation_id: op, @@ -519,7 +587,11 @@ mod tests { /// Acquire and complete the drain directly, as the write drain will: the controller ends in /// `SOURCE_WRITE_FENCED`. - async fn fence_source(meta: &MetaStore, controller: &Arc, op: Uuid) { + pub(crate) async fn fence_source( + meta: &MetaStore, + controller: &Arc, + op: Uuid, + ) { let commit = controller .apply_command(meta, acquire("ns", op, 1), ctx()) .await diff --git a/libsql-server/src/namespace/fence/outcome.rs b/libsql-server/src/namespace/fence/outcome.rs index 6a9b6a8b57..cf4df0c9af 100644 --- a/libsql-server/src/namespace/fence/outcome.rs +++ b/libsql-server/src/namespace/fence/outcome.rs @@ -201,6 +201,8 @@ pub enum FenceDetail { IncompleteTargetCreation, MetastoreBehindMarker, IndeterminateCommit, + // MIGRATION_WRITE_FENCED + StaleTransaction, } impl FenceDetail { @@ -223,6 +225,7 @@ impl FenceDetail { FenceDetail::IncompleteTargetCreation => "incomplete_target_creation", FenceDetail::MetastoreBehindMarker => "metastore_behind_marker", FenceDetail::IndeterminateCommit => "indeterminate_commit", + FenceDetail::StaleTransaction => "stale_transaction", } } } From 628ef48ce8be25cbd8661f739cc6c553042935e8 Mon Sep 17 00:00:00 2001 From: River Date: Tue, 29 Sep 2026 16:03:47 +0000 Subject: [PATCH 09/33] libsql-server: class writer-queue entries and wake them on fence changes Queue entries and the write slot now carry the operation class. A write transaction waiting for the slot re-checks the connection's fence admission every time it takes the manager's lock, and the fence controller wakes every registered write queue after each change of the write generation, so a writer queued before a fence leaves the queue with MIGRATION_WRITE_FENCED instead of waiting for the slot and then writing. Checkpoints ask for the slot as maintenance: they are never refused and queue again when woken. The manager exposes what the positive write drain needs: the active writer and its class, a notification on every release, and abort_active(), which uses the registered rollback handle. Abort no longer panics when the connection has already closed. VACUUM is skipped, and reported as skipped rather than failed, while the fence denies normal writes, including when the fence closes between the check and the statement. TRUNCATE checkpoints run in every state. Co-authored-by: Tomasz Szymczyszyn --- docs/NAMESPACE_FENCE.md | 13 +- .../src/connection/connection_core.rs | 43 +- .../src/connection/connection_manager.rs | 425 +++++++++++++++++- libsql-server/src/connection/legacy.rs | 12 +- .../src/namespace/fence/controller.rs | 94 +++- 5 files changed, 542 insertions(+), 45 deletions(-) diff --git a/docs/NAMESPACE_FENCE.md b/docs/NAMESPACE_FENCE.md index 278f472c62..02a4d92d26 100644 --- a/docs/NAMESPACE_FENCE.md +++ b/docs/NAMESPACE_FENCE.md @@ -407,7 +407,7 @@ Per namespace: - `gate`: a `tokio::sync::watch` of `GateSnapshot { fence, write_generation, indeterminate }`, where `fence` is the durable fence as last published (a record, no record, or `UNKNOWN_UNAVAILABLE` with its detail). State, revision, owning operation and every admission (`permits(class)`, `write()`, `read()`) are derived from it through the permission matrix. The WAL wrapper, `CoreConnection`, dump, replication and lifecycle code read it without locks. Phase 5 adds the live capability set. - `write_generation: u64`, in the snapshot, **incremented on every publication that changes the fence state, the owning operation, or the indeterminate flag**. That covers every transition that closes or opens write admission (acquire, release, create, seal, publish, enable, abort, adopt), and is conservative for the others. A replay that publishes the same durable state does not move it. - `indeterminate: Option<(operation_id, command_id)>`: set when a command's `COMMIT` failed (or the task running it died) so that whether it applied is unknown. While set, every class except `Maintenance` and `Observability` is denied with `FENCE_STATE_UNAVAILABLE` / `indeterminate_commit`, and every other command is refused with `FENCE_COMMIT_INDETERMINATE`. A replay of the same command is answered by the metastore from the durable row (replayed if it had committed, applied if it had not) and clears it (section 8.4). -- writer tracking, provided by the namespace's `ManagedConnectionWalManager` (section 8.2). +- the write queues of the namespace's connection managers: each `MakeLegacyConnection` registers a waker (`register_write_queue`) that the controller calls after every publication that moves `write_generation`, after the new gate is visible. A waker holds its manager weakly and is dropped once the manager is gone (for example after eviction). Writer tracking itself is the connection manager's (section 8.2). - `read_leases`: counters and cancel handles per lease class (`sql`, `dump`, `replication`), with a `Notify` on every release. - `capabilities`: the live `MigrationCapability` set and an import-writer counter. - in `cfg(test)` builds only, a `FenceTestHooks` (section 16). @@ -422,7 +422,7 @@ Every write-transaction request at the WAL, every read lease and every lifecycle |---|---|---| | `NormalWrite` | SQL over HTTP, Hrana, RPC, proxy; admin shell; schema migration; dump load outside the capability | normal-write column allows | | `Maintenance` | `TRUNCATE` checkpoint, the manager's checkpoint slot, storage monitor | always | -| `Vacuum` | `vacuum_if_needed`, `Namespace::checkpoint`, snapshot at shutdown | normal-write column allows; otherwise skipped with a debug log | +| `Vacuum` | `vacuum_if_needed`, `Namespace::checkpoint`, snapshot at shutdown | normal-write column allows; otherwise skipped with a debug log. `CoreConnection::vacuum_if_needed_above` checks the gate first and also reports a WAL refusal of the `VACUUM` itself (the fence closing between the check and the statement) as skipped, not failed | | `CapabilityImport` | import session writes | `TARGET_QUARANTINED`, matching `operation_id`, capability revision equal to the record's, capability not invalidated | | `CapabilityValidate` | validation reads (never writes) | `TARGET_VALIDATING`, `TARGET_WRITE_FENCED` | | `NormalRead` | SQL programs, Hrana cursors, `/beta/listen`, ATTACH of this namespace | normal-read column allows | @@ -450,9 +450,10 @@ This makes the WAL gate independent of statement classification: DDL, misclassif ### 8.2 Connection manager changes -- Queue entries carry the operation class (`NormalWrite`, `Maintenance`, `CapabilityImport`) and the connection id. `acquire()` for checkpoints becomes `acquire(Maintenance)`. -- On every write-generation change the controller wakes the whole write queue using the existing `sync_token` shape; each woken waiter re-checks the gate and, if denied, returns the typed error instead of waiting for a release. -- The manager exposes `active_writer() -> Option<(ConnId, OperationClass)>`, notifies a `Notify` from `release()`, and offers `abort_active()` that uses the registered rollback handle. `Abort::abort` currently panics if the connection is gone; the drain path tolerates a concurrently closing connection. +- Queue entries and the write slot carry the operation class and the connection id. A write transaction asks for the slot with its connection's class (`NormalWrite`, later `CapabilityImport`); checkpoints ask with `acquire(Maintenance)`. +- `acquire(class)` re-checks the connection's write admission (`admit_write`, section 8.1) every time it takes the manager's `current` lock, for every class but `Maintenance`. A refusal fills the denial slot and returns `SQLITE_AUTH`; if the slot had already been handed to this connection it is passed on first (and the release notification fires), so a refused waiter never holds or leaks the slot. +- On every write-generation change the controller wakes the whole write queue, in the shape of the existing `sync_token` queue sync: under the `current` lock the manager increments a `fence_token`, steals every queue entry and unparks it. Each woken waiter re-checks as above and, if denied, returns the typed error instead of waiting for a release; one that is still admitted (a checkpoint) sees the token moved and queues again. Because the wake takes the `current` lock after the gate is published, a writer admitted under the old gate is either stolen and woken, or sees the new gate when it next takes the lock, or already holds the slot and is the active writer the drain waits for. The checkpoint's own queue sync tolerates entries a concurrent fence wake has already taken. +- The manager exposes `active_writer() -> Option<(ConnId, OperationClass)>` (a slot handed to a queued connection that has not taken it yet counts as held), `released()`, a `tokio::sync::Notify` notified with `notify_waiters` on every release or hand-on (a waiter enables its `Notified` before it reads `active_writer`), and `abort_active()`, which calls the active writer's registered rollback handle and returns its id. `Abort::abort` no longer panics when the connection is gone: a connection that has closed has released (or is releasing) its slot, so there is nothing to roll back. `MakeLegacyConnection::connection_manager()` hands the manager to the drain. - The active writer's lease lasts until `release()`. Because `ReplicationLoggerWalWrapper::insert_frames` commits the log and publishes the new frame number before `end_write_txn` → `release()`, observing "no `NormalWrite` or `CapabilityImport` holder" under the manager's `current` lock means the committed `log_id` and `frame_no` are final. ### 8.3 Source write drain, step by step @@ -695,7 +696,7 @@ Planned test names; the table is updated as tests land. |---|---|---| | 1 | Concurrent acquisition by two operations: one owner, typed conflict for the loser | `fence::tests::acquire_race_single_owner`; `tests::fence::admin::concurrent_acquire_one_owner` | | 2 | Active writer commits or is rolled back before freeze acknowledgement; nothing commits after | `fence::drain::tests::active_writer_commits_before_ack`, `forced_rollback_before_ack`, `no_commit_after_ack` | -| 3 | Autocommit, explicit transactions, queued writers, batches, DDL, schema jobs, old WebSockets, read-to-write upgrades cannot bypass | landed: `connection::connection_manager::fence_tests::{fence_rejects_read_to_write_upgrade, fence_rejects_ddl_and_pragma, fence_rejects_raw_with_raw_write}` (autocommit, explicit transactions, DDL, header-writing pragma, `BEGIN IMMEDIATE`, `VACUUM`, `with_raw` users); planned: `connection_manager::tests::fence_rejects_queued_writer`; `tests::fence::protocol::old_ws_session_cannot_write`, `batch_denied_mid_batch`; `fence::tests::acquire_rejects_shared_schema` | +| 3 | Autocommit, explicit transactions, queued writers, batches, DDL, schema jobs, old WebSockets, read-to-write upgrades cannot bypass | landed: `connection::connection_manager::fence_tests::{fence_rejects_read_to_write_upgrade, fence_rejects_ddl_and_pragma, fence_rejects_raw_with_raw_write}` (autocommit, explicit transactions, DDL, header-writing pragma, `BEGIN IMMEDIATE`, `VACUUM`, `with_raw` users); `connection::connection_manager::fence_tests::fence_rejects_queued_writer` (a writer parked in the queue behind an open transaction leaves it with `MIGRATION_WRITE_FENCED` when the fence changes, and the holder keeps the slot); maintenance and vacuum under a fence: `queued_checkpoint_survives_fence_wake`, `checkpoint_allowed_while_fenced`, `vacuum_skipped_while_fenced`; drain primitives: `abort_active_tolerates_closed_connection`, `release_notifies_drain_waiters`, `fence::controller::tests::write_queues_are_woken_on_every_generation_change`; planned: `tests::fence::protocol::old_ws_session_cannot_write`, `batch_denied_mid_batch`; `fence::tests::acquire_rejects_shared_schema` | | 4 | Program that captured config before the fence is rejected at the WAL | landed: `connection::connection_manager::fence_tests::wal_gate_rejects_program_admitted_before_fence` (a SQL function parks the program between admission and its write while the fence is acquired and released) | | 5 | Pre-fence transactions cannot write after release or publication | landed: `connection::connection_manager::fence_tests::stale_generation_cannot_write_after_release`; unfenced behaviour unchanged: `unfenced_namespace_is_unchanged`; planned: `stale_generation_cannot_write_after_enable_writes` | | 6 | Acquisition timeout returns `DRAINING`, admission stays closed | `fence::drain::tests::deadline_returns_draining_and_stays_closed` | diff --git a/libsql-server/src/connection/connection_core.rs b/libsql-server/src/connection/connection_core.rs index ad2075166b..212e5c2b3d 100644 --- a/libsql-server/src/connection/connection_core.rs +++ b/libsql-server/src/connection/connection_core.rs @@ -12,6 +12,7 @@ use crate::error::Error; use crate::metrics::{PROGRAM_EXEC_COUNT, QUERY_CANCELED, VACUUM_COUNT, WAL_CHECKPOINT_COUNT}; use crate::namespace::broadcasters::BroadcasterHandle; use crate::namespace::fence::controller::FenceConnState; +use crate::namespace::fence::state::OperationClass; use crate::namespace::meta_store::MetaStoreHandle; use crate::namespace::ResolveNamespacePathFn; use crate::query_analysis::StmtKind; @@ -25,6 +26,15 @@ use super::program::{DescribeCol, DescribeParam, DescribeResponse, Program, Vm}; pub type GetCurrentFrameNo = Arc Option + Send + Sync + 'static>; +/// What [`CoreConnection::vacuum_if_needed_above`] did. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(crate) enum VacuumOutcome { + Vacuumed, + NotNeeded, + /// Skipped because the namespace fence denies normal writes. + Fenced, +} + /// The base connection type, shared between legacy and libsql-wal implementations pub(super) struct CoreConnection { conn: libsql_sys::Connection, @@ -306,22 +316,45 @@ impl CoreConnection { } pub(super) fn vacuum_if_needed(&self) -> Result<()> { + // NOTICE: don't bother vacuuming if we don't have at least 256MiB of data + self.vacuum_if_needed_above(65536).map(|_| ()) + } + + /// `VACUUM` if the database has at least `min_pages` pages and more than half of them are + /// free. `VACUUM` is not maintenance: it takes a write transaction and produces replicated + /// frames, so it is skipped whenever the namespace fence denies normal writes + /// (`docs/NAMESPACE_FENCE.md` section 7.3), including when the fence closes between the + /// check and the `VACUUM` itself. + pub(super) fn vacuum_if_needed_above(&self, min_pages: i64) -> Result { self.fence.begin_program(); + if let Err(e) = self.fence.controller().permits(OperationClass::Vacuum) { + tracing::debug!("skipping vacuum: {e}"); + return Ok(VacuumOutcome::Fenced); + } let page_count = self .conn .query_row("PRAGMA page_count", (), |row| row.get::<_, i64>(0))?; let freelist_count = self .conn .query_row("PRAGMA freelist_count", (), |row| row.get::<_, i64>(0))?; - // NOTICE: don't bother vacuuming if we don't have at least 256MiB of data - if page_count >= 65536 && freelist_count * 2 > page_count { + let outcome = if page_count >= min_pages && freelist_count * 2 > page_count { tracing::info!("Vacuuming: pages={page_count} freelist={freelist_count}"); - self.conn.execute("VACUUM", ())?; + if let Err(e) = self.conn.execute("VACUUM", ()) { + return match self.fence.take_denial() { + Some(denial) => { + tracing::debug!("skipping vacuum: {denial}"); + Ok(VacuumOutcome::Fenced) + } + None => Err(e.into()), + }; + } + VacuumOutcome::Vacuumed } else { tracing::trace!("Not vacuuming: pages={page_count} freelist={freelist_count}"); - } + VacuumOutcome::NotNeeded + }; VACUUM_COUNT.increment(1); - Ok(()) + Ok(outcome) } pub(super) fn describe(&self, sql: &str) -> crate::Result { diff --git a/libsql-server/src/connection/connection_manager.rs b/libsql-server/src/connection/connection_manager.rs index 4cf4675092..009adfeb45 100644 --- a/libsql-server/src/connection/connection_manager.rs +++ b/libsql-server/src/connection/connection_manager.rs @@ -15,6 +15,7 @@ use rusqlite::ErrorCode; use super::connection_core::CoreConnection; use super::TXN_TIMEOUT; use crate::namespace::fence::controller::FenceConnState; +use crate::namespace::fence::state::OperationClass; pub type ConnId = u64; pub type InnerWalManager = Sqlite3WalManager; @@ -25,10 +26,16 @@ pub type ManagedConnectionWal = WrappedWal); @@ -36,10 +43,13 @@ impl Abort { fn from_conn(conn: &Arc>>) -> Self { let conn = Arc::downgrade(conn); Self(Arc::new(move || { - conn.upgrade() - .expect("connection still owns the slot, so it must exist") - .lock() - .force_rollback(); + // The connection can be closing concurrently: a drain aborting the active writer + // races with the client going away. A connection that is gone has already released + // its slot (or is about to, from `close`), so there is nothing left to roll back. + match conn.upgrade() { + Some(conn) => conn.lock().force_rollback(), + None => tracing::debug!("connection closed before it could be rolled back"), + } })) } @@ -62,6 +72,89 @@ impl ConnectionManager { let abort = Abort::from_conn(conn); self.inner.abort_handle.lock().insert(id, abort); } + + /// The connection holding the write slot, and the class it holds it for. A slot that has + /// been handed to a queued connection that has not taken it yet counts as held. + // Used by the positive source write drain (section 8.3). + #[cfg_attr(not(test), allow(dead_code))] + pub(crate) fn active_writer(&self) -> Option<(ConnId, OperationClass)> { + self.inner.current.lock().map(|slot| (slot.id, slot.class)) + } + + /// Notified (with `notify_waiters`) every time the write slot is released or handed on. A + /// waiter registers interest (`Notified::enable`) before it checks + /// [`active_writer`](Self::active_writer), so a release in between is not missed. + // Used by the positive source write drain (section 8.3). + #[cfg_attr(not(test), allow(dead_code))] + pub(crate) fn released(&self) -> &tokio::sync::Notify { + &self.inner.released + } + + /// Roll back the transaction of the connection holding the write slot, using the rollback + /// handle it registered. Returns the connection that was asked to roll back, if any. The + /// slot is released by the rollback itself (`end_read_txn`/`end_write_txn`), which notifies + /// [`released`](Self::released); a connection closing at the same time is tolerated. + // Used by the positive source write drain (section 8.3). + #[cfg_attr(not(test), allow(dead_code))] + pub(crate) fn abort_active(&self) -> Option { + let id = self.active_writer()?.0; + let handle = self.inner.abort_handle.lock().get(&id).cloned(); + match handle { + Some(handle) => { + tracing::debug!("aborting the active writer {id}"); + handle.abort(); + Some(id) + } + None => { + tracing::debug!("the active writer {id} is closing; nothing to abort"); + None + } + } + } + + /// A waker for the fence controller, which calls it after every change of the write + /// generation (`docs/NAMESPACE_FENCE.md` section 8.2): every connection waiting in the write + /// queue is woken and re-checks the fence. One the gate denies returns the typed denial + /// instead of waiting for the slot; one it admits (a checkpoint) queues again. The waker + /// holds the manager weakly and reports `false` once it is gone, so the controller can + /// forget it. + pub(crate) fn fence_waker(&self) -> Box bool + Send + Sync> { + let inner = Arc::downgrade(&self.inner); + Box::new(move || match inner.upgrade() { + Some(inner) => { + wake_queue_for_fence(&inner); + true + } + None => false, + }) + } + + #[cfg(test)] + pub(crate) fn queued_writers(&self) -> usize { + self.inner.write_queue.len() + } +} + +fn wake_queue_for_fence(inner: &ConnectionManagerInner) { + // Under the `current` lock, so that a waiter either queued before this (and is stolen and + // woken here) or observes the new token when it takes the lock. + let _current = inner.current.lock(); + inner.fence_token.fetch_add(1, Ordering::SeqCst); + let mut woken = 0; + loop { + match inner.write_queue.steal() { + Steal::Empty => break, + Steal::Success((id, _, unparker)) => { + tracing::debug!("fence changed, waking queued connection id={id}"); + unparker.unpark(); + woken += 1; + } + Steal::Retry => (), + } + } + if woken > 0 { + tracing::debug!("fence changed, woke {woken} queued connections"); + } } impl Deref for ConnectionManager { @@ -92,12 +185,17 @@ pub struct ConnectionManagerInner { abort_handle: Mutex>, /// threads waiting to acquire the lock /// todo: limit how many can be push - write_queue: crossbeam::deque::Injector<(ConnId, Unparker)>, + write_queue: crossbeam::deque::Injector, txn_timeout_duration: Duration, /// the time we are given to acquire a transaction after we were given a slot acquire_timeout_duration: Duration, next_conn_id: AtomicU64, sync_token: AtomicU64, + /// Incremented, under the `current` lock, every time the queue is woken for a fence change. + /// A waiter that sees it move knows it was taken off the queue. + fence_token: AtomicU64, + /// Notified whenever the write slot is released or handed on. + released: tokio::sync::Notify, } impl Default for ConnectionManagerInner { @@ -110,6 +208,8 @@ impl Default for ConnectionManagerInner { acquire_timeout_duration: Duration::from_millis(15), next_conn_id: Default::default(), sync_token: AtomicU64::new(0), + fence_token: AtomicU64::new(0), + released: Default::default(), } } } @@ -133,13 +233,37 @@ impl ManagedConnectionWalWrapper { self.id } - fn acquire(&self) -> libsql_sys::wal::Result<()> { + /// Wait for the write slot on behalf of work of `class`. `Maintenance` (checkpoints) is never + /// refused by the fence; every other class re-checks the connection's write admission each + /// time it takes the `current` lock, so a waiter queued before a fence change leaves the + /// queue with the typed denial (`docs/NAMESPACE_FENCE.md` section 8.2). + fn acquire(&self, class: OperationClass) -> libsql_sys::wal::Result<()> { let parker = Parker::new(); let mut enqueued = false; let enqueued_at = Instant::now(); let sync_token = self.manager.sync_token.load(Ordering::SeqCst); + let mut fence_token = self.manager.fence_token.load(Ordering::SeqCst); loop { let mut current = self.manager.current.lock(); + let ours = current.as_ref().map_or(false, |slot| slot.id == self.id); + if class != OperationClass::Maintenance && self.fence.admit_write().is_err() { + // The denial is in the connection's fence state. If the slot had already been + // handed to us, pass it on: we are not going to use it. + if ours { + self.hand_off(&mut current); + } + tracing::debug!("write slot request refused by the namespace fence"); + return Err(rusqlite::ffi::Error::new(rusqlite::ffi::SQLITE_AUTH)); + } + let observed = self.manager.fence_token.load(Ordering::SeqCst); + if observed != fence_token { + fence_token = observed; + // A fence change emptied the queue. Unless the slot was handed to us before + // that, we are no longer queued and must queue again. + if enqueued && !ours { + enqueued = false; + } + } // if current is not currently us, and we havent enqueued yet, then enqueue // current can be us in two cases: // - in previous iteration, the queue was empty, and we popped ourselves @@ -194,7 +318,7 @@ impl ManagedConnectionWalWrapper { if current.as_mut().map_or(true, |slot| slot.id != self.id) && !enqueued { self.manager .write_queue - .push((self.id, parker.unparker().clone())); + .push((self.id, class, parker.unparker().clone())); enqueued = true; tracing::debug!("enqueued"); } @@ -284,6 +408,7 @@ impl ManagedConnectionWalWrapper { None => { *current = Some(Slot { id: self.id, + class, started_at: Instant::now(), state: SlotState::Acquiring, }); @@ -321,6 +446,7 @@ impl ManagedConnectionWalWrapper { None => { *current = Some(Slot { id: self.id, + class, started_at: Instant::now(), state: SlotState::Acquiring, }) @@ -344,10 +470,11 @@ impl ManagedConnectionWalWrapper { }; match next { - Some((id, unpaker)) => { + Some((id, class, unpaker)) => { tracing::debug!(line = line!(), "unparking id={id}"); **current = Some(Slot { id, + class, started_at: Instant::now(), state: SlotState::Notified, }); @@ -369,12 +496,15 @@ impl ManagedConnectionWalWrapper { assert_eq!(slot.id, self.id); tracing::debug!("transaction finished after {:?}", slot.started_at.elapsed()); - match self.schedule_next(&mut current) { - Some(_) => (), - None => { - *current = None; - } - } + self.hand_off(&mut current); + } + + /// Give the (already vacated or ours) slot to the next queued connection, or leave it free, + /// and tell drain waiters. + fn hand_off(&self, current: &mut MutexGuard>) { + **current = None; + self.schedule_next(current); + self.manager.released.notify_waiters(); } } @@ -421,7 +551,7 @@ impl WrapWal for ManagedConnectionWalWrapper { if self.fence.admit_write().is_err() { return Err(rusqlite::ffi::Error::new(rusqlite::ffi::SQLITE_AUTH)); } - self.acquire()?; + self.acquire(self.fence.class())?; match wrapped.begin_write_txn() { Ok(_) => { tracing::debug!("transaction acquired"); @@ -460,7 +590,7 @@ impl WrapWal for ManagedConnectionWalWrapper { backfilled: Option<&mut i32>, ) -> libsql_sys::wal::Result<()> { let before = Instant::now(); - self.acquire()?; + self.acquire(OperationClass::Maintenance)?; self.manager.current.lock().as_mut().unwrap().state = SlotState::Acquired(SlotType::Checkpoint); @@ -473,11 +603,19 @@ impl WrapWal for ManagedConnectionWalWrapper { if mode as i32 >= CheckpointMode::Restart as i32 { tracing::debug!("forcing queue sync"); self.manager.sync_token.fetch_add(1, Ordering::SeqCst); - let queue_len = self.manager.write_queue.len(); - for _ in 0..queue_len { - let (id, unparker) = self.manager.write_queue.steal().success().unwrap(); - tracing::debug!("forcing queue sync for id={id}"); - unparker.unpark(); + // A fence change can empty the queue concurrently (`wake_queue_for_fence`), so an + // entry counted here may already be gone. + let mut queue_len = self.manager.write_queue.len(); + while queue_len > 0 { + match self.manager.write_queue.steal() { + Steal::Success((id, _, unparker)) => { + tracing::debug!("forcing queue sync for id={id}"); + unparker.unpark(); + queue_len -= 1; + } + Steal::Empty => break, + Steal::Retry => (), + } } } @@ -577,6 +715,7 @@ impl WrapWal for ManagedConnectionWalWrapper { mod fence_tests { use std::path::Path; use std::sync::Arc; + use std::time::Duration; use libsql_sys::wal::wrapper::PassthroughWalWrapper; use libsql_sys::wal::Sqlite3WalManager; @@ -585,6 +724,7 @@ mod fence_tests { use tempfile::tempdir; use crate::connection::connection_core::CoreConnection; + use crate::connection::connection_core::VacuumOutcome; use crate::connection::legacy::{LegacyConnection, MakeLegacyConnection}; use crate::connection::program::Program; use crate::connection::Connection as _; @@ -594,7 +734,7 @@ mod fence_tests { }; use crate::namespace::fence::controller::FenceController; use crate::namespace::fence::outcome::{FenceDetail, FenceError, FenceOutcome}; - use crate::namespace::fence::state::FenceState; + use crate::namespace::fence::state::{FenceState, OperationClass}; use crate::namespace::meta_store::{MetaStore, MetaStoreHandle}; use crate::query_result_builder::test::{StepResult, TestBuilder}; use crate::query_result_builder::QueryResultBuilder as _; @@ -611,6 +751,13 @@ mod fence_tests { impl Harness { async fn new() -> Self { + Self::with_txn_timeout(None).await + } + + /// A harness whose connections steal the write slot from a transaction only after + /// `txn_timeout` (the test default is 100 ms). Queue tests hold a writer open far longer + /// than that and must not have it stolen. + async fn with_txn_timeout(txn_timeout: Option) -> Self { let dir = tempdir().unwrap(); let meta_dir = dir.path().join("meta"); let db_dir = dir.path().join("db"); @@ -619,7 +766,13 @@ mod fence_tests { let meta = open_metastore(&meta_dir).await; create_namespace(&meta, "ns").await; let controller = FenceController::unfenced("ns".into()); - let maker = make_connections(&db_dir, controller.clone()).await; + let config = MetaStoreHandle::load(&db_dir).unwrap(); + if txn_timeout.is_some() { + let mut c = (*config.get()).clone(); + c.txn_timeout = txn_timeout; + config.store(c).await.unwrap(); + } + let maker = make_connections(&db_dir, config, controller.clone()).await; let this = Self { _dir: dir, meta, @@ -657,6 +810,7 @@ mod fence_tests { async fn make_connections( path: &Path, + config: MetaStoreHandle, fence: Arc, ) -> MakeLegacyConnection { MakeLegacyConnection::new( @@ -664,7 +818,7 @@ mod fence_tests { PassthroughWalWrapper, Default::default(), Default::default(), - MetaStoreHandle::load(path).unwrap(), + config, Arc::new([]), 100000000, 100000000, @@ -939,4 +1093,227 @@ mod fence_tests { } assert!(conn.fence.take_denial().is_none()); } + + /// Long enough that no test transaction is ever stolen by the manager's own timeout. + const LONG_TXN: Option = Some(Duration::from_secs(600)); + /// Upper bound for a test waiting on something that happens promptly; reaching it is a + /// failure, never the expected path. + const PROMPT: Duration = Duration::from_secs(30); + + /// Wait until `n` connections are parked in the write queue. This polls a condition; it + /// does not use elapsed time as evidence of anything. + async fn until_queued(h: &Harness, n: usize) { + tokio::time::timeout(PROMPT, async { + while h.maker.connection_manager().queued_writers() != n { + tokio::time::sleep(Duration::from_millis(1)).await; + } + }) + .await + .unwrap_or_else(|_| panic!("{n} connections never queued for the write slot")); + } + + fn freelist(conn: &Conn) -> i64 { + conn.with_raw(|c| c.query_row("pragma freelist_count", (), |r| r.get(0))) + .unwrap() + } + + /// A writer parked in the write queue behind an open transaction leaves the queue with the + /// typed denial as soon as the fence changes the write generation: it neither waits for the + /// slot nor, once the holder commits, writes. The holder itself, admitted before the fence, + /// keeps the slot and commits, which is what the positive drain waits for. + #[tokio::test(flavor = "multi_thread")] + async fn fence_rejects_queued_writer() { + let h = Harness::with_txn_timeout(LONG_TXN).await; + let manager = h.maker.connection_manager().clone(); + let holder = h.conn().await; + assert_ok(&run(&holder, &["begin immediate", "insert into t values (1)"]).await); + let (_, class) = manager + .active_writer() + .expect("the holder has the write slot"); + assert_eq!(class, OperationClass::NormalWrite); + + let queued = h.conn().await; + let waiting = tokio::spawn({ + let queued = queued.clone(); + async move { run(&queued, &["insert into t values (2)"]).await } + }); + until_queued(&h, 1).await; + assert!(!waiting.is_finished()); + + h.fence().await; + let steps = tokio::time::timeout(PROMPT, waiting) + .await + .expect("the queued writer was not woken by the fence") + .unwrap(); + assert_eq!( + fence_error(&steps[0]).outcome(), + FenceOutcome::MigrationWriteFenced + ); + assert_eq!(manager.queued_writers(), 0); + assert_eq!( + manager.active_writer().map(|(_, c)| c), + Some(OperationClass::NormalWrite) + ); + + assert_ok(&run(&holder, &["commit"]).await); + assert_eq!(manager.active_writer(), None); + assert_eq!(count(&holder).await, 1); + // The refused connection stays refused while the fence holds. + let steps = run(&queued, &["insert into t values (3)"]).await; + assert_eq!( + fence_error(&steps[0]).outcome(), + FenceOutcome::MigrationWriteFenced + ); + assert_eq!(count(&holder).await, 1); + } + + /// A checkpoint is maintenance: one queued behind an open transaction when the fence + /// changes queues again instead of being refused, and runs once the holder commits. + #[tokio::test(flavor = "multi_thread")] + async fn queued_checkpoint_survives_fence_wake() { + let h = Harness::with_txn_timeout(LONG_TXN).await; + let manager = h.maker.connection_manager().clone(); + let holder = h.conn().await; + assert_ok(&run(&holder, &["begin immediate", "insert into t values (1)"]).await); + + let checkpointer = h.conn().await; + let checkpoint = tokio::task::spawn_blocking({ + let inner = checkpointer.inner.clone(); + move || inner.lock().checkpoint() + }); + until_queued(&h, 1).await; + + h.fence().await; + // Woken by both transitions, it put itself back in the queue. + until_queued(&h, 1).await; + assert!(!checkpoint.is_finished()); + + assert_ok(&run(&holder, &["commit"]).await); + tokio::time::timeout(PROMPT, checkpoint) + .await + .expect("the checkpoint never got the slot") + .unwrap() + .unwrap(); + assert_eq!(manager.active_writer(), None); + assert_eq!(count(&holder).await, 1); + } + + /// `TRUNCATE` checkpoints run in every fence state. + #[tokio::test(flavor = "multi_thread")] + async fn checkpoint_allowed_while_fenced() { + let h = Harness::new().await; + let conn = h.conn().await; + assert_ok( + &run( + &conn, + &["insert into t values (1)", "insert into t values (2)"], + ) + .await, + ); + h.fence().await; + + conn.checkpoint().await.unwrap(); + let (busy, log, checkpointed): (i64, i64, i64) = conn + .with_raw(|c| { + c.query_row("pragma wal_checkpoint(truncate)", (), |r| { + Ok((r.get(0)?, r.get(1)?, r.get(2)?)) + }) + }) + .unwrap(); + assert_eq!((busy, log, checkpointed), (0, 0, 0)); + assert_eq!(count(&conn).await, 2); + } + + /// `VACUUM` is not maintenance. While normal writes are denied it is skipped, and reported as + /// skipped rather than failed; once writes are admitted again it runs. + #[tokio::test(flavor = "multi_thread")] + async fn vacuum_skipped_while_fenced() { + let h = Harness::new().await; + let conn = h.conn().await; + assert_ok( + &run( + &conn, + &[ + "insert into t select randomblob(4096) from \ + (with recursive n(i) as (select 1 union all select i + 1 from n where i < 200) \ + select i from n)", + "delete from t", + ], + ) + .await, + ); + let free = freelist(&conn); + assert!(free > 100, "freelist {free}"); + + h.fence().await; + let outcome = conn.inner.lock().vacuum_if_needed_above(0).unwrap(); + assert_eq!(outcome, VacuumOutcome::Fenced); + assert_eq!(freelist(&conn), free); + // The periodic path reports success, not a failed vacuum. + conn.vacuum_if_needed().await.unwrap(); + assert_eq!(freelist(&conn), free); + + h.release().await; + let outcome = conn.inner.lock().vacuum_if_needed_above(0).unwrap(); + assert_eq!(outcome, VacuumOutcome::Vacuumed); + assert_eq!(freelist(&conn), 0); + } + + /// `abort_active` rolls back the connection holding the write slot, which releases it and + /// notifies drain waiters; a rollback handle whose connection has already closed does + /// nothing instead of panicking, and there is nothing to abort without a writer. + #[tokio::test(flavor = "multi_thread")] + async fn abort_active_tolerates_closed_connection() { + let h = Harness::with_txn_timeout(LONG_TXN).await; + let manager = h.maker.connection_manager().clone(); + assert_eq!(manager.abort_active(), None); + + let holder = h.conn().await; + assert_ok(&run(&holder, &["begin immediate", "insert into t values (1)"]).await); + let (id, _) = manager.active_writer().unwrap(); + let released = manager.released().notified(); + tokio::pin!(released); + released.as_mut().enable(); + assert_eq!(manager.abort_active(), Some(id)); + tokio::time::timeout(PROMPT, released) + .await + .expect("the rollback did not notify drain waiters"); + assert_eq!(manager.active_writer(), None); + assert_eq!(count(&h.conn().await).await, 0); + + // A handle taken while its connection was open, used after the connection closed. + let closing = h.conn().await; + let closing_id = *manager.inner.abort_handle.lock().keys().max().unwrap(); + let handle = manager.inner.abort_handle.lock()[&closing_id].clone(); + drop(closing); + assert!(!manager.inner.abort_handle.lock().contains_key(&closing_id)); + handle.abort(); + assert_eq!(manager.abort_active(), None); + } + + /// Committing releases the slot and wakes a waiter that registered before it looked at the + /// active writer, which is the drain's wait (section 8.3 step 5). + #[tokio::test(flavor = "multi_thread")] + async fn release_notifies_drain_waiters() { + let h = Harness::with_txn_timeout(LONG_TXN).await; + let manager = h.maker.connection_manager().clone(); + let holder = h.conn().await; + assert_ok(&run(&holder, &["begin immediate", "insert into t values (1)"]).await); + h.fence().await; + + let released = manager.released().notified(); + tokio::pin!(released); + released.as_mut().enable(); + assert!(manager.active_writer().is_some()); + let commit = tokio::spawn({ + let holder = holder.clone(); + async move { run(&holder, &["commit"]).await } + }); + tokio::time::timeout(PROMPT, released) + .await + .expect("the commit did not notify drain waiters"); + assert_eq!(manager.active_writer(), None); + assert_ok(&commit.await.unwrap()); + assert_eq!(count(&holder).await, 1); + } } diff --git a/libsql-server/src/connection/legacy.rs b/libsql-server/src/connection/legacy.rs index 93ab86cb49..88e7115b7a 100644 --- a/libsql-server/src/connection/legacy.rs +++ b/libsql-server/src/connection/legacy.rs @@ -77,6 +77,9 @@ where fence: Arc, ) -> Result { let txn_timeout = config_store.get().txn_timeout.unwrap_or(TXN_TIMEOUT); + let connection_manager = ConnectionManager::new(txn_timeout); + // Queued writers re-check the fence whenever its write generation changes. + fence.register_write_queue(connection_manager.fence_waker()); let mut this = Self { db_path, @@ -93,7 +96,7 @@ where encryption_config, block_writes, resolve_attach_path, - connection_manager: ConnectionManager::new(txn_timeout), + connection_manager, make_wal_manager, fence, }; @@ -104,6 +107,13 @@ where Ok(this) } + /// The write-slot manager shared by every connection this maker opens. + // Used by the positive source write drain (section 8.3). + #[cfg_attr(not(test), allow(dead_code))] + pub(crate) fn connection_manager(&self) -> &ConnectionManager { + &self.connection_manager + } + /// Tries to create a database, retrying if the database is busy. async fn try_create_db(&self) -> Result> { // try 100 times to acquire initial db connection. diff --git a/libsql-server/src/namespace/fence/controller.rs b/libsql-server/src/namespace/fence/controller.rs index a9a31dc03a..02b9a3db8c 100644 --- a/libsql-server/src/namespace/fence/controller.rs +++ b/libsql-server/src/namespace/fence/controller.rs @@ -110,11 +110,17 @@ impl GateSnapshot { } } +/// Wakes one connection manager's write queue after a write-generation change; returns `false` +/// once the manager is gone. +pub type WriteQueueWaker = Box bool + Send + Sync>; + /// The fence controller of one namespace. pub struct FenceController { namespace: NamespaceName, transition_lock: Arc>, gate: watch::Sender, + /// The write queues of the namespace's connection managers (section 8.2). + write_queues: Mutex>, #[cfg(test)] hooks: FenceTestHooks, } @@ -136,6 +142,7 @@ impl FenceController { namespace, transition_lock: Default::default(), gate, + write_queues: Mutex::new(Vec::new()), #[cfg(test)] hooks: FenceTestHooks::default(), }) @@ -174,6 +181,13 @@ impl FenceController { self.gate.borrow().permits(class) } + /// Register a connection manager's write queue, to be woken after every change of the write + /// generation so that queued writers re-check the gate instead of waiting for the slot. + /// Wakers whose manager is gone are dropped on the next change. + pub fn register_write_queue(&self, waker: WriteQueueWaker) { + self.write_queues.lock().push(waker); + } + /// Take the namespace's transition lock. Every fence command on the namespace runs while /// holding it, from its first check to its response. pub async fn begin_transition(self: &Arc) -> Transition { @@ -224,6 +238,7 @@ impl FenceController { /// Publish a new gate. `fence: None` keeps the published fence. The write generation moves /// whenever the state, the owning operation or the indeterminate flag changes. fn publish(&self, fence: Option, indeterminate: Option) { + let mut generation_changed = false; self.gate.send_modify(|gate| { let fence = fence.unwrap_or_else(|| gate.fence.clone()); let changed = fence.state() != gate.fence.state() @@ -234,17 +249,24 @@ impl FenceController { gate.indeterminate = indeterminate; if changed { gate.write_generation += 1; + generation_changed = true; } }); - let gate = self.gate.borrow(); - tracing::debug!( - namespace = %self.namespace, - state = %gate.state(), - revision = gate.revision(), - write_generation = gate.write_generation, - indeterminate = gate.indeterminate.is_some(), - "published namespace fence gate" - ); + { + let gate = self.gate.borrow(); + tracing::debug!( + namespace = %self.namespace, + state = %gate.state(), + revision = gate.revision(), + write_generation = gate.write_generation, + indeterminate = gate.indeterminate.is_some(), + "published namespace fence gate" + ); + } + // After the gate is published, so that every woken writer re-checks against it. + if generation_changed { + self.write_queues.lock().retain(|wake| wake()); + } } } @@ -682,6 +704,60 @@ pub(crate) mod tests { assert_eq!(controller.gate(), before); } + /// Registered write queues are woken after every change of the write generation, and only + /// then; a waker whose manager is gone is forgotten. + #[tokio::test] + async fn write_queues_are_woken_on_every_generation_change() { + let tmp = tempdir().unwrap(); + let meta = open_metastore(tmp.path()).await; + create_namespace(&meta, "ns").await; + let controller = FenceController::unfenced("ns".into()); + + let woken = Arc::new(AtomicU64::new(0)); + let seen_generation = Arc::new(AtomicU64::new(0)); + controller.register_write_queue(Box::new({ + let woken = woken.clone(); + let seen_generation = seen_generation.clone(); + let controller = Arc::downgrade(&controller); + move || { + // The gate is already published when the queue is woken. + if let Some(c) = controller.upgrade() { + seen_generation.store(c.write_generation(), Ordering::SeqCst); + } + woken.fetch_add(1, Ordering::SeqCst); + true + } + })); + let gone = Arc::new(AtomicU64::new(0)); + controller.register_write_queue(Box::new({ + let gone = gone.clone(); + move || { + gone.fetch_add(1, Ordering::SeqCst); + false + } + })); + assert_eq!(controller.write_queues.lock().len(), 2); + + fence_source(&meta, &controller, OP).await; + assert_eq!(woken.load(Ordering::SeqCst), 2); + assert_eq!(seen_generation.load(Ordering::SeqCst), 2); + assert_eq!(gone.load(Ordering::SeqCst), 1); + assert_eq!(controller.write_queues.lock().len(), 1); + + // A replay does not move the generation and wakes nobody. + let revision = controller.gate().revision(); + controller + .apply_command(&meta, release("ns", OP, 2, revision), ctx()) + .await + .unwrap(); + assert_eq!(woken.load(Ordering::SeqCst), 3); + controller + .apply_command(&meta, release("ns", OP, 2, revision), ctx()) + .await + .unwrap(); + assert_eq!(woken.load(Ordering::SeqCst), 3); + } + #[tokio::test] async fn failed_before_commit_leaves_gate_unchanged() { let tmp = tempdir().unwrap(); From 95f8ba681a15f3ca44e02b32b47c2e251a659070 Mon Sep 17 00:00:00 2001 From: River Date: Tue, 29 Sep 2026 16:32:59 +0000 Subject: [PATCH 10/33] libsql-server: positive source write drain AcquireSourceWriteFence now runs the drain of docs/NAMESPACE_FENCE.md section 8.3 end to end, under the namespace's transition lock and on a task of its own: - an in-memory INSTALLING gate closes write admission (and moves the write generation, waking queued writers) before SOURCE_DRAINING is persisted; a command proven not to have committed removes it again; - the drain waits on the connection manager's release notification for the writer that held the slot when admission closed, never on elapsed time or the transaction timeout; at the deadline it answers DRAINING (admission stays closed) or, with force_rollback, rolls the writer back and waits for the actual release; - the frozen boundary (log id, last committed frame) is read under the write-slot lock once no writer holds it, and SOURCE_WRITE_FENCED is committed with it; - replaying a DRAINING command resumes the same drain. Primary connection makers register a write-drain source (their connection manager, held weakly, and their replication log) with the namespace's controller; NamespaceStore::execute_fence_command loads the namespace before an acquisition so that the source exists. The default drain deadline is --namespace-fence-default-write-drain-ms (30 s). FrozenBoundary.frame_no becomes optional, for a log without frames. Co-authored-by: Tomasz Szymczyszyn --- docs/NAMESPACE_FENCE.md | 40 +- libsql-server/proto/namespace_fence.proto | 3 +- libsql-server/src/config.rs | 3 + .../src/connection/connection_manager.rs | 44 +- libsql-server/src/connection/legacy.rs | 2 - .../src/generated/namespace_fence.rs | 5 +- libsql-server/src/main.rs | 9 + .../src/namespace/configurator/helpers.rs | 47 +- .../src/namespace/fence/controller.rs | 144 +++- libsql-server/src/namespace/fence/drain.rs | 771 ++++++++++++++++++ libsql-server/src/namespace/fence/mod.rs | 6 +- libsql-server/src/namespace/fence/record.rs | 5 +- .../src/namespace/fence/transition.rs | 16 +- libsql-server/src/namespace/meta_store.rs | 23 +- libsql-server/src/namespace/store.rs | 35 +- 15 files changed, 1076 insertions(+), 77 deletions(-) create mode 100644 libsql-server/src/namespace/fence/drain.rs diff --git a/docs/NAMESPACE_FENCE.md b/docs/NAMESPACE_FENCE.md index 02a4d92d26..17e0920170 100644 --- a/docs/NAMESPACE_FENCE.md +++ b/docs/NAMESPACE_FENCE.md @@ -185,6 +185,8 @@ Success (`APPLIED`, `ALREADY_APPLIED`) is `200`; `DRAINING` is `202`. Every resp } ``` +`frozen_boundary.frame_no` is the last frame committed to the source's replication log, or `null` when the log has no frames. + Errors use the same shape with `"outcome": ""`, plus `"error": ""` and, where useful, `"detail"` (a bounded reason string such as `role_mismatch` or `namespace_identity_mismatch`). ### 4.4 Routes @@ -404,9 +406,10 @@ This is additive in proto3: older peers skip the unknown field; a newer replica Per namespace: - `transition_lock`: a `tokio::sync::Mutex` serialising commands on this namespace. A command holds it from its first check to its response (`FenceController::begin_transition` returns a `Transition` that owns the guard). -- `gate`: a `tokio::sync::watch` of `GateSnapshot { fence, write_generation, indeterminate }`, where `fence` is the durable fence as last published (a record, no record, or `UNKNOWN_UNAVAILABLE` with its detail). State, revision, owning operation and every admission (`permits(class)`, `write()`, `read()`) are derived from it through the permission matrix. The WAL wrapper, `CoreConnection`, dump, replication and lifecycle code read it without locks. Phase 5 adds the live capability set. +- `gate`: a `tokio::sync::watch` of `GateSnapshot { fence, write_generation, indeterminate, installing }`, where `fence` is the durable fence as last published (a record, no record, or `UNKNOWN_UNAVAILABLE` with its detail) and `installing` is the in-memory `INSTALLING` gate of a closing command being persisted (section 8.3), which denies normal writes, vacuum, import writes and lifecycle work on top of `fence`. State, revision, owning operation and every admission (`permits(class)`, `write()`, `read()`) are derived from it through the permission matrix. The WAL wrapper, `CoreConnection`, dump, replication and lifecycle code read it without locks. Phase 5 adds the live capability set. - `write_generation: u64`, in the snapshot, **incremented on every publication that changes the fence state, the owning operation, or the indeterminate flag**. That covers every transition that closes or opens write admission (acquire, release, create, seal, publish, enable, abort, adopt), and is conservative for the others. A replay that publishes the same durable state does not move it. - `indeterminate: Option<(operation_id, command_id)>`: set when a command's `COMMIT` failed (or the task running it died) so that whether it applied is unknown. While set, every class except `Maintenance` and `Observability` is denied with `FENCE_STATE_UNAVAILABLE` / `indeterminate_commit`, and every other command is refused with `FENCE_COMMIT_INDETERMINATE`. A replay of the same command is answered by the metastore from the durable row (replayed if it had committed, applied if it had not) and clears it (section 8.4). +- the write-drain sources of the namespace's primary connection makers (`register_write_drain`): each maker's connection manager, held weakly, and its replication log id and last-committed-frame reader, which the write drain waits on and reads the boundary from (section 8.3). - the write queues of the namespace's connection managers: each `MakeLegacyConnection` registers a waker (`register_write_queue`) that the controller calls after every publication that moves `write_generation`, after the new gate is visible. A waker holds its manager weakly and is dropped once the manager is gone (for example after eviction). Writer tracking itself is the connection manager's (section 8.2). - `read_leases`: counters and cancel handles per lease class (`sql`, `dump`, `replication`), with a `Notify` on every release. - `capabilities`: the live `MigrationCapability` set and an import-writer counter. @@ -458,24 +461,27 @@ This makes the WAL gate independent of statement classification: DDL, misclassif ### 8.3 Source write drain, step by step -`AcquireSourceWriteFence`, under the transition lock: +`AcquireSourceWriteFence` runs through `FenceController::execute` (`namespace/fence/drain.rs`), on a task of its own and under the transition lock for all seven steps. `NamespaceStore::execute_fence_command` loads the namespace first, so that its primary connection maker has registered a *write-drain source* with the controller: the maker's connection manager (held weakly) and its replication log (`log_id` and the last committed frame). Sources whose manager is gone are dropped; after an eviction and a lazy reload there can be more than one live source for a while, and the drain waits for all of them. The drain policy is the request's, or `--namespace-fence-default-write-drain-ms` (default 30 s) with `on_deadline: fail`. -1. Replay handling and checks (section 5.3), including `expected_namespace_identity.log_id == ReplicationLogger::log_id()` and the shared-schema rejection. -2. Publish the in-memory `INSTALLING` gate: write admission closed, `write_generation += 1`. New write admissions now fail with `MIGRATION_WRITE_FENCED`. -3. Wake the write queue (section 8.2). Queued writers fail with `MIGRATION_WRITE_FENCED`. +1. Replay handling and checks (section 5.3), including `expected_namespace_identity.log_id == ReplicationLogger::log_id()` (the log id of the newest live source) and the shared-schema rejection. These run inside the metastore transaction of step 4. +2. If the gate still admits normal writes, publish the in-memory `INSTALLING` gate (`GateSnapshot::installing`): normal writes, vacuum, import writes and lifecycle work are refused with `MIGRATION_WRITE_FENCED`, and `write_generation += 1`. Where writes are already closed (a resumed drain, a command being reconciled after an indeterminate commit, a fenced namespace) nothing is installed. +3. The publication wakes the write queues (section 8.2). Queued writers fail with `MIGRATION_WRITE_FENCED`. 4. CAS `SOURCE_DRAINING` in the metastore with receipt outcome `DRAINING`. - - Committed: continue. - - Proven not committed (the transaction failed before `COMMIT` for a precondition or constraint reason): remove the `INSTALLING` gate, bump the generation, return the error. - - Unknown (error on `COMMIT`, task cancelled, timeout): the gate stays closed, the controller enters `Indeterminate(command_id)`, the response is `FENCE_COMMIT_INDETERMINATE`. It is never treated as not applied. -5. Wait for the active pre-cutoff writer to commit or roll back, on the manager's release notification. Elapsed time and `txn_timeout` are never evidence. - - Deadline reached with `on_deadline: fail`: respond `DRAINING`. The durable state stays `SOURCE_DRAINING` and admission stays closed. - - Deadline reached with `on_deadline: force_rollback`: `abort_active()`, then keep waiting for the release notification. -6. Under the manager's `current` lock, observe no writer holding the slot and read the frozen boundary (`log_id`, current `frame_no`). -7. CAS `SOURCE_WRITE_FENCED` with the boundary; update the receipt to `APPLIED`; write the marker; respond. + - Committed: the commit publishes `SOURCE_DRAINING` in place of the `INSTALLING` gate (write admission never reopens in between). Continue. + - A replay of a finished acquisition, or `ALREADY_APPLIED`: publish the durable state, respond with the stored result. + - A replay of the `DRAINING` receipt, or a new command of the owner joining the drain the record is in: nothing is written; continue at step 5. + - Proven not committed (the transition function refused it, or the transaction failed before `COMMIT`): remove the `INSTALLING` gate, which moves the generation again, and return the error. A transaction opened while it was up therefore cannot write afterwards. + - Unknown (error on `COMMIT`, the task died): the gate closes as indeterminate (section 7.2) and the response is `FENCE_COMMIT_INDETERMINATE`. It is never treated as not applied. +5. For each live source, wait until its connection manager has no connection holding the write slot for a write (a checkpoint, `Maintenance`, may hold it): enable the manager's release `Notify`, check `has_writer()`, and wait for the notification. Elapsed time and `txn_timeout` are never evidence; with admission closed nobody queues behind the holder, so its slot is not stolen by the timeout either. Because admission is closed, a manager seen without a writer stays without one, so the managers are waited for in turn. + - Deadline reached with `on_deadline: fail`: respond with the `DRAINING` commit. The durable state stays `SOURCE_DRAINING` and admission stays closed. + - Deadline reached with `on_deadline: force_rollback`: `abort_active()` on every manager that still has a writer (on a blocking task: the rollback takes the connection's lock, which a running program holds), then keep waiting for the actual release, for one more deadline but at least 10 seconds (a rollback releases at once unless a program is still running on the connection); if the writer has still not released, respond `DRAINING`. That bound only decides when to answer `DRAINING`; it is never evidence of a drain. + - No live source (no loaded primary, so the replication log cannot be read): respond `DRAINING`; a replay once the namespace is loaded completes it. +6. Hook point `BeforeBoundaryCapture`. Then, under each manager's `current` lock (`ConnectionManager::with_no_writer`), observe that no writer holds the slot and read that source's last committed frame. The frozen boundary is the newest source's `log_id` and the highest of those frames; `frame_no` is absent when the log has no frames. If a writer were seen here the drain would wait again rather than guess. +7. CAS `SOURCE_WRITE_FENCED` with the boundary (`complete_fence_drain`, keyed by the `DRAINING` receipt); the receipt becomes `APPLIED`; the marker is written; the gate is published; respond. ### 8.4 Reconciliation and resumption -- Replay of a command whose receipt says `DRAINING` resumes at step 5. After a restart there is no pre-cutoff writer (SQLite recovery discards uncommitted work), so it completes at once. +- Replay of a command whose receipt says `DRAINING` resumes at step 5, with the request's own deadline counted from the replay. After a restart there is no pre-cutoff writer (SQLite recovery discards uncommitted work), so it completes at once. - Replay of a command held `Indeterminate` re-reads the metastore: if the row shows the command applied, it continues from that durable point; if it shows it did not, it retries the same CAS. Other commands receive `FENCE_COMMIT_INDETERMINATE` until then. After a restart the gate reflects whatever is durable, which by definition was never acknowledged as open. - Opening transitions (`ReleaseSourceWriteFence`, `EnableTargetWrites`) follow **commit → publish the exact revision to the gate → respond `APPLIED`**. A crash after commit and before publication sends no success, and startup recovers the committed gate before exposing the namespace. @@ -694,12 +700,12 @@ Planned test names; the table is updated as tests land. | # | Requirement | Planned tests | |---|---|---| -| 1 | Concurrent acquisition by two operations: one owner, typed conflict for the loser | `fence::tests::acquire_race_single_owner`; `tests::fence::admin::concurrent_acquire_one_owner` | -| 2 | Active writer commits or is rolled back before freeze acknowledgement; nothing commits after | `fence::drain::tests::active_writer_commits_before_ack`, `forced_rollback_before_ack`, `no_commit_after_ack` | +| 1 | Concurrent acquisition by two operations: one owner, typed conflict for the loser | landed: `namespace::fence::drain::tests::acquire_race_single_owner` (the first acquisition is parked after closing admission while the second waits on the transition lock); planned: `tests::fence::admin::concurrent_acquire_one_owner` | +| 2 | Active writer commits or is rolled back before freeze acknowledgement; nothing commits after | landed: `namespace::fence::drain::tests::{active_writer_commits_before_ack, forced_rollback_before_ack, no_commit_after_ack}` (the boundary equals the last committed replication frame and no frame follows it; autocommit, `BEGIN IMMEDIATE`, DDL and a pre-fence read transaction upgrading are refused), `installing_gate_closes_writes_before_persisting`, `refused_acquire_reopens_writes`, `release_reopens_with_new_generation` | | 3 | Autocommit, explicit transactions, queued writers, batches, DDL, schema jobs, old WebSockets, read-to-write upgrades cannot bypass | landed: `connection::connection_manager::fence_tests::{fence_rejects_read_to_write_upgrade, fence_rejects_ddl_and_pragma, fence_rejects_raw_with_raw_write}` (autocommit, explicit transactions, DDL, header-writing pragma, `BEGIN IMMEDIATE`, `VACUUM`, `with_raw` users); `connection::connection_manager::fence_tests::fence_rejects_queued_writer` (a writer parked in the queue behind an open transaction leaves it with `MIGRATION_WRITE_FENCED` when the fence changes, and the holder keeps the slot); maintenance and vacuum under a fence: `queued_checkpoint_survives_fence_wake`, `checkpoint_allowed_while_fenced`, `vacuum_skipped_while_fenced`; drain primitives: `abort_active_tolerates_closed_connection`, `release_notifies_drain_waiters`, `fence::controller::tests::write_queues_are_woken_on_every_generation_change`; planned: `tests::fence::protocol::old_ws_session_cannot_write`, `batch_denied_mid_batch`; `fence::tests::acquire_rejects_shared_schema` | | 4 | Program that captured config before the fence is rejected at the WAL | landed: `connection::connection_manager::fence_tests::wal_gate_rejects_program_admitted_before_fence` (a SQL function parks the program between admission and its write while the fence is acquired and released) | | 5 | Pre-fence transactions cannot write after release or publication | landed: `connection::connection_manager::fence_tests::stale_generation_cannot_write_after_release`; unfenced behaviour unchanged: `unfenced_namespace_is_unchanged`; planned: `stale_generation_cannot_write_after_enable_writes` | -| 6 | Acquisition timeout returns `DRAINING`, admission stays closed | `fence::drain::tests::deadline_returns_draining_and_stays_closed` | +| 6 | Acquisition timeout returns `DRAINING`, admission stays closed | landed: `namespace::fence::drain::tests::{deadline_returns_draining_and_stays_closed, replay_of_draining_resumes_and_completes}` | | 7 | Restart at every persistence boundary; indeterminate persistence keeps the gate closed until same-command reconciliation | `fence::tests::restart_at_each_boundary` (parameterised over hook points), `indeterminate_commit_keeps_gate_closed`; landed: `fence::controller::tests::{indeterminate_commit_keeps_writes_closed_until_replayed, indeterminate_commit_that_did_not_apply_is_retried_by_replay, failed_before_commit_leaves_gate_unchanged, publication_happens_before_the_response, committed_command_is_published_when_the_caller_goes_away}`, `namespace::store::fence_tests::restart_installs_the_durable_gate_before_serving` | | 8 | Evict and lazily reload a fenced namespace; identical admission | `tests::fence::lifecycle::evicted_namespace_reloads_same_gate`; landed at unit level: `namespace::store::fence_tests::evicted_namespace_reloads_with_the_same_controller`, `fence::registry::tests::seeded_from_load_fences_including_recovered_names` | | 9 | Filesystem recovery, `destroy_on_error`, undecodable records, missing target quarantine, metastore backup rollback fail closed with provenance | `meta_store::fence_tests::recovery::{fs_recovery_with_marker_unavailable, destroy_on_error_keeps_fenced_unavailable, undecodable_row_unavailable, incomplete_target_unavailable, metastore_rollback_detected_by_marker, lookup_never_creates, undecodable_name_with_fence_fails_startup, marker_in_invalid_directory_fails_startup}`; legacy behaviour kept: `destroy_on_error_without_fences_is_unchanged`, `undecodable_row_without_fences_is_skipped_as_before`; `meta_store::fence_tests::corrupt_fence_row_fails_closed` | diff --git a/libsql-server/proto/namespace_fence.proto b/libsql-server/proto/namespace_fence.proto index efdee1ab81..73250dfb0d 100644 --- a/libsql-server/proto/namespace_fence.proto +++ b/libsql-server/proto/namespace_fence.proto @@ -75,7 +75,8 @@ message DrainPolicy { message FrozenBoundary { string log_id = 1; - uint64 frame_no = 2; + // Absent when the replication log has no frames. + optional uint64 frame_no = 2; } message LegacyBlocks { diff --git a/libsql-server/src/config.rs b/libsql-server/src/config.rs index 9ac7add98b..4457232d03 100644 --- a/libsql-server/src/config.rs +++ b/libsql-server/src/config.rs @@ -193,6 +193,9 @@ pub struct MetaStoreConfig { /// How long receipts of finished fence operations are kept. `None` is the default of /// 30 days. pub namespace_fence_receipt_retention: Option, + /// How long `AcquireSourceWriteFence` waits for active writers when the request names no + /// drain policy. `None` is the default of 30 seconds. + pub namespace_fence_default_write_drain: Option, } #[derive(Debug, Clone)] diff --git a/libsql-server/src/connection/connection_manager.rs b/libsql-server/src/connection/connection_manager.rs index 009adfeb45..aab548461b 100644 --- a/libsql-server/src/connection/connection_manager.rs +++ b/libsql-server/src/connection/connection_manager.rs @@ -75,17 +75,41 @@ impl ConnectionManager { /// The connection holding the write slot, and the class it holds it for. A slot that has /// been handed to a queued connection that has not taken it yet counts as held. - // Used by the positive source write drain (section 8.3). - #[cfg_attr(not(test), allow(dead_code))] pub(crate) fn active_writer(&self) -> Option<(ConnId, OperationClass)> { self.inner.current.lock().map(|slot| (slot.id, slot.class)) } + /// Whether a connection holds the write slot for a write transaction: any holder but a + /// checkpoint (`Maintenance`), which cannot change logical contents. + pub(crate) fn has_writer(&self) -> bool { + self.active_writer() + .is_some_and(|(_, class)| class != OperationClass::Maintenance) + } + + /// Run `f` under the write-slot lock if no connection holds the slot for a write + /// transaction (a checkpoint may). While `f` runs no write transaction can start or end on + /// this manager, so, once the fence has closed write admission, anything `f` reads about the + /// committed log is final (`docs/NAMESPACE_FENCE.md` section 8.3, step 6). Returns the + /// holder otherwise. + pub(crate) fn with_no_writer( + &self, + f: impl FnOnce() -> R, + ) -> Result { + let current = self.inner.current.lock(); + match *current { + Some(slot) if slot.class != OperationClass::Maintenance => Err((slot.id, slot.class)), + _ => Ok(f()), + } + } + + /// A handle that does not keep the manager alive. + pub(crate) fn downgrade(&self) -> WeakConnectionManager { + WeakConnectionManager(Arc::downgrade(&self.inner)) + } + /// Notified (with `notify_waiters`) every time the write slot is released or handed on. A /// waiter registers interest (`Notified::enable`) before it checks /// [`active_writer`](Self::active_writer), so a release in between is not missed. - // Used by the positive source write drain (section 8.3). - #[cfg_attr(not(test), allow(dead_code))] pub(crate) fn released(&self) -> &tokio::sync::Notify { &self.inner.released } @@ -94,8 +118,6 @@ impl ConnectionManager { /// handle it registered. Returns the connection that was asked to roll back, if any. The /// slot is released by the rollback itself (`end_read_txn`/`end_write_txn`), which notifies /// [`released`](Self::released); a connection closing at the same time is tolerated. - // Used by the positive source write drain (section 8.3). - #[cfg_attr(not(test), allow(dead_code))] pub(crate) fn abort_active(&self) -> Option { let id = self.active_writer()?.0; let handle = self.inner.abort_handle.lock().get(&id).cloned(); @@ -135,6 +157,16 @@ impl ConnectionManager { } } +/// A [`ConnectionManager`] that is not kept alive by this handle. +#[derive(Clone)] +pub(crate) struct WeakConnectionManager(std::sync::Weak); + +impl WeakConnectionManager { + pub(crate) fn upgrade(&self) -> Option { + self.0.upgrade().map(|inner| ConnectionManager { inner }) + } +} + fn wake_queue_for_fence(inner: &ConnectionManagerInner) { // Under the `current` lock, so that a waiter either queued before this (and is stolen and // woken here) or observes the new token when it takes the lock. diff --git a/libsql-server/src/connection/legacy.rs b/libsql-server/src/connection/legacy.rs index 88e7115b7a..71763e199e 100644 --- a/libsql-server/src/connection/legacy.rs +++ b/libsql-server/src/connection/legacy.rs @@ -108,8 +108,6 @@ where } /// The write-slot manager shared by every connection this maker opens. - // Used by the positive source write drain (section 8.3). - #[cfg_attr(not(test), allow(dead_code))] pub(crate) fn connection_manager(&self) -> &ConnectionManager { &self.connection_manager } diff --git a/libsql-server/src/generated/namespace_fence.rs b/libsql-server/src/generated/namespace_fence.rs index dcbbcafe96..7012a0dbee 100644 --- a/libsql-server/src/generated/namespace_fence.rs +++ b/libsql-server/src/generated/namespace_fence.rs @@ -12,8 +12,9 @@ pub struct DrainPolicy { pub struct FrozenBoundary { #[prost(string, tag = "1")] pub log_id: ::prost::alloc::string::String, - #[prost(uint64, tag = "2")] - pub frame_no: u64, + /// Absent when the replication log has no frames. + #[prost(uint64, optional, tag = "2")] + pub frame_no: ::core::option::Option, } #[allow(clippy::derive_partial_eq_without_eq)] #[derive(Clone, PartialEq, ::prost::Message)] diff --git a/libsql-server/src/main.rs b/libsql-server/src/main.rs index 5d738a1ed5..a1ab51c4c4 100644 --- a/libsql-server/src/main.rs +++ b/libsql-server/src/main.rs @@ -268,6 +268,12 @@ struct Cli { #[clap(long, env = "SQLD_NAMESPACE_FENCE_RECEIPT_RETENTION_S")] namespace_fence_receipt_retention_s: Option, + /// How long, in milliseconds, acquiring a namespace write fence waits for active writers + /// when the request names no drain policy (the deadline then answers `DRAINING`). + /// Defaults to 30 seconds. + #[clap(long, env = "SQLD_NAMESPACE_FENCE_DEFAULT_WRITE_DRAIN_MS")] + namespace_fence_default_write_drain_ms: Option, + /// Shutdown timeout duration in seconds, defaults to 30 seconds. #[clap(long, env = "SQLD_SHUTDOWN_TIMEOUT")] shutdown_timeout: Option, @@ -664,6 +670,9 @@ fn make_meta_store_config(config: &Cli) -> anyhow::Result { namespace_fence_receipt_retention: config .namespace_fence_receipt_retention_s .map(Duration::from_secs), + namespace_fence_default_write_drain: config + .namespace_fence_default_write_drain_ms + .map(Duration::from_millis), }) } diff --git a/libsql-server/src/namespace/configurator/helpers.rs b/libsql-server/src/namespace/configurator/helpers.rs index 1f2524cade..a10ef89d6b 100644 --- a/libsql-server/src/namespace/configurator/helpers.rs +++ b/libsql-server/src/namespace/configurator/helpers.rs @@ -24,7 +24,7 @@ use crate::connection::{Connection as _, MakeConnection, MakeThrottledConnection use crate::database::{PrimaryConnection, PrimaryConnectionMaker}; use crate::error::LoadDumpError; use crate::namespace::broadcasters::BroadcasterHandle; -use crate::namespace::fence::controller::FenceController; +use crate::namespace::fence::controller::{FenceController, WriteDrainSource}; use crate::namespace::meta_store::MetaStoreHandle; use crate::namespace::replication_wal::{make_replication_wal_wrapper, ReplicationWalWrapper}; use crate::namespace::{ @@ -160,26 +160,33 @@ pub(super) async fn make_primary_connection_maker( let rcv = logger.new_frame_notifier.subscribe(); move || *rcv.borrow() }); + let legacy_maker = MakeLegacyConnection::new( + db_path.to_path_buf(), + wal_wrapper.clone(), + stats.clone(), + broadcaster, + meta_store_handle.clone(), + base_config.extensions.clone(), + base_config.max_response_size, + base_config.max_total_response_size, + auto_checkpoint, + get_current_frame_no.clone(), + encryption_config, + block_writes, + resolve_attach_path, + make_wal_manager.clone(), + fence.clone(), + ) + .await?; + // The positive write drain waits on this maker's write slot and reads the frozen boundary + // from its replication log (`docs/NAMESPACE_FENCE.md` section 8.3). + fence.register_write_drain(WriteDrainSource::new( + legacy_maker.connection_manager(), + logger.log_id(), + get_current_frame_no, + )); let connection_maker = Arc::new( - MakeLegacyConnection::new( - db_path.to_path_buf(), - wal_wrapper.clone(), - stats.clone(), - broadcaster, - meta_store_handle.clone(), - base_config.extensions.clone(), - base_config.max_response_size, - base_config.max_total_response_size, - auto_checkpoint, - get_current_frame_no, - encryption_config, - block_writes, - resolve_attach_path, - make_wal_manager.clone(), - fence, - ) - .await? - .throttled( + legacy_maker.throttled( base_config.max_concurrent_connections.clone(), base_config .connection_creation_timeout diff --git a/libsql-server/src/namespace/fence/controller.rs b/libsql-server/src/namespace/fence/controller.rs index 02b9a3db8c..cc6a274195 100644 --- a/libsql-server/src/namespace/fence/controller.rs +++ b/libsql-server/src/namespace/fence/controller.rs @@ -17,9 +17,11 @@ use parking_lot::Mutex; use tokio::sync::{watch, OwnedMutexGuard}; use uuid::Uuid; +use crate::connection::connection_manager::{ConnectionManager, WeakConnectionManager}; use crate::error::Error; use crate::namespace::meta_store::{FenceCommit, FenceContext, MetaStore}; use crate::namespace::NamespaceName; +use crate::replication::FrameNo; use super::command::FenceRequest; #[cfg(test)] @@ -46,6 +48,10 @@ pub struct GateSnapshot { /// A command whose commit outcome is unknown. While set, every class except maintenance /// and observability is denied, and every other command is refused. pub indeterminate: Option, + /// The in-memory `INSTALLING` gate of a closing transition that is being persisted + /// (section 8.3, step 2): write admission is closed on top of whatever `fence` allows. + /// Never persisted. + pub installing: Option, } impl GateSnapshot { @@ -54,6 +60,7 @@ impl GateSnapshot { fence, write_generation: 0, indeterminate: None, + installing: None, } } @@ -90,7 +97,30 @@ impl GateSnapshot { .with_detail(FenceDetail::IndeterminateCommit)); } } - self.fence.permits(class) + self.fence.permits(class)?; + if let Some((operation_id, command_id)) = self.installing { + if matches!( + class, + OperationClass::NormalWrite + | OperationClass::Vacuum + | OperationClass::CapabilityImport + | OperationClass::Lifecycle + ) { + return Err(FenceError::new( + FenceOutcome::MigrationWriteFenced, + format!( + "{class:?} is not permitted: fence command {command_id} of operation \ + {operation_id} is closing write admission" + ), + )); + } + } + Ok(()) + } + + /// Whether a closing transition is being installed. + pub fn is_installing(&self) -> bool { + self.installing.is_some() } /// Normal write admission. @@ -114,6 +144,39 @@ impl GateSnapshot { /// once the manager is gone. pub type WriteQueueWaker = Box bool + Send + Sync>; +/// The last frame committed to a namespace's replication log, `None` while it has none. +pub type GetCurrentFrameNo = Arc Option + Send + Sync + 'static>; + +/// What the write drain needs from one primary connection maker of the namespace: its +/// connection manager (held weakly, so an evicted namespace's manager goes away with it) and +/// its replication log. +pub struct WriteDrainSource { + pub(crate) manager: WeakConnectionManager, + pub(crate) log_id: Uuid, + pub(crate) current_frame_no: GetCurrentFrameNo, +} + +impl WriteDrainSource { + pub(crate) fn new( + manager: &ConnectionManager, + log_id: Uuid, + current_frame_no: GetCurrentFrameNo, + ) -> Self { + Self { + manager: manager.downgrade(), + log_id, + current_frame_no, + } + } +} + +/// A [`WriteDrainSource`] whose manager is alive, held for the length of a drain. +pub(crate) struct LiveWriteDrain { + pub(crate) manager: ConnectionManager, + pub(crate) log_id: Uuid, + pub(crate) current_frame_no: GetCurrentFrameNo, +} + /// The fence controller of one namespace. pub struct FenceController { namespace: NamespaceName, @@ -121,6 +184,9 @@ pub struct FenceController { gate: watch::Sender, /// The write queues of the namespace's connection managers (section 8.2). write_queues: Mutex>, + /// What the write drain needs from each of the namespace's primary connection makers + /// (section 8.3). + write_drains: Mutex>, #[cfg(test)] hooks: FenceTestHooks, } @@ -143,6 +209,7 @@ impl FenceController { transition_lock: Default::default(), gate, write_queues: Mutex::new(Vec::new()), + write_drains: Mutex::new(Vec::new()), #[cfg(test)] hooks: FenceTestHooks::default(), }) @@ -188,6 +255,32 @@ impl FenceController { self.write_queues.lock().push(waker); } + /// Register what the write drain needs from a primary connection maker of this namespace: + /// its connection manager and its replication log. Sources whose manager is gone are + /// dropped the next time the drain looks. + pub fn register_write_drain(&self, source: WriteDrainSource) { + self.write_drains.lock().push(source); + } + + /// The write-drain sources whose manager is still alive, oldest first. The drain holds them + /// (and so their managers) for as long as it runs. + pub(crate) fn live_write_drains(&self) -> Vec { + let mut sources = self.write_drains.lock(); + let mut live = Vec::with_capacity(sources.len()); + sources.retain(|source| match source.manager.upgrade() { + Some(manager) => { + live.push(LiveWriteDrain { + manager, + log_id: source.log_id, + current_frame_no: source.current_frame_no.clone(), + }); + true + } + None => false, + }); + live + } + /// Take the namespace's transition lock. Every fence command on the namespace runs while /// holding it, from its first check to its response. pub async fn begin_transition(self: &Arc) -> Transition { @@ -236,17 +329,25 @@ impl FenceController { } /// Publish a new gate. `fence: None` keeps the published fence. The write generation moves - /// whenever the state, the owning operation or the indeterminate flag changes. - fn publish(&self, fence: Option, indeterminate: Option) { + /// whenever the state, the owning operation, the indeterminate flag or the installing gate + /// changes. + fn publish( + &self, + fence: Option, + indeterminate: Option, + installing: Option, + ) { let mut generation_changed = false; self.gate.send_modify(|gate| { let fence = fence.unwrap_or_else(|| gate.fence.clone()); let changed = fence.state() != gate.fence.state() || fence.record().map(|r| r.operation_id) != gate.fence.record().map(|r| r.operation_id) - || indeterminate != gate.indeterminate; + || indeterminate != gate.indeterminate + || installing != gate.installing; gate.fence = fence; gate.indeterminate = indeterminate; + gate.installing = installing; if changed { gate.write_generation += 1; generation_changed = true; @@ -260,6 +361,7 @@ impl FenceController { revision = gate.revision(), write_generation = gate.write_generation, indeterminate = gate.indeterminate.is_some(), + installing = gate.installing.is_some(), "published namespace fence gate" ); } @@ -282,6 +384,28 @@ impl Transition { &self.controller } + /// Publish the in-memory `INSTALLING` gate for the closing command `key` (section 8.3, + /// step 2): write admission closes and the write generation moves, which wakes the write + /// queues. It is replaced by whatever the command's commit publishes, or removed with + /// [`remove_installing`](Self::remove_installing) when the command is proven not to have + /// committed. + pub fn install_closing_gate(&mut self, key: CommandKey) { + let indeterminate = self.controller.gate.borrow().indeterminate; + self.controller.publish(None, indeterminate, Some(key)); + } + + /// Remove the `INSTALLING` gate of a command that was proven not to have committed. The + /// write generation moves again, so nothing admitted before it closed can write. + pub fn remove_installing(&mut self) { + let (indeterminate, installing) = { + let gate = self.controller.gate.borrow(); + (gate.indeterminate, gate.installing) + }; + if installing.is_some() { + self.controller.publish(None, indeterminate, None); + } + } + /// Commit `request` in the metastore and publish the result. pub async fn apply( &mut self, @@ -353,7 +477,7 @@ impl Transition { match result { Ok(commit) => { let _ = controller.hook(HookPoint::BeforeGatePublish).await; - controller.publish(commit.record.clone().map(StoredFence::Record), None); + controller.publish(commit.record.clone().map(StoredFence::Record), None, None); let _ = controller.hook(HookPoint::BeforeResponse).await; Ok(commit) } @@ -365,7 +489,7 @@ impl Transition { "fence commit outcome unknown; the namespace stays closed until the command \ is replayed: {e}" ); - controller.publish(None, Some(key)); + controller.publish(None, Some(key), None); Err(e.into()) } } @@ -627,7 +751,7 @@ pub(crate) mod tests { DrainCompletion::SourceWrites { boundary: FrozenBoundary { log_id: LOG, - frame_no: 0, + frame_no: Some(0), }, }, ctx(), @@ -867,7 +991,7 @@ pub(crate) mod tests { // Simulate an indeterminate outcome for a command that never reached the metastore: // the replay applies it. - controller.publish(None, Some((OP, Uuid::from_u128(1)))); + controller.publish(None, Some((OP, Uuid::from_u128(1))), None); assert!(controller.permits(OperationClass::NormalWrite).is_err()); let r = controller .apply_command(&meta, acquire("ns", OP, 1), ctx()) @@ -1001,8 +1125,8 @@ pub(crate) mod tests { #[test] fn conn_state_starts_at_the_current_generation() { let controller = FenceController::unfenced("ns".into()); - controller.publish(None, Some((OP, OP))); - controller.publish(None, None); + controller.publish(None, Some((OP, OP)), None); + controller.publish(None, None, None); let state = FenceConnState::new(controller.clone(), OperationClass::NormalWrite); assert_eq!(state.program_generation(), 2); assert_eq!(state.txn_generation(), 2); diff --git a/libsql-server/src/namespace/fence/drain.rs b/libsql-server/src/namespace/fence/drain.rs new file mode 100644 index 0000000000..4f77a95d4a --- /dev/null +++ b/libsql-server/src/namespace/fence/drain.rs @@ -0,0 +1,771 @@ +//! The positive source write drain (`docs/NAMESPACE_FENCE.md` sections 8.3 and 8.4). +//! +//! `AcquireSourceWriteFence` closes write admission, persists `SOURCE_DRAINING`, and then waits +//! for the writer that was already holding the write slot when admission closed to commit or +//! roll back. It waits on the connection manager's release notification: neither elapsed time +//! nor the transaction timeout is ever taken as evidence that a writer has finished. Once no +//! connection holds the slot for a write, it reads the frozen boundary (`log_id`, last committed +//! frame) under the slot lock and persists `SOURCE_WRITE_FENCED` with it. Every step runs under +//! the namespace's transition lock, on a task of its own, so a caller that goes away does not +//! interrupt it. + +use std::sync::Arc; +use std::time::Duration; + +use tokio::time::Instant; + +use crate::namespace::meta_store::{FenceCommit, FenceContext, MetaStore}; + +use super::command::{DrainPolicy, FenceCommand, FenceRequest, OnDeadline}; +use super::controller::{FenceController, LiveWriteDrain, Transition}; +use super::hooks::HookPoint; +use super::outcome::FenceOutcome; +use super::record::FrozenBoundary; +use super::state::OperationClass; +use super::transition::DrainCompletion; + +/// How long `AcquireSourceWriteFence` waits for active writers when neither the request nor +/// `--namespace-fence-default-write-drain-ms` names a deadline. +pub const DEFAULT_WRITE_DRAIN: Duration = Duration::from_secs(30); + +/// After a forced rollback, how long the drain waits at least for the rolled-back writer to +/// release the slot before it answers `DRAINING` (the request's own deadline, if longer, is +/// used instead). A rollback normally releases at once; it is delayed only while a program is +/// still running on the connection. Reaching this bound is never taken as proof of anything: +/// the answer is `DRAINING` and admission stays closed. +pub const FORCED_ROLLBACK_GRACE: Duration = Duration::from_secs(10); + +impl FenceController { + /// Run one fence command to completion under the namespace's transition lock, including + /// the drain it starts, and return its result. + /// + /// The command runs on its own task: a caller that goes away (a lost response) interrupts + /// neither the commit nor the drain, and the result can be recovered by replaying the same + /// command or inspecting the fence. + pub async fn execute( + self: &Arc, + meta: &MetaStore, + request: FenceRequest, + ctx: FenceContext, + ) -> crate::Result { + let this = self.clone(); + let meta = meta.clone(); + tokio::spawn(async move { + let mut transition = this.begin_transition().await; + match request.command { + FenceCommand::AcquireSourceWriteFence { .. } => { + acquire_source_write_fence(&mut transition, &meta, request, ctx).await + } + _ => transition.apply(&meta, request, ctx).await, + } + }) + .await? + } +} + +/// `AcquireSourceWriteFence`, steps 1 to 7 of section 8.3, under `transition`. +/// +/// Returns the `APPLIED` commit of `SOURCE_WRITE_FENCED` once the drain is proven, the +/// `DRAINING` commit of `SOURCE_DRAINING` when the deadline passes first (write admission stays +/// closed and a replay of the same command resumes the drain), or the stored result of a replay. +pub async fn acquire_source_write_fence( + transition: &mut Transition, + meta: &MetaStore, + request: FenceRequest, + mut ctx: FenceContext, +) -> crate::Result { + let controller = transition.controller().clone(); + let key = (request.operation_id, request.command_id); + let policy = match &request.command { + FenceCommand::AcquireSourceWriteFence { drain_policy, .. } => { + drain_policy.unwrap_or_else(|| meta.fence_default_write_drain()) + } + _ => return transition.apply(meta, request, ctx).await, + }; + + // Step 1 is the metastore's (replay, owner, identity, shared schema, expectation). The + // identity it checks is the namespace's own replication log id. + if let Some(source) = controller.live_write_drains().last() { + ctx.namespace_log_id = Some(source.log_id); + } + + // Steps 2 and 3: close write admission in memory. Publishing moves the write generation and + // wakes the write queues, whose waiters then fail with MIGRATION_WRITE_FENCED. Where writes + // are already closed (a resumed drain, an indeterminate commit being reconciled, a fenced + // namespace) there is nothing to install. + if controller.permits(OperationClass::NormalWrite).is_ok() { + transition.install_closing_gate(key); + let _ = controller.hook(HookPoint::AfterInstallingGate).await; + } + + // Step 4: persist SOURCE_DRAINING. The commit publishes it in place of the INSTALLING gate. A + // command proven not to have committed removes the INSTALLING gate again; one whose commit + // is unknown has already closed the gate as indeterminate. + let commit = match transition.apply(meta, request, ctx.clone()).await { + Ok(commit) => commit, + Err(e) => { + transition.remove_installing(); + return Err(e); + } + }; + if commit.receipt.outcome != FenceOutcome::Draining { + // A replay of a finished acquisition, or ALREADY_APPLIED. + return Ok(commit); + } + let drain_key = (commit.receipt.operation_id, commit.receipt.command_id); + + // Steps 5 and 6. + let boundary = match drain_writers(&controller, policy).await { + Some(boundary) => boundary, + None => return Ok(commit), + }; + + // Step 7. + ctx.now_ms = now_ms(); + transition + .complete_drain( + meta, + drain_key, + DrainCompletion::SourceWrites { boundary }, + ctx, + ) + .await +} + +/// Wait until no connection manager of the namespace has a writer holding its write slot, and +/// read the frozen boundary. `None` when the drain could not be proven within the policy: the +/// deadline passed (after a forced rollback, the same deadline again, or at least +/// [`FORCED_ROLLBACK_GRACE`]), or the namespace has no +/// loaded primary whose replication log could be read. +async fn drain_writers( + controller: &FenceController, + policy: DrainPolicy, +) -> Option { + let namespace = controller.namespace().clone(); + let deadline_after = Duration::from_millis(policy.deadline_ms); + let mut deadline = Instant::now() + deadline_after; + let mut forced = false; + loop { + // Every manager registered from now on belongs to a maker opened after write admission + // closed, so it has no pre-cutoff writer; sources are re-read on each round anyway. + let sources = controller.live_write_drains(); + if sources.is_empty() { + tracing::warn!( + %namespace, + "the namespace has no loaded primary; the write drain cannot read its replication \ + log and stays DRAINING until the command is replayed" + ); + return None; + } + + if !wait_for_writers(&sources, deadline).await { + match policy.on_deadline { + OnDeadline::ForceRollback if !forced => { + forced = true; + for source in &sources { + let manager = source.manager.clone(); + // The rollback takes the connection's lock, which a running program + // holds; it releases the slot when it happens, and that release is what + // the drain keeps waiting for. + tokio::task::spawn_blocking(move || { + if let Some(id) = manager.abort_active() { + tracing::info!( + connection = id, + "write drain deadline passed; rolling back the active writer" + ); + } + }); + } + deadline = Instant::now() + deadline_after.max(FORCED_ROLLBACK_GRACE); + continue; + } + _ => { + tracing::info!( + %namespace, + deadline_ms = policy.deadline_ms, + on_deadline = policy.on_deadline.as_str(), + forced, + "write drain deadline passed with a writer still active; answering DRAINING" + ); + return None; + } + } + } + + let _ = controller.hook(HookPoint::BeforeBoundaryCapture).await; + match capture_boundary(&sources) { + Ok(boundary) => { + tracing::info!( + %namespace, + log_id = %boundary.log_id, + frame_no = ?boundary.frame_no, + "write drain proven" + ); + return Some(boundary); + } + // Cannot happen while write admission is closed; wait again rather than guess. + Err((id, class)) => { + tracing::warn!( + %namespace, + connection = id, + ?class, + "a writer holds the write slot at boundary capture; waiting again" + ); + } + } + } +} + +/// Wait, on each manager's release notification, until none of them has a connection holding +/// the write slot for a write. `false` when `deadline` passes first. +/// +/// With write admission closed a manager that has been seen without a writer stays without one +/// (only checkpoints can take the slot), so the managers are waited for one after the other. +async fn wait_for_writers(sources: &[LiveWriteDrain], deadline: Instant) -> bool { + for source in sources { + loop { + let released = source.manager.released().notified(); + tokio::pin!(released); + // Registered before the check, so a release in between is not missed. + released.as_mut().enable(); + if !source.manager.has_writer() { + break; + } + tokio::select! { + _ = &mut released => {} + _ = tokio::time::sleep_until(deadline) => return false, + } + } + } + true +} + +/// Step 6: under each manager's write-slot lock, observe that no connection holds the slot for a +/// write, and read the last committed frame of the replication log. The replication logger +/// commits a transaction's frames and publishes its frame number before the transaction +/// releases the slot, so what is read here is final. +fn capture_boundary( + sources: &[LiveWriteDrain], +) -> Result< + FrozenBoundary, + ( + crate::connection::connection_manager::ConnId, + OperationClass, + ), +> { + let mut frame_no = None; + for source in sources { + let frame = source + .manager + .with_no_writer(|| (source.current_frame_no)())?; + frame_no = frame_no.max(frame); + } + let log_id = sources + .last() + .expect("capture_boundary is called with at least one source") + .log_id; + Ok(FrozenBoundary { log_id, frame_no }) +} + +fn now_ms() -> i64 { + std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .map_or(0, |d| i64::try_from(d.as_millis()).unwrap_or(i64::MAX)) +} + +#[cfg(test)] +mod tests { + use std::sync::Arc; + use std::time::Duration; + + use rusqlite::ErrorCode; + use tempfile::{tempdir, TempDir}; + use uuid::Uuid; + + use super::*; + use crate::connection::config::DatabaseConfig; + use crate::connection::Connection as _; + use crate::database::{Connection, Database}; + use crate::error::Error; + use crate::namespace::fence::hooks::HookPoint; + use crate::namespace::fence::record::ServerIdentity; + use crate::namespace::fence::state::FenceState; + use crate::namespace::meta_store::FenceCommitKind; + use crate::namespace::store::fence_tests::open_store; + use crate::namespace::store::NamespaceStore; + use crate::namespace::RestoreOption; + use crate::replication::primary::logger::ReplicationLogger; + + const OP: Uuid = Uuid::from_u128(0xa); + const OTHER_OP: Uuid = Uuid::from_u128(0xb); + /// Long enough that no test ever reaches it: a drain must never finish because of time. + const LONG: DrainPolicy = DrainPolicy { + deadline_ms: 600_000, + on_deadline: OnDeadline::Fail, + }; + const PROMPT: Duration = Duration::from_secs(30); + + /// A primary namespace `ns` with a table `t`, served by a real `NamespaceStore`, so that the + /// drain goes through the connection manager and replication logger the configurator + /// registered. + struct Source { + _dir: TempDir, + store: NamespaceStore, + fence: Arc, + logger: Arc, + } + + impl Source { + async fn new() -> Self { + let dir = tempdir().unwrap(); + let store = open_store(dir.path()).await; + store + .create( + "ns".into(), + RestoreOption::Latest, + DatabaseConfig { + // Held writers must never have the slot stolen by the timeout. + txn_timeout: Some(Duration::from_secs(600)), + ..Default::default() + }, + ) + .await + .unwrap(); + let (fence, logger) = store + .with("ns".into(), |ns| { + let logger = match &ns.db { + Database::Primary(p) => p.wal_wrapper.wrapper().logger(), + _ => unreachable!(), + }; + (ns.fence().clone(), logger) + }) + .await + .unwrap(); + let this = Self { + _dir: dir, + store, + fence, + logger, + }; + let conn = this.conn().await; + raw(&conn, "create table t (x)").await.unwrap(); + this + } + + async fn conn(&self) -> Arc { + let maker = self + .store + .with("ns".into(), |ns| ns.db.connection_maker()) + .await + .unwrap(); + Arc::new(maker.create().await.unwrap()) + } + + fn log_id(&self) -> Uuid { + self.logger.log_id() + } + + /// The last frame the replication log has committed. + fn frame_no(&self) -> Option { + *self.logger.new_frame_notifier.borrow() + } + + fn acquire(&self, op: Uuid, command_id: u128, policy: DrainPolicy) -> FenceRequest { + FenceRequest { + namespace: "ns".into(), + operation_id: op, + command_id: Uuid::from_u128(command_id), + expected_state: FenceState::Unfenced, + expected_revision: 0, + command: FenceCommand::AcquireSourceWriteFence { + expected_log_id: self.log_id(), + drain_policy: Some(policy), + }, + } + } + + /// Run `request` through the store, on a task of its own. + fn execute( + &self, + request: FenceRequest, + ) -> tokio::task::JoinHandle> { + let store = self.store.clone(); + tokio::spawn(async move { + store + .execute_fence_command( + request, + ServerIdentity { + build: "test".into(), + instance_id: Uuid::from_u128(0x99), + }, + ) + .await + }) + } + + /// Wait until the published gate is in `state`. + async fn until_state(&self, state: FenceState) { + let mut rx = self.fence.subscribe(); + tokio::time::timeout(PROMPT, rx.wait_for(|g| g.state() == state)) + .await + .expect("the gate never reached the state") + .unwrap(); + } + + async fn count(&self) -> i64 { + let conn = self.conn().await; + tokio::task::spawn_blocking(move || { + conn.with_raw(|c| c.query_row("select count(*) from t", (), |r| r.get(0))) + }) + .await + .unwrap() + .unwrap() + } + } + + /// Run `sql` as one raw program on `conn`, off the async runtime (it can block on the write + /// slot). + async fn raw(conn: &Arc, sql: &'static str) -> rusqlite::Result<()> { + let conn = conn.clone(); + tokio::task::spawn_blocking(move || conn.with_raw(|c| c.execute_batch(sql))) + .await + .unwrap() + } + + fn assert_fenced(result: rusqlite::Result<()>) { + match result { + Err(rusqlite::Error::SqliteFailure(e, _)) => { + assert_eq!(e.code, ErrorCode::AuthorizationForStatementDenied, "{e}") + } + other => panic!("expected the WAL gate to refuse the write, got {other:?}"), + } + } + + fn fence_outcome(result: &crate::Result) -> FenceOutcome { + match result { + Ok(c) => c.receipt.outcome, + Err(Error::NamespaceFence(e)) => e.outcome(), + Err(e) => panic!("unexpected error: {e}"), + } + } + + fn boundary(commit: &FenceCommit) -> FrozenBoundary { + commit + .record + .as_ref() + .unwrap() + .frozen_boundary + .expect("a write-fenced source has a frozen boundary") + } + + /// Two operations acquire the same namespace at once: exactly one owns it, and the other + /// gets the typed ownership conflict. The first is parked after closing write admission so + /// that the second is certainly waiting on the transition lock. + #[tokio::test(flavor = "multi_thread")] + async fn acquire_race_single_owner() { + let s = Source::new().await; + let paused = s.fence.hooks().pause_at(HookPoint::AfterInstallingGate); + let first = s.execute(s.acquire(OP, 1, LONG)); + paused.reached().await; + let second = s.execute(s.acquire(OTHER_OP, 2, LONG)); + paused.resume(); + + let first = first.await.unwrap(); + let second = second.await.unwrap(); + let outcomes = [fence_outcome(&first), fence_outcome(&second)]; + assert_eq!( + outcomes, + [ + FenceOutcome::Applied, + FenceOutcome::FenceOwnedByAnotherOperation + ] + ); + let gate = s.fence.gate(); + assert_eq!(gate.state(), FenceState::SourceWriteFenced); + assert_eq!(gate.operation_id(), Some(OP)); + assert!(!gate.is_installing()); + } + + /// A writer that holds the write slot when the fence arrives commits, and only then is the + /// freeze acknowledged, with a boundary that includes its commit. Writes attempted while + /// the drain waits are refused. + #[tokio::test(flavor = "multi_thread")] + async fn active_writer_commits_before_ack() { + let s = Source::new().await; + let holder = s.conn().await; + raw(&holder, "begin immediate; insert into t values (1);") + .await + .unwrap(); + + let acquire = s.execute(s.acquire(OP, 1, LONG)); + s.until_state(FenceState::SourceDraining).await; + // Admission is closed while the pre-cutoff writer is still active. + assert_fenced(raw(&s.conn().await, "insert into t values (2)").await); + assert!(!acquire.is_finished()); + + raw(&holder, "commit").await.unwrap(); + let committed_frame = s.frame_no(); + assert!(committed_frame.is_some()); + + let commit = acquire.await.unwrap().unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + assert_eq!(commit.receipt.state_after, FenceState::SourceWriteFenced); + assert_eq!( + boundary(&commit), + FrozenBoundary { + log_id: s.log_id(), + frame_no: committed_frame, + } + ); + assert_eq!(s.count().await, 1); + } + + /// Under `force_rollback`, a writer still active at the deadline is rolled back, and the + /// freeze is acknowledged only after its slot was actually released; its write is not in + /// the database or the boundary. + #[tokio::test(flavor = "multi_thread")] + async fn forced_rollback_before_ack() { + let s = Source::new().await; + let before = s.frame_no(); + let holder = s.conn().await; + raw(&holder, "begin immediate; insert into t values (1);") + .await + .unwrap(); + + let commit = s + .execute(s.acquire( + OP, + 1, + DrainPolicy { + deadline_ms: 0, + on_deadline: OnDeadline::ForceRollback, + }, + )) + .await + .unwrap() + .unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + assert_eq!( + boundary(&commit), + FrozenBoundary { + log_id: s.log_id(), + frame_no: before, + } + ); + assert_eq!(s.frame_no(), before); + assert_eq!(s.count().await, 0); + // The rolled-back transaction is gone; its connection cannot write either. + assert!(raw(&holder, "commit").await.is_err()); + assert_fenced(raw(&holder, "insert into t values (3)").await); + assert_eq!(s.frame_no(), before); + } + + /// After the acknowledgement nothing commits: the boundary is the last committed frame, and + /// autocommit writes, explicit transactions, DDL and a read transaction opened before the + /// fence that tries to upgrade are all refused without adding a frame. + #[tokio::test(flavor = "multi_thread")] + async fn no_commit_after_ack() { + let s = Source::new().await; + let writer = s.conn().await; + for _ in 0..3 { + raw(&writer, "insert into t values (1)").await.unwrap(); + } + let last = s.frame_no(); + // A reader whose transaction predates the fence. + let reader = s.conn().await; + raw(&reader, "begin; select count(*) from t;") + .await + .unwrap(); + + let commit = s.execute(s.acquire(OP, 1, LONG)).await.unwrap().unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + assert_eq!( + boundary(&commit), + FrozenBoundary { + log_id: s.log_id(), + frame_no: last, + } + ); + + assert_fenced(raw(&writer, "insert into t values (2)").await); + assert_fenced(raw(&writer, "begin immediate").await); + assert_fenced(raw(&writer, "create table u (y)").await); + assert_fenced(raw(&reader, "insert into t values (2)").await); + assert_fenced(raw(&s.conn().await, "insert into t values (2)").await); + assert_eq!(s.frame_no(), last); + assert_eq!(s.count().await, 3); + assert_eq!(boundary(&commit).frame_no, s.frame_no()); + } + + /// With `on_deadline: fail`, a writer still active at the deadline makes the command answer + /// `DRAINING`: the durable state is `SOURCE_DRAINING` and write admission stays closed. + #[tokio::test(flavor = "multi_thread")] + async fn deadline_returns_draining_and_stays_closed() { + let s = Source::new().await; + let holder = s.conn().await; + raw(&holder, "begin immediate; insert into t values (1);") + .await + .unwrap(); + + let policy = DrainPolicy { + deadline_ms: 0, + on_deadline: OnDeadline::Fail, + }; + let commit = s.execute(s.acquire(OP, 1, policy)).await.unwrap().unwrap(); + assert_eq!(commit.kind, FenceCommitKind::Committed); + assert_eq!(commit.receipt.outcome, FenceOutcome::Draining); + assert_eq!(s.fence.gate().state(), FenceState::SourceDraining); + let inspected = s + .store + .meta_store() + .inspect_fence("ns".into()) + .await + .unwrap(); + assert_eq!(inspected.fence.state(), FenceState::SourceDraining); + assert_fenced(raw(&s.conn().await, "insert into t values (2)").await); + // Nothing reopens by itself; the writer that was active before the fence is still the + // only one that can commit. + raw(&holder, "commit").await.unwrap(); + assert_fenced(raw(&holder, "insert into t values (3)").await); + assert_eq!(s.fence.gate().state(), FenceState::SourceDraining); + assert_eq!(s.count().await, 1); + } + + /// Replaying a command whose receipt is `DRAINING` resumes the same drain: it completes once + /// the writer has finished, and a further replay returns that stored result. + #[tokio::test(flavor = "multi_thread")] + async fn replay_of_draining_resumes_and_completes() { + let s = Source::new().await; + let holder = s.conn().await; + raw(&holder, "begin immediate; insert into t values (1);") + .await + .unwrap(); + let policy = DrainPolicy { + deadline_ms: 0, + on_deadline: OnDeadline::Fail, + }; + let first = s.execute(s.acquire(OP, 1, policy)).await.unwrap().unwrap(); + assert_eq!(first.receipt.outcome, FenceOutcome::Draining); + let draining_revision = s.fence.gate().revision(); + + // Replayed while the writer is still active: the same drain resumes, writes nothing and, + // with the same zero deadline, answers DRAINING again. + let replay = s.execute(s.acquire(OP, 1, policy)).await.unwrap().unwrap(); + assert_eq!(replay.kind, FenceCommitKind::Resumed); + assert_eq!(replay.receipt.outcome, FenceOutcome::Draining); + assert_eq!(s.fence.gate().revision(), draining_revision); + + raw(&holder, "commit").await.unwrap(); + let committed = s.frame_no(); + let done = s.execute(s.acquire(OP, 1, policy)).await.unwrap().unwrap(); + assert_eq!(done.receipt.outcome, FenceOutcome::Applied); + assert_eq!(done.receipt.command_id, Uuid::from_u128(1)); + assert_eq!( + boundary(&done), + FrozenBoundary { + log_id: s.log_id(), + frame_no: committed, + } + ); + assert_eq!(s.fence.gate().state(), FenceState::SourceWriteFenced); + assert_eq!(s.fence.gate().revision(), draining_revision + 1); + + let again = s.execute(s.acquire(OP, 1, policy)).await.unwrap().unwrap(); + assert_eq!(again.kind, FenceCommitKind::Replayed); + assert_eq!(again.receipt, done.receipt); + assert_eq!(boundary(&again), boundary(&done)); + } + + /// Releasing the write fence commits, publishes a new write generation and only then + /// answers: new programs write again, and a transaction that began under the fence cannot. + #[tokio::test(flavor = "multi_thread")] + async fn release_reopens_with_new_generation() { + let s = Source::new().await; + s.execute(s.acquire(OP, 1, LONG)).await.unwrap().unwrap(); + let fenced = s.fence.gate(); + let reader = s.conn().await; + raw(&reader, "begin; select count(*) from t;") + .await + .unwrap(); + + let release = FenceRequest { + namespace: "ns".into(), + operation_id: OP, + command_id: Uuid::from_u128(2), + expected_state: FenceState::SourceWriteFenced, + expected_revision: fenced.revision(), + command: FenceCommand::ReleaseSourceWriteFence, + }; + let paused = s.fence.hooks().pause_at(HookPoint::BeforeResponse); + let released = s.execute(release); + paused.reached().await; + // Published before the response is sent. + let gate = s.fence.gate(); + assert_eq!(gate.state(), FenceState::Released); + assert!(gate.write_generation > fenced.write_generation); + assert!(!released.is_finished()); + paused.resume(); + let commit = released.await.unwrap().unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + + raw(&s.conn().await, "insert into t values (1)") + .await + .unwrap(); + assert_fenced(raw(&reader, "insert into t values (2)").await); + raw(&reader, "rollback").await.unwrap(); + raw(&reader, "insert into t values (3)").await.unwrap(); + assert_eq!(s.count().await, 2); + } + + /// An acquisition refused by the metastore (here: the caller observed a different + /// replication log) removes the INSTALLING gate it had published; writes are admitted again + /// under a new generation. + #[tokio::test(flavor = "multi_thread")] + async fn refused_acquire_reopens_writes() { + let s = Source::new().await; + let before = s.fence.gate().write_generation; + let mut request = s.acquire(OP, 1, LONG); + request.command = FenceCommand::AcquireSourceWriteFence { + expected_log_id: Uuid::from_u128(0x77), + drain_policy: Some(LONG), + }; + let result = s.execute(request).await.unwrap(); + assert_eq!( + fence_outcome(&result), + FenceOutcome::FencePreconditionFailed + ); + let gate = s.fence.gate(); + assert_eq!(gate.state(), FenceState::Unfenced); + assert!(!gate.is_installing()); + assert_eq!(gate.write_generation, before + 2); + raw(&s.conn().await, "insert into t values (1)") + .await + .unwrap(); + } + + /// While the INSTALLING gate is up, before anything is persisted, writes are already + /// refused and reads are served. + #[tokio::test(flavor = "multi_thread")] + async fn installing_gate_closes_writes_before_persisting() { + let s = Source::new().await; + let paused = s.fence.hooks().pause_at(HookPoint::AfterInstallingGate); + let acquire = s.execute(s.acquire(OP, 1, LONG)); + paused.reached().await; + let gate = s.fence.gate(); + assert!(gate.is_installing()); + assert_eq!(gate.state(), FenceState::Unfenced); + let inspected = s + .store + .meta_store() + .inspect_fence("ns".into()) + .await + .unwrap(); + assert_eq!(inspected.fence.state(), FenceState::Unfenced); + assert_fenced(raw(&s.conn().await, "insert into t values (1)").await); + assert_eq!(s.count().await, 0); + paused.resume(); + let commit = acquire.await.unwrap().unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + assert!(!s.fence.gate().is_installing()); + } +} diff --git a/libsql-server/src/namespace/fence/mod.rs b/libsql-server/src/namespace/fence/mod.rs index d5ac933e2e..6b445a9ba6 100644 --- a/libsql-server/src/namespace/fence/mod.rs +++ b/libsql-server/src/namespace/fence/mod.rs @@ -8,8 +8,9 @@ //! markers with their strict durable encoding ([`record`]), the pure transition function //! ([`transition`]), the metastore tables, compare-and-swap and marker file that persist them //! ([`store`], driven by `MetaStore::apply_fence_command`), and the in-memory authority built -//! on them: the per-namespace [`controller`] with its gate, the [`registry`] that holds the -//! controllers outside the namespace cache, and the test [`hooks`] on their paths. +//! on them: the per-namespace [`controller`] with its gate, the positive write [`drain`], the +//! [`registry`] that holds the controllers outside the namespace cache, and the test [`hooks`] +//! on their paths. // The persistence, controller and protocol layers that consume these types land in the // following commits of this series; until then most of the module is unused by the rest of @@ -18,6 +19,7 @@ pub mod command; pub mod controller; +pub mod drain; pub mod hooks; pub mod outcome; pub mod record; diff --git a/libsql-server/src/namespace/fence/record.rs b/libsql-server/src/namespace/fence/record.rs index a2d9f7f5dd..247b481ad2 100644 --- a/libsql-server/src/namespace/fence/record.rs +++ b/libsql-server/src/namespace/fence/record.rs @@ -32,7 +32,8 @@ pub struct NamespaceIdentity { #[derive(Debug, Clone, Copy, PartialEq, Eq)] pub struct FrozenBoundary { pub log_id: Uuid, - pub frame_no: u64, + /// The last frame committed to the replication log, or `None` when the log has no frames. + pub frame_no: Option, } /// The pre-fence values of the legacy `block_*` configuration fields, restored when the @@ -700,7 +701,7 @@ pub(super) mod tests { drain_started_at_ms: Some(100), frozen_boundary: Some(FrozenBoundary { log_id: Uuid::from_u128(10), - frame_no: 1234, + frame_no: Some(1234), }), validation: None, legacy_blocks: LegacyBlocks { diff --git a/libsql-server/src/namespace/fence/transition.rs b/libsql-server/src/namespace/fence/transition.rs index f1de43f747..d589ce8859 100644 --- a/libsql-server/src/namespace/fence/transition.rs +++ b/libsql-server/src/namespace/fence/transition.rs @@ -903,7 +903,7 @@ mod tests { let boundary = DrainCompletion::SourceWrites { boundary: FrozenBoundary { log_id: LOG, - frame_no: 42, + frame_no: Some(42), }, }; let mut h = if state.role() == Some(Role::Target) || state == S::Absent { @@ -1083,14 +1083,14 @@ mod tests { h.complete(DrainCompletion::SourceWrites { boundary: FrozenBoundary { log_id: LOG, - frame_no: 7, + frame_no: Some(7), }, }); assert_eq!(h.state(), S::SourceWriteFenced); assert_eq!(h.revision(), 2); assert_eq!( h.record.as_ref().unwrap().frozen_boundary.unwrap().frame_no, - 7 + Some(7) ); let acquire_receipt = h .receipts @@ -1255,7 +1255,7 @@ mod tests { h.complete(DrainCompletion::SourceWrites { boundary: FrozenBoundary { log_id: LOG, - frame_no: 1, + frame_no: Some(1), }, }); h.run(OP, CommandKind::SetSourceReadFence).unwrap(); @@ -1464,7 +1464,7 @@ mod tests { DrainCompletion::SourceWrites { boundary: FrozenBoundary { log_id: LOG, - frame_no: 3, + frame_no: Some(3), }, }, &env(), @@ -1492,7 +1492,7 @@ mod tests { DrainCompletion::SourceWrites { boundary: FrozenBoundary { log_id: Uuid::from_u128(0x77), - frame_no: 1, + frame_no: Some(1), }, }, &env(), @@ -1506,7 +1506,7 @@ mod tests { let boundary = DrainCompletion::SourceWrites { boundary: FrozenBoundary { log_id: LOG, - frame_no: 1, + frame_no: Some(1), }, }; assert!(complete_drain(&record, &final_receipt, boundary, &env()).is_err()); @@ -1732,7 +1732,7 @@ mod tests { h.complete(DrainCompletion::SourceWrites { boundary: FrozenBoundary { log_id: LOG, - frame_no: 5, + frame_no: Some(5), }, }); assert_eq!(h.state(), S::SourceWriteFenced); diff --git a/libsql-server/src/namespace/meta_store.rs b/libsql-server/src/namespace/meta_store.rs index 6ed7b29fd5..1a16994c3f 100644 --- a/libsql-server/src/namespace/meta_store.rs +++ b/libsql-server/src/namespace/meta_store.rs @@ -31,7 +31,7 @@ use crate::{ config::MetaStoreConfig, connection::legacy::open_conn_active_checkpoint, error::Error, Result, }; -use super::fence::command::{FenceCommand, FenceRequest}; +use super::fence::command::{DrainPolicy, FenceCommand, FenceRequest, OnDeadline}; use super::fence::outcome::{FenceDetail, FenceError, FenceOutcome}; use super::fence::record::{ CommandReceipt, NamespaceFenceRecord, ServerIdentity, ValidationSnapshot, @@ -110,6 +110,8 @@ struct FenceSettings { /// namespace directory holds a marker. fail_closed: bool, receipt_retention: Duration, + /// The write drain deadline of an `AcquireSourceWriteFence` that names no drain policy. + default_write_drain: Duration, } fn setup_connection(conn: &rusqlite::Connection) -> Result<()> { @@ -235,6 +237,9 @@ impl MetaStoreInner { receipt_retention: config .namespace_fence_receipt_retention .unwrap_or(fence_store::DEFAULT_RECEIPT_RETENTION), + default_write_drain: config + .namespace_fence_default_write_drain + .unwrap_or(crate::namespace::fence::drain::DEFAULT_WRITE_DRAIN), }; let mut this = MetaStoreInner { @@ -1300,6 +1305,16 @@ impl MetaStore { self.inner.fence.enabled } + /// The drain policy of an `AcquireSourceWriteFence` that names none: the configured + /// deadline, then `DRAINING`. + pub fn fence_default_write_drain(&self) -> DrainPolicy { + DrainPolicy { + deadline_ms: u64::try_from(self.inner.fence.default_write_drain.as_millis()) + .unwrap_or(u64::MAX), + on_deadline: OnDeadline::Fail, + } + } + /// Whether this metastore holds fence state, so fences are loaded and enforced. pub fn fence_enforced(&self) -> bool { self.inner.fence.tables @@ -1749,7 +1764,7 @@ mod fence_tests { let boundary = FrozenBoundary { log_id: LOG, - frame_no: 42, + frame_no: Some(42), }; let commit = store .complete_fence_drain( @@ -2124,7 +2139,7 @@ mod fence_tests { DrainCompletion::SourceWrites { boundary: FrozenBoundary { log_id: LOG, - frame_no: 1, + frame_no: Some(1), }, }, ctx(2_000), @@ -2175,7 +2190,7 @@ mod fence_tests { DrainCompletion::SourceWrites { boundary: FrozenBoundary { log_id: LOG, - frame_no: 7, + frame_no: Some(7), }, }, ctx(2_000), diff --git a/libsql-server/src/namespace/store.rs b/libsql-server/src/namespace/store.rs index 3a132cf10f..af406dfd96 100644 --- a/libsql-server/src/namespace/store.rs +++ b/libsql-server/src/namespace/store.rs @@ -21,8 +21,10 @@ use crate::stats::Stats; use super::broadcasters::{BroadcasterHandle, BroadcasterRegistry}; use super::configurator::{DynConfigurator, NamespaceConfigurators}; +use super::fence::command::{FenceCommand, FenceRequest}; +use super::fence::record::ServerIdentity; use super::fence::registry::FenceRegistry; -use super::meta_store::{MetaStore, MetaStoreHandle}; +use super::meta_store::{FenceCommit, FenceContext, MetaStore, MetaStoreHandle}; use super::schema_lock::SchemaLocksRegistry; use super::{Namespace, ResetCb, ResetOp, ResolveNamespacePathFn, RestoreOption}; @@ -531,6 +533,33 @@ impl NamespaceStore { &self.inner.metadata } + /// Run one fence command on its namespace, including the drain it starts + /// (`docs/NAMESPACE_FENCE.md` sections 5.3 and 8). `AcquireSourceWriteFence` loads the + /// namespace first, so that its connection manager and replication log are registered with + /// the namespace's controller before the drain needs them. + // The admin routes that call this are not part of the server yet. + #[cfg_attr(not(test), allow(dead_code))] + pub(crate) async fn execute_fence_command( + &self, + request: FenceRequest, + server: ServerIdentity, + ) -> crate::Result { + let controller = match request.command { + FenceCommand::AcquireSourceWriteFence { .. } => { + self.with(request.namespace.clone(), |ns| ns.fence().clone()) + .await? + } + _ => self.inner.fences.controller(&request.namespace), + }; + controller + .execute( + &self.inner.metadata, + request, + FenceContext::now(server, None), + ) + .await + } + pub(crate) fn schema_locks(&self) -> &SchemaLocksRegistry { &self.inner.schema_locks } @@ -559,7 +588,7 @@ impl NamespaceStore { } #[cfg(test)] -mod fence_tests { +pub(crate) mod fence_tests { use std::path::Path; use libsql_sys::wal::Sqlite3WalManager; @@ -580,7 +609,7 @@ mod fence_tests { const LOG: Uuid = Uuid::from_u128(0x10); const OP: Uuid = Uuid::from_u128(0xa); - async fn open_store(dir: &Path) -> NamespaceStore { + pub(crate) async fn open_store(dir: &Path) -> NamespaceStore { let (maker, manager) = metastore_connection_maker(None, dir).await.unwrap(); let meta = MetaStore::new( MetaStoreConfig { From a1c4f6f8242f71814283db73cca9dab8ba8a82cb Mon Sep 17 00:00:00 2001 From: River Date: Tue, 29 Sep 2026 16:53:59 +0000 Subject: [PATCH 11/33] libsql-server: reconcile indeterminate fence commits and restart at each boundary Add crash-restart tests for the namespace fence: each server lifetime runs on its own runtime and is ended without any shutdown code while a fence command is parked at a hook point, so the next start takes the real dirty-recovery path on the same directory. They cover every persistence boundary of AcquireSourceWriteFence and ReleaseSourceWriteFence (including a marker that lags the metastore commit), a restart in SOURCE_DRAINING with a writer active at the crash, indeterminate commits through the drain path (applied and not applied), and lost acquisition responses resolved by replay and inspection. The tests exposed that a source restarted while draining could never finish its drain: dirty recovery rebuilds the replication log under a new log id, and completing the drain refused a boundary on a log other than the one the fence was acquired against, leaving the namespace in SOURCE_DRAINING for good. The frozen boundary now names the log that is live when the drain is proven, the record's identity keeps the acquisition log id, and the server warns when the two differ. Write admission was durably closed throughout, so the data at the boundary is unchanged. The BeforeMetastoreCommit test hook can now report a commit as indeterminate without running it. The contract document describes the restart and log rebuild semantics. Co-authored-by: Tomasz Szymczyszyn --- docs/NAMESPACE_FENCE.md | 9 +- .../src/namespace/fence/controller.rs | 14 +- libsql-server/src/namespace/fence/drain.rs | 10 + libsql-server/src/namespace/fence/hooks.rs | 4 +- libsql-server/src/namespace/fence/mod.rs | 3 + libsql-server/src/namespace/fence/tests.rs | 898 ++++++++++++++++++ .../src/namespace/fence/transition.rs | 34 +- 7 files changed, 950 insertions(+), 22 deletions(-) create mode 100644 libsql-server/src/namespace/fence/tests.rs diff --git a/docs/NAMESPACE_FENCE.md b/docs/NAMESPACE_FENCE.md index 17e0920170..aac7954a35 100644 --- a/docs/NAMESPACE_FENCE.md +++ b/docs/NAMESPACE_FENCE.md @@ -491,6 +491,11 @@ This makes the WAL gate independent of statement classification: DDL, misclassif - `NamespaceStore::with` and `make_namespace` check the registry before `lookup()`, `handle()` or any setup: `UNKNOWN_UNAVAILABLE` is refused before any setup work. - Idle or capacity eviction shuts the namespace down but leaves the controller in the registry; a lazy reload reinstalls the identical gate, revision and generation. A drain waiter that holds the evicted manager sees its connections close and is notified. - Namespaces in `SOURCE_DRAINING` or `TARGET_IMPORT_DRAINING` after a restart stay closed until the same command is replayed. Nothing advances in the background. +- A restart at any persistence boundary recovers either the state before the command or the state it committed, never anything in between, and never an open namespace unless an opening transition had committed: the `INSTALLING` gate and an indeterminate flag are in memory only, so a crash before the commit recovers the prior state (which was never acknowledged as closed), and a crash after it recovers the committed one. A marker that fell behind (a crash between the metastore commit and the marker write) is repaired when the fence is loaded. +- **A crash rebuilds the source's replication log.** A namespace that was not shut down cleanly is recovered by rebuilding its replication log from the database file under a new `log_id` (existing behaviour, not specific to fences). For the fence this means: + - Crash before `SOURCE_DRAINING` committed: nothing was written or acknowledged. A replay of the same acquisition is refused with `FENCE_PRECONDITION_FAILED`/`namespace_identity_mismatch`, before anything is written, because the log id the caller observed is gone; the caller reads the new identity and acquires with a new command. + - Crash in `SOURCE_DRAINING`: the replay completes the drain at once, and the frozen boundary names the rebuilt log (its `log_id` and last frame). The record's `identity.log_id` keeps the log the caller acquired against; the two differ exactly when the log was rebuilt during the drain. Write admission was durably closed from the `SOURCE_DRAINING` commit on and the lifecycle paths that could replace the database are denied, so the data at the boundary is what was committed before the cutoff. The server logs a warning when it records such a boundary. + - Crash after `SOURCE_WRITE_FENCED` committed: the stored boundary names the log that was live when the drain was proven. After the restart the live log has a new id but the same data (writes stayed closed). A caller that compares the boundary's `log_id` with the source's current log id sees the rebuild and copies from a snapshot of the unchanged database rather than from the old log's frames. ## 9. Source read fence (Design) @@ -690,7 +695,9 @@ Every transition emits one structured log event (target `libsql_server::fence::a - **No new dependency.** The crate has no failpoint library. Race tests use `#[cfg(test)]` hooks: `FenceTestHooks` holds named points (`AfterInstallingGate`, `BeforeMetastoreCommit`, `AfterMetastoreCommit`, `BeforeGatePublish`, `InBeginWriteTxnAfterCheck`, `AfterManagerRelease`, `BeforeBoundaryCapture`, `AfterTargetRowsCommitted`, `BeforeResponse`), each able to park the task on a pair of `Notify`s (`pause_at` returns handles to wait until the task arrives and to release it) or to inject an error or an indeterminate commit. An armed point fires once. `BeforeMetastoreCommit` is reached immediately before the metastore transaction is started (an injected error there is a failure before commit); an injected indeterminate outcome at `AfterMetastoreCommit` is a commit that happened but was not acknowledged. Hooks compile only in the library's own test build, so they cost nothing in release builds; integration tests under `tests/` cover protocol behaviour and do not rely on hooks. - **Restart** at a boundary: reopen `MetaStore` and rebuild the registry on the same temporary directory, the way existing metastore tests do; integration tests stop and start a `TestServer` on the same path. +- **Restart** at a boundary, as a crash: `namespace::fence::tests` runs each server lifetime on a runtime of its own and ends it by shutting the runtime down without any shutdown code and leaking the `NamespaceStore`, with the command parked at the hook point; the next lifetime opens a new `NamespaceStore` (metastore and registry) on the same directory. The namespace's `.sentinel` stays behind, so the restart takes the real dirty-recovery path. - **Response loss**: the test drops the command future after `AfterMetastoreCommit` and then replays or inspects. +- An armed `Indeterminate` at `BeforeMetastoreCommit` reports a commit as indeterminate without running it: a commit that failed without applying but whose outcome the controller cannot know. - **Representative schemas**: synthetic multi-table schemas with indexes, triggers, views and an FTS5 table (FTS5 is compiled in). - `TXN_TIMEOUT` is 100 ms in test builds; drain tests that hold a writer longer than that use their own `txn_timeout` or a hook, never a sleep. @@ -706,7 +713,7 @@ Planned test names; the table is updated as tests land. | 4 | Program that captured config before the fence is rejected at the WAL | landed: `connection::connection_manager::fence_tests::wal_gate_rejects_program_admitted_before_fence` (a SQL function parks the program between admission and its write while the fence is acquired and released) | | 5 | Pre-fence transactions cannot write after release or publication | landed: `connection::connection_manager::fence_tests::stale_generation_cannot_write_after_release`; unfenced behaviour unchanged: `unfenced_namespace_is_unchanged`; planned: `stale_generation_cannot_write_after_enable_writes` | | 6 | Acquisition timeout returns `DRAINING`, admission stays closed | landed: `namespace::fence::drain::tests::{deadline_returns_draining_and_stays_closed, replay_of_draining_resumes_and_completes}` | -| 7 | Restart at every persistence boundary; indeterminate persistence keeps the gate closed until same-command reconciliation | `fence::tests::restart_at_each_boundary` (parameterised over hook points), `indeterminate_commit_keeps_gate_closed`; landed: `fence::controller::tests::{indeterminate_commit_keeps_writes_closed_until_replayed, indeterminate_commit_that_did_not_apply_is_retried_by_replay, failed_before_commit_leaves_gate_unchanged, publication_happens_before_the_response, committed_command_is_published_when_the_caller_goes_away}`, `namespace::store::fence_tests::restart_installs_the_durable_gate_before_serving` | +| 7 | Restart at every persistence boundary; indeterminate persistence keeps the gate closed until same-command reconciliation | landed: `namespace::fence::tests::restart_at_each_boundary` (a crash at each of 17 boundaries of `AcquireSourceWriteFence` and `ReleaseSourceWriteFence`: after the `INSTALLING` gate, before, at and after each metastore commit, with and without a lagging marker, before publication, before boundary capture and before the response; the restart recovers the prior or the committed state, admits writes only if an opening transition committed, repairs the marker, and a replay finishes the command), `restart_in_draining_waits_for_the_same_command` (a writer active at the crash; nothing advances until the same command is replayed, which completes at once), `indeterminate_commit_keeps_gate_closed` (through the drain path, both when the commit happened and when it did not), `acquire_response_loss_resolved_by_replay_and_inspect`; `fence::controller::tests::{indeterminate_commit_keeps_writes_closed_until_replayed, indeterminate_commit_that_did_not_apply_is_retried_by_replay, failed_before_commit_leaves_gate_unchanged, publication_happens_before_the_response, committed_command_is_published_when_the_caller_goes_away}`, `namespace::store::fence_tests::restart_installs_the_durable_gate_before_serving` | | 8 | Evict and lazily reload a fenced namespace; identical admission | `tests::fence::lifecycle::evicted_namespace_reloads_same_gate`; landed at unit level: `namespace::store::fence_tests::evicted_namespace_reloads_with_the_same_controller`, `fence::registry::tests::seeded_from_load_fences_including_recovered_names` | | 9 | Filesystem recovery, `destroy_on_error`, undecodable records, missing target quarantine, metastore backup rollback fail closed with provenance | `meta_store::fence_tests::recovery::{fs_recovery_with_marker_unavailable, destroy_on_error_keeps_fenced_unavailable, undecodable_row_unavailable, incomplete_target_unavailable, metastore_rollback_detected_by_marker, lookup_never_creates, undecodable_name_with_fence_fails_startup, marker_in_invalid_directory_fails_startup}`; legacy behaviour kept: `destroy_on_error_without_fences_is_unchanged`, `undecodable_row_without_fences_is_skipped_as_before`; `meta_store::fence_tests::corrupt_fence_row_fails_closed` | | 10 | Wrong owner, stale revision, invalid role/state, replay, command-id reuse; replay before revision check | `fence::transition::tests::*` (exhaustive over states × commands) | diff --git a/libsql-server/src/namespace/fence/controller.rs b/libsql-server/src/namespace/fence/controller.rs index cc6a274195..7e718e5516 100644 --- a/libsql-server/src/namespace/fence/controller.rs +++ b/libsql-server/src/namespace/fence/controller.rs @@ -458,11 +458,17 @@ impl Transition { } } - if let HookOutcome::Fail(e) = controller.hook(HookPoint::BeforeMetastoreCommit).await { - return Err(e.into()); - } + let result = match controller.hook(HookPoint::BeforeMetastoreCommit).await { + HookOutcome::Continue => run.await, + HookOutcome::Fail(e) => return Err(e.into()), + // A commit that failed without applying, but whose outcome the controller cannot + // know (test hook). + HookOutcome::Indeterminate => { + Err(indeterminate(key, "the commit was not acknowledged (test hook)").into()) + } + }; - let result = match run.await { + let result = match result { Ok(commit) => match controller.hook(HookPoint::AfterMetastoreCommit).await { HookOutcome::Continue => Ok(commit), HookOutcome::Indeterminate | HookOutcome::Fail(_) => Err(indeterminate( diff --git a/libsql-server/src/namespace/fence/drain.rs b/libsql-server/src/namespace/fence/drain.rs index 4f77a95d4a..6165784f60 100644 --- a/libsql-server/src/namespace/fence/drain.rs +++ b/libsql-server/src/namespace/fence/drain.rs @@ -121,6 +121,16 @@ pub async fn acquire_source_write_fence( }; // Step 7. + let acquired_on = commit.record.as_ref().and_then(|r| r.identity.log_id); + if acquired_on.is_some_and(|log_id| log_id != boundary.log_id) { + tracing::warn!( + namespace = %controller.namespace(), + acquired_on = ?acquired_on, + boundary_log_id = %boundary.log_id, + "the replication log was rebuilt since the write fence was acquired (the source \ + restarted while draining); the frozen boundary names the rebuilt log" + ); + } ctx.now_ms = now_ms(); transition .complete_drain( diff --git a/libsql-server/src/namespace/fence/hooks.rs b/libsql-server/src/namespace/fence/hooks.rs index 90ef0d1685..28a00c673e 100644 --- a/libsql-server/src/namespace/fence/hooks.rs +++ b/libsql-server/src/namespace/fence/hooks.rs @@ -56,7 +56,9 @@ pub enum HookAction { /// Fail at this point with `error`, as if the step had failed before it took effect. Fail(FenceError), /// At `AfterMetastoreCommit`: report the commit as indeterminate even though it happened, - /// which is what a lost commit acknowledgement looks like to the controller. + /// which is what a lost commit acknowledgement looks like to the controller. At + /// `BeforeMetastoreCommit`: report it as indeterminate without running it, which is a + /// commit that failed without applying but whose outcome the controller cannot know. Indeterminate, } diff --git a/libsql-server/src/namespace/fence/mod.rs b/libsql-server/src/namespace/fence/mod.rs index 6b445a9ba6..7e64e0cd7d 100644 --- a/libsql-server/src/namespace/fence/mod.rs +++ b/libsql-server/src/namespace/fence/mod.rs @@ -28,6 +28,9 @@ pub mod state; pub mod store; pub mod transition; +#[cfg(test)] +mod tests; + #[allow(clippy::all)] pub(crate) mod proto { include!("../../generated/namespace_fence.rs"); diff --git a/libsql-server/src/namespace/fence/tests.rs b/libsql-server/src/namespace/fence/tests.rs new file mode 100644 index 0000000000..bb41fbeecd --- /dev/null +++ b/libsql-server/src/namespace/fence/tests.rs @@ -0,0 +1,898 @@ +//! Restart at every persistence boundary, indeterminate commits and lost responses +//! (`docs/NAMESPACE_FENCE.md` sections 8.4, 8.5 and 16; section 17 row 7). +//! +//! A restart here is a crash: each server lifetime runs on a runtime of its own, and ending it +//! shuts that runtime down without running any shutdown code and leaks the `NamespaceStore`, so +//! nothing is flushed, the namespace's `.sentinel` stays behind and the next start takes the +//! dirty-recovery path. The next lifetime opens a new `NamespaceStore` (and so a new +//! `MetaStore` and fence registry) on the same directory. The task running the fence command is +//! parked at a hook point when the crash happens, so the crash lands exactly on that boundary. + +use std::future::Future; +use std::path::Path; +use std::sync::Arc; +use std::time::Duration; + +use rusqlite::ErrorCode; +use tempfile::tempdir; +use tokio::runtime::Runtime; +use uuid::Uuid; + +use super::command::{DrainPolicy, FenceCommand, FenceRequest, OnDeadline}; +use super::controller::FenceController; +use super::hooks::{HookAction, HookPoint, Paused}; +use super::outcome::{FenceDetail, FenceError, FenceOutcome}; +use super::record::{FrozenBoundary, ServerIdentity}; +use super::state::{FenceState, OperationClass}; +use super::store as fence_store; +use crate::connection::config::DatabaseConfig; +use crate::connection::Connection as _; +use crate::database::{Connection, Database}; +use crate::error::Error; +use crate::namespace::meta_store::{FenceCommit, FenceCommitKind, FenceInspection}; +use crate::namespace::store::fence_tests::open_store; +use crate::namespace::store::NamespaceStore; +use crate::namespace::RestoreOption; + +const OP: Uuid = Uuid::from_u128(0xa); +const OTHER_OP: Uuid = Uuid::from_u128(0xb); +/// Long enough that no test reaches it: a drain must never finish because of time. +const LONG: DrainPolicy = DrainPolicy { + deadline_ms: 600_000, + on_deadline: OnDeadline::Fail, +}; +/// A deadline that has already passed: an acquisition with a writer still active answers +/// `DRAINING` at once. +const NOW: DrainPolicy = DrainPolicy { + deadline_ms: 0, + on_deadline: OnDeadline::Fail, +}; +/// Bound on waiting for a task to reach a hook point; a test that hits it has failed. +const PROMPT: Duration = Duration::from_secs(30); +/// Rows committed to `t` before any fence command runs. +const ROWS: i64 = 3; + +fn server_identity() -> ServerIdentity { + ServerIdentity { + build: "test".into(), + instance_id: Uuid::from_u128(0x99), + } +} + +/// One lifetime of a server process on `dir`. +struct Server { + rt: Runtime, + store: NamespaceStore, +} + +impl Server { + fn boot(dir: &Path) -> Self { + let rt = tokio::runtime::Builder::new_multi_thread() + .worker_threads(4) + .enable_all() + .build() + .unwrap(); + let store = rt.block_on(open_store(dir)); + Self { rt, store } + } + + /// End the lifetime the way a crash does: no shutdown code runs, nothing is flushed or + /// closed, and every task (including a fence command parked at a hook point) stops where + /// it is. + fn crash(self) { + let Self { rt, store } = self; + std::mem::forget(store); + rt.shutdown_background(); + } + + fn run(&self, f: F) -> F::Output { + self.rt.block_on(f) + } + + /// Create `ns` with a table `t` holding [`ROWS`] rows. + fn create_source(&self) { + self.run(async { + self.store + .create( + "ns".into(), + RestoreOption::Latest, + DatabaseConfig { + // Held writers must never have the slot stolen by the timeout. + txn_timeout: Some(Duration::from_secs(600)), + ..Default::default() + }, + ) + .await + .unwrap(); + let conn = self.conn().await; + raw(&conn, "create table t (x)").await.unwrap(); + for _ in 0..ROWS { + raw(&conn, "insert into t values (1)").await.unwrap(); + } + }) + } + + /// The namespace's controller, loading the namespace if it is not loaded. + async fn fence(&self) -> Arc { + self.store + .with("ns".into(), |ns| ns.fence().clone()) + .await + .unwrap() + } + + async fn conn(&self) -> Arc { + let maker = self + .store + .with("ns".into(), |ns| ns.db.connection_maker()) + .await + .unwrap(); + Arc::new(maker.create().await.unwrap()) + } + + /// The live replication log's id and last committed frame. + async fn log(&self) -> (Uuid, Option) { + self.store + .with("ns".into(), |ns| match &ns.db { + Database::Primary(p) => { + let logger = p.wal_wrapper.wrapper().logger(); + let frame_no = *logger.new_frame_notifier.borrow(); + (logger.log_id(), frame_no) + } + _ => unreachable!(), + }) + .await + .unwrap() + } + + fn execute( + &self, + request: FenceRequest, + ) -> tokio::task::JoinHandle> { + let store = self.store.clone(); + self.rt.spawn(async move { + store + .execute_fence_command(request, server_identity()) + .await + }) + } + + async fn inspect(&self) -> FenceInspection { + self.store + .meta_store() + .inspect_fence("ns".into()) + .await + .unwrap() + } + + async fn count(&self) -> i64 { + let conn = self.conn().await; + tokio::task::spawn_blocking(move || { + conn.with_raw(|c| c.query_row("select count(*) from t", (), |r| r.get(0))) + }) + .await + .unwrap() + .unwrap() + } + + /// Whether a new connection may begin a write transaction. Writes nothing. + async fn writes_admitted(&self) -> bool { + let conn = self.conn().await; + match raw(&conn, "begin immediate; rollback;").await { + Ok(()) => true, + Err(rusqlite::Error::SqliteFailure(e, _)) + if e.code == ErrorCode::AuthorizationForStatementDenied => + { + false + } + Err(e) => panic!("unexpected error probing write admission: {e}"), + } + } + + /// Acquire and complete the source write fence under `OP`, command 1. + fn fence_source(&self) -> FenceCommit { + self.run(async { + let (log_id, _) = self.log().await; + let commit = self + .execute(acquire(log_id, 1, LONG)) + .await + .unwrap() + .unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + commit + }) + } +} + +/// Run `sql` as one raw program on `conn`, off the async runtime (it can block on the write +/// slot). +async fn raw(conn: &Arc, sql: &'static str) -> rusqlite::Result<()> { + let conn = conn.clone(); + tokio::task::spawn_blocking(move || conn.with_raw(|c| c.execute_batch(sql))) + .await + .unwrap() +} + +fn acquire(log_id: Uuid, command_id: u128, policy: DrainPolicy) -> FenceRequest { + FenceRequest { + namespace: "ns".into(), + operation_id: OP, + command_id: Uuid::from_u128(command_id), + expected_state: FenceState::Unfenced, + expected_revision: 0, + command: FenceCommand::AcquireSourceWriteFence { + expected_log_id: log_id, + drain_policy: Some(policy), + }, + } +} + +fn release(command_id: u128, revision: u64) -> FenceRequest { + FenceRequest { + namespace: "ns".into(), + operation_id: OP, + command_id: Uuid::from_u128(command_id), + expected_state: FenceState::SourceWriteFenced, + expected_revision: revision, + command: FenceCommand::ReleaseSourceWriteFence, + } +} + +fn fence_error(result: &crate::Result) -> &FenceError { + match result { + Err(Error::NamespaceFence(e)) => e, + other => panic!("expected a fence error, got {other:?}"), + } +} + +fn boundary(commit: &FenceCommit) -> FrozenBoundary { + commit + .record + .as_ref() + .and_then(|r| r.frozen_boundary) + .expect("a write-fenced source has a frozen boundary") +} + +/// The marker file's bytes, if there is one. +fn read_marker_bytes(dbs: &Path) -> Option> { + match std::fs::read(fence_store::marker_path(dbs, &"ns".into())) { + Ok(bytes) => Some(bytes), + Err(e) if e.kind() == std::io::ErrorKind::NotFound => None, + Err(e) => panic!("{e}"), + } +} + +/// Put the marker file back to `bytes` (`None`: no marker). +fn restore_marker_bytes(dbs: &Path, bytes: Option<&[u8]>) { + let path = fence_store::marker_path(dbs, &"ns".into()); + match bytes { + Some(bytes) => std::fs::write(path, bytes).unwrap(), + None => std::fs::remove_file(path).unwrap(), + } +} + +async fn reached(paused: &Paused, case: &str, point: HookPoint) { + tokio::time::timeout(PROMPT, paused.reached()) + .await + .unwrap_or_else(|_| panic!("{case}: the command never reached {point:?}")); +} + +#[derive(Debug, Clone, Copy)] +enum Command { + Acquire, + Release, +} + +/// What the metastore holds for the command when the process dies. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +enum Durable { + /// Nothing of the command. + Nothing, + /// The `SOURCE_DRAINING` record and the command's `DRAINING` receipt (acquisition only). + Draining, + /// The command's final result. + Final, +} + +#[derive(Debug, Clone, Copy)] +struct Boundary { + name: &'static str, + command: Command, + point: HookPoint, + /// The point is the one reached on the acquisition's second commit (the completion of the + /// drain), not its first. + second_commit: bool, + /// The marker file is put back to what it held before the commit, as a crash between the + /// metastore commit and the marker write leaves it. + marker_lags: bool, + durable: Durable, +} + +const BOUNDARIES: &[Boundary] = &[ + Boundary { + name: "acquire/after-installing-gate", + command: Command::Acquire, + point: HookPoint::AfterInstallingGate, + second_commit: false, + marker_lags: false, + durable: Durable::Nothing, + }, + Boundary { + name: "acquire/before-draining-commit", + command: Command::Acquire, + point: HookPoint::BeforeMetastoreCommit, + second_commit: false, + marker_lags: false, + durable: Durable::Nothing, + }, + Boundary { + name: "acquire/after-draining-commit", + command: Command::Acquire, + point: HookPoint::AfterMetastoreCommit, + second_commit: false, + marker_lags: false, + durable: Durable::Draining, + }, + Boundary { + name: "acquire/after-draining-commit/marker-lags", + command: Command::Acquire, + point: HookPoint::AfterMetastoreCommit, + second_commit: false, + marker_lags: true, + durable: Durable::Draining, + }, + Boundary { + name: "acquire/before-draining-publish", + command: Command::Acquire, + point: HookPoint::BeforeGatePublish, + second_commit: false, + marker_lags: false, + durable: Durable::Draining, + }, + Boundary { + name: "acquire/draining-published", + command: Command::Acquire, + point: HookPoint::BeforeResponse, + second_commit: false, + marker_lags: false, + durable: Durable::Draining, + }, + Boundary { + name: "acquire/before-boundary-capture", + command: Command::Acquire, + point: HookPoint::BeforeBoundaryCapture, + second_commit: false, + marker_lags: false, + durable: Durable::Draining, + }, + Boundary { + name: "acquire/before-fenced-commit", + command: Command::Acquire, + point: HookPoint::BeforeMetastoreCommit, + second_commit: true, + marker_lags: false, + durable: Durable::Draining, + }, + Boundary { + name: "acquire/after-fenced-commit", + command: Command::Acquire, + point: HookPoint::AfterMetastoreCommit, + second_commit: true, + marker_lags: false, + durable: Durable::Final, + }, + Boundary { + name: "acquire/after-fenced-commit/marker-lags", + command: Command::Acquire, + point: HookPoint::AfterMetastoreCommit, + second_commit: true, + marker_lags: true, + durable: Durable::Final, + }, + Boundary { + name: "acquire/before-fenced-publish", + command: Command::Acquire, + point: HookPoint::BeforeGatePublish, + second_commit: true, + marker_lags: false, + durable: Durable::Final, + }, + Boundary { + name: "acquire/before-fenced-response", + command: Command::Acquire, + point: HookPoint::BeforeResponse, + second_commit: true, + marker_lags: false, + durable: Durable::Final, + }, + Boundary { + name: "release/before-commit", + command: Command::Release, + point: HookPoint::BeforeMetastoreCommit, + second_commit: false, + marker_lags: false, + durable: Durable::Nothing, + }, + Boundary { + name: "release/after-commit", + command: Command::Release, + point: HookPoint::AfterMetastoreCommit, + second_commit: false, + marker_lags: false, + durable: Durable::Final, + }, + Boundary { + name: "release/after-commit/marker-lags", + command: Command::Release, + point: HookPoint::AfterMetastoreCommit, + second_commit: false, + marker_lags: true, + durable: Durable::Final, + }, + Boundary { + name: "release/before-publish", + command: Command::Release, + point: HookPoint::BeforeGatePublish, + second_commit: false, + marker_lags: false, + durable: Durable::Final, + }, + Boundary { + name: "release/before-response", + command: Command::Release, + point: HookPoint::BeforeResponse, + second_commit: false, + marker_lags: false, + durable: Durable::Final, + }, +]; + +/// The state a restart recovers for `case`: the one before the command, or the one it +/// committed. Never anything else. +fn recovered_state(case: &Boundary) -> FenceState { + match (case.command, case.durable) { + (Command::Acquire, Durable::Nothing) => FenceState::Unfenced, + (Command::Acquire, Durable::Draining) => FenceState::SourceDraining, + (Command::Acquire, Durable::Final) => FenceState::SourceWriteFenced, + (Command::Release, Durable::Nothing) => FenceState::SourceWriteFenced, + (Command::Release, Durable::Final) => FenceState::Released, + (Command::Release, Durable::Draining) => unreachable!(), + } +} + +/// Kill the process at every point where a fence command persists, publishes or answers, and +/// restart it on the same directory. The restarted server recovers exactly the state before the +/// command or the state it committed, installs that gate before it serves the namespace (so a +/// namespace is never open unless an opening transition committed), and a replay of the same +/// command then finishes it with the stored or the expected result. +#[test] +fn restart_at_each_boundary() { + for case in BOUNDARIES { + restart_at(case); + } +} + +fn restart_at(case: &Boundary) { + let name = case.name; + let dir = tempdir().unwrap(); + let dbs = dir.path().join("dbs"); + + // First lifetime: run the command until it reaches the boundary, then crash. + let server = Server::boot(dir.path()); + server.create_source(); + let (log_before, _) = server.run(server.log()); + let (request, revision_before) = match case.command { + Command::Acquire => (acquire(log_before, 1, LONG), 0), + Command::Release => { + let fenced = server.fence_source(); + let revision = fenced.record.as_ref().unwrap().revision; + (release(2, revision), revision) + } + }; + let committed_boundary = server.run(async { + let fence = server.fence().await; + let hooks = fence.hooks(); + let mut marker_before = read_marker_bytes(&dbs); + let paused = if case.second_commit { + let capture = hooks.pause_at(HookPoint::BeforeBoundaryCapture); + let task = server.execute(request.clone()); + reached(&capture, name, HookPoint::BeforeBoundaryCapture).await; + marker_before = read_marker_bytes(&dbs); + let paused = hooks.pause_at(case.point); + capture.resume(); + reached(&paused, name, case.point).await; + drop(task); + paused + } else { + let paused = hooks.pause_at(case.point); + let task = server.execute(request.clone()); + reached(&paused, name, case.point).await; + drop(task); + paused + }; + if case.marker_lags { + restore_marker_bytes(&dbs, marker_before.as_deref()); + } + // What the metastore holds at the moment of the crash. + let inspected = server.inspect().await; + assert_eq!(inspected.fence.state(), recovered_state(case), "{name}"); + drop(paused); + inspected.fence.record().and_then(|r| r.frozen_boundary) + }); + server.crash(); + + // Second lifetime. + let server = Server::boot(dir.path()); + server.run(async { + let expected = recovered_state(case); + let fence = server.fence().await; + let gate = fence.gate(); + assert_eq!(gate.state(), expected, "{name}: recovered state"); + assert!( + gate.indeterminate.is_none() && !gate.is_installing(), + "{name}" + ); + let open = matches!(expected, FenceState::Unfenced | FenceState::Released); + assert_eq!( + server.writes_admitted().await, + open, + "{name}: write admission" + ); + // Committed data survived, nothing else was written. + assert_eq!(server.count().await, ROWS, "{name}"); + // The marker was repaired if it had fallen behind. + let marker = fence_store::read_marker(&dbs, &"ns".into()).unwrap(); + assert_eq!( + marker.and_then(|m| m.ok()).map(|m| m.record.revision), + gate.fence.record().map(|r| r.revision), + "{name}: marker" + ); + + // A crash leaves the namespace dirty, so its replication log was rebuilt from the + // database file under a new log id. + let (log_after, frame_after) = server.log().await; + assert_ne!(log_after, log_before, "{name}: the log was not rebuilt"); + + let replay = server.execute(request.clone()).await.unwrap(); + match (case.command, case.durable) { + (Command::Acquire, Durable::Nothing) => { + // Nothing was acknowledged and nothing was written, and the identity the caller + // observed is gone: the replay is refused before anything is written, and the + // caller acquires again under the identity it reads now. + let e = fence_error(&replay); + assert_eq!(e.outcome(), FenceOutcome::FencePreconditionFailed, "{name}"); + assert_eq!(e.detail(), Some(FenceDetail::NamespaceIdentityMismatch)); + assert_eq!(fence.gate().state(), FenceState::Unfenced, "{name}"); + assert!(server.writes_admitted().await, "{name}"); + let commit = server + .execute(acquire(log_after, 3, LONG)) + .await + .unwrap() + .unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied, "{name}"); + assert_eq!( + boundary(&commit), + FrozenBoundary { + log_id: log_after, + frame_no: frame_after, + }, + "{name}" + ); + } + (Command::Acquire, Durable::Draining) => { + // The drain that was requested resumes and completes at once: recovery + // discarded any uncommitted work. The boundary is on the live, rebuilt log. + let commit = replay.unwrap(); + assert_eq!(commit.kind, FenceCommitKind::Committed, "{name}"); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied, "{name}"); + assert_eq!(commit.receipt.command_id, request.command_id); + assert_eq!(commit.receipt.revision_after, 2, "{name}"); + let record = commit.record.as_ref().unwrap(); + assert_eq!(record.identity.log_id, Some(log_before), "{name}"); + assert_eq!( + boundary(&commit), + FrozenBoundary { + log_id: log_after, + frame_no: frame_after, + }, + "{name}" + ); + } + (Command::Acquire, Durable::Final) => { + // The stored result, boundary included: it names the log that was live when + // the drain was proven. + let commit = replay.unwrap(); + assert_eq!(commit.kind, FenceCommitKind::Replayed, "{name}"); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied, "{name}"); + assert_eq!(Some(boundary(&commit)), committed_boundary, "{name}"); + assert_eq!(boundary(&commit).log_id, log_before, "{name}"); + } + (Command::Release, durable) => { + let commit = replay.unwrap(); + let kind = if durable == Durable::Final { + FenceCommitKind::Replayed + } else { + FenceCommitKind::Committed + }; + assert_eq!(commit.kind, kind, "{name}"); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied, "{name}"); + assert_eq!(commit.receipt.revision_before, revision_before, "{name}"); + assert_eq!(commit.receipt.state_after, FenceState::Released, "{name}"); + } + } + + // Settled: the gate is the durable state, and a further replay answers the same. + let gate = fence.gate(); + let durable = server.inspect().await; + assert_eq!(gate.fence, durable.fence, "{name}"); + let open = gate.state() == FenceState::Released; + assert_eq!(server.writes_admitted().await, open, "{name}"); + assert_eq!(server.count().await, ROWS, "{name}"); + }); + server.crash(); +} + +/// After a restart in `SOURCE_DRAINING` with a writer that was active at the crash, nothing +/// advances by itself: the namespace stays closed, other commands cannot move it on, and only +/// the replay of the same acquisition completes the drain, at once. +#[test] +fn restart_in_draining_waits_for_the_same_command() { + let dir = tempdir().unwrap(); + + let server = Server::boot(dir.path()); + server.create_source(); + let (log_before, _) = server.run(server.log()); + let request = acquire(log_before, 1, NOW); + server.run(async { + let holder = server.conn().await; + raw(&holder, "begin immediate; insert into t values (2);") + .await + .unwrap(); + let commit = server.execute(request.clone()).await.unwrap().unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Draining); + // The writer never finishes: its transaction dies with the process (closing the + // connection rolls it back, which is what SQLite recovery does to it on disk). + drop(holder); + }); + server.crash(); + + let server = Server::boot(dir.path()); + server.run(async { + let fence = server.fence().await; + assert_eq!(fence.gate().state(), FenceState::SourceDraining); + assert!(!server.writes_admitted().await); + // The uncommitted write is gone. + assert_eq!(server.count().await, ROWS); + + // Reads are served and move nothing; other commands cannot advance the namespace. + let set_read_fence = FenceRequest { + namespace: "ns".into(), + operation_id: OP, + command_id: Uuid::from_u128(2), + expected_state: FenceState::SourceWriteFenced, + expected_revision: 2, + command: FenceCommand::SetSourceReadFence { drain_policy: None }, + }; + let r = server.execute(set_read_fence).await.unwrap(); + assert!(fence_error(&r).outcome() != FenceOutcome::Applied); + let (log_after, frame_after) = server.log().await; + let mut other = acquire(log_after, 3, LONG); + other.operation_id = OTHER_OP; + let r = server.execute(other).await.unwrap(); + assert_eq!( + fence_error(&r).outcome(), + FenceOutcome::FenceOwnedByAnotherOperation + ); + let inspected = server.inspect().await; + assert_eq!(inspected.fence.state(), FenceState::SourceDraining); + assert_eq!(inspected.fence.revision(), 1); + assert_eq!(fence.gate().state(), FenceState::SourceDraining); + + // The same command, with the same zero deadline, completes at once. + let commit = server.execute(request.clone()).await.unwrap().unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + assert_eq!(commit.receipt.command_id, request.command_id); + assert_eq!( + boundary(&commit), + FrozenBoundary { + log_id: log_after, + frame_no: frame_after, + } + ); + assert_eq!(fence.gate().state(), FenceState::SourceWriteFenced); + assert!(!server.writes_admitted().await); + assert_eq!(server.count().await, ROWS); + }); + server.crash(); +} + +/// A commit whose outcome is unknown closes the namespace (every class but maintenance and +/// observability), makes every other command answer `FENCE_COMMIT_INDETERMINATE`, and is +/// reconciled by replaying the same command from the durable row: both when the commit had +/// happened and when it had not. +#[test] +fn indeterminate_commit_keeps_gate_closed() { + for committed in [true, false] { + let dir = tempdir().unwrap(); + let server = Server::boot(dir.path()); + server.create_source(); + server.run(async { + let (log_id, frame_no) = server.log().await; + let fence = server.fence().await; + let request = acquire(log_id, 1, LONG); + fence.hooks().arm( + if committed { + HookPoint::AfterMetastoreCommit + } else { + HookPoint::BeforeMetastoreCommit + }, + HookAction::Indeterminate, + ); + let r = server.execute(request.clone()).await.unwrap(); + let e = fence_error(&r); + assert_eq!(e.outcome(), FenceOutcome::FenceCommitIndeterminate); + assert_eq!(e.detail(), Some(FenceDetail::IndeterminateCommit)); + + let durable = server.inspect().await.fence.state(); + assert_eq!( + durable, + if committed { + FenceState::SourceDraining + } else { + FenceState::Unfenced + } + ); + let gate = fence.gate(); + assert_eq!(gate.indeterminate, Some((OP, request.command_id))); + assert!(!gate.is_installing()); + for class in OperationClass::ALL { + let permitted = fence.permits(class); + match class { + OperationClass::Maintenance | OperationClass::Observability => { + assert!(permitted.is_ok()) + } + _ => assert_eq!( + permitted.unwrap_err().outcome(), + FenceOutcome::FenceStateUnavailable, + "{class:?}" + ), + } + } + assert!(!server.writes_admitted().await); + + // Any other command is refused with the indeterminate code, nothing is written. + let mut other = acquire(log_id, 2, LONG); + other.operation_id = OTHER_OP; + let r = server.execute(other).await.unwrap(); + assert_eq!( + fence_error(&r).outcome(), + FenceOutcome::FenceCommitIndeterminate + ); + let r = server.execute(acquire(log_id, 3, LONG)).await.unwrap(); + assert_eq!( + fence_error(&r).outcome(), + FenceOutcome::FenceCommitIndeterminate + ); + assert_eq!(server.inspect().await.fence.state(), durable); + + // The replay reconciles from the durable row: it resumes the drain that committed, + // or runs the acquisition that did not, and completes it. + let commit = server.execute(request.clone()).await.unwrap().unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + assert_eq!(commit.receipt.command_id, request.command_id); + assert_eq!(commit.receipt.revision_after, 2); + assert_eq!(boundary(&commit), FrozenBoundary { log_id, frame_no }); + let gate = fence.gate(); + assert_eq!(gate.indeterminate, None); + assert_eq!(gate.state(), FenceState::SourceWriteFenced); + assert_eq!(gate.fence, server.inspect().await.fence); + assert!(!server.writes_admitted().await); + // Maintenance and reads are served again. + assert_eq!(server.count().await, ROWS); + }); + server.crash(); + } +} + +/// The caller of an acquisition goes away before it hears the answer, once while the drain is +/// still waiting and once just before the final response. The acquisition finishes on its own +/// task regardless, `InspectFence` shows the result, and a replay of the same command returns +/// it. +#[test] +fn acquire_response_loss_resolved_by_replay_and_inspect() { + for lose_at in [HookPoint::AfterMetastoreCommit, HookPoint::BeforeResponse] { + let dir = tempdir().unwrap(); + let server = Server::boot(dir.path()); + server.create_source(); + server.run(async { + let (log_id, _) = server.log().await; + let fence = server.fence().await; + let request = acquire(log_id, 1, LONG); + let holder = server.conn().await; + raw(&holder, "begin immediate; insert into t values (2);") + .await + .unwrap(); + + let hooks = fence.hooks(); + let capture = hooks.pause_at(HookPoint::BeforeBoundaryCapture); + let first = (lose_at == HookPoint::AfterMetastoreCommit) + .then(|| hooks.pause_at(HookPoint::AfterMetastoreCommit)); + let caller = server.execute(request.clone()); + if let Some(first) = first { + // The caller is gone right after SOURCE_DRAINING commits. + reached(&first, "response loss", HookPoint::AfterMetastoreCommit).await; + caller.abort(); + first.resume(); + } + + // The drain waits for the writer, with or without a caller. + let mut rx = fence.subscribe(); + tokio::time::timeout( + PROMPT, + rx.wait_for(|g| g.state() == FenceState::SourceDraining), + ) + .await + .unwrap() + .unwrap(); + assert_eq!( + server.inspect().await.fence.state(), + FenceState::SourceDraining + ); + // A replay while it runs waits for the transition lock rather than racing it. + let replay_during = server.execute(request.clone()); + + raw(&holder, "commit").await.unwrap(); + let (_, committed) = server.log().await; + reached(&capture, "response loss", HookPoint::BeforeBoundaryCapture).await; + if lose_at == HookPoint::BeforeResponse { + // The caller is gone after the final commit was published, before the answer. + let last = hooks.pause_at(HookPoint::BeforeResponse); + capture.resume(); + reached(&last, "response loss", HookPoint::BeforeResponse).await; + assert_eq!(fence.gate().state(), FenceState::SourceWriteFenced); + caller.abort(); + last.resume(); + } else { + capture.resume(); + } + assert!(caller.await.unwrap_err().is_cancelled()); + tokio::time::timeout( + PROMPT, + rx.wait_for(|g| g.state() == FenceState::SourceWriteFenced), + ) + .await + .unwrap() + .unwrap(); + + let inspected = server.inspect().await; + assert_eq!(inspected.fence.state(), FenceState::SourceWriteFenced); + let record = inspected.fence.record().unwrap(); + assert_eq!( + record.frozen_boundary, + Some(FrozenBoundary { + log_id, + frame_no: committed, + }) + ); + let receipt = inspected + .receipts + .iter() + .filter_map(|r| r.receipt.as_ref().ok()) + .find(|r| r.command_id == request.command_id) + .unwrap() + .clone(); + assert_eq!(receipt.outcome, FenceOutcome::Applied); + + for replay in [ + replay_during.await.unwrap().unwrap(), + server.execute(request.clone()).await.unwrap().unwrap(), + ] { + assert_eq!(replay.kind, FenceCommitKind::Replayed); + assert_eq!(replay.receipt, receipt); + assert_eq!(replay.record.as_ref(), Some(record)); + } + assert_eq!(server.count().await, ROWS + 1); + assert!(!server.writes_admitted().await); + }); + server.crash(); + } +} diff --git a/libsql-server/src/namespace/fence/transition.rs b/libsql-server/src/namespace/fence/transition.rs index d589ce8859..9c664b320c 100644 --- a/libsql-server/src/namespace/fence/transition.rs +++ b/libsql-server/src/namespace/fence/transition.rs @@ -673,12 +673,13 @@ pub fn complete_drain( let mut next = record.clone(); if let DrainCompletion::SourceWrites { boundary } = completion { - if record.identity.log_id != Some(boundary.log_id) { - return Err(precondition( - FenceDetail::NamespaceIdentityMismatch, - "the frozen boundary belongs to a different replication log", - )); - } + // The boundary names the replication log that is live when the drain is proven. It is + // not the log the identity was captured on when the source was restarted while + // draining: crash recovery rebuilds the log from the database file under a new id + // (section 8.5). Write admission has been durably closed since `SOURCE_DRAINING` + // committed, and the lifecycle paths that could replace the database are denied, so + // the data is what was committed before the cutoff. The identity keeps the log id the + // caller acquired against. next.frozen_boundary = Some(boundary); } next.state = to; @@ -1485,20 +1486,21 @@ mod tests { assert!(complete_drain(&record, &receipt, DrainCompletion::SourceReads, &env()).is_err()); assert!(complete_drain(&record, &receipt, DrainCompletion::TargetImport, &env()).is_err()); - // A boundary from another log. - let err = complete_drain( + // A boundary on a log rebuilt since acquisition (a restart while draining) is recorded + // as it is; the identity keeps the log the caller acquired against. + let rebuilt = FrozenBoundary { + log_id: Uuid::from_u128(0x77), + frame_no: Some(1), + }; + let (next, _) = complete_drain( &record, &receipt, - DrainCompletion::SourceWrites { - boundary: FrozenBoundary { - log_id: Uuid::from_u128(0x77), - frame_no: Some(1), - }, - }, + DrainCompletion::SourceWrites { boundary: rebuilt }, &env(), ) - .unwrap_err(); - assert_eq!(err.detail(), Some(FenceDetail::NamespaceIdentityMismatch)); + .unwrap(); + assert_eq!(next.frozen_boundary, Some(rebuilt)); + assert_eq!(next.identity.log_id, Some(LOG)); // A final receipt, or another operation's. let mut final_receipt = receipt.clone(); From 5a3f983bf38bee79b0ed43f149dbaf9dd38c1125 Mon Sep 17 00:00:00 2001 From: River Date: Tue, 29 Sep 2026 17:20:13 +0000 Subject: [PATCH 12/33] libsql-server: source read fence for SQL with read leases SetSourceReadFence now closes read admission in memory, persists SOURCE_READ_DRAINING, and waits for every read lease already held before it persists SOURCE_READ_FENCED. Leases are taken only after checking the gate under the controller's lease lock, so once admission is closed the set of leases can only shrink. Each SQL program holds a lease for as long as it runs (including a Hrana cursor still producing rows), as does describe and each admin shell query. ATTACH of a namespace is a read of that namespace: the attaching connection keeps a lease on it for every later program until it is detached. A connection idle inside a transaction holds no lease; its next program is refused with MIGRATION_READ_FENCED and rolled back. /beta/listen is refused where reads are denied and ends when reads are fenced. At the deadline (--namespace-fence-default-read-drain-ms, 30s) running programs are cancelled through the connection's progress-handler cancel flag and report the read fence; the drain still waits for the actual releases and answers DRAINING if they do not come, and a replay of the same command resumes it. ClearSourceReadFence reopens reads with writes still fenced. Dump and replication stream leases follow. Co-authored-by: Tomasz Szymczyszyn --- docs/NAMESPACE_FENCE.md | 32 +- libsql-server/src/admin_shell.rs | 114 +++- libsql-server/src/config.rs | 3 + .../src/connection/connection_core.rs | 67 +- libsql-server/src/connection/legacy.rs | 2 +- libsql-server/src/connection/program.rs | 9 +- libsql-server/src/http/user/listen.rs | 40 +- libsql-server/src/main.rs | 9 + .../src/namespace/fence/controller.rs | 332 +++++++++- libsql-server/src/namespace/fence/drain.rs | 42 +- libsql-server/src/namespace/fence/hooks.rs | 4 + libsql-server/src/namespace/fence/mod.rs | 7 +- libsql-server/src/namespace/fence/read.rs | 588 ++++++++++++++++++ libsql-server/src/namespace/meta_store.rs | 15 + libsql-server/src/namespace/mod.rs | 13 +- libsql-server/src/namespace/store.rs | 26 +- 16 files changed, 1248 insertions(+), 55 deletions(-) create mode 100644 libsql-server/src/namespace/fence/read.rs diff --git a/docs/NAMESPACE_FENCE.md b/docs/NAMESPACE_FENCE.md index aac7954a35..a6a6bd5748 100644 --- a/docs/NAMESPACE_FENCE.md +++ b/docs/NAMESPACE_FENCE.md @@ -406,12 +406,13 @@ This is additive in proto3: older peers skip the unknown field; a newer replica Per namespace: - `transition_lock`: a `tokio::sync::Mutex` serialising commands on this namespace. A command holds it from its first check to its response (`FenceController::begin_transition` returns a `Transition` that owns the guard). -- `gate`: a `tokio::sync::watch` of `GateSnapshot { fence, write_generation, indeterminate, installing }`, where `fence` is the durable fence as last published (a record, no record, or `UNKNOWN_UNAVAILABLE` with its detail) and `installing` is the in-memory `INSTALLING` gate of a closing command being persisted (section 8.3), which denies normal writes, vacuum, import writes and lifecycle work on top of `fence`. State, revision, owning operation and every admission (`permits(class)`, `write()`, `read()`) are derived from it through the permission matrix. The WAL wrapper, `CoreConnection`, dump, replication and lifecycle code read it without locks. Phase 5 adds the live capability set. +- `gate`: a `tokio::sync::watch` of `GateSnapshot { fence, write_generation, indeterminate, installing, closing_reads }`, where `fence` is the durable fence as last published (a record, no record, or `UNKNOWN_UNAVAILABLE` with its detail) and `installing` is the in-memory `INSTALLING` gate of a closing command being persisted (section 8.3), which denies normal writes, vacuum, import writes and lifecycle work on top of `fence`. State, revision, owning operation and every admission (`permits(class)`, `write()`, `read()`) are derived from it through the permission matrix. The WAL wrapper, `CoreConnection`, dump, replication and lifecycle code read it without locks. Phase 5 adds the live capability set. - `write_generation: u64`, in the snapshot, **incremented on every publication that changes the fence state, the owning operation, or the indeterminate flag**. That covers every transition that closes or opens write admission (acquire, release, create, seal, publish, enable, abort, adopt), and is conservative for the others. A replay that publishes the same durable state does not move it. - `indeterminate: Option<(operation_id, command_id)>`: set when a command's `COMMIT` failed (or the task running it died) so that whether it applied is unknown. While set, every class except `Maintenance` and `Observability` is denied with `FENCE_STATE_UNAVAILABLE` / `indeterminate_commit`, and every other command is refused with `FENCE_COMMIT_INDETERMINATE`. A replay of the same command is answered by the metastore from the durable row (replayed if it had committed, applied if it had not) and clears it (section 8.4). - the write-drain sources of the namespace's primary connection makers (`register_write_drain`): each maker's connection manager, held weakly, and its replication log id and last-committed-frame reader, which the write drain waits on and reads the boundary from (section 8.3). - the write queues of the namespace's connection managers: each `MakeLegacyConnection` registers a waker (`register_write_queue`) that the controller calls after every publication that moves `write_generation`, after the new gate is visible. A waker holds its manager weakly and is dropped once the manager is gone (for example after eviction). Writer tracking itself is the connection manager's (section 8.2). -- `read_leases`: counters and cancel handles per lease class (`sql`, `dump`, `replication`), with a `Notify` on every release. +- `closing_reads` (in the snapshot): the in-memory read-closing gate of a `SetSourceReadFence` being persisted (section 9, step 2), which denies `NormalRead` and `Stream` with `MIGRATION_READ_FENCED` on top of `fence`. Never persisted; every publication clears it; it does not move `write_generation` (writes are already closed wherever a read fence can be set). +- `read_leases`: the live read leases, each with its kind (`sql`, `dump`, `replication`), a cancel handle and a cancelled flag, and a `Notify` on every release. `FenceController::acquire_read_lease(class, kind, cancel)` checks the gate **under the lease lock** and registers the lease, so a lease is either refused by a read-closing gate published before it, or counted by a drain that closes admission after it; once admission is closed the set can only shrink. A `ReadLease` is released when dropped. - `capabilities`: the live `MigrationCapability` set and an import-writer counter. - in `cfg(test)` builds only, a `FenceTestHooks` (section 16). @@ -499,16 +500,18 @@ This makes the WAL gate independent of statement classification: DDL, misclassif ## 9. Source read fence (Design) -`SetSourceReadFence`, under the transition lock: +`SetSourceReadFence` runs through `FenceController::execute` (`namespace/fence/read.rs`), on a task of its own and under the transition lock for all five steps. The drain policy is the request's, or `--namespace-fence-default-read-drain-ms` (default 30 s); `on_deadline` does not apply to reads, which are always cancelled at the deadline. 1. Checks (section 5.3). -2. Close read admission in memory. New SQL programs, dump requests, replication calls and ATTACHes of this namespace fail with `MIGRATION_READ_FENCED`. -3. CAS `SOURCE_READ_DRAINING` (receipt `DRAINING`). +2. If the source is `SOURCE_WRITE_FENCED` and the gate still admits reads, publish the in-memory read-closing gate (`GateSnapshot::closing_reads`). New SQL programs, dump requests, replication calls and ATTACHes of this namespace fail with `MIGRATION_READ_FENCED`. Where reads are already closed (a resumed drain, a command being reconciled) nothing is closed. A command the checks refuse reopens it. +3. CAS `SOURCE_READ_DRAINING` (receipt `DRAINING`); its publication replaces the read-closing gate. 4. Wait for all read leases to be released: - - **SQL:** a lease is held for the duration of each running program (including a Hrana cursor that is still producing rows). A connection that is idle with an open transaction holds no lease; its next program consults the live gate, fails, and rolls the transaction back. Idle upgraded Hrana WebSocket and HTTP streams may therefore stay open. + - **SQL:** a lease is held for the duration of each running program (`CoreConnection::run`, `FenceConnState::begin_read_program`), including a Hrana cursor that is still producing rows, and for each `describe`. A connection that is idle with an open transaction holds no lease; its next program consults the live gate, fails with `MIGRATION_READ_FENCED`, and rolls the transaction back. Idle upgraded Hrana WebSocket and HTTP streams may therefore stay open. The admin shell, which runs raw SQL, takes a lease per query and is cancelled through the connection's interrupt handle. + - **ATTACH:** attaching a namespace is a `NormalRead` of the attached namespace (its controller comes with the resolved path). The attaching program takes a lease on it; because an attachment outlives the program that made it, the connection remembers it (by schema alias, pruned against `PRAGMA database_list` when a program starts) and every later program on the connection takes a lease on each attached namespace as well, and is refused while any of them denies reads. - **Dump:** a lease is held for the dump stream. The exporter checks a cancel flag between rows. A cancelled dump aborts the HTTP body (the chunked transfer is not completed), so a client never receives a dump that looks complete; the dump text also never reaches its final `COMMIT;`. - **Replication:** `log_entries` and `snapshot` streams register a lease when created. The stream wrapper selects on the gate; on read fence it yields a terminal `FAILED_PRECONDITION` status carrying `MIGRATION_READ_FENCED` and ends. On cancel the wrapper drops the inner stream synchronously, so the lease is released even if the peer never reads again. `hello` and `batch_log_entries` are unary and are simply denied. - - At the deadline, SQL programs are cancelled through the connection's existing progress-handler cancel flag, dumps are cancelled, and streams are terminated. The command keeps waiting for the actual releases; if a lease does not release, the result stays `DRAINING`. + - At the deadline, SQL programs are cancelled through the connection's existing progress-handler cancel flag (a cancelled program rolls back any transaction and reports `MIGRATION_READ_FENCED`, not an interruption), dumps are cancelled, and streams are terminated. The command keeps waiting for the actual releases, for one more deadline but at least 10 seconds; if a lease still does not release, the result stays `DRAINING`, read admission stays closed, and a replay of the same command resumes the wait. That bound only decides when to answer `DRAINING`. + - **`/beta/listen`:** refused at request start where reads are denied (the gate is read without loading the namespace), and the event stream ends with an error event as soon as the gate denies reads. It holds no lease: it serves change notifications, not data, and no change can happen while writes are fenced. 5. CAS `SOURCE_READ_FENCED`; respond. `ClearSourceReadFence` reopens reads (writes stay fenced) with a new revision. @@ -655,7 +658,7 @@ How each path that can reach namespace data or lifecycle is covered. File refere | Path | Coverage | |---|---| -| `connection/connection_core.rs` `CoreConnection::run` | Early live-gate check and `program_generation` capture at program start; SQL read lease for the program; `Error::Fence` from the WAL denial slot in `Vm::try_step`. | +| `connection/connection_core.rs` `CoreConnection::run`, `describe` | Early live-gate check and `program_generation` capture at program start; SQL read lease for the program (and on each attached namespace), refused with `MIGRATION_READ_FENCED` and the open transaction rolled back; cancellation through the progress-handler flag; `Error::NamespaceFence` from the WAL denial slot in `Vm::try_step`. `describe` holds a read lease. | | `connection_core.rs` `checkpoint`, `vacuum_if_needed`, `force_rollback` | Checkpoint is `Maintenance`; vacuum is `Vacuum` and skipped while writes are fenced; `force_rollback` is the drain's abort. | | `connection/connection_manager.rs` | Authoritative gate in `begin_write_txn` before `acquire()`, generation check, non-`BUSY` refusal; classed queue entries; queue wake on generation change; active-writer query, release notification, `abort_active()`. | | `connection/legacy.rs` | `FenceConnState` wired into every `LegacyConnection`; the controller is passed to `MakeLegacyConnection::new` before the first connection. `with_raw` users are covered by the WAL gate. | @@ -667,11 +670,11 @@ How each path that can reach namespace data or lifecycle is covered. File refere | `http/admin/mod.rs` config, create, fork, delete, checkpoint, stats | Config POST, create, fork, delete follow the lifecycle column; config GET, stats and checkpoint are allowed. Fence routes live in `http/admin/fence.rs`. | | `http/user/dump.rs` | Gate check before connection creation (typed, no panic on create error); dump stream lease; cancel flag in the exporter; aborted body on termination. | | `rpc/replication/replication_log.rs` `hello`, `log_entries`, `batch_log_entries`, `snapshot` | Denied at request start; stream leases; typed terminal status for open streams; `ReplicatedFence` in `hello`'s config. | -| `admin_shell.rs` | Writes denied at the WAL (no capability); reads checked against the gate per query. | +| `admin_shell.rs` | Writes denied at the WAL (no capability); reads checked against the gate, and a read lease held, per query (cancelled through the connection's interrupt handle). | | `schema/scheduler.rs`, `database/schema.rs` | Shared schema excluded from fencing; migration writes are WAL-gated; the scheduler's `block_writes` flag is not treated as drain evidence. | | `namespace/configurator/helpers.rs` `load_dump`, `http/admin/mod.rs` `dump_stream_from_url` | Restore options and dump URLs are refused for fenced namespaces; import goes through `ImportSession`. | -| `connection/program.rs` ATTACH resolution | The attached namespace's gate is checked (`NormalRead`), through a non-creating lookup. | -| `http/user/listen.rs` `/beta/listen` | `NormalRead`; denied where reads are denied; ended by the read fence. | +| `connection/program.rs` ATTACH resolution | `check_program_auth` uses a non-creating lookup; the resolver returns the attached namespace's controller, the attachment is admitted as `NormalRead` of it with a read lease, and the connection keeps a lease on it for every later program until it is detached. | +| `http/user/listen.rs` `/beta/listen` | `NormalRead`, read from the registry without loading the namespace; denied where reads are denied; the stream ends with an error event when reads are fenced. | | Raw internal connections (storage monitor, periodic checkpoint, shutdown checkpoint, replication logger, `checkpoint_db`, bottomless) | `Maintenance` / `Observability`; not leases; never blocked. | | DDL, autocommit, explicit transactions, batches, PRAGMAs, read-to-write upgrades | All converge on `begin_write_txn`; classification is only the early check. | @@ -693,7 +696,7 @@ Every transition emits one structured log event (target `libsql_server::fence::a ## 16. Test strategy (Design) -- **No new dependency.** The crate has no failpoint library. Race tests use `#[cfg(test)]` hooks: `FenceTestHooks` holds named points (`AfterInstallingGate`, `BeforeMetastoreCommit`, `AfterMetastoreCommit`, `BeforeGatePublish`, `InBeginWriteTxnAfterCheck`, `AfterManagerRelease`, `BeforeBoundaryCapture`, `AfterTargetRowsCommitted`, `BeforeResponse`), each able to park the task on a pair of `Notify`s (`pause_at` returns handles to wait until the task arrives and to release it) or to inject an error or an indeterminate commit. An armed point fires once. `BeforeMetastoreCommit` is reached immediately before the metastore transaction is started (an injected error there is a failure before commit); an injected indeterminate outcome at `AfterMetastoreCommit` is a commit that happened but was not acknowledged. Hooks compile only in the library's own test build, so they cost nothing in release builds; integration tests under `tests/` cover protocol behaviour and do not rely on hooks. +- **No new dependency.** The crate has no failpoint library. Race tests use `#[cfg(test)]` hooks: `FenceTestHooks` holds named points (`AfterInstallingGate`, `AfterClosingReads`, `BeforeReadLeaseCancel`, `BeforeMetastoreCommit`, `AfterMetastoreCommit`, `BeforeGatePublish`, `InBeginWriteTxnAfterCheck`, `AfterManagerRelease`, `BeforeBoundaryCapture`, `AfterTargetRowsCommitted`, `BeforeResponse`), each able to park the task on a pair of `Notify`s (`pause_at` returns handles to wait until the task arrives and to release it) or to inject an error or an indeterminate commit. An armed point fires once. `BeforeMetastoreCommit` is reached immediately before the metastore transaction is started (an injected error there is a failure before commit); an injected indeterminate outcome at `AfterMetastoreCommit` is a commit that happened but was not acknowledged. Hooks compile only in the library's own test build, so they cost nothing in release builds; integration tests under `tests/` cover protocol behaviour and do not rely on hooks. - **Restart** at a boundary: reopen `MetaStore` and rebuild the registry on the same temporary directory, the way existing metastore tests do; integration tests stop and start a `TestServer` on the same path. - **Restart** at a boundary, as a crash: `namespace::fence::tests` runs each server lifetime on a runtime of its own and ends it by shutting the runtime down without any shutdown code and leaking the `NamespaceStore`, with the command parked at the hook point; the next lifetime opens a new `NamespaceStore` (metastore and registry) on the same directory. The namespace's `.sentinel` stays behind, so the restart takes the real dirty-recovery path. - **Response loss**: the test drops the command future after `AfterMetastoreCommit` and then replays or inspects. @@ -722,7 +725,7 @@ Planned test names; the table is updated as tests land. | 13 | Seal enters `TARGET_IMPORT_DRAINING`, waits, reaches `TARGET_VALIDATING`, cannot resume import; only a durable validation receipt permits idempotent publication | `fence::target::tests::seal_waits_for_import_writers`, `sealed_target_rejects_import`, `publish_requires_validation_receipt`, `publish_is_idempotent` | | 14 | Enable writes idempotent, survives restart and response loss, irreversible | `fence::target::tests::enable_writes_idempotent_and_irreversible`, `enable_writes_survives_restart` | | 15 | Lost `EnableTargetWrites` response resolved from receipt/state | `fence::target::tests::enable_writes_response_loss_resolved` | -| 16 | Read fence drains SQL, dump, `log_entries`, `snapshot`, including dead peers and forced termination | `fence::read::tests::*`; `tests::fence::protocol::read_fence_ends_dump`, `read_fence_ends_log_entries_typed`, `read_fence_ends_snapshot`, `read_fence_forced_termination` | +| 16 | Read fence drains SQL, dump, `log_entries`, `snapshot`, including dead peers and forced termination | SQL landed: `namespace::fence::read::tests::{read_fence_waits_for_running_program, program_after_closing_gate_is_refused (parked after the read-closing gate, before the CAS), read_fence_cancels_at_deadline, unreleased_lease_answers_draining_and_replay_completes, idle_txn_fails_on_next_program, clear_read_fence_reopens_reads_not_writes, refused_read_fence_reopens_reads, attach_of_read_fenced_namespace_denied}`, `admin_shell::fence_tests::admin_shell_read_denied`; planned for dump and replication: `fence::read::tests::{dump_lease_released_on_cancel, log_entries_stream_ends_typed, snapshot_stream_ends_typed, stream_lease_released_without_peer_read, read_fence_forced_termination}`, and an integration test that an interrupted dump never ends with `COMMIT;` | | 17 | Delete, reset, fork, restore, config, schema mutation rejected | `tests::fence::lifecycle::lifecycle_rejected_while_fenced` | | 18 | Codes through HTTP, Hrana, RPC, dump, replication, replica write proxy; distinguishable from auth/timeout/not-found; old peers compatible; no retry loops | `tests::fence::protocol::{http_codes, hrana_http_codes, hrana_ws_codes, rpc_codes, dump_codes, replication_codes, replica_proxy_preserves_code, auth_and_not_found_distinct, denial_not_retried}`; `libsql-replication` `proxy_error_stable_code_is_additive` | | 19 | Corrupt or unknown durable fence state fails closed | `fence::store::tests::corrupt_payload_fails_closed`, `unknown_format_version_fails_closed` | @@ -759,7 +762,8 @@ libsql-server/src/namespace/fence/ store.rs metastore tables, fence CAS, marker file registry.rs FenceRegistry controller.rs FenceController, GateSnapshot, generations, leases - drain.rs write, read and import drains + drain.rs write and import drains + read.rs source read fence and its drain target.rs target lifecycle, MigrationCapability, ImportSession, ValidationSession audit.rs audit events and metrics hooks.rs cfg(test) FenceTestHooks diff --git a/libsql-server/src/admin_shell.rs b/libsql-server/src/admin_shell.rs index e97b272d72..84f11e7fe8 100644 --- a/libsql-server/src/admin_shell.rs +++ b/libsql-server/src/admin_shell.rs @@ -10,6 +10,8 @@ use tonic::metadata::{AsciiMetadataValue, BinaryMetadataValue}; use crate::connection::Connection as _; use crate::database::Connection; +use crate::namespace::fence::controller::{FenceController, LeaseKind}; +use crate::namespace::fence::state::OperationClass; use crate::namespace::{NamespaceName, NamespaceStore}; use self::rpc::admin_shell_service_server::{AdminShellService, AdminShellServiceServer}; @@ -41,17 +43,20 @@ impl AdminShell { queries: impl Stream>, ) -> anyhow::Result>> { let namespace = NamespaceName::from_bytes(ns).unwrap(); - let connection_maker = self + let (connection_maker, fence) = self .namespace_store - .with(namespace, |ns| ns.db.connection_maker()) + .with(namespace, |ns| { + (ns.db.connection_maker(), ns.fence().clone()) + }) .await?; let connection = connection_maker.create().await?; - Ok(run_shell(connection, queries)) + Ok(run_shell(connection, fence, queries)) } } fn run_shell( conn: Connection, + fence: std::sync::Arc, queries: impl Stream>, ) -> impl Stream> { async_stream::stream! { @@ -59,9 +64,7 @@ fn run_shell( while let Some(q) = queries.next().await { let Ok(q) = q else { break }; let res = tokio::task::block_in_place(|| { - conn.with_raw(move |conn| { - run_one(conn, q.query) - }) + conn.with_raw(|conn| run_admitted(&fence, conn, q.query)) }); yield res @@ -69,6 +72,39 @@ fn run_shell( } } +/// Run one shell query as a read of the namespace (`docs/NAMESPACE_FENCE.md` section 9): the +/// shell runs raw SQL, so the read gate is checked, and a read lease held, per query. Writes are +/// refused by the WAL gate like any other connection's. +fn run_admitted( + fence: &std::sync::Arc, + conn: &mut rusqlite::Connection, + q: String, +) -> Result { + let interrupt = conn.get_interrupt_handle(); + let lease = + match fence.acquire_read_lease(OperationClass::NormalRead, LeaseKind::Sql, move || { + interrupt.interrupt() + }) { + Ok(lease) => lease, + Err(e) => { + return Ok(rpc::Response { + resp: Some(Resp::Error(rpc::Error { + error: e.to_string(), + })), + }) + } + }; + let res = run_one(conn, q); + if lease.cancelled_by_fence() { + return Ok(rpc::Response { + resp: Some(Resp::Error(rpc::Error { + error: "the query was cancelled by the namespace read fence".into(), + })), + }); + } + res +} + fn run_one(conn: &mut rusqlite::Connection, q: String) -> Result { match try_run_one(conn, q) { Ok(resp) => Ok(resp), @@ -242,3 +278,69 @@ impl Display for RowsFormatter { Ok(()) } } + +#[cfg(test)] +mod fence_tests { + use uuid::Uuid; + + use super::*; + use crate::namespace::fence::command::{FenceCommand, FenceRequest}; + use crate::namespace::fence::drain::tests::{fence_outcome, raw, Source, LONG, OP}; + use crate::namespace::fence::outcome::FenceOutcome; + + fn error(resp: &rpc::Response) -> &str { + match &resp.resp { + Some(Resp::Error(e)) => &e.error, + other => panic!("expected an error response, got {other:?}"), + } + } + + /// The shell runs raw SQL, so it checks the read gate per query: reads are served while + /// the source is only write-fenced (writes are refused by the WAL gate), and refused once + /// its reads are fenced. + #[tokio::test(flavor = "multi_thread")] + async fn admin_shell_read_denied() { + let s = Source::new().await; + raw(&s.conn().await, "insert into t values (1)") + .await + .unwrap(); + let acquired = s.execute(s.acquire(OP, 1, LONG)).await.unwrap(); + assert_eq!(fence_outcome(&acquired), FenceOutcome::Applied); + + let conn = s.conn().await; + let rows = conn + .with_raw(|c| run_admitted(&s.fence, c, "select count(*) from t".into())) + .unwrap(); + assert!(matches!(rows.resp, Some(Resp::Rows(ref r)) if r.rows.len() == 1)); + let write = conn + .with_raw(|c| run_admitted(&s.fence, c, "insert into t values (2)".into())) + .unwrap(); + assert!(error(&write).contains("authoriz"), "{}", error(&write)); + + let gate = s.fence.gate(); + let fenced = s + .execute(FenceRequest { + namespace: "ns".into(), + operation_id: OP, + command_id: Uuid::from_u128(2), + expected_state: gate.state(), + expected_revision: gate.revision(), + command: FenceCommand::SetSourceReadFence { + drain_policy: Some(LONG), + }, + }) + .await + .unwrap(); + assert_eq!(fence_outcome(&fenced), FenceOutcome::Applied); + + let refused = conn + .with_raw(|c| run_admitted(&s.fence, c, "select count(*) from t".into())) + .unwrap(); + assert!( + error(&refused).starts_with("MIGRATION_READ_FENCED"), + "{}", + error(&refused) + ); + assert_eq!(s.fence.read_lease_counts().total(), 0); + } +} diff --git a/libsql-server/src/config.rs b/libsql-server/src/config.rs index 4457232d03..48dee38dfa 100644 --- a/libsql-server/src/config.rs +++ b/libsql-server/src/config.rs @@ -196,6 +196,9 @@ pub struct MetaStoreConfig { /// How long `AcquireSourceWriteFence` waits for active writers when the request names no /// drain policy. `None` is the default of 30 seconds. pub namespace_fence_default_write_drain: Option, + /// How long `SetSourceReadFence` waits for running reads and streams before it cancels + /// them, when the request names no drain policy. `None` is the default of 30 seconds. + pub namespace_fence_default_read_drain: Option, } #[derive(Debug, Clone)] diff --git a/libsql-server/src/connection/connection_core.rs b/libsql-server/src/connection/connection_core.rs index 212e5c2b3d..821428ecc9 100644 --- a/libsql-server/src/connection/connection_core.rs +++ b/libsql-server/src/connection/connection_core.rs @@ -11,7 +11,8 @@ use crate::connection::legacy::open_conn_active_checkpoint; use crate::error::Error; use crate::metrics::{PROGRAM_EXEC_COUNT, QUERY_CANCELED, VACUUM_COUNT, WAL_CHECKPOINT_COUNT}; use crate::namespace::broadcasters::BroadcasterHandle; -use crate::namespace::fence::controller::FenceConnState; +use crate::namespace::fence::controller::{FenceConnState, LeaseKind}; +use crate::namespace::fence::outcome::{FenceError, FenceOutcome}; use crate::namespace::fence::state::OperationClass; use crate::namespace::meta_store::MetaStoreHandle; use crate::namespace::ResolveNamespacePathFn; @@ -102,6 +103,8 @@ impl CoreConnection { ); let canceled = Arc::new(AtomicBool::new(false)); + // The read drain cancels a running program at its deadline through the same flag. + fence.set_cancel_flag(canceled.clone()); conn.progress_handler(100, { let canceled = canceled.clone(); @@ -218,6 +221,22 @@ impl CoreConnection { // The program is admitted under the gate's current write generation; a write // transaction it opens must start under the same one (section 8.1). fence.begin_program(); + // ...and admitted for reading, holding a read lease for as long as it runs, including + // while a cursor is still producing rows (section 9). A connection left idle in a + // transaction held no lease; its next program is refused here and its transaction is + // rolled back. + let read_lease = { + let lock = this.lock(); + match fence.begin_read_program(|| attached_schemas(lock.raw())) { + Ok(lease) => lease, + Err(e) => { + if !lock.conn.is_autocommit() { + lock.rollback(); + } + return Err(Error::NamespaceFence(e)); + } + } + }; builder.init(&this.lock().builder_config)?; let mut vm = Vm::new( @@ -272,16 +291,40 @@ impl CoreConnection { vm.step(&conn.raw())?; } + if read_lease.cancelled_by_fence() { + // The read drain reached its deadline and interrupted the program: report the fence, + // not the interruption, and leave no transaction behind. + let lock = this.lock(); + if !lock.conn.is_autocommit() { + lock.rollback(); + } + return Err(Error::NamespaceFence(FenceError::new( + FenceOutcome::MigrationReadFenced, + "the program was cancelled by the namespace read fence", + ))); + } + { let lock = this.lock(); let is_autocommit = lock.conn.is_autocommit(); let current_fno = (lock.get_current_frame_no)(); vm.builder().finish(current_fno, is_autocommit)?; } + drop(read_lease); Ok(vm.into_builder()) } + pub(super) fn describe_admitted(&self, sql: &str) -> crate::Result { + // Describing prepares the statement, which reads the schema: it is a read. + let _lease = self.fence.controller().acquire_read_lease( + self.fence.read_class(), + LeaseKind::Sql, + || (), + )?; + self.describe(sql) + } + fn rollback(&self) { if let Err(e) = self.conn.execute("ROLLBACK", ()) { tracing::error!("failed to rollback: {e}"); @@ -425,6 +468,28 @@ impl CoreConnection { } } +/// The schema aliases attached on `conn` (other than `main` and `temp`). +/// `None` when they cannot be listed. +fn attached_schemas(conn: &libsql_sys::Connection) -> Option> { + let mut aliases = Vec::new(); + let result = conn.prepare("PRAGMA database_list").and_then(|mut stmt| { + let mut rows = stmt.query(())?; + while let Some(row) = rows.next()? { + let name: String = row.get(1)?; + if name != "main" && name != "temp" { + aliases.push(name); + } + } + Ok(()) + }); + if let Err(e) = result { + // Keep every recorded attachment: over-leasing is safe, missing one is not. + tracing::warn!("could not list attached schemas: {e}"); + return None; + } + Some(aliases) +} + #[cfg(test)] mod test { use itertools::Itertools; diff --git a/libsql-server/src/connection/legacy.rs b/libsql-server/src/connection/legacy.rs index 71763e199e..efc176cae8 100644 --- a/libsql-server/src/connection/legacy.rs +++ b/libsql-server/src/connection/legacy.rs @@ -438,7 +438,7 @@ where DESCRIBE_COUNT.increment(1); check_describe_auth(ctx)?; let conn = self.inner.clone(); - let res = tokio::task::spawn_blocking(move || conn.lock().describe(&sql)) + let res = tokio::task::spawn_blocking(move || conn.lock().describe_admitted(&sql)) .await .unwrap(); diff --git a/libsql-server/src/connection/program.rs b/libsql-server/src/connection/program.rs index 8cafe681e7..f285343137 100644 --- a/libsql-server/src/connection/program.rs +++ b/libsql-server/src/connection/program.rs @@ -205,10 +205,15 @@ where let attached = attached.strip_prefix('"').unwrap_or(attached); let attached = attached.strip_suffix('"').unwrap_or(attached); let attached = NamespaceName::from_string(attached.into())?; - let path = (self.resolve_attach_path)(&attached)?; + let target = (self.resolve_attach_path)(&attached)?; + // Attaching a namespace reads it: the attachment is admitted by that namespace's fence, + // and the connection holds a read lease on it while its programs run (section 9). + if let Some(fence) = &self.fence { + fence.attach(attached_alias.trim_matches('"'), target.fence)?; + } let query = format!( "ATTACH DATABASE 'file:{}?mode=ro' AS \"{attached_alias}\"", - path.join("data").display() + target.path.join("data").display() ); tracing::trace!("ATTACH rewritten to: {query}"); Ok(query) diff --git a/libsql-server/src/http/user/listen.rs b/libsql-server/src/http/user/listen.rs index 04f95fdfc2..3a63ad32a4 100644 --- a/libsql-server/src/http/user/listen.rs +++ b/libsql-server/src/http/user/listen.rs @@ -1,6 +1,9 @@ use crate::broadcaster::BroadcastMsg; use crate::error::Error; use crate::metrics::{LISTEN_EVENTS_DROPPED, LISTEN_EVENTS_SENT}; +use crate::namespace::fence::controller::{FenceController, GateSnapshot}; +use crate::namespace::fence::outcome::FenceError; +use crate::namespace::fence::state::OperationClass; use crate::{ auth::Authenticated, namespace::{NamespaceName, NamespaceStore}, @@ -18,7 +21,9 @@ use serde::{Deserialize, Serialize}; use std::boxed::Box; use std::convert::Infallible; use std::pin::Pin; +use std::sync::Arc; use std::time::Duration; +use tokio::sync::watch; use tokio_stream::wrappers::errors::BroadcastStreamRecvError; use super::db_factory::namespace_from_headers; @@ -93,11 +98,19 @@ pub(super) async fn handle_listen( return Ok(ListenResponse::Redirect(Redirect::temporary(&url))); } + // Change notifications are a read of the namespace: refused where reads are denied, and + // ended when the namespace's reads are fenced (`docs/NAMESPACE_FENCE.md` section 9). + let fence = state.namespaces.fence_gate(&namespace).await?; + if let Some(fence) = &fence { + fence.permits(OperationClass::NormalRead)?; + } + let stream = sse_stream( state.namespaces.clone(), namespace, query.table.clone(), query.action.clone(), + fence, ) .await; @@ -115,9 +128,10 @@ async fn sse_stream( namespace: NamespaceName, table: String, actions: Option>, + fence: Option>, ) -> SseStream { Box::pin( - listen_stream(store, namespace, table, actions) + listen_stream(store, namespace, table, actions, fence) .await .map(|result| { Ok(match result { @@ -136,12 +150,19 @@ async fn listen_stream( namespace: NamespaceName, table: String, actions: Option>, + fence: Option>, ) -> impl Stream> { async_stream::try_stream! { let _sub = Subscription::new(store.clone(), namespace.clone(), table.clone()); let mut stream = store.subscribe(namespace.clone(), table.clone()); + let mut gate = fence.as_ref().map(|fence| fence.subscribe()); - while let Some(item) = stream.next().await { + loop { + let item = tokio::select! { + item = stream.next() => Ok(item), + denied = read_denied(&mut gate) => Err(Error::NamespaceFence(denied)), + }; + let Some(item) = item? else { break }; match item { Ok(msg) => if filter_actions(&msg, &actions) { LISTEN_EVENTS_SENT.increment(1); @@ -156,6 +177,21 @@ async fn listen_stream( } } +/// Resolves once the gate denies normal reads, with the denial; never without a gate. +async fn read_denied(gate: &mut Option>) -> FenceError { + let Some(gate) = gate else { + return std::future::pending().await; + }; + loop { + if let Err(e) = gate.borrow_and_update().permits(OperationClass::NormalRead) { + return e; + } + if gate.changed().await.is_err() { + return std::future::pending().await; + } + } +} + fn filter_actions(msg: &BroadcastMsg, actions: &Option>) -> bool { actions.as_ref().map_or(true, |actions| { actions.iter().any(|action| { diff --git a/libsql-server/src/main.rs b/libsql-server/src/main.rs index a1ab51c4c4..f8f56e7494 100644 --- a/libsql-server/src/main.rs +++ b/libsql-server/src/main.rs @@ -274,6 +274,12 @@ struct Cli { #[clap(long, env = "SQLD_NAMESPACE_FENCE_DEFAULT_WRITE_DRAIN_MS")] namespace_fence_default_write_drain_ms: Option, + /// How long, in milliseconds, setting a namespace read fence waits for running reads and + /// streams when the request names no drain policy, before it cancels them. Defaults to 30 + /// seconds. + #[clap(long, env = "SQLD_NAMESPACE_FENCE_DEFAULT_READ_DRAIN_MS")] + namespace_fence_default_read_drain_ms: Option, + /// Shutdown timeout duration in seconds, defaults to 30 seconds. #[clap(long, env = "SQLD_SHUTDOWN_TIMEOUT")] shutdown_timeout: Option, @@ -673,6 +679,9 @@ fn make_meta_store_config(config: &Cli) -> anyhow::Result { namespace_fence_default_write_drain: config .namespace_fence_default_write_drain_ms .map(Duration::from_millis), + namespace_fence_default_read_drain: config + .namespace_fence_default_read_drain_ms + .map(Duration::from_millis), }) } diff --git a/libsql-server/src/namespace/fence/controller.rs b/libsql-server/src/namespace/fence/controller.rs index 7e718e5516..ff946554c6 100644 --- a/libsql-server/src/namespace/fence/controller.rs +++ b/libsql-server/src/namespace/fence/controller.rs @@ -10,11 +10,12 @@ //! namespace cache, so evicting and reloading a namespace hands the reloaded namespace the //! same controller. -use std::sync::atomic::{AtomicU64, Ordering}; +use std::collections::HashMap; +use std::sync::atomic::{AtomicBool, AtomicU64, Ordering}; use std::sync::Arc; use parking_lot::Mutex; -use tokio::sync::{watch, OwnedMutexGuard}; +use tokio::sync::{watch, Notify, OwnedMutexGuard}; use uuid::Uuid; use crate::connection::connection_manager::{ConnectionManager, WeakConnectionManager}; @@ -52,6 +53,11 @@ pub struct GateSnapshot { /// (section 8.3, step 2): write admission is closed on top of whatever `fence` allows. /// Never persisted. pub installing: Option, + /// The in-memory read-closing gate of a `SetSourceReadFence` that is being persisted + /// (section 9, step 2): normal reads and streams are refused on top of whatever `fence` + /// allows. Never persisted, and cleared by every publication of a commit. It does not move + /// the write generation: write admission is already closed wherever a read fence can be set. + pub closing_reads: Option, } impl GateSnapshot { @@ -61,6 +67,7 @@ impl GateSnapshot { write_generation: 0, indeterminate: None, installing: None, + closing_reads: None, } } @@ -115,6 +122,17 @@ impl GateSnapshot { )); } } + if let Some((operation_id, command_id)) = self.closing_reads { + if matches!(class, OperationClass::NormalRead | OperationClass::Stream) { + return Err(FenceError::new( + FenceOutcome::MigrationReadFenced, + format!( + "{class:?} is not permitted: fence command {command_id} of operation \ + {operation_id} is closing read admission" + ), + )); + } + } Ok(()) } @@ -177,6 +195,84 @@ pub(crate) struct LiveWriteDrain { pub(crate) current_frame_no: GetCurrentFrameNo, } +/// What a read lease covers (section 9). +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] +pub enum LeaseKind { + /// A running SQL program (including a Hrana cursor producing rows), or an ATTACH of the + /// namespace by a program running on another namespace. + Sql, + /// A `/dump` stream. + Dump, + /// A replication `log_entries` or `snapshot` stream. + Replication, +} + +/// The number of read leases held on a namespace, by kind. +#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)] +pub struct ReadLeaseCounts { + pub sql: usize, + pub dump: usize, + pub replication: usize, +} + +impl ReadLeaseCounts { + pub fn total(&self) -> usize { + self.sql + self.dump + self.replication + } +} + +struct LeaseEntry { + kind: LeaseKind, + cancel: Box, + cancelled: Arc, +} + +#[derive(Default)] +struct ReadLeaseSet { + next_id: u64, + live: HashMap, +} + +/// Read work admitted by the gate and counted by the read drain until it is dropped +/// ([`FenceController::acquire_read_lease`]). +pub struct ReadLease { + controller: Arc, + id: u64, + kind: LeaseKind, + cancelled: Arc, +} + +impl ReadLease { + pub fn kind(&self) -> LeaseKind { + self.kind + } + + pub fn controller(&self) -> &Arc { + &self.controller + } + + /// Whether the read drain cancelled this lease's work at its deadline. + pub fn cancelled_by_fence(&self) -> bool { + self.cancelled.load(Ordering::Acquire) + } +} + +impl std::fmt::Debug for ReadLease { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("ReadLease") + .field("namespace", &self.controller.namespace) + .field("id", &self.id) + .field("kind", &self.kind) + .finish() + } +} + +impl Drop for ReadLease { + fn drop(&mut self) { + self.controller.release_read_lease(self.id); + } +} + /// The fence controller of one namespace. pub struct FenceController { namespace: NamespaceName, @@ -187,6 +283,10 @@ pub struct FenceController { /// What the write drain needs from each of the namespace's primary connection makers /// (section 8.3). write_drains: Mutex>, + /// The read leases held on the namespace (section 9). + read_leases: Mutex, + /// Notified whenever a read lease is released. + read_released: Notify, #[cfg(test)] hooks: FenceTestHooks, } @@ -210,6 +310,8 @@ impl FenceController { gate, write_queues: Mutex::new(Vec::new()), write_drains: Mutex::new(Vec::new()), + read_leases: Mutex::new(ReadLeaseSet::default()), + read_released: Notify::new(), #[cfg(test)] hooks: FenceTestHooks::default(), }) @@ -281,6 +383,82 @@ impl FenceController { live } + /// Admit read work of `class` and hold a read lease of `kind` for it (section 9). The gate + /// is checked under the lease lock, so a read fence that closed admission before this call + /// is seen here, and one that closes after it counts this lease and waits for it. `cancel` + /// is how the read drain stops the work at its deadline: it must make the work end and + /// drop the lease, never wait for it to end. The lease is released when dropped. + pub fn acquire_read_lease( + self: &Arc, + class: OperationClass, + kind: LeaseKind, + cancel: impl Fn() + Send + Sync + 'static, + ) -> Result { + let cancelled = Arc::new(AtomicBool::new(false)); + let id = { + let mut leases = self.read_leases.lock(); + self.permits(class)?; + let id = leases.next_id; + leases.next_id += 1; + leases.live.insert( + id, + LeaseEntry { + kind, + cancel: Box::new(cancel), + cancelled: cancelled.clone(), + }, + ); + id + }; + Ok(ReadLease { + controller: self.clone(), + id, + kind, + cancelled, + }) + } + + /// The read leases currently held, by kind. + pub fn read_lease_counts(&self) -> ReadLeaseCounts { + let leases = self.read_leases.lock(); + let mut counts = ReadLeaseCounts::default(); + for entry in leases.live.values() { + match entry.kind { + LeaseKind::Sql => counts.sql += 1, + LeaseKind::Dump => counts.dump += 1, + LeaseKind::Replication => counts.replication += 1, + } + } + counts + } + + /// Cancel every read lease held now (the read drain's deadline). Each lease's work is asked + /// to stop once; the leases stay counted until they are actually released. Returns how many + /// were asked. + pub(crate) fn cancel_read_leases(&self) -> usize { + let leases = self.read_leases.lock(); + let mut asked = 0; + for entry in leases.live.values() { + if !entry.cancelled.swap(true, Ordering::AcqRel) { + (entry.cancel)(); + asked += 1; + } + } + asked + } + + /// Notified on every read-lease release. Enable the notification before checking + /// [`read_lease_counts`](Self::read_lease_counts), so a release in between is not missed. + pub(crate) fn read_released(&self) -> &Notify { + &self.read_released + } + + fn release_read_lease(&self, id: u64) { + let removed = self.read_leases.lock().live.remove(&id); + drop(removed); + self.read_released.notify_waiters(); + } + /// Take the namespace's transition lock. Every fence command on the namespace runs while /// holding it, from its first check to its response. pub async fn begin_transition(self: &Arc) -> Transition { @@ -330,7 +508,8 @@ impl FenceController { /// Publish a new gate. `fence: None` keeps the published fence. The write generation moves /// whenever the state, the owning operation, the indeterminate flag or the installing gate - /// changes. + /// changes. Every publication removes the read-closing gate: the commit that follows it + /// either persists the read fence or proves that nothing changed. fn publish( &self, fence: Option, @@ -348,6 +527,7 @@ impl FenceController { gate.fence = fence; gate.indeterminate = indeterminate; gate.installing = installing; + gate.closing_reads = None; if changed { gate.write_generation += 1; generation_changed = true; @@ -406,6 +586,35 @@ impl Transition { } } + /// Publish the in-memory read-closing gate for `SetSourceReadFence` `key` (section 9, + /// step 2): new SQL programs, dumps, replication calls and ATTACHes of the namespace are + /// refused with `MIGRATION_READ_FENCED`. A read lease is only ever taken after checking the + /// gate under the lease lock, so every lease taken once this returns was refused, and the + /// drain only has to wait for the leases already held. It is replaced by whatever the + /// command's commit publishes, or removed with + /// [`reopen_read_admission`](Self::reopen_read_admission) when the command is proven not to + /// have committed. + pub fn close_read_admission(&mut self, key: CommandKey) { + self.controller + .gate + .send_modify(|gate| gate.closing_reads = Some(key)); + tracing::debug!( + namespace = %self.controller.namespace, + operation_id = %key.0, + command_id = %key.1, + "closed namespace read admission" + ); + } + + /// Remove the read-closing gate of a command that was proven not to have committed. + pub fn reopen_read_admission(&mut self) { + self.controller.gate.send_if_modified(|gate| { + let was_closing = gate.closing_reads.is_some(); + gate.closing_reads = None; + was_closing + }); + } + /// Commit `request` in the metastore and publish the result. pub async fn apply( &mut self, @@ -535,6 +744,34 @@ fn pending_indeterminate((operation_id, command_id): CommandKey) -> FenceError { .with_detail(FenceDetail::IndeterminateCommit) } +/// The read leases of one running program ([`FenceConnState::begin_read_program`]), released +/// when dropped. +#[derive(Debug)] +pub struct ProgramReadLease { + conn: Arc, + lease: ReadLease, +} + +impl ProgramReadLease { + /// Whether the read drain cancelled the program at its deadline. + pub fn cancelled_by_fence(&self) -> bool { + self.lease.cancelled_by_fence() + || self + .conn + .attached_leases + .lock() + .iter() + .any(ReadLease::cancelled_by_fence) + } +} + +impl Drop for ProgramReadLease { + fn drop(&mut self) { + let attached = std::mem::take(&mut *self.conn.attached_leases.lock()); + drop(attached); + } +} + /// The fence state of one connection, shared by its WAL wrapper and its `CoreConnection` /// (section 7.4). /// @@ -554,6 +791,14 @@ pub struct FenceConnState { txn_generation: AtomicU64, /// The typed outcome of the last refusal at the WAL. denial: Mutex>, + /// The connection's cancel flag (its progress handler interrupts the running statement + /// while it is set), through which the read drain cancels a program at its deadline. + cancel: std::sync::OnceLock>, + /// The namespaces attached on this connection, by schema alias. An attachment outlives the + /// program that made it, so every later program takes a read lease on each of them too. + attached: Mutex)>>, + /// The read leases the running program holds on attached namespaces. + attached_leases: Mutex>, } impl FenceConnState { @@ -565,9 +810,90 @@ impl FenceConnState { program_generation: AtomicU64::new(generation), txn_generation: AtomicU64::new(generation), denial: Mutex::new(None), + cancel: std::sync::OnceLock::new(), + attached: Mutex::new(Vec::new()), + attached_leases: Mutex::new(Vec::new()), }) } + /// The flag that cancels the statement running on this connection. Set once, by the + /// connection that owns this state. + pub fn set_cancel_flag(&self, flag: Arc) { + let _ = self.cancel.set(flag); + } + + /// The class this connection's reads are admitted as: a normal connection reads as + /// `NormalRead`; a capability connection reads under its capability. + pub fn read_class(&self) -> OperationClass { + match self.class { + OperationClass::NormalWrite => OperationClass::NormalRead, + class => class, + } + } + + fn lease_canceller(&self) -> impl Fn() + Send + Sync + 'static { + let flag = self.cancel.get().cloned(); + move || { + if let Some(flag) = &flag { + flag.store(true, Ordering::Relaxed); + } + } + } + + /// Admit a program for reading and hold its read leases until the returned guard is + /// dropped (section 9): one on this connection's namespace and one on each namespace + /// attached on the connection. `still_attached` lists the aliases attached on the + /// connection now (it is only asked when [`attach`](Self::attach) recorded one); `None` + /// keeps every recorded attachment. + pub fn begin_read_program( + self: &Arc, + still_attached: impl FnOnce() -> Option>, + ) -> Result { + let lease = self.controller.acquire_read_lease( + self.read_class(), + LeaseKind::Sql, + self.lease_canceller(), + )?; + let guard = ProgramReadLease { + conn: self.clone(), + lease, + }; + let attached = { + let mut attached = self.attached.lock(); + if !attached.is_empty() { + if let Some(live) = still_attached() { + attached.retain(|(alias, _)| live.iter().any(|a| a == alias)); + } + } + attached.clone() + }; + for (_, controller) in attached { + let lease = controller.acquire_read_lease( + OperationClass::NormalRead, + LeaseKind::Sql, + self.lease_canceller(), + )?; + self.attached_leases.lock().push(lease); + } + Ok(guard) + } + + /// The running program is attaching `controller`'s namespace as `alias`: admit it as a + /// normal read of that namespace, hold a lease on it for the rest of the program, and + /// remember the attachment for the connection's later programs. + pub fn attach(&self, alias: &str, controller: Arc) -> Result<(), FenceError> { + let lease = controller.acquire_read_lease( + OperationClass::NormalRead, + LeaseKind::Sql, + self.lease_canceller(), + )?; + self.attached_leases.lock().push(lease); + let mut attached = self.attached.lock(); + attached.retain(|(a, _)| a != alias); + attached.push((alias.to_owned(), controller)); + Ok(()) + } + pub fn controller(&self) -> &Arc { &self.controller } diff --git a/libsql-server/src/namespace/fence/drain.rs b/libsql-server/src/namespace/fence/drain.rs index 6165784f60..5cc70563b8 100644 --- a/libsql-server/src/namespace/fence/drain.rs +++ b/libsql-server/src/namespace/fence/drain.rs @@ -56,6 +56,9 @@ impl FenceController { FenceCommand::AcquireSourceWriteFence { .. } => { acquire_source_write_fence(&mut transition, &meta, request, ctx).await } + FenceCommand::SetSourceReadFence { .. } => { + super::read::set_source_read_fence(&mut transition, &meta, request, ctx).await + } _ => transition.apply(&meta, request, ctx).await, } }) @@ -277,14 +280,14 @@ fn capture_boundary( Ok(FrozenBoundary { log_id, frame_no }) } -fn now_ms() -> i64 { +pub(super) fn now_ms() -> i64 { std::time::SystemTime::now() .duration_since(std::time::UNIX_EPOCH) .map_or(0, |d| i64::try_from(d.as_millis()).unwrap_or(i64::MAX)) } #[cfg(test)] -mod tests { +pub(crate) mod tests { use std::sync::Arc; use std::time::Duration; @@ -306,27 +309,27 @@ mod tests { use crate::namespace::RestoreOption; use crate::replication::primary::logger::ReplicationLogger; - const OP: Uuid = Uuid::from_u128(0xa); + pub(crate) const OP: Uuid = Uuid::from_u128(0xa); const OTHER_OP: Uuid = Uuid::from_u128(0xb); /// Long enough that no test ever reaches it: a drain must never finish because of time. - const LONG: DrainPolicy = DrainPolicy { + pub(crate) const LONG: DrainPolicy = DrainPolicy { deadline_ms: 600_000, on_deadline: OnDeadline::Fail, }; - const PROMPT: Duration = Duration::from_secs(30); + pub(crate) const PROMPT: Duration = Duration::from_secs(30); /// A primary namespace `ns` with a table `t`, served by a real `NamespaceStore`, so that the /// drain goes through the connection manager and replication logger the configurator /// registered. - struct Source { + pub(crate) struct Source { _dir: TempDir, - store: NamespaceStore, - fence: Arc, + pub(crate) store: NamespaceStore, + pub(crate) fence: Arc, logger: Arc, } impl Source { - async fn new() -> Self { + pub(crate) async fn new() -> Self { let dir = tempdir().unwrap(); let store = open_store(dir.path()).await; store @@ -362,7 +365,7 @@ mod tests { this } - async fn conn(&self) -> Arc { + pub(crate) async fn conn(&self) -> Arc { let maker = self .store .with("ns".into(), |ns| ns.db.connection_maker()) @@ -380,7 +383,12 @@ mod tests { *self.logger.new_frame_notifier.borrow() } - fn acquire(&self, op: Uuid, command_id: u128, policy: DrainPolicy) -> FenceRequest { + pub(crate) fn acquire( + &self, + op: Uuid, + command_id: u128, + policy: DrainPolicy, + ) -> FenceRequest { FenceRequest { namespace: "ns".into(), operation_id: op, @@ -395,7 +403,7 @@ mod tests { } /// Run `request` through the store, on a task of its own. - fn execute( + pub(crate) fn execute( &self, request: FenceRequest, ) -> tokio::task::JoinHandle> { @@ -414,7 +422,7 @@ mod tests { } /// Wait until the published gate is in `state`. - async fn until_state(&self, state: FenceState) { + pub(crate) async fn until_state(&self, state: FenceState) { let mut rx = self.fence.subscribe(); tokio::time::timeout(PROMPT, rx.wait_for(|g| g.state() == state)) .await @@ -422,7 +430,7 @@ mod tests { .unwrap(); } - async fn count(&self) -> i64 { + pub(crate) async fn count(&self) -> i64 { let conn = self.conn().await; tokio::task::spawn_blocking(move || { conn.with_raw(|c| c.query_row("select count(*) from t", (), |r| r.get(0))) @@ -435,14 +443,14 @@ mod tests { /// Run `sql` as one raw program on `conn`, off the async runtime (it can block on the write /// slot). - async fn raw(conn: &Arc, sql: &'static str) -> rusqlite::Result<()> { + pub(crate) async fn raw(conn: &Arc, sql: &'static str) -> rusqlite::Result<()> { let conn = conn.clone(); tokio::task::spawn_blocking(move || conn.with_raw(|c| c.execute_batch(sql))) .await .unwrap() } - fn assert_fenced(result: rusqlite::Result<()>) { + pub(crate) fn assert_fenced(result: rusqlite::Result<()>) { match result { Err(rusqlite::Error::SqliteFailure(e, _)) => { assert_eq!(e.code, ErrorCode::AuthorizationForStatementDenied, "{e}") @@ -451,7 +459,7 @@ mod tests { } } - fn fence_outcome(result: &crate::Result) -> FenceOutcome { + pub(crate) fn fence_outcome(result: &crate::Result) -> FenceOutcome { match result { Ok(c) => c.receipt.outcome, Err(Error::NamespaceFence(e)) => e.outcome(), diff --git a/libsql-server/src/namespace/fence/hooks.rs b/libsql-server/src/namespace/fence/hooks.rs index 28a00c673e..75631e122c 100644 --- a/libsql-server/src/namespace/fence/hooks.rs +++ b/libsql-server/src/namespace/fence/hooks.rs @@ -27,6 +27,10 @@ pub enum HookPoint { AfterInstallingGate, /// Under the transition lock, immediately before the metastore transaction runs. BeforeMetastoreCommit, + /// The in-memory read-closing gate of `SetSourceReadFence` has been published. + AfterClosingReads, + /// The read drain's deadline passed; the leases still held are about to be cancelled. + BeforeReadLeaseCancel, /// The metastore transaction returned a committed result. AfterMetastoreCommit, /// The committed result is about to be published to the gate. diff --git a/libsql-server/src/namespace/fence/mod.rs b/libsql-server/src/namespace/fence/mod.rs index 7e64e0cd7d..5c6f5d4382 100644 --- a/libsql-server/src/namespace/fence/mod.rs +++ b/libsql-server/src/namespace/fence/mod.rs @@ -8,9 +8,9 @@ //! markers with their strict durable encoding ([`record`]), the pure transition function //! ([`transition`]), the metastore tables, compare-and-swap and marker file that persist them //! ([`store`], driven by `MetaStore::apply_fence_command`), and the in-memory authority built -//! on them: the per-namespace [`controller`] with its gate, the positive write [`drain`], the -//! [`registry`] that holds the controllers outside the namespace cache, and the test [`hooks`] -//! on their paths. +//! on them: the per-namespace [`controller`] with its gate and read leases, the positive write +//! [`drain`], the source [`read`] fence, the [`registry`] that holds the controllers outside the +//! namespace cache, and the test [`hooks`] on their paths. // The persistence, controller and protocol layers that consume these types land in the // following commits of this series; until then most of the module is unused by the rest of @@ -22,6 +22,7 @@ pub mod controller; pub mod drain; pub mod hooks; pub mod outcome; +pub mod read; pub mod record; pub mod registry; pub mod state; diff --git a/libsql-server/src/namespace/fence/read.rs b/libsql-server/src/namespace/fence/read.rs new file mode 100644 index 0000000000..cc4a60b906 --- /dev/null +++ b/libsql-server/src/namespace/fence/read.rs @@ -0,0 +1,588 @@ +//! The source read fence (`docs/NAMESPACE_FENCE.md` section 9). +//! +//! `SetSourceReadFence` closes read admission in memory, persists `SOURCE_READ_DRAINING`, and +//! then waits for every read lease that was already held to be released: running SQL programs +//! (including cursors still producing rows and ATTACHes of the namespace from other +//! namespaces), dumps and replication streams. Leases are taken only after checking the gate +//! under the controller's lease lock, so once admission is closed the set can only shrink. At the +//! deadline the remaining work is cancelled and the drain keeps waiting for the actual +//! releases; it never takes elapsed time as proof that a reader has finished. Once no lease is +//! held it persists `SOURCE_READ_FENCED`. + +use std::time::Duration; + +use tokio::time::Instant; + +use crate::namespace::meta_store::{FenceCommit, FenceContext, MetaStore}; + +use super::command::{DrainPolicy, FenceCommand, FenceRequest}; +use super::controller::{FenceController, Transition}; +use super::drain::{now_ms, FORCED_ROLLBACK_GRACE}; +use super::hooks::HookPoint; +use super::outcome::FenceOutcome; +use super::state::{FenceState, OperationClass}; +use super::transition::DrainCompletion; + +/// How long `SetSourceReadFence` waits for running reads and streams before it cancels them, +/// when neither the request nor `--namespace-fence-default-read-drain-ms` names a deadline. +pub const DEFAULT_READ_DRAIN: Duration = Duration::from_secs(30); + +/// `SetSourceReadFence`, steps 1 to 5 of section 9, under `transition`. +/// +/// Returns the `APPLIED` commit of `SOURCE_READ_FENCED` once every read lease is released, the +/// `DRAINING` commit of `SOURCE_READ_DRAINING` when leases are still held after the deadline and +/// the cancellation that follows it (read admission stays closed and a replay of the same +/// command resumes the drain), or the stored result of a replay. +pub async fn set_source_read_fence( + transition: &mut Transition, + meta: &MetaStore, + request: FenceRequest, + mut ctx: FenceContext, +) -> crate::Result { + let controller = transition.controller().clone(); + let key = (request.operation_id, request.command_id); + let policy = match &request.command { + FenceCommand::SetSourceReadFence { drain_policy } => { + drain_policy.unwrap_or_else(|| meta.fence_default_read_drain()) + } + _ => return transition.apply(meta, request, ctx).await, + }; + + // Step 2: close read admission in memory. From here on every new read lease is refused, so + // the drain only waits for the leases already held. Where reads are already closed (a + // resumed drain, a command being reconciled, a read-fenced namespace) nothing is closed. + // A command the checks refuse (step 1, the metastore's) reopens it. + if controller.permits(OperationClass::NormalRead).is_ok() + && controller.gate().state() == FenceState::SourceWriteFenced + { + transition.close_read_admission(key); + let _ = controller.hook(HookPoint::AfterClosingReads).await; + } + + // Step 3: persist SOURCE_READ_DRAINING; its publication replaces the in-memory gate. + let commit = match transition.apply(meta, request, ctx.clone()).await { + Ok(commit) => commit, + Err(e) => { + transition.reopen_read_admission(); + return Err(e); + } + }; + if commit.receipt.outcome != FenceOutcome::Draining { + return Ok(commit); + } + let drain_key = (commit.receipt.operation_id, commit.receipt.command_id); + + // Step 4. + if !drain_readers(&controller, policy).await { + return Ok(commit); + } + + // Step 5. + ctx.now_ms = now_ms(); + transition + .complete_drain(meta, drain_key, DrainCompletion::SourceReads, ctx) + .await +} + +/// Wait until every read lease of the namespace is released. At the deadline the leases still +/// held are cancelled (SQL programs through their connection's cancel flag, dumps and streams +/// through their own), and the drain keeps waiting for the actual releases for one more +/// deadline, but at least [`FORCED_ROLLBACK_GRACE`]. `false` when leases are still held then: +/// the answer is `DRAINING`, never a guess. +async fn drain_readers(controller: &FenceController, policy: DrainPolicy) -> bool { + let namespace = controller.namespace().clone(); + let deadline_after = Duration::from_millis(policy.deadline_ms); + let deadline = Instant::now() + deadline_after; + if wait_for_read_leases(controller, deadline).await { + return true; + } + let _ = controller.hook(HookPoint::BeforeReadLeaseCancel).await; + let held = controller.read_lease_counts(); + let asked = controller.cancel_read_leases(); + tracing::info!( + %namespace, + deadline_ms = policy.deadline_ms, + sql = held.sql, + dump = held.dump, + replication = held.replication, + cancelled = asked, + "read drain deadline passed; cancelling the reads and streams still running" + ); + let grace = Instant::now() + deadline_after.max(FORCED_ROLLBACK_GRACE); + if wait_for_read_leases(controller, grace).await { + return true; + } + let held = controller.read_lease_counts(); + tracing::warn!( + %namespace, + sql = held.sql, + dump = held.dump, + replication = held.replication, + "read leases still held after cancellation; answering DRAINING" + ); + false +} + +/// Wait, on the controller's release notification, until no read lease is held. `false` when +/// `deadline` passes first. +async fn wait_for_read_leases(controller: &FenceController, deadline: Instant) -> bool { + loop { + let released = controller.read_released().notified(); + tokio::pin!(released); + // Registered before the check, so a release in between is not missed. + released.as_mut().enable(); + if controller.read_lease_counts().total() == 0 { + return true; + } + tokio::select! { + _ = &mut released => {} + _ = tokio::time::sleep_until(deadline) => return false, + } + } +} + +#[cfg(test)] +mod tests { + use std::sync::Arc; + use std::time::Duration; + + use rusqlite::functions::FunctionFlags; + use uuid::Uuid; + + use super::*; + use crate::auth::Authenticated; + use crate::connection::config::DatabaseConfig; + use crate::connection::program::Program; + use crate::connection::{Connection as _, RequestContext}; + use crate::database::Connection; + use crate::error::Error; + use crate::namespace::fence::command::OnDeadline; + use crate::namespace::fence::drain::tests::{fence_outcome, raw, Source, LONG, OP, PROMPT}; + use crate::namespace::fence::outcome::FenceError; + use crate::namespace::RestoreOption; + use crate::query_result_builder::test::{StepResult, TestBuilder}; + use crate::query_result_builder::QueryResultBuilder as _; + + const OTHER_OP: Uuid = Uuid::from_u128(0xb); + /// A deadline that has already passed: the drain cancels at once. + const NOW: DrainPolicy = DrainPolicy { + deadline_ms: 0, + on_deadline: OnDeadline::Fail, + }; + + /// A write-fenced source. + async fn fenced_source() -> Source { + let s = Source::new().await; + raw(&s.conn().await, "insert into t values (1), (2)") + .await + .unwrap(); + let acquired = s.execute(s.acquire(OP, 1, LONG)).await.unwrap(); + assert_eq!(fence_outcome(&acquired), FenceOutcome::Applied); + s + } + + fn request(s: &Source, op: Uuid, command_id: u128, command: FenceCommand) -> FenceRequest { + let gate = s.fence.gate(); + FenceRequest { + namespace: "ns".into(), + operation_id: op, + command_id: Uuid::from_u128(command_id), + expected_state: gate.state(), + expected_revision: gate.revision(), + command, + } + } + + fn read_fence(s: &Source, command_id: u128, policy: DrainPolicy) -> FenceRequest { + request( + s, + OP, + command_id, + FenceCommand::SetSourceReadFence { + drain_policy: Some(policy), + }, + ) + } + + fn clear_read_fence(s: &Source, command_id: u128) -> FenceRequest { + request(s, OP, command_id, FenceCommand::ClearSourceReadFence) + } + + async fn conn_to(s: &Source, ns: &'static str) -> Arc { + let maker = s + .store + .with(ns.into(), |ns| ns.db.connection_maker()) + .await + .unwrap(); + Arc::new(maker.create().await.unwrap()) + } + + /// Run `stmts` as one program on `conn`, the way a SQL request runs. + async fn program( + s: &Source, + ns: &'static str, + conn: &Arc, + stmts: &'static [&'static str], + ) -> crate::Result> { + let ctx = RequestContext::new( + Authenticated::FullAccess, + ns.into(), + s.store.meta_store().clone(), + ); + conn.execute_program(Program::seq(stmts), ctx, TestBuilder::default(), None) + .await + .map(|b| b.into_ret()) + } + + /// [`program`], on a task of its own. + fn spawn_program( + s: &Source, + ns: &'static str, + conn: &Arc, + stmts: &'static [&'static str], + ) -> tokio::task::JoinHandle>> { + let ctx = RequestContext::new( + Authenticated::FullAccess, + ns.into(), + s.store.meta_store().clone(), + ); + let conn = conn.clone(); + tokio::spawn(async move { + conn.execute_program(Program::seq(stmts), ctx, TestBuilder::default(), None) + .await + .map(|b| b.into_ret()) + }) + } + + fn read_fenced(e: &Error) -> &FenceError { + match e { + Error::NamespaceFence(f) if f.outcome() == FenceOutcome::MigrationReadFenced => f, + other => panic!("expected MIGRATION_READ_FENCED, got {other:?}"), + } + } + + fn step_read_fenced(step: &StepResult) { + match step { + Err(e) => { + read_fenced(e); + } + Ok(rows) => panic!("expected MIGRATION_READ_FENCED, got rows {rows:?}"), + } + } + + fn assert_ok(steps: &[StepResult]) { + for (i, step) in steps.iter().enumerate() { + assert!(step.is_ok(), "step {i} failed: {step:?}"); + } + } + + /// A `park()` SQL function on `conn`: it signals the returned receiver when a program + /// reaches it, and returns once the returned sender is used. SQLite cannot interrupt it, + /// so a program parked in it holds its read lease until the test lets it go. + fn park( + conn: &Connection, + ) -> ( + tokio::sync::mpsc::UnboundedReceiver<()>, + std::sync::mpsc::Sender<()>, + ) { + let (reached_tx, reached_rx) = tokio::sync::mpsc::unbounded_channel::<()>(); + let (resume_tx, resume_rx) = std::sync::mpsc::channel::<()>(); + let parked = std::panic::AssertUnwindSafe((reached_tx, std::sync::Mutex::new(resume_rx))); + conn.with_raw(move |c| { + c.create_scalar_function("park", 0, FunctionFlags::SQLITE_UTF8, move |_| { + let (reached, resume) = &*parked; + reached.send(()).unwrap(); + resume.lock().unwrap().recv().unwrap(); + Ok(1) + }) + }) + .unwrap(); + (reached_rx, resume_tx) + } + + /// A `reached()` SQL function on `conn` that signals the returned receiver and returns. + fn signal(conn: &Connection) -> tokio::sync::mpsc::UnboundedReceiver<()> { + let (tx, rx) = tokio::sync::mpsc::unbounded_channel::<()>(); + let tx = std::panic::AssertUnwindSafe(tx); + conn.with_raw(move |c| { + c.create_scalar_function("reached", 0, FunctionFlags::SQLITE_UTF8, move |_| { + tx.send(()).unwrap(); + Ok(1) + }) + }) + .unwrap(); + rx + } + + /// A program that is running when the read fence arrives keeps its lease: the drain waits + /// for it and acknowledges only after it finished. Programs that start meanwhile are + /// refused. + #[tokio::test(flavor = "multi_thread")] + async fn read_fence_waits_for_running_program() { + let s = fenced_source().await; + let conn = s.conn().await; + let (mut reached, resume) = park(&conn); + let running = spawn_program( + &s, + "ns", + &conn, + &["select park()", "select count(*) from t"], + ); + reached.recv().await.unwrap(); + assert_eq!(s.fence.read_lease_counts().sql, 1); + + let fence = s.execute(read_fence(&s, 2, LONG)); + s.until_state(FenceState::SourceReadDraining).await; + // New work is refused while the old program still runs. + let refused = program(&s, "ns", &s.conn().await, &["select count(*) from t"]).await; + read_fenced(&refused.unwrap_err()); + assert!(!fence.is_finished()); + + resume.send(()).unwrap(); + let steps = running.await.unwrap().unwrap(); + assert_ok(&steps); + let fenced = fence.await.unwrap(); + assert_eq!(fence_outcome(&fenced), FenceOutcome::Applied); + assert_eq!(s.fence.gate().state(), FenceState::SourceReadFenced); + assert_eq!(s.fence.read_lease_counts().total(), 0); + } + + /// A program admitted after the read-closing gate is published, but before the state is + /// persisted, is refused: the drain never has to wait for work that started after it closed + /// admission. + #[tokio::test(flavor = "multi_thread")] + async fn program_after_closing_gate_is_refused() { + let s = fenced_source().await; + let closed = s.fence.hooks().pause_at(HookPoint::AfterClosingReads); + let fence = s.execute(read_fence(&s, 2, LONG)); + closed.reached().await; + assert_eq!(s.fence.gate().state(), FenceState::SourceWriteFenced); + let refused = program(&s, "ns", &s.conn().await, &["select count(*) from t"]).await; + read_fenced(&refused.unwrap_err()); + assert_eq!(s.fence.read_lease_counts().total(), 0); + closed.resume(); + assert_eq!(fence_outcome(&fence.await.unwrap()), FenceOutcome::Applied); + } + + /// At the deadline a running program is cancelled through its connection's cancel flag; + /// the drain waits for the actual release and then acknowledges. The program reports the + /// read fence. + #[tokio::test(flavor = "multi_thread")] + async fn read_fence_cancels_at_deadline() { + let s = fenced_source().await; + let conn = s.conn().await; + let mut reached = signal(&conn); + let running = spawn_program( + &s, + "ns", + &conn, + &[ + "select reached()", + "with recursive c(x) as (select 1 union all select x + 1 from c) \ + select count(*) from c", + ], + ); + reached.recv().await.unwrap(); + assert_eq!(s.fence.read_lease_counts().sql, 1); + + let fenced = tokio::time::timeout(PROMPT, s.execute(read_fence(&s, 2, NOW))) + .await + .expect("the cancelled program never released its lease") + .unwrap(); + assert_eq!(fence_outcome(&fenced), FenceOutcome::Applied); + let result = tokio::time::timeout(PROMPT, running) + .await + .unwrap() + .unwrap(); + read_fenced(&result.unwrap_err()); + assert!(conn.is_autocommit().await.unwrap()); + } + + /// A lease that is not released after cancellation keeps the drain from acknowledging: the + /// answer is `DRAINING`, reads stay closed, and a replay of the same command completes the + /// drain once the lease is gone. + #[tokio::test(flavor = "multi_thread")] + async fn unreleased_lease_answers_draining_and_replay_completes() { + let s = fenced_source().await; + let conn = s.conn().await; + let (mut reached, resume) = park(&conn); + let running = spawn_program(&s, "ns", &conn, &["select park()"]); + reached.recv().await.unwrap(); + + let request = read_fence(&s, 2, NOW); + let draining = s.execute(request.clone()).await.unwrap(); + assert_eq!(fence_outcome(&draining), FenceOutcome::Draining); + assert_eq!(s.fence.gate().state(), FenceState::SourceReadDraining); + let refused = program(&s, "ns", &s.conn().await, &["select 1"]).await; + read_fenced(&refused.unwrap_err()); + + resume.send(()).unwrap(); + // The program was cancelled by the fence; it reports the fence, not its rows. + read_fenced(&running.await.unwrap().unwrap_err()); + let replayed = s.execute(request).await.unwrap(); + assert_eq!(fence_outcome(&replayed), FenceOutcome::Applied); + assert_eq!(s.fence.gate().state(), FenceState::SourceReadFenced); + } + + /// A connection left idle inside a transaction holds no lease, so the drain does not wait + /// for it; its next program is refused and its transaction rolled back. + #[tokio::test(flavor = "multi_thread")] + async fn idle_txn_fails_on_next_program() { + let s = fenced_source().await; + let conn = s.conn().await; + assert_ok( + &program(&s, "ns", &conn, &["begin", "select count(*) from t"]) + .await + .unwrap(), + ); + assert!(!conn.is_autocommit().await.unwrap()); + assert_eq!(s.fence.read_lease_counts().total(), 0); + + let fenced = s.execute(read_fence(&s, 2, LONG)).await.unwrap(); + assert_eq!(fence_outcome(&fenced), FenceOutcome::Applied); + + let refused = program(&s, "ns", &conn, &["select count(*) from t", "commit"]).await; + read_fenced(&refused.unwrap_err()); + assert!(conn.is_autocommit().await.unwrap()); + } + + /// Clearing the read fence reopens reads and leaves writes fenced. + #[tokio::test(flavor = "multi_thread")] + async fn clear_read_fence_reopens_reads_not_writes() { + let s = fenced_source().await; + let fenced = s.execute(read_fence(&s, 2, LONG)).await.unwrap(); + assert_eq!(fence_outcome(&fenced), FenceOutcome::Applied); + let conn = s.conn().await; + read_fenced(&program(&s, "ns", &conn, &["select 1"]).await.unwrap_err()); + + let revision = s.fence.gate().revision(); + let cleared = s.execute(clear_read_fence(&s, 3)).await.unwrap(); + assert_eq!(fence_outcome(&cleared), FenceOutcome::Applied); + let gate = s.fence.gate(); + assert_eq!(gate.state(), FenceState::SourceWriteFenced); + assert!(gate.revision() > revision); + + let steps = program( + &s, + "ns", + &conn, + &["select count(*) from t", "insert into t values (3)"], + ) + .await + .unwrap(); + assert!(matches!( + steps[0].as_ref().unwrap().as_slice(), + [row] if matches!(row.as_slice(), [crate::query::Value::Integer(2)]) + )); + match &steps[1] { + Err(Error::NamespaceFence(e)) => { + assert_eq!(e.outcome(), FenceOutcome::MigrationWriteFenced) + } + other => panic!("expected MIGRATION_WRITE_FENCED, got {other:?}"), + } + assert_eq!(s.count().await, 2); + } + + /// A read fence the checks refuse (another operation owns the source) reopens read + /// admission it had closed, and reads are served again. + #[tokio::test(flavor = "multi_thread")] + async fn refused_read_fence_reopens_reads() { + let s = fenced_source().await; + let request = request( + &s, + OTHER_OP, + 2, + FenceCommand::SetSourceReadFence { + drain_policy: Some(LONG), + }, + ); + let refused = s.execute(request).await.unwrap(); + assert_eq!( + fence_outcome(&refused), + FenceOutcome::FenceOwnedByAnotherOperation + ); + let gate = s.fence.gate(); + assert_eq!(gate.state(), FenceState::SourceWriteFenced); + assert!(gate.closing_reads.is_none()); + assert_ok( + &program(&s, "ns", &s.conn().await, &["select count(*) from t"]) + .await + .unwrap(), + ); + } + + /// ATTACH of a namespace is a read of that namespace: a program on another namespace that + /// has it attached holds a lease on it, so its read fence waits for that program; once it is + /// read-fenced, a new ATTACH and any program on a connection that still has it attached are + /// refused, and a connection that detached it is served again. + #[tokio::test(flavor = "multi_thread")] + async fn attach_of_read_fenced_namespace_denied() { + let s = Source::new().await; + raw(&s.conn().await, "insert into t values (1)") + .await + .unwrap(); + let config = s + .store + .with("ns".into(), |ns| ns.db_config_store.clone()) + .await + .unwrap(); + config + .store(DatabaseConfig { + allow_attach: true, + txn_timeout: Some(Duration::from_secs(600)), + ..Default::default() + }) + .await + .unwrap(); + s.store + .create( + "other".into(), + RestoreOption::Latest, + DatabaseConfig::default(), + ) + .await + .unwrap(); + let acquired = s.execute(s.acquire(OP, 1, LONG)).await.unwrap(); + assert_eq!(fence_outcome(&acquired), FenceOutcome::Applied); + + let other = conn_to(&s, "other").await; + let steps = program( + &s, + "other", + &other, + &["attach ns as a", "select count(*) from a.t"], + ) + .await + .unwrap(); + assert_ok(&steps); + assert_eq!(s.fence.read_lease_counts().total(), 0); + + // A later program on the connection that has `ns` attached holds a lease on it. + let (mut reached, resume) = park(&other); + let running = spawn_program(&s, "other", &other, &["select park()"]); + reached.recv().await.unwrap(); + assert_eq!(s.fence.read_lease_counts().sql, 1); + let fence = s.execute(read_fence(&s, 2, LONG)); + s.until_state(FenceState::SourceReadDraining).await; + assert!(!fence.is_finished()); + resume.send(()).unwrap(); + assert_ok(&running.await.unwrap().unwrap()); + assert_eq!(fence_outcome(&fence.await.unwrap()), FenceOutcome::Applied); + + // The connection that still has it attached is refused outright... + read_fenced( + &program(&s, "other", &other, &["select 1"]) + .await + .unwrap_err(), + ); + // ...a fresh ATTACH is refused as a step... + let fresh = conn_to(&s, "other").await; + let steps = program(&s, "other", &fresh, &["attach ns as b"]) + .await + .unwrap(); + step_read_fenced(&steps[0]); + // ...and a connection that detached it is served again. + let steps = program(&s, "other", &fresh, &["select 1"]).await.unwrap(); + assert_ok(&steps); + } +} diff --git a/libsql-server/src/namespace/meta_store.rs b/libsql-server/src/namespace/meta_store.rs index 1a16994c3f..ebd1ba2a64 100644 --- a/libsql-server/src/namespace/meta_store.rs +++ b/libsql-server/src/namespace/meta_store.rs @@ -112,6 +112,7 @@ struct FenceSettings { receipt_retention: Duration, /// The write drain deadline of an `AcquireSourceWriteFence` that names no drain policy. default_write_drain: Duration, + default_read_drain: Duration, } fn setup_connection(conn: &rusqlite::Connection) -> Result<()> { @@ -240,6 +241,9 @@ impl MetaStoreInner { default_write_drain: config .namespace_fence_default_write_drain .unwrap_or(crate::namespace::fence::drain::DEFAULT_WRITE_DRAIN), + default_read_drain: config + .namespace_fence_default_read_drain + .unwrap_or(crate::namespace::fence::read::DEFAULT_READ_DRAIN), }; let mut this = MetaStoreInner { @@ -1315,6 +1319,17 @@ impl MetaStore { } } + /// The drain policy of a `SetSourceReadFence` that names none: the configured deadline, + /// after which running reads are cancelled and streams terminated (`on_deadline` does not + /// apply to reads). + pub fn fence_default_read_drain(&self) -> DrainPolicy { + DrainPolicy { + deadline_ms: u64::try_from(self.inner.fence.default_read_drain.as_millis()) + .unwrap_or(u64::MAX), + on_deadline: OnDeadline::Fail, + } + } + /// Whether this metastore holds fence state, so fences are loaded and enforced. pub fn fence_enforced(&self) -> bool { self.inner.fence.tables diff --git a/libsql-server/src/namespace/mod.rs b/libsql-server/src/namespace/mod.rs index 28ba60ea26..f75dbd700f 100644 --- a/libsql-server/src/namespace/mod.rs +++ b/libsql-server/src/namespace/mod.rs @@ -29,8 +29,16 @@ mod schema_lock; mod store; pub type ResetCb = Box; +/// Resolves a namespace that a program ATTACHes: its directory, and its fence controller, which +/// admits the attachment as a read of that namespace (`docs/NAMESPACE_FENCE.md` section 9). pub type ResolveNamespacePathFn = - Arc crate::Result> + Sync + Send + 'static>; + Arc crate::Result + Sync + Send + 'static>; + +/// A namespace resolved for ATTACH. +pub struct AttachTarget { + pub path: Arc, + pub fence: Arc, +} pub enum ResetOp { Reset(NamespaceName), @@ -103,9 +111,6 @@ impl Namespace { Ok(()) } - // Read by the protocol layers that consult the gate outside a connection (dump, - // replication, lifecycle), which land later in this series. - #[allow(dead_code)] pub(crate) fn fence(&self) -> &Arc { &self.fence } diff --git a/libsql-server/src/namespace/store.rs b/libsql-server/src/namespace/store.rs index af406dfd96..a472fc39c3 100644 --- a/libsql-server/src/namespace/store.rs +++ b/libsql-server/src/namespace/store.rs @@ -22,6 +22,7 @@ use crate::stats::Stats; use super::broadcasters::{BroadcasterHandle, BroadcasterRegistry}; use super::configurator::{DynConfigurator, NamespaceConfigurators}; use super::fence::command::{FenceCommand, FenceRequest}; +use super::fence::controller::FenceController; use super::fence::record::ServerIdentity; use super::fence::registry::FenceRegistry; use super::meta_store::{FenceCommit, FenceContext, MetaStore, MetaStoreHandle}; @@ -376,8 +377,12 @@ impl NamespaceStore { Arc::new({ let store = self.clone(); move |ns: &NamespaceName| { - tokio::runtime::Handle::current() - .block_on(store.with(ns.clone(), |ns| ns.path.clone())) + tokio::runtime::Handle::current().block_on(store.with(ns.clone(), |ns| { + super::AttachTarget { + path: ns.path.clone(), + fence: ns.fence().clone(), + } + })) } }) }) @@ -560,6 +565,23 @@ impl NamespaceStore { .await } + /// The fence controller that admits reads of `namespace` without loading it: `None` when + /// the namespace does not exist (and has no fence state). A namespace whose fence state is + /// unavailable is refused. + pub(crate) async fn fence_gate( + &self, + namespace: &NamespaceName, + ) -> crate::Result>> { + self.inner.fences.check_available(namespace)?; + if let Some(controller) = self.inner.fences.get(namespace) { + return Ok(Some(controller)); + } + if self.inner.metadata.exists(namespace).await { + return Ok(Some(self.inner.fences.controller(namespace))); + } + Ok(None) + } + pub(crate) fn schema_locks(&self) -> &SchemaLocksRegistry { &self.inner.schema_locks } From e4d0ab767aa54c7c2be7de9c6f187549824a549e Mon Sep 17 00:00:00 2001 From: River Date: Tue, 29 Sep 2026 18:02:32 +0000 Subject: [PATCH 13/33] libsql-server: end dump and replication streams on the read fence Dump and replication now hold read leases on the namespace fence, so the source read fence drains them positively: - /dump is admitted by the fence gate before a connection is created (a failure to create one is an error, not a panic) and the export holds a dump lease. At the read drain's deadline the export is cancelled before its next row, and a write blocked on a peer that stopped reading fails at once, so the lease is released without the peer. A cancelled dump ends its body with the fence error and never reaches its final COMMIT;. - Replication hello, log_entries, batch_log_entries and snapshot are refused at their start with FAILED_PRECONDITION and x-libsql-fence-code while streams are denied; refusals are counted and logged at most once a minute per namespace. Streams (and the frames of a batch) are served through FencedStream, whose watcher ends the stream as soon as the gate closes or the drain cancels it, dropping the inner stream and the lease itself; the next poll yields the typed terminal status. - With --enable-namespace-fence, the RPC server and the user HTTP server send HTTP/2 keepalive pings (--namespace-fence-keepalive-interval-s, default 30 s, 20 s timeout) so dead peers are detected. Co-authored-by: Tomasz Szymczyszyn --- docs/NAMESPACE_FENCE.md | 17 +- libsql-server/src/connection/dump/exporter.rs | 35 +- libsql-server/src/h2c.rs | 38 +- libsql-server/src/http/user/dump.rs | 111 +++- libsql-server/src/http/user/mod.rs | 21 +- libsql-server/src/lib.rs | 9 + libsql-server/src/main.rs | 10 + libsql-server/src/namespace/fence/drain.rs | 12 +- libsql-server/src/namespace/fence/mod.rs | 4 +- libsql-server/src/namespace/fence/read.rs | 8 +- libsql-server/src/namespace/fence/stream.rs | 594 ++++++++++++++++++ libsql-server/src/namespace/store.rs | 9 +- libsql-server/src/rpc/mod.rs | 19 +- .../src/rpc/replication/replication_log.rs | 136 +++- 14 files changed, 957 insertions(+), 66 deletions(-) create mode 100644 libsql-server/src/namespace/fence/stream.rs diff --git a/docs/NAMESPACE_FENCE.md b/docs/NAMESPACE_FENCE.md index a6a6bd5748..1723996586 100644 --- a/docs/NAMESPACE_FENCE.md +++ b/docs/NAMESPACE_FENCE.md @@ -508,8 +508,8 @@ This makes the WAL gate independent of statement classification: DDL, misclassif 4. Wait for all read leases to be released: - **SQL:** a lease is held for the duration of each running program (`CoreConnection::run`, `FenceConnState::begin_read_program`), including a Hrana cursor that is still producing rows, and for each `describe`. A connection that is idle with an open transaction holds no lease; its next program consults the live gate, fails with `MIGRATION_READ_FENCED`, and rolls the transaction back. Idle upgraded Hrana WebSocket and HTTP streams may therefore stay open. The admin shell, which runs raw SQL, takes a lease per query and is cancelled through the connection's interrupt handle. - **ATTACH:** attaching a namespace is a `NormalRead` of the attached namespace (its controller comes with the resolved path). The attaching program takes a lease on it; because an attachment outlives the program that made it, the connection remembers it (by schema alias, pruned against `PRAGMA database_list` when a program starts) and every later program on the connection takes a lease on each attached namespace as well, and is refused while any of them denies reads. - - **Dump:** a lease is held for the dump stream. The exporter checks a cancel flag between rows. A cancelled dump aborts the HTTP body (the chunked transfer is not completed), so a client never receives a dump that looks complete; the dump text also never reaches its final `COMMIT;`. - - **Replication:** `log_entries` and `snapshot` streams register a lease when created. The stream wrapper selects on the gate; on read fence it yields a terminal `FAILED_PRECONDITION` status carrying `MIGRATION_READ_FENCED` and ends. On cancel the wrapper drops the inner stream synchronously, so the lease is released even if the peer never reads again. `hello` and `batch_log_entries` are unary and are simply denied. + - **Dump:** `/dump` is admitted as a `Stream` by the gate before any connection is created (a refusal is the typed `MIGRATION_READ_FENCED`, and a connection that cannot be created is an error, not a panic), and the export holds a `Dump` lease until it has stopped and its read transaction is gone (`http/user/dump.rs` `dump_stream`). A dump that is running when the read fence starts is allowed to finish, like a SQL program, and the drain waits for it. At the deadline it is cancelled: the exporter (`export_dump_cancellable`) checks the cancel before every row and before its final `COMMIT;`, and the pipe to the HTTP body fails its pending write at once, so an export blocked because the peer is not reading stops too and the lease is released without the peer's help. A cancelled dump ends its body stream with the fence error, which aborts the HTTP response (the chunked transfer is not completed), so a client never receives a dump that looks complete; the dump text never reaches its final `COMMIT;`. + - **Replication:** every call (`hello`, `log_entries`, `batch_log_entries`, `snapshot`) is refused at its start with `FAILED_PRECONDITION` carrying `MIGRATION_READ_FENCED` (and `x-libsql-fence-code`) when the gate does not admit streams. `log_entries`, `snapshot` and the frames of `batch_log_entries` are served through a `FencedStream` (`namespace/fence/stream.rs`) that holds a `Replication` lease. A watcher task per stream ends it as soon as the gate stops admitting streams (the read-closing gate of step 2 does), or when the drain cancels it at its deadline: the watcher drops the inner stream and the lease itself, so the lease is released even if the peer never polls again, and the stream's next poll yields the terminal status and ends. Frames the stream had not yet handed to the transport are never served. Tailing streams never finish on their own, so ending them on the gate is what lets the drain complete before its deadline. Both replication services use the same code: the internal one used by replica servers and the external one on the user port. - At the deadline, SQL programs are cancelled through the connection's existing progress-handler cancel flag (a cancelled program rolls back any transaction and reports `MIGRATION_READ_FENCED`, not an interruption), dumps are cancelled, and streams are terminated. The command keeps waiting for the actual releases, for one more deadline but at least 10 seconds; if a lease still does not release, the result stays `DRAINING`, read admission stays closed, and a replay of the same command resumes the wait. That bound only decides when to answer `DRAINING`. - **`/beta/listen`:** refused at request start where reads are denied (the gate is read without loading the namespace), and the event stream ends with an error event as soon as the gate denies reads. It holds no lease: it serves change notifications, not data, and no change can happen while writes are fenced. 5. CAS `SOURCE_READ_FENCED`; respond. @@ -518,9 +518,9 @@ This makes the WAL gate independent of statement classification: DDL, misclassif Covered surfaces: HTTP (`/`, `/v1/execute`, `/v1/batch`), Hrana over HTTP (`/v2`, `/v3`, cursors), Hrana over WebSocket, the gRPC proxy (`execute`, `stream_exec`, `describe`), the admin shell (which runs raw SQL and is checked per query), `/beta/listen`, ATTACH from other namespaces, `/dump`, and both replication services (the internal one used by replica servers and the external one on the user port). A replica server that receives the terminal status installs a local read denial for that namespace, so it stops serving its local copy. -Transport keepalive: when the fence is enabled, the RPC server and the user-port gRPC service set HTTP/2 keepalive (`--namespace-fence-keepalive-interval`, default 30 s; timeout 20 s) so dead peers are detected; lease release does not depend on it. +Transport keepalive: when the fence is enabled (`--enable-namespace-fence`), the RPC server and the user-port HTTP server (which carries the user-facing gRPC services), including connections upgraded to `h2c`, send HTTP/2 keepalive pings (`--namespace-fence-keepalive-interval-s`, default 30 s; a ping unanswered for 20 s closes the connection) so dead peers are detected; lease release does not depend on it. -Denied replication calls from old replicas are logged at most once per namespace per minute and counted, so repeated reconnects are observable rather than noisy. Newer replicas back off on the typed code (capped exponential, at most 60 s). +Replication calls denied by the fence are counted (`libsql_server_fence_denials_total{code, surface = "replication"}`) and logged at most once per namespace per minute, so repeated reconnects of replicas that do not understand the typed code are observable rather than noisy. Replica-side handling of the typed code (a local read denial, capped back-off) is part of the protocol work of section 6. Internal work that must keep running is classed `Maintenance` or `Observability` and holds no read lease: bottomless WAL upload, the storage monitor's read transaction, stats and metrics. @@ -621,7 +621,7 @@ The server cannot verify who the approvers are: the admin API has one shared key - Off, but fence tables or markers exist (the flag was turned off after use): fences are still loaded and enforced; mutating routes are disabled. - On: tables are created, routes are served, and the fail-closed recovery rules of section 13.3 apply. -Related flags: `--namespace-fence-receipt-retention-s`, `--namespace-fence-adoption-key`, `--namespace-fence-keepalive-interval`, and default drain deadlines `--namespace-fence-default-write-drain-ms` and `--namespace-fence-default-read-drain-ms` (used when a request has no `drain_policy`). +Related flags: `--namespace-fence-receipt-retention-s`, `--namespace-fence-adoption-key`, `--namespace-fence-keepalive-interval-s`, and default drain deadlines `--namespace-fence-default-write-drain-ms` and `--namespace-fence-default-read-drain-ms` (used when a request has no `drain_policy`). Upgrade order: deploy a binary with capability discovery and proxy `stable_code` support on every primary and replica; confirm with `GET /v1/fence/capabilities`; then enable the flag; then use fences. Rollback to a binary without fence support is refused by deployment tooling while `active_fences > 0`. @@ -668,8 +668,8 @@ How each path that can reach namespace data or lifecycle is covered. File refere | `namespace/store.rs` `with`, `load_namespace`, `make_namespace`, eviction | Registry check before setup; controller passed into setup; eviction keeps the registry entry. | | `store.rs` `create`, `destroy`, `reset`, `fork`, `checkpoint`, restore options | Create refuses names with a record; `CreateTargetQuarantined` is the atomic quarantined create; destroy, reset, fork (either side) and any restore are denied while lifecycle is denied; checkpoint uses a non-creating lookup and skips vacuum. | | `http/admin/mod.rs` config, create, fork, delete, checkpoint, stats | Config POST, create, fork, delete follow the lifecycle column; config GET, stats and checkpoint are allowed. Fence routes live in `http/admin/fence.rs`. | -| `http/user/dump.rs` | Gate check before connection creation (typed, no panic on create error); dump stream lease; cancel flag in the exporter; aborted body on termination. | -| `rpc/replication/replication_log.rs` `hello`, `log_entries`, `batch_log_entries`, `snapshot` | Denied at request start; stream leases; typed terminal status for open streams; `ReplicatedFence` in `hello`'s config. | +| `http/user/dump.rs`, `connection/dump/exporter.rs` | Gate check (`Stream`) before connection creation (typed, no panic on create error); `Dump` lease held by the export; cancel checked before every row and by the pipe's pending write; the body stream ends with the fence error (aborted response) on cancellation. | +| `rpc/replication/replication_log.rs` `hello`, `log_entries`, `batch_log_entries`, `snapshot` | Denied at request start (`FAILED_PRECONDITION` + `x-libsql-fence-code`, counted, rate-limited log); `FencedStream` replication leases for both streams and the batch; typed terminal status when the gate closes or at the deadline, lease released without the peer; `ReplicatedFence` in `hello`'s config (planned, section 6.2). | | `admin_shell.rs` | Writes denied at the WAL (no capability); reads checked against the gate, and a read lease held, per query (cancelled through the connection's interrupt handle). | | `schema/scheduler.rs`, `database/schema.rs` | Shared schema excluded from fencing; migration writes are WAL-gated; the scheduler's `block_writes` flag is not treated as drain evidence. | | `namespace/configurator/helpers.rs` `load_dump`, `http/admin/mod.rs` `dump_stream_from_url` | Restore options and dump URLs are refused for fenced namespaces; import goes through `ImportSession`. | @@ -725,7 +725,7 @@ Planned test names; the table is updated as tests land. | 13 | Seal enters `TARGET_IMPORT_DRAINING`, waits, reaches `TARGET_VALIDATING`, cannot resume import; only a durable validation receipt permits idempotent publication | `fence::target::tests::seal_waits_for_import_writers`, `sealed_target_rejects_import`, `publish_requires_validation_receipt`, `publish_is_idempotent` | | 14 | Enable writes idempotent, survives restart and response loss, irreversible | `fence::target::tests::enable_writes_idempotent_and_irreversible`, `enable_writes_survives_restart` | | 15 | Lost `EnableTargetWrites` response resolved from receipt/state | `fence::target::tests::enable_writes_response_loss_resolved` | -| 16 | Read fence drains SQL, dump, `log_entries`, `snapshot`, including dead peers and forced termination | SQL landed: `namespace::fence::read::tests::{read_fence_waits_for_running_program, program_after_closing_gate_is_refused (parked after the read-closing gate, before the CAS), read_fence_cancels_at_deadline, unreleased_lease_answers_draining_and_replay_completes, idle_txn_fails_on_next_program, clear_read_fence_reopens_reads_not_writes, refused_read_fence_reopens_reads, attach_of_read_fenced_namespace_denied}`, `admin_shell::fence_tests::admin_shell_read_denied`; planned for dump and replication: `fence::read::tests::{dump_lease_released_on_cancel, log_entries_stream_ends_typed, snapshot_stream_ends_typed, stream_lease_released_without_peer_read, read_fence_forced_termination}`, and an integration test that an interrupted dump never ends with `COMMIT;` | +| 16 | Read fence drains SQL, dump, `log_entries`, `snapshot`, including dead peers and forced termination | SQL landed: `namespace::fence::read::tests::{read_fence_waits_for_running_program, program_after_closing_gate_is_refused (parked after the read-closing gate, before the CAS), read_fence_cancels_at_deadline, unreleased_lease_answers_draining_and_replay_completes, idle_txn_fails_on_next_program, clear_read_fence_reopens_reads_not_writes, refused_read_fence_reopens_reads, attach_of_read_fenced_namespace_denied}`, `admin_shell::fence_tests::admin_shell_read_denied`; dump and replication landed: `namespace::fence::stream::tests::{dump_lease_released_on_cancel (a dump blocked mid-row on a peer that stopped reading is cancelled at the deadline, its lease released without the peer, the fence acknowledged, and the body ends with the fence error and no `COMMIT;`), read_fence_waits_for_dump, dump_refused_while_read_fenced, log_entries_stream_ends_typed, stream_lease_released_without_peer_read (a dead peer), snapshot_stream_ends_typed, replication_calls_denied_while_read_fenced, read_fence_forced_termination}`; planned: the HTTP-level check that an interrupted `/dump` response is aborted rather than completed, with `tests::fence::protocol::dump_codes` | | 17 | Delete, reset, fork, restore, config, schema mutation rejected | `tests::fence::lifecycle::lifecycle_rejected_while_fenced` | | 18 | Codes through HTTP, Hrana, RPC, dump, replication, replica write proxy; distinguishable from auth/timeout/not-found; old peers compatible; no retry loops | `tests::fence::protocol::{http_codes, hrana_http_codes, hrana_ws_codes, rpc_codes, dump_codes, replication_codes, replica_proxy_preserves_code, auth_and_not_found_distinct, denial_not_retried}`; `libsql-replication` `proxy_error_stable_code_is_additive` | | 19 | Corrupt or unknown durable fence state fails closed | `fence::store::tests::corrupt_payload_fails_closed`, `unknown_format_version_fails_closed` | @@ -764,6 +764,7 @@ libsql-server/src/namespace/fence/ controller.rs FenceController, GateSnapshot, generations, leases drain.rs write and import drains read.rs source read fence and its drain + stream.rs stream leases: FencedStream for replication, dump cancel target.rs target lifecycle, MigrationCapability, ImportSession, ValidationSession audit.rs audit events and metrics hooks.rs cfg(test) FenceTestHooks diff --git a/libsql-server/src/connection/dump/exporter.rs b/libsql-server/src/connection/dump/exporter.rs index 1a1c69728b..66ae566edc 100644 --- a/libsql-server/src/connection/dump/exporter.rs +++ b/libsql-server/src/connection/dump/exporter.rs @@ -7,15 +7,30 @@ use anyhow::bail; use rusqlite::types::ValueRef; use rusqlite::OptionalExtension; -struct DumpState { +struct DumpState<'a, W: Write> { /// true if db is in writable_schema mode writable_schema: bool, writer: W, + /// Checked before every row: once it returns `true` the export stops with [`DumpCancelled`]. + cancelled: &'a dyn Fn() -> bool, } +/// The export was stopped by its caller before it completed (for example by a namespace read +/// fence). The output is incomplete: it never reaches its final `COMMIT;`. +#[derive(Debug, thiserror::Error)] +#[error("the dump was cancelled before it completed")] +pub struct DumpCancelled; + use rusqlite::ffi::{sqlite3_keyword_check, sqlite3_table_column_metadata, SQLITE_OK}; -impl DumpState { +impl DumpState<'_, W> { + fn check_cancelled(&self) -> anyhow::Result<()> { + if (self.cancelled)() { + return Err(DumpCancelled.into()); + } + Ok(()) + } + fn run_schema_dump_query( &mut self, txn: &rusqlite::Connection, @@ -25,6 +40,7 @@ impl DumpState { let mut stmt = txn.prepare(stmt)?; let mut rows = stmt.query(())?; while let Some(row) = rows.next()? { + self.check_cancelled()?; let ValueRef::Text(table) = row.get_ref(0)? else { bail!("invalid schema table") }; @@ -103,6 +119,7 @@ impl DumpState { let mut stmt = txn.prepare(&select)?; let mut rows = stmt.query(())?; while let Some(row) = rows.next()? { + self.check_cancelled()?; write!(self.writer, "{insert}")?; if row_id_col.is_some() { write_value_ref(&mut self.writer, row.get_ref(0)?)?; @@ -128,6 +145,7 @@ impl DumpState { let col_count = stmt.column_count(); let mut rows = stmt.query(())?; while let Some(row) = rows.next()? { + self.check_cancelled()?; let ValueRef::Text(sql) = row.get_ref(0)? else { bail!("the first row in a table dump query should be of type text") }; @@ -437,6 +455,17 @@ pub fn export_dump( db: &mut rusqlite::Connection, writer: impl Write, preserve_rowids: bool, +) -> anyhow::Result<()> { + export_dump_cancellable(db, writer, preserve_rowids, &|| false) +} + +/// [`export_dump`], stopped with a [`DumpCancelled`] error as soon as `cancelled` returns `true` +/// (it is checked before every row and before the final `COMMIT;`). +pub fn export_dump_cancellable( + db: &mut rusqlite::Connection, + writer: impl Write, + preserve_rowids: bool, + cancelled: &dyn Fn() -> bool, ) -> anyhow::Result<()> { let mut txn = db.transaction()?; txn.execute("PRAGMA writable_schema=ON", ())?; @@ -444,6 +473,7 @@ pub fn export_dump( let mut state = DumpState { writable_schema: false, writer, + cancelled, }; writeln!(state.writer, "PRAGMA foreign_keys=OFF;")?; @@ -469,6 +499,7 @@ AND type IN ('index','trigger','view')"; writeln!(state.writer, "PRAGMA writable_schema=OFF;")?; } + state.check_cancelled()?; writeln!(state.writer, "COMMIT;")?; let _ = savepoint.execute("PRAGMA writable_schema = OFF;", ()); diff --git a/libsql-server/src/h2c.rs b/libsql-server/src/h2c.rs index 93d4999543..2f85f6f256 100644 --- a/libsql-server/src/h2c.rs +++ b/libsql-server/src/h2c.rs @@ -40,6 +40,7 @@ use std::marker::PhantomData; use std::pin::Pin; +use std::time::Duration; use axum::{body::BoxBody, http::HeaderValue}; use bytes::Bytes; @@ -56,6 +57,7 @@ type BoxError = Box; #[derive(Debug, Clone)] pub struct H2cMaker { s: S, + http2_keepalive_interval: Option, _pd: PhantomData, } @@ -63,9 +65,32 @@ impl H2cMaker { pub fn new(s: S) -> Self { Self { s, + http2_keepalive_interval: None, _pd: PhantomData, } } + + /// Send HTTP/2 keepalive pings on upgraded `h2c` connections at this interval. + pub fn with_http2_keepalive(mut self, interval: Option) -> Self { + self.http2_keepalive_interval = interval; + self + } +} + +/// How long a keepalive ping may go unanswered before the connection is closed. +pub const HTTP2_KEEPALIVE_TIMEOUT: Duration = Duration::from_secs(20); + +/// Configure HTTP/2 keepalive on a server builder; `None` leaves it unchanged. +pub fn with_http2_keepalive( + builder: hyper::server::Builder, + interval: Option, +) -> hyper::server::Builder { + match interval { + Some(interval) => builder + .http2_keep_alive_interval(interval) + .http2_keep_alive_timeout(HTTP2_KEEPALIVE_TIMEOUT), + None => builder, + } } impl Service<&C> for H2cMaker @@ -95,10 +120,12 @@ where fn call(&mut self, conn: &C) -> Self::Future { let connect_info = conn.connect_info(); let s = self.s.clone(); + let http2_keepalive_interval = self.http2_keepalive_interval; Box::pin(async move { Ok(H2c { s, connect_info, + http2_keepalive_interval, _pd: PhantomData, }) }) @@ -112,6 +139,7 @@ where pub struct H2c { s: S, connect_info: TcpConnectInfo, + http2_keepalive_interval: Option, _pd: PhantomData, } @@ -139,6 +167,7 @@ where fn call(&mut self, mut req: hyper::Request) -> Self::Future { let mut svc = self.s.clone(); let connect_info = self.connect_info.clone(); + let http2_keepalive_interval = self.http2_keepalive_interval; Box::pin(async move { req.extensions_mut().insert(connect_info.clone()); @@ -169,8 +198,13 @@ where tracing::debug!("Successfully upgraded the connection, speaking h2 now"); - if let Err(e) = hyper::server::conn::Http::new() - .http2_only(true) + let mut http = hyper::server::conn::Http::new(); + http.http2_only(true); + if let Some(interval) = http2_keepalive_interval { + http.http2_keep_alive_interval(interval) + .http2_keep_alive_timeout(HTTP2_KEEPALIVE_TIMEOUT); + } + if let Err(e) = http .serve_connection( upgraded_io, tower::service_fn(move |mut r: hyper::Request| { diff --git a/libsql-server/src/http/user/dump.rs b/libsql-server/src/http/user/dump.rs index 16efcc52a7..ae0260482e 100644 --- a/libsql-server/src/http/user/dump.rs +++ b/libsql-server/src/http/user/dump.rs @@ -1,5 +1,7 @@ use std::future::Future; +use std::io::Write; use std::pin::Pin; +use std::sync::Arc; use std::task; use axum::extract::{Query, State as AxumState}; @@ -9,9 +11,14 @@ use pin_project_lite::pin_project; use serde::Deserialize; use crate::auth::Authenticated; -use crate::connection::dump::exporter::export_dump; -use crate::connection::Connection as _; +use crate::connection::dump::exporter::export_dump_cancellable; +use crate::connection::{Connection as _, MakeConnection}; +use crate::database::Connection; use crate::error::Error; +use crate::namespace::fence::controller::{FenceController, LeaseKind}; +use crate::namespace::fence::stream::{ + acquire_stream_lease, cancelled_by_read_fence, StreamCancel, +}; use crate::BLOCKING_RT; use super::db_factory::namespace_from_headers; @@ -95,36 +102,114 @@ pub(super) async fn handle_dump( return Err(Error::NamespaceDoesntExist(namespace.to_string())); } - let conn_maker = state + let (conn_maker, fence) = state .namespaces .with(namespace, |ns| { if !ns.db.is_primary() { return Err(Error::NotAPrimary); } - Ok::<_, crate::Error>(ns.db.connection_maker()) + Ok::<_, crate::Error>((ns.db.connection_maker(), ns.fence().clone())) }) .await??; - let conn = conn_maker.create().await.unwrap(); + let stream = dump_stream(&fence, conn_maker, query.preserve_row_ids.unwrap_or(false)).await?; + + Ok(axum::body::StreamBody::new(stream)) +} + +/// The dump of one namespace as a byte stream (`docs/NAMESPACE_FENCE.md` section 9). +/// +/// The dump is admitted by the namespace's fence gate before any connection is created, and +/// holds a `Dump` read lease until the export has stopped. When the read drain cancels it, the +/// export stops before its next row, and also if it is blocked because the peer is not reading, +/// so the lease is released without the peer's help; the stream then ends with the fence error +/// instead of ending cleanly, so the response body is aborted and the client never receives a +/// dump that looks complete. +pub(crate) async fn dump_stream( + fence: &Arc, + conn_maker: Arc>, + preserve_row_ids: bool, +) -> crate::Result>> { + let (lease, cancel) = + acquire_stream_lease(fence, LeaseKind::Dump).map_err(Error::NamespaceFence)?; + + let conn = conn_maker.create().await?; let (reader, writer) = tokio::io::duplex(8 * 1024); + let writer = CancellableWriter { + inner: writer, + cancel: cancel.clone(), + handle: tokio::runtime::Handle::current(), + }; let join_handle = BLOCKING_RT.spawn_blocking(move || { - let writer = tokio_util::io::SyncIoBridge::new(writer); - conn.with_raw(|conn| { - export_dump(conn, writer, query.preserve_row_ids.unwrap_or(false)).map_err(Into::into) - }) + // Released once the export has stopped and its read transaction is gone. + let _lease = lease; + let result = conn.with_raw(|conn| { + export_dump_cancellable(conn, writer, preserve_row_ids, &|| cancel.is_cancelled()) + }); + match result { + Ok(()) => Ok(()), + Err(_) if cancel.is_cancelled() => Err(Error::NamespaceFence(cancelled_by_read_fence( + LeaseKind::Dump, + ))), + Err(e) => Err(e.into()), + } }); let stream = tokio_util::io::ReaderStream::new(reader); - let stream = DumpStream { + Ok(DumpStream { stream: stream.fuse(), join_handle: Some(join_handle), - }; + }) +} - let stream = axum::body::StreamBody::new(stream); +/// The export's side of the dump pipe. A write waits for the reader (the HTTP body), unless +/// the dump is cancelled, in which case it fails at once. +struct CancellableWriter { + inner: tokio::io::DuplexStream, + cancel: Arc, + handle: tokio::runtime::Handle, +} - Ok(stream) +impl CancellableWriter { + fn cancelled() -> std::io::Error { + std::io::Error::new(std::io::ErrorKind::Other, "dump cancelled") + } +} + +impl Write for CancellableWriter { + fn write(&mut self, buf: &[u8]) -> std::io::Result { + use tokio::io::AsyncWriteExt as _; + let Self { + inner, + cancel, + handle, + } = self; + handle.block_on(async { + tokio::select! { + biased; + _ = cancel.cancelled() => Err(Self::cancelled()), + r = inner.write(buf) => r, + } + }) + } + + fn flush(&mut self) -> std::io::Result<()> { + use tokio::io::AsyncWriteExt as _; + let Self { + inner, + cancel, + handle, + } = self; + handle.block_on(async { + tokio::select! { + biased; + _ = cancel.cancelled() => Err(Self::cancelled()), + r = inner.flush() => r, + } + }) + } } diff --git a/libsql-server/src/http/user/mod.rs b/libsql-server/src/http/user/mod.rs index 1575a3574b..7d46dce8cd 100644 --- a/libsql-server/src/http/user/mod.rs +++ b/libsql-server/src/http/user/mod.rs @@ -1,5 +1,5 @@ pub mod db_factory; -mod dump; +pub(crate) mod dump; mod extract; mod hrana_over_http_1; mod listen; @@ -10,6 +10,7 @@ mod types; pub mod timing; use std::sync::Arc; +use std::time::Duration; use anyhow::Context; use axum::extract::{FromRef, FromRequest, FromRequestParts, Path as AxumPath, State as AxumState}; @@ -256,6 +257,8 @@ pub struct UserApi { pub enable_console: bool, pub self_url: Option, pub primary_url: Option, + /// HTTP/2 keepalive interval (see `Server::http2_keepalive_interval`). + pub http2_keepalive_interval: Option, } impl UserApi @@ -443,14 +446,18 @@ where ); let router = router.fallback(handle_fallback); - let h2c = crate::h2c::H2cMaker::new(router); + let keepalive = self.http2_keepalive_interval; + let h2c = crate::h2c::H2cMaker::new(router).with_http2_keepalive(keepalive); task_manager.spawn_with_shutdown_notify(|shutdown| async move { - hyper::server::Server::builder(acceptor) - .serve(h2c) - .with_graceful_shutdown(shutdown.notified()) - .await - .context("http server")?; + crate::h2c::with_http2_keepalive( + hyper::server::Server::builder(acceptor), + keepalive, + ) + .serve(h2c) + .with_graceful_shutdown(shutdown.notified()) + .await + .context("http server")?; Ok(()) }); } diff --git a/libsql-server/src/lib.rs b/libsql-server/src/lib.rs index 1642ad951a..66fbbcf762 100644 --- a/libsql-server/src/lib.rs +++ b/libsql-server/src/lib.rs @@ -151,6 +151,10 @@ pub struct Server anyhow::Result<()> + Send + Sync + 'static>>, + /// HTTP/2 keepalive interval of the RPC server and the user-port gRPC services, set when + /// namespace fences are enabled so that the streams of dead peers are detected + /// (`docs/NAMESPACE_FENCE.md` section 9). `None` keeps hyper's default (no keepalive). + pub http2_keepalive_interval: Option, } impl Default for Server { @@ -180,6 +184,7 @@ impl Default for Server { force_load_wals: false, sync_conccurency: 8, set_log_level: None, + http2_keepalive_interval: None, } } } @@ -196,6 +201,7 @@ struct Services { db_config: DbConfig, user_auth_strategy: Auth, pub set_log_level: Option anyhow::Result<()> + Send + Sync + 'static>>, + http2_keepalive_interval: Option, } struct TaskManager { @@ -290,6 +296,7 @@ where enable_console: self.user_api_config.enable_http_console, self_url: self.user_api_config.self_url, primary_url: self.user_api_config.primary_url, + http2_keepalive_interval: self.http2_keepalive_interval, }; let user_http_service = user_http.configure(task_manager); @@ -529,6 +536,7 @@ where db_config: self.db_config, user_auth_strategy, set_log_level: self.set_log_level.take(), + http2_keepalive_interval: self.http2_keepalive_interval, } } @@ -677,6 +685,7 @@ where config.tls_config, idle_shutdown_kicker.clone(), replication_service, // internal replicaton service + self.http2_keepalive_interval, )); } diff --git a/libsql-server/src/main.rs b/libsql-server/src/main.rs index f8f56e7494..02a8011c0f 100644 --- a/libsql-server/src/main.rs +++ b/libsql-server/src/main.rs @@ -280,6 +280,13 @@ struct Cli { #[clap(long, env = "SQLD_NAMESPACE_FENCE_DEFAULT_READ_DRAIN_MS")] namespace_fence_default_read_drain_ms: Option, + /// HTTP/2 keepalive interval, in seconds, of the RPC server and the user-port gRPC + /// services when namespace fences are enabled, so that replication streams of dead peers + /// are detected (a ping unanswered for 20 seconds closes the connection). Defaults to 30 + /// seconds; ignored unless `--enable-namespace-fence` is set. + #[clap(long, env = "SQLD_NAMESPACE_FENCE_KEEPALIVE_INTERVAL_S")] + namespace_fence_keepalive_interval_s: Option, + /// Shutdown timeout duration in seconds, defaults to 30 seconds. #[clap(long, env = "SQLD_SHUTDOWN_TIMEOUT")] shutdown_timeout: Option, @@ -756,6 +763,9 @@ async fn build_server( force_load_wals: config.force_load_wals, sync_conccurency: config.sync_conccurency, set_log_level: Some(Box::new(set_log_level)), + http2_keepalive_interval: config.enable_namespace_fence.then(|| { + Duration::from_secs(config.namespace_fence_keepalive_interval_s.unwrap_or(30)) + }), }) } diff --git a/libsql-server/src/namespace/fence/drain.rs b/libsql-server/src/namespace/fence/drain.rs index 5cc70563b8..3e5c8ba5b8 100644 --- a/libsql-server/src/namespace/fence/drain.rs +++ b/libsql-server/src/namespace/fence/drain.rs @@ -304,7 +304,7 @@ pub(crate) mod tests { use crate::namespace::fence::record::ServerIdentity; use crate::namespace::fence::state::FenceState; use crate::namespace::meta_store::FenceCommitKind; - use crate::namespace::store::fence_tests::open_store; + use crate::namespace::store::fence_tests::open_store_with_max_log_size; use crate::namespace::store::NamespaceStore; use crate::namespace::RestoreOption; use crate::replication::primary::logger::ReplicationLogger; @@ -325,13 +325,19 @@ pub(crate) mod tests { _dir: TempDir, pub(crate) store: NamespaceStore, pub(crate) fence: Arc, - logger: Arc, + pub(crate) logger: Arc, } impl Source { pub(crate) async fn new() -> Self { + Self::with_max_log_size(1_000_000_000).await + } + + /// A source whose replication log is compacted into a snapshot once it holds more than + /// `max_log_size` MB (`0`: at the next compaction). + pub(crate) async fn with_max_log_size(max_log_size: u64) -> Self { let dir = tempdir().unwrap(); - let store = open_store(dir.path()).await; + let store = open_store_with_max_log_size(dir.path(), max_log_size).await; store .create( "ns".into(), diff --git a/libsql-server/src/namespace/fence/mod.rs b/libsql-server/src/namespace/fence/mod.rs index 5c6f5d4382..74f227cf23 100644 --- a/libsql-server/src/namespace/fence/mod.rs +++ b/libsql-server/src/namespace/fence/mod.rs @@ -9,7 +9,8 @@ //! ([`transition`]), the metastore tables, compare-and-swap and marker file that persist them //! ([`store`], driven by `MetaStore::apply_fence_command`), and the in-memory authority built //! on them: the per-namespace [`controller`] with its gate and read leases, the positive write -//! [`drain`], the source [`read`] fence, the [`registry`] that holds the controllers outside the +//! [`drain`], the source [`read`] fence and its +//! [`stream`] leases for dump and replication, the [`registry`] that holds the controllers outside the //! namespace cache, and the test [`hooks`] on their paths. // The persistence, controller and protocol layers that consume these types land in the @@ -27,6 +28,7 @@ pub mod record; pub mod registry; pub mod state; pub mod store; +pub mod stream; pub mod transition; #[cfg(test)] diff --git a/libsql-server/src/namespace/fence/read.rs b/libsql-server/src/namespace/fence/read.rs index cc4a60b906..50feb0b3b1 100644 --- a/libsql-server/src/namespace/fence/read.rs +++ b/libsql-server/src/namespace/fence/read.rs @@ -142,7 +142,7 @@ async fn wait_for_read_leases(controller: &FenceController, deadline: Instant) - } #[cfg(test)] -mod tests { +pub(crate) mod tests { use std::sync::Arc; use std::time::Duration; @@ -165,13 +165,13 @@ mod tests { const OTHER_OP: Uuid = Uuid::from_u128(0xb); /// A deadline that has already passed: the drain cancels at once. - const NOW: DrainPolicy = DrainPolicy { + pub(crate) const NOW: DrainPolicy = DrainPolicy { deadline_ms: 0, on_deadline: OnDeadline::Fail, }; /// A write-fenced source. - async fn fenced_source() -> Source { + pub(crate) async fn fenced_source() -> Source { let s = Source::new().await; raw(&s.conn().await, "insert into t values (1), (2)") .await @@ -193,7 +193,7 @@ mod tests { } } - fn read_fence(s: &Source, command_id: u128, policy: DrainPolicy) -> FenceRequest { + pub(crate) fn read_fence(s: &Source, command_id: u128, policy: DrainPolicy) -> FenceRequest { request( s, OP, diff --git a/libsql-server/src/namespace/fence/stream.rs b/libsql-server/src/namespace/fence/stream.rs new file mode 100644 index 0000000000..e57ba2f181 --- /dev/null +++ b/libsql-server/src/namespace/fence/stream.rs @@ -0,0 +1,594 @@ +//! Read leases for streams that serve namespace data outside SQL programs: `/dump` and the +//! replication `log_entries`, `batch_log_entries` and `snapshot` calls (`docs/NAMESPACE_FENCE.md` +//! section 9). +//! +//! A [`FencedStream`] holds a read lease of the stream's kind for as long as it can produce +//! items. A watcher task ends it when the gate stops admitting streams, or when the read drain +//! cancels it at its deadline: the watcher drops the inner stream and the lease itself, so the +//! lease is released even if the peer never polls the stream again, and the next poll yields +//! one terminal error carrying the typed fence code and then ends. +//! +//! A dump is not a stream of this kind (its data is produced by a blocking export on a +//! connection), so it holds its lease in the export itself and uses a [`StreamCancel`] to stop +//! it ([`crate::http::user::dump`]). + +use std::pin::Pin; +use std::sync::atomic::{AtomicBool, Ordering}; +use std::sync::Arc; +use std::task::{Context, Poll}; + +use futures::task::AtomicWaker; +use futures::Stream; +use parking_lot::Mutex; +use tokio::sync::{oneshot, Notify}; + +use super::controller::{FenceController, LeaseKind, ReadLease}; +use super::outcome::{FenceError, FenceOutcome}; +use super::state::OperationClass; + +/// How the read drain stops a stream at its deadline: a flag for code that polls it, and a +/// notification for code that waits. Setting it never blocks. +#[derive(Debug, Default)] +pub struct StreamCancel { + cancelled: AtomicBool, + notify: Notify, +} + +impl StreamCancel { + pub fn cancel(&self) { + self.cancelled.store(true, Ordering::Release); + // `notify_one` keeps a permit when nobody waits yet, so a waiter that arrives later + // still wakes. + self.notify.notify_one(); + } + + pub fn is_cancelled(&self) -> bool { + self.cancelled.load(Ordering::Acquire) + } + + /// Resolves once [`cancel`](Self::cancel) has been called. + pub async fn cancelled(&self) { + while !self.is_cancelled() { + self.notify.notified().await; + } + } +} + +/// The error a stream ends with when the read drain cancels it at its deadline. +pub fn cancelled_by_read_fence(kind: LeaseKind) -> FenceError { + FenceError::new( + FenceOutcome::MigrationReadFenced, + format!("the {kind:?} stream was ended by the namespace read fence"), + ) +} + +/// Admit a stream of `kind` and hold a read lease for it; the returned cancel is what the read +/// drain uses at its deadline. +pub fn acquire_stream_lease( + fence: &Arc, + kind: LeaseKind, +) -> Result<(ReadLease, Arc), FenceError> { + let cancel = Arc::new(StreamCancel::default()); + let lease = fence.acquire_read_lease(OperationClass::Stream, kind, { + let cancel = cancel.clone(); + move || cancel.cancel() + })?; + Ok((lease, cancel)) +} + +/// Resolves with the gate's refusal once the gate stops admitting streams. +async fn gate_denies_streams(fence: &FenceController) -> FenceError { + let mut gate = fence.subscribe(); + loop { + if let Err(e) = gate.borrow_and_update().permits(OperationClass::Stream) { + return e; + } + if gate.changed().await.is_err() { + // The controller is gone, which only happens when the namespace is destroyed: end + // the stream rather than serve it without a gate. + return FenceError::new( + FenceOutcome::FenceStateUnavailable, + "the namespace fence controller is gone", + ); + } + } +} + +enum Slot { + Live { + stream: S, + _lease: ReadLease, + /// Dropped with the stream, which stops the watcher. + _stop_watcher: oneshot::Sender<()>, + }, + /// Ended by the fence; the error is yielded once. + Terminated(Option), + /// Ended on its own. + Finished, +} + +struct Shared { + slot: Mutex>, + waker: AtomicWaker, +} + +impl Shared { + /// End the stream with `error`: drop the inner stream and the lease now, and wake the + /// consumer so that it sees the error when it next polls. + fn terminate(&self, error: FenceError) { + let ended = { + let mut slot = self.slot.lock(); + match &*slot { + Slot::Live { .. } => std::mem::replace(&mut *slot, Slot::Terminated(Some(error))), + _ => return, + } + }; + // Drop the inner stream and release the lease outside the slot lock. + drop(ended); + self.waker.wake(); + } +} + +/// A stream that holds a read lease and ends with a typed fence error when the gate stops +/// admitting streams or the read drain cancels it (see the module documentation). +pub struct FencedStream { + shared: Arc>, + map_err: F, +} + +impl FencedStream +where + S: Stream> + Unpin + Send + 'static, + F: Fn(FenceError) -> E + Unpin, +{ + /// Admit `stream` as a stream of `kind` on `fence`. Refused, with the gate's error, when the + /// gate does not admit streams. `map_err` turns the terminal fence error into the stream's + /// error type. Must be called within a Tokio runtime (the watcher is a task). + pub fn new( + fence: &Arc, + kind: LeaseKind, + stream: S, + map_err: F, + ) -> Result { + let (lease, cancel) = acquire_stream_lease(fence, kind)?; + let (stop_tx, stop_rx) = oneshot::channel(); + let shared = Arc::new(Shared { + slot: Mutex::new(Slot::Live { + stream, + _lease: lease, + _stop_watcher: stop_tx, + }), + waker: AtomicWaker::new(), + }); + let watched = Arc::downgrade(&shared); + let fence = fence.clone(); + tokio::spawn(async move { + let error = tokio::select! { + e = gate_denies_streams(&fence) => e, + _ = cancel.cancelled() => cancelled_by_read_fence(kind), + _ = stop_rx => return, + }; + if let Some(shared) = watched.upgrade() { + tracing::debug!( + namespace = %fence.namespace(), + "{kind:?} stream ended by the namespace fence: {error}" + ); + shared.terminate(error); + } + }); + Ok(Self { shared, map_err }) + } +} + +impl Stream for FencedStream +where + S: Stream> + Unpin, + F: Fn(FenceError) -> E + Unpin, +{ + type Item = Result; + + fn poll_next(self: Pin<&mut Self>, cx: &mut Context<'_>) -> Poll> { + let this = self.get_mut(); + this.shared.waker.register(cx.waker()); + let mut slot = this.shared.slot.lock(); + match &mut *slot { + Slot::Live { stream, .. } => match Pin::new(stream).poll_next(cx) { + Poll::Ready(None) => { + // Release the lease as soon as the stream has nothing more to serve. + let ended = std::mem::replace(&mut *slot, Slot::Finished); + drop(slot); + drop(ended); + Poll::Ready(None) + } + other => other, + }, + Slot::Terminated(error) => match error.take() { + Some(error) => Poll::Ready(Some(Err((this.map_err)(error)))), + None => Poll::Ready(None), + }, + Slot::Finished => Poll::Ready(None), + } + } +} + +/// The gRPC status of a fence error on a replication call. +pub fn fence_status(error: FenceError) -> tonic::Status { + error + .to_grpc_status() + .unwrap_or_else(|| tonic::Status::failed_precondition(error.to_string())) +} + +#[cfg(test)] +mod tests { + use bytes::Bytes; + use futures::{Stream, StreamExt}; + use libsql_replication::rpc::replication::replication_log_server::ReplicationLog; + use libsql_replication::rpc::replication::{ + Frame, HelloRequest, LogOffset, NAMESPACE_METADATA_KEY, SESSION_TOKEN_KEY, + }; + use tonic::metadata::{AsciiMetadataValue, BinaryMetadataValue}; + + use super::*; + use crate::error::Error; + use crate::http::user::dump::dump_stream; + use crate::namespace::fence::drain::tests::{fence_outcome, raw, Source, LONG, OP, PROMPT}; + use crate::namespace::fence::read::tests::{fenced_source, read_fence, NOW}; + use crate::namespace::fence::state::FenceState; + use crate::rpc::replication::replication_log::ReplicationLogService; + + /// A write-fenced source whose table holds one row of a megabyte (two in the dump's hex), + /// far more than the dump pipe buffers, so a dump whose peer stops reading blocks in the + /// middle of writing that row, where only the pipe's cancel can stop it. + async fn large_fenced_source() -> Source { + let s = Source::new().await; + raw( + &s.conn().await, + "insert into t values (randomblob(1000000))", + ) + .await + .unwrap(); + let acquired = s.execute(s.acquire(OP, 1, LONG)).await.unwrap(); + assert_eq!(fence_outcome(&acquired), FenceOutcome::Applied); + s + } + + async fn dump(s: &Source) -> crate::Result>> { + let maker = s + .store + .with("ns".into(), |ns| ns.db.connection_maker()) + .await + .unwrap(); + dump_stream(&s.fence, maker, false).await + } + + /// Read `stream` to its end: the bytes, and the error it ended with, if any. + async fn read_to_end( + mut stream: impl Stream> + Unpin, + ) -> (Vec, Option) { + let mut out = Vec::new(); + while let Some(chunk) = tokio::time::timeout(PROMPT, stream.next()) + .await + .expect("the dump stream stalled") + { + match chunk { + Ok(bytes) => out.extend_from_slice(&bytes), + Err(e) => return (out, Some(e)), + } + } + (out, None) + } + + fn assert_read_fenced(e: &Error) { + match e { + Error::NamespaceFence(f) => assert_eq!(f.outcome(), FenceOutcome::MigrationReadFenced), + other => panic!("expected MIGRATION_READ_FENCED, got {other:?}"), + } + } + + fn ends_with_commit(dump: &[u8]) -> bool { + String::from_utf8_lossy(dump) + .trim_end() + .ends_with("COMMIT;") + } + + /// A dump whose peer has stopped reading is cancelled at the read drain's deadline: its + /// lease is released without the peer reading anything more, the fence is acknowledged, + /// and the body ends with the fence error, never with `COMMIT;`. + #[tokio::test(flavor = "multi_thread")] + async fn dump_lease_released_on_cancel() { + let s = large_fenced_source().await; + let mut stream = Box::pin(dump(&s).await.unwrap()); + // Read until the row has begun: the export is then inside the write of a row that does + // not fit in the pipe, and blocked on it. + let mut head = Vec::new(); + while !String::from_utf8_lossy(&head).contains("INSERT INTO") { + head.extend_from_slice(&stream.next().await.unwrap().unwrap()); + } + assert_eq!(s.fence.read_lease_counts().dump, 1); + + let fenced = tokio::time::timeout(PROMPT, s.execute(read_fence(&s, 2, NOW))) + .await + .expect("the dump's lease was never released") + .unwrap(); + assert_eq!(fence_outcome(&fenced), FenceOutcome::Applied); + assert_eq!(s.fence.gate().state(), FenceState::SourceReadFenced); + assert_eq!(s.fence.read_lease_counts().total(), 0); + + let (rest, error) = read_to_end(stream).await; + assert_read_fenced(&error.expect("a cancelled dump must end with an error")); + let mut body = head; + body.extend_from_slice(&rest); + assert!(!ends_with_commit(&body)); + assert!(!String::from_utf8_lossy(&body).contains("COMMIT;")); + } + + /// The read drain waits for a running dump rather than for time: the fence is acknowledged + /// only once the dump has completed and released its lease. + #[tokio::test(flavor = "multi_thread")] + async fn read_fence_waits_for_dump() { + let s = large_fenced_source().await; + let mut stream = Box::pin(dump(&s).await.unwrap()); + let first = stream.next().await.unwrap().unwrap(); + + let fencing = s.execute(read_fence(&s, 2, LONG)); + s.until_state(FenceState::SourceReadDraining).await; + assert!(!fencing.is_finished()); + assert_eq!(s.fence.read_lease_counts().dump, 1); + // New dumps are refused while the drain waits for this one. + assert_read_fenced(&dump(&s).await.err().unwrap()); + + let (rest, error) = read_to_end(stream).await; + assert!(error.is_none(), "{error:?}"); + let mut body = first.to_vec(); + body.extend_from_slice(&rest); + assert!(ends_with_commit(&body)); + + let fenced = tokio::time::timeout(PROMPT, fencing) + .await + .unwrap() + .unwrap(); + assert_eq!(fence_outcome(&fenced), FenceOutcome::Applied); + assert_eq!(s.fence.read_lease_counts().total(), 0); + } + + /// A dump requested while reads are fenced is refused before any connection is created. + #[tokio::test] + async fn dump_refused_while_read_fenced() { + let s = fenced_source().await; + let fenced = s.execute(read_fence(&s, 2, LONG)).await.unwrap(); + assert_eq!(fence_outcome(&fenced), FenceOutcome::Applied); + assert_read_fenced(&dump(&s).await.err().unwrap()); + assert_eq!(s.fence.read_lease_counts().total(), 0); + } + + struct Replication { + service: ReplicationLogService, + token: Option, + } + + impl Replication { + /// The internal replication service on `s`, after a `hello`. + async fn new(s: &Source) -> Self { + let mut this = Self { + service: ReplicationLogService::new( + s.store.clone(), + None, + None, + false, + false, + true, + ), + token: None, + }; + let hello = this + .service + .hello(this.request(HelloRequest { + handshake_version: Some(1), + })) + .await + .unwrap() + .into_inner(); + this.token = Some(AsciiMetadataValue::try_from(&hello.session_token[..]).unwrap()); + this + } + + fn request(&self, msg: T) -> tonic::Request { + let mut req = tonic::Request::new(msg); + req.metadata_mut().insert_bin( + NAMESPACE_METADATA_KEY, + BinaryMetadataValue::from_bytes(b"ns"), + ); + if let Some(token) = &self.token { + req.metadata_mut().insert(SESSION_TOKEN_KEY, token.clone()); + } + req + } + + fn offset(&self, next_offset: u64) -> tonic::Request { + self.request(LogOffset { + next_offset, + wal_flavor: None, + }) + } + } + + fn assert_read_fenced_status(status: &tonic::Status) { + assert_eq!(status.code(), tonic::Code::FailedPrecondition, "{status:?}"); + assert_eq!( + FenceError::outcome_from_grpc_status(status), + Some(FenceOutcome::MigrationReadFenced), + "{status:?}" + ); + } + + async fn next_frame( + stream: &mut (impl Stream> + Unpin), + ) -> Option> { + tokio::time::timeout(PROMPT, stream.next()) + .await + .expect("the replication stream stalled") + } + + async fn until_released(fence: &FenceController) { + tokio::time::timeout(PROMPT, async { + loop { + let released = fence.read_released().notified(); + if fence.read_lease_counts().total() == 0 { + break; + } + released.await; + } + }) + .await + .expect("the stream's lease was never released") + } + + /// A tailing `log_entries` stream on a write-fenced source is served, holds a replication + /// lease, and is ended by the read fence with a typed terminal status, without the drain + /// having to wait for its deadline. + #[tokio::test(flavor = "multi_thread")] + async fn log_entries_stream_ends_typed() { + let s = fenced_source().await; + let r = Replication::new(&s).await; + let mut stream = r + .service + .log_entries(r.offset(1)) + .await + .unwrap() + .into_inner(); + assert!(next_frame(&mut stream).await.unwrap().is_ok()); + assert_eq!(s.fence.read_lease_counts().replication, 1); + + let fenced = tokio::time::timeout(PROMPT, s.execute(read_fence(&s, 2, LONG))) + .await + .expect("the tailing stream kept the drain waiting") + .unwrap(); + assert_eq!(fence_outcome(&fenced), FenceOutcome::Applied); + assert_eq!(s.fence.read_lease_counts().total(), 0); + + let last = next_frame(&mut stream).await.unwrap(); + assert_read_fenced_status(&last.unwrap_err()); + assert!(next_frame(&mut stream).await.is_none()); + } + + /// A stream whose peer never reads again (a dead peer) releases its lease anyway: the + /// watcher drops the inner stream and the lease without the stream being polled. + #[tokio::test(flavor = "multi_thread")] + async fn stream_lease_released_without_peer_read() { + let s = fenced_source().await; + let r = Replication::new(&s).await; + let stream = r + .service + .log_entries(r.offset(1)) + .await + .unwrap() + .into_inner(); + assert_eq!(s.fence.read_lease_counts().replication, 1); + + let fenced = tokio::time::timeout(PROMPT, s.execute(read_fence(&s, 2, LONG))) + .await + .expect("the unread stream kept the drain waiting") + .unwrap(); + assert_eq!(fence_outcome(&fenced), FenceOutcome::Applied); + until_released(&s.fence).await; + + // The frames the stream had not yet produced are never served. + let mut stream = stream; + assert_read_fenced_status(&next_frame(&mut stream).await.unwrap().unwrap_err()); + assert!(next_frame(&mut stream).await.is_none()); + } + + /// A `snapshot` stream is ended by the read fence the same way. + #[tokio::test(flavor = "multi_thread")] + async fn snapshot_stream_ends_typed() { + // Compact at once, so the log's frames are in a snapshot. + let s = Source::with_max_log_size(0).await; + raw(&s.conn().await, "insert into t values (1), (2)") + .await + .unwrap(); + // The periodic compaction may already have run; either way there is nothing left to + // compact once this returns. + let logger = s.logger.clone(); + tokio::task::spawn_blocking(move || logger.maybe_compact()) + .await + .unwrap() + .unwrap(); + tokio::time::timeout(PROMPT, async { + // The snapshot is written by the compactor's own thread. + while s.logger.get_snapshot_file(1).await.unwrap().is_none() { + tokio::time::sleep(std::time::Duration::from_millis(10)).await; + } + }) + .await + .expect("the snapshot was never written"); + let acquired = s.execute(s.acquire(OP, 1, LONG)).await.unwrap(); + assert_eq!(fence_outcome(&acquired), FenceOutcome::Applied); + + let r = Replication::new(&s).await; + let stream = r.service.snapshot(r.offset(1)).await.unwrap().into_inner(); + assert_eq!(s.fence.read_lease_counts().replication, 1); + + let fenced = tokio::time::timeout(PROMPT, s.execute(read_fence(&s, 2, LONG))) + .await + .expect("the snapshot stream kept the drain waiting") + .unwrap(); + assert_eq!(fence_outcome(&fenced), FenceOutcome::Applied); + until_released(&s.fence).await; + + let mut stream = stream; + assert_read_fenced_status(&next_frame(&mut stream).await.unwrap().unwrap_err()); + assert!(next_frame(&mut stream).await.is_none()); + } + + /// While reads are fenced every replication call is refused at its start with the typed + /// status (never `UNAVAILABLE`), and no lease is taken. + #[tokio::test] + async fn replication_calls_denied_while_read_fenced() { + let s = fenced_source().await; + let r = Replication::new(&s).await; + let fenced = s.execute(read_fence(&s, 2, LONG)).await.unwrap(); + assert_eq!(fence_outcome(&fenced), FenceOutcome::Applied); + + let hello = r + .service + .hello(r.request(HelloRequest { + handshake_version: Some(1), + })) + .await; + assert_read_fenced_status(&hello.unwrap_err()); + assert_read_fenced_status(&r.service.log_entries(r.offset(1)).await.err().unwrap()); + assert_read_fenced_status( + &r.service + .batch_log_entries(r.offset(1)) + .await + .err() + .unwrap(), + ); + assert_read_fenced_status(&r.service.snapshot(r.offset(1)).await.err().unwrap()); + assert_eq!(s.fence.read_lease_counts().total(), 0); + } + + /// The deadline cancel of a stream (the backstop when the gate has not ended it) also ends + /// it with the typed status and releases the lease without the stream being polled. + #[tokio::test] + async fn read_fence_forced_termination() { + let s = fenced_source().await; + let (_tx, rx) = tokio::sync::mpsc::channel::>(1); + let mut stream = FencedStream::new( + &s.fence, + LeaseKind::Replication, + tokio_stream::wrappers::ReceiverStream::new(rx), + fence_status as fn(FenceError) -> tonic::Status, + ) + .unwrap(); + assert_eq!(s.fence.cancel_read_leases(), 1); + until_released(&s.fence).await; + let status = tokio::time::timeout(PROMPT, stream.next()) + .await + .unwrap() + .unwrap() + .unwrap_err(); + assert_read_fenced_status(&status); + assert!(stream.next().await.is_none()); + } +} diff --git a/libsql-server/src/namespace/store.rs b/libsql-server/src/namespace/store.rs index a472fc39c3..ca8e7031be 100644 --- a/libsql-server/src/namespace/store.rs +++ b/libsql-server/src/namespace/store.rs @@ -632,6 +632,13 @@ pub(crate) mod fence_tests { const OP: Uuid = Uuid::from_u128(0xa); pub(crate) async fn open_store(dir: &Path) -> NamespaceStore { + open_store_with_max_log_size(dir, 1_000_000_000).await + } + + pub(crate) async fn open_store_with_max_log_size( + dir: &Path, + max_log_size: u64, + ) -> NamespaceStore { let (maker, manager) = metastore_connection_maker(None, dir).await.unwrap(); let meta = MetaStore::new( MetaStoreConfig { @@ -660,7 +667,7 @@ pub(crate) mod fence_tests { disable_intelligent_throttling: false, }, PrimaryConfig { - max_log_size: 1_000_000_000, + max_log_size, max_log_duration: None, bottomless_replication: None, scripted_backup: None, diff --git a/libsql-server/src/rpc/mod.rs b/libsql-server/src/rpc/mod.rs index 2936a8742c..af74fa4b4c 100644 --- a/libsql-server/src/rpc/mod.rs +++ b/libsql-server/src/rpc/mod.rs @@ -29,6 +29,7 @@ pub async fn run_rpc_server( maybe_tls: Option, idle_shutdown_layer: Option, service: BoxReplicationService, + http2_keepalive_interval: Option, ) -> anyhow::Result<()> { if let Some(tls_config) = maybe_tls { let cert_pem = tokio::fs::read_to_string(&tls_config.cert).await?; @@ -81,8 +82,13 @@ pub async fn run_rpc_server( .service(router); tracing::info!("serving internal rpc server with tls"); - let h2c = crate::h2c::H2cMaker::new(svc); - hyper::server::Server::builder(acceptor).serve(h2c).await?; + let h2c = crate::h2c::H2cMaker::new(svc).with_http2_keepalive(http2_keepalive_interval); + crate::h2c::with_http2_keepalive( + hyper::server::Server::builder(acceptor), + http2_keepalive_interval, + ) + .serve(h2c) + .await?; } else { let proxy = ProxyServer::new(proxy_service); let replication = ReplicationLogServer::new(service); @@ -105,11 +111,16 @@ pub async fn run_rpc_server( ) .service(router); - let h2c = crate::h2c::H2cMaker::new(svc); + let h2c = crate::h2c::H2cMaker::new(svc).with_http2_keepalive(http2_keepalive_interval); tracing::info!("serving internal rpc server without tls"); - hyper::server::Server::builder(acceptor).serve(h2c).await?; + crate::h2c::with_http2_keepalive( + hyper::server::Server::builder(acceptor), + http2_keepalive_interval, + ) + .serve(h2c) + .await?; } Ok(()) } diff --git a/libsql-server/src/rpc/replication/replication_log.rs b/libsql-server/src/rpc/replication/replication_log.rs index 1f7585f34f..4457c49e7a 100644 --- a/libsql-server/src/rpc/replication/replication_log.rs +++ b/libsql-server/src/rpc/replication/replication_log.rs @@ -1,7 +1,8 @@ -use std::collections::HashSet; +use std::collections::{HashMap, HashSet}; use std::net::SocketAddr; use std::pin::Pin; use std::sync::{Arc, RwLock}; +use std::time::{Duration, Instant}; use bytes::Bytes; use chrono::{DateTime, Utc}; @@ -22,6 +23,10 @@ use uuid::Uuid; use crate::auth::Auth; use crate::connection::config::DatabaseConfig; +use crate::namespace::fence::controller::{FenceController, LeaseKind}; +use crate::namespace::fence::outcome::FenceError; +use crate::namespace::fence::state::OperationClass; +use crate::namespace::fence::stream::{fence_status, FencedStream}; use crate::namespace::{NamespaceName, NamespaceStore}; use crate::replication::primary::frame_stream::FrameStream; use crate::replication::{LogReadError, ReplicationLogger}; @@ -44,10 +49,16 @@ pub struct ReplicationLogService { //deprecated: generation_id: Uuid, replicas_with_hello: RwLock>, + /// When a fence denial of a replication call was last logged, per namespace. + fence_denials_logged: parking_lot::Mutex>, } pub const MAX_FRAMES_PER_BATCH: usize = 1024; +/// Replication calls denied by a namespace fence are logged at most this often per namespace +/// (they are all counted): replicas that do not understand the typed code reconnect in a loop. +const FENCE_DENIAL_LOG_INTERVAL: Duration = Duration::from_secs(60); + impl ReplicationLogService { pub fn new( namespaces: NamespaceStore, @@ -68,7 +79,60 @@ impl ReplicationLogService { generation_id: Uuid::new_v4(), replicas_with_hello: Default::default(), service_internal, + fence_denials_logged: Default::default(), + } + } + + /// The status of a replication call (or stream) that the namespace fence refuses. Counted + /// every time, and logged at most once per namespace per [`FENCE_DENIAL_LOG_INTERVAL`]. + fn fence_denied(&self, namespace: &NamespaceName, call: &str, error: FenceError) -> Status { + metrics::increment_counter!( + "libsql_server_fence_denials_total", + "code" => error.outcome().as_str(), + "surface" => "replication", + ); + let now = Instant::now(); + let log = { + let mut logged = self.fence_denials_logged.lock(); + match logged.get(namespace) { + Some(at) if now.duration_since(*at) < FENCE_DENIAL_LOG_INTERVAL => false, + _ => { + logged.insert(namespace.clone(), now); + true + } + } + }; + if log { + tracing::warn!( + namespace = %namespace, + internal = self.service_internal, + "replication {call} refused by the namespace fence: {error} \ + (further refusals for this namespace are not logged for a minute)" + ); } + fence_status(error) + } + + /// Admit `stream`, a replication stream on `namespace`, under the namespace's fence + /// (`docs/NAMESPACE_FENCE.md` section 9): it holds a replication read lease and ends with a + /// typed `FAILED_PRECONDITION` when the gate stops admitting streams. + fn fenced_stream( + &self, + namespace: &NamespaceName, + call: &str, + fence: &Arc, + stream: S, + ) -> Result Status>, Status> + where + S: futures::Stream> + Unpin + Send + 'static, + { + FencedStream::new( + fence, + LeaseKind::Replication, + stream, + fence_status as fn(FenceError) -> Status, + ) + .map_err(|e| self.fence_denied(namespace, call, e)) } async fn authenticate( @@ -124,9 +188,12 @@ impl ReplicationLogService { Ok(()) } + /// The namespace's replication log and what goes with it, for a replication `call`. Refused + /// with the typed fence status when the namespace's fence does not admit streams. async fn logger_from_namespace( &self, namespace: NamespaceName, + call: &str, req: &tonic::Request, verify_session: bool, ) -> Result< @@ -136,12 +203,13 @@ impl ReplicationLogService { usize, Arc, impl Future, + Arc, ), Status, > { - let (logger, config, version, stats, config_changed) = self + let (logger, config, version, stats, config_changed, fence) = self .namespaces - .with(namespace, |ns| -> Result<_, Status> { + .with(namespace.clone(), |ns| -> Result<_, Status> { let logger = ns .db .logger() @@ -150,23 +218,30 @@ impl ReplicationLogService { let config = ns.config(); let version = ns.config_version(); let stats = ns.stats(); + let fence = ns.fence().clone(); - Ok((logger, config, version, stats, config_changed)) + Ok((logger, config, version, stats, config_changed, fence)) }) .await - .map_err(|e| { - if let crate::error::Error::NamespaceDoesntExist(_) = e.as_ref() { + .map_err(|e| match e.as_ref() { + crate::error::Error::NamespaceDoesntExist(_) => { Status::failed_precondition(NAMESPACE_DOESNT_EXIST) - } else { - Status::internal(e.to_string()) } + crate::error::Error::NamespaceFence(f) => { + self.fence_denied(&namespace, call, f.clone()) + } + _ => Status::internal(e.to_string()), })??; + fence + .permits(OperationClass::Stream) + .map_err(|e| self.fence_denied(&namespace, call, e))?; + if verify_session { self.verify_session_token(req, version)?; } - Ok((logger, config, version, stats, config_changed)) + Ok((logger, config, version, stats, config_changed, fence)) } fn encode_session_token(&self, version: usize) -> Uuid { @@ -254,8 +329,9 @@ impl ReplicationLog for ReplicationLogService { self.authenticate(&req, namespace.clone()).await?; - let (logger, _, _, stats, config_changed) = - self.logger_from_namespace(namespace, &req, true).await?; + let (logger, _, _, stats, config_changed, fence) = self + .logger_from_namespace(namespace.clone(), "log_entries", &req, true) + .await?; let stats = if self.collect_stats { Some(stats) @@ -288,6 +364,8 @@ impl ReplicationLog for ReplicationLogService { } }; + let stream = self.fenced_stream(&namespace, "log_entries", &fence, Box::pin(stream))?; + Ok(tonic::Response::new(Box::pin(stream))) } @@ -301,7 +379,9 @@ impl ReplicationLog for ReplicationLogService { let namespace = super::super::extract_namespace(self.disable_namespaces, &req)?; self.authenticate(&req, namespace.clone()).await?; - let (logger, _, _, stats, _) = self.logger_from_namespace(namespace, &req, true).await?; + let (logger, _, _, stats, _, fence) = self + .logger_from_namespace(namespace.clone(), "batch_log_entries", &req, true) + .await?; let stats = if self.collect_stats { Some(stats) @@ -322,9 +402,12 @@ impl ReplicationLog for ReplicationLogService { .map_err(|e| Status::internal(e.to_string()))?, self.idle_shutdown_layer.clone(), ) - .map(map_frame_stream_output) - .collect::, _>>() - .await?; + .map(map_frame_stream_output); + // The batch is read under a replication read lease, so a read drain waits for it. + let frames = self + .fenced_stream(&namespace, "batch_log_entries", &fence, frames)? + .collect::, _>>() + .await?; Ok(tonic::Response::new(Frames { frames })) } @@ -348,8 +431,9 @@ impl ReplicationLog for ReplicationLogService { guard.insert((replica_addr, namespace.clone())); } } - let (logger, config, version, _, _) = - self.logger_from_namespace(namespace, &req, false).await?; + let (logger, config, version, _, _, _) = self + .logger_from_namespace(namespace, "hello", &req, false) + .await?; // If we are a shared schema and serving externally (aka to embedded replica's) then // return an error. @@ -383,7 +467,9 @@ impl ReplicationLog for ReplicationLogService { let namespace = super::super::extract_namespace(self.disable_namespaces, &req)?; self.authenticate(&req, namespace.clone()).await?; - let (logger, _, _, stats, _) = self.logger_from_namespace(namespace, &req, true).await?; + let (logger, _, _, stats, _, fence) = self + .logger_from_namespace(namespace.clone(), "snapshot", &req, true) + .await?; let stats = if self.collect_stats { Some(stats) @@ -395,9 +481,17 @@ impl ReplicationLog for ReplicationLogService { let offset = req.next_offset; match logger.get_snapshot_file(offset).await { - Ok(Some(snapshot)) => Ok(tonic::Response::new(Box::pin( - snapshot_stream::make_snapshot_stream(snapshot, offset, stats), - ))), + Ok(Some(snapshot)) => { + let stream = self.fenced_stream( + &namespace, + "snapshot", + &fence, + Box::pin(snapshot_stream::make_snapshot_stream( + snapshot, offset, stats, + )), + )?; + Ok(tonic::Response::new(Box::pin(stream))) + } Ok(None) => Err(Status::new(tonic::Code::Unavailable, "snapshot not found")), Err(e) => Err(Status::new(tonic::Code::Internal, e.to_string())), } From 88a618f16c668cf1785b1e75fe64f17a9fe508dd Mon Sep 17 00:00:00 2001 From: River Date: Tue, 29 Sep 2026 18:20:21 +0000 Subject: [PATCH 14/33] libsql-server: create migration targets in quarantine Add `NamespaceStore::create_target_quarantined` (and route the admin `CreateTargetQuarantined` command through it), which creates a namespace as a quarantined migration target atomically with namespace creation: - A name the server already knows (config in memory, or a namespace cache entry) is refused with FENCE_PRECONDITION_FAILED/namespace_exists without touching its gate. - Otherwise the controller publishes an in-memory target-creation gate before the metastore transaction writes the marker, config row, record and receipt. The gate refuses every class but maintenance and observability, and `check_available` refuses the name, so create, fork and `with()` neither store nor set anything up for it. - The committed record's quarantine gate replaces it; only then is the config published into the in-memory map (from the durable row, with the record's own block values) and the namespace loaded, so its first connection maker is created behind the quarantine gate. - The command runs on its own task under the transition lock; a replay returns the stored result and completes the publication and load of a commit that was not acknowledged, or of a creation interrupted between its marker and its commit. `CreateTargetRequest` is the typed entry point for the admin route and bulk import. The target's log id is not written back into the record, to keep the marker rule for same-revision records intact (documented). Co-authored-by: Tomasz Szymczyszyn --- docs/NAMESPACE_FENCE.md | 37 +- .../src/namespace/fence/controller.rs | 80 ++- libsql-server/src/namespace/fence/mod.rs | 6 +- libsql-server/src/namespace/fence/registry.rs | 6 +- libsql-server/src/namespace/fence/target.rs | 546 ++++++++++++++++++ libsql-server/src/namespace/meta_store.rs | 52 +- libsql-server/src/namespace/store.rs | 159 ++++- 7 files changed, 863 insertions(+), 23 deletions(-) create mode 100644 libsql-server/src/namespace/fence/target.rs diff --git a/docs/NAMESPACE_FENCE.md b/docs/NAMESPACE_FENCE.md index 1723996586..ffe64a2e43 100644 --- a/docs/NAMESPACE_FENCE.md +++ b/docs/NAMESPACE_FENCE.md @@ -406,11 +406,12 @@ This is additive in proto3: older peers skip the unknown field; a newer replica Per namespace: - `transition_lock`: a `tokio::sync::Mutex` serialising commands on this namespace. A command holds it from its first check to its response (`FenceController::begin_transition` returns a `Transition` that owns the guard). -- `gate`: a `tokio::sync::watch` of `GateSnapshot { fence, write_generation, indeterminate, installing, closing_reads }`, where `fence` is the durable fence as last published (a record, no record, or `UNKNOWN_UNAVAILABLE` with its detail) and `installing` is the in-memory `INSTALLING` gate of a closing command being persisted (section 8.3), which denies normal writes, vacuum, import writes and lifecycle work on top of `fence`. State, revision, owning operation and every admission (`permits(class)`, `write()`, `read()`) are derived from it through the permission matrix. The WAL wrapper, `CoreConnection`, dump, replication and lifecycle code read it without locks. Phase 5 adds the live capability set. +- `gate`: a `tokio::sync::watch` of `GateSnapshot { fence, write_generation, indeterminate, installing, closing_reads, creating_target }`, where `fence` is the durable fence as last published (a record, no record, or `UNKNOWN_UNAVAILABLE` with its detail) and `installing` is the in-memory `INSTALLING` gate of a closing command being persisted (section 8.3), which denies normal writes, vacuum, import writes and lifecycle work on top of `fence`. State, revision, owning operation and every admission (`permits(class)`, `write()`, `read()`) are derived from it through the permission matrix. The WAL wrapper, `CoreConnection`, dump, replication and lifecycle code read it without locks. Phase 5 adds the live capability set. - `write_generation: u64`, in the snapshot, **incremented on every publication that changes the fence state, the owning operation, or the indeterminate flag**. That covers every transition that closes or opens write admission (acquire, release, create, seal, publish, enable, abort, adopt), and is conservative for the others. A replay that publishes the same durable state does not move it. - `indeterminate: Option<(operation_id, command_id)>`: set when a command's `COMMIT` failed (or the task running it died) so that whether it applied is unknown. While set, every class except `Maintenance` and `Observability` is denied with `FENCE_STATE_UNAVAILABLE` / `indeterminate_commit`, and every other command is refused with `FENCE_COMMIT_INDETERMINATE`. A replay of the same command is answered by the metastore from the durable row (replayed if it had committed, applied if it had not) and clears it (section 8.4). - the write-drain sources of the namespace's primary connection makers (`register_write_drain`): each maker's connection manager, held weakly, and its replication log id and last-committed-frame reader, which the write drain waits on and reads the boundary from (section 8.3). - the write queues of the namespace's connection managers: each `MakeLegacyConnection` registers a waker (`register_write_queue`) that the controller calls after every publication that moves `write_generation`, after the new gate is visible. A waker holds its manager weakly and is dropped once the manager is gone (for example after eviction). Writer tracking itself is the connection manager's (section 8.2). +- `creating_target` (in the snapshot): the in-memory target-creation gate of a `CreateTargetQuarantined` being persisted (section 10.1), which denies every class except `Maintenance` and `Observability` with `MIGRATION_TARGET_QUARANTINED`, and makes `FenceRegistry::check_available` refuse the name, so nothing sets the namespace up, serves it or stores a config for it. Never persisted; it moves `write_generation`; the publication of the committed record replaces it, it is removed when the command is proven not to have committed, and it stays (with `indeterminate`) when the outcome is unknown. - `closing_reads` (in the snapshot): the in-memory read-closing gate of a `SetSourceReadFence` being persisted (section 9, step 2), which denies `NormalRead` and `Stream` with `MIGRATION_READ_FENCED` on top of `fence`. Never persisted; every publication clears it; it does not move `write_generation` (writes are already closed wherever a read fence can be set). - `read_leases`: the live read leases, each with its kind (`sql`, `dump`, `replication`), a cancel handle and a cancelled flag, and a `Notify` on every release. `FenceController::acquire_read_lease(class, kind, cancel)` checks the gate **under the lease lock** and registers the lease, so a lease is either refused by a read-closing gate published before it, or counted by a drain that closes admission after it; once admission is closed the set can only shrink. A `ReadLease` is released when dropped. - `capabilities`: the live `MigrationCapability` set and an import-writer counter. @@ -530,14 +531,17 @@ This fence stops future service from the source. It cannot recall bytes a peer h ### 10.1 CreateTargetQuarantined -1. Checks (section 5.3). The name must have no config row, no fence row and no marker (`FENCE_PRECONDITION_FAILED`, `namespace_exists`). -2. Create `dbs//` and write the marker (`TARGET_QUARANTINED`, revision 1). -3. One metastore transaction: insert the config row (with the legacy mirror `block_reads = block_writes = true`), the fence row (`TARGET_QUARANTINED`, revision 1, new `target_incarnation_id`) and the receipt. -4. Install the controller in the registry with the quarantine gate. -5. Only now insert the config into the metastore's in-memory map (which is what makes `exists()` true) and call `load_namespace`. The first connection maker is created with the quarantine gate already in place. -6. Record the new `log_id` in the record (same revision, informational) and respond `APPLIED`. +`NamespaceStore::create_target_quarantined(CreateTargetRequest, ServerIdentity)` (the admin route calls the same code through `execute_fence_command`) runs on its own task under the namespace's transition lock: -A crash after step 2 leaves a marker with no rows: `UNKNOWN_UNAVAILABLE` until the same command is replayed, which completes it. A crash after step 3 recovers `TARGET_QUARANTINED` from the metastore. +1. Checks. A name without fence state that the server already knows — its config is in the in-memory map, or the namespace cache holds it (loaded, or a fork in flight) — is refused with `FENCE_PRECONDITION_FAILED`, `namespace_exists`, without touching its gate. Otherwise the controller (get-or-create in the registry) publishes the in-memory **target-creation gate** (section 7.2): from here on every class but maintenance and observability is refused and `check_available` refuses the name, so `with()`, `make_namespace`, `create` and the fork destination refuse it before storing or setting anything up. The in-use check is repeated with the gate in place, which closes the race with a create or fork that had not stored anything yet. The metastore then checks section 5.3 and requires no config row, no fence row and no marker (`namespace_exists`). A name that already has fence state (a replay, a creation interrupted after its marker, a refusal) keeps the gate its state implies and goes straight to the metastore. +2. Inside the metastore transaction, create `dbs//` and write the marker (`TARGET_QUARANTINED`, revision 1). +3. In the same transaction: insert the config row (with the legacy mirror `block_reads = block_writes = true`), the fence row (`TARGET_QUARANTINED`, revision 1, new `target_incarnation_id`) and the receipt; commit. +4. The controller publishes the committed record, whose quarantine gate replaces the creation gate (commit → publish, as for every command). A command proven not to have committed removes the creation gate; one whose commit is unknown keeps it with the indeterminate flag. +5. Only now `MetaStore::publish_target_config` puts the config into the in-memory map (which is what makes `exists()` and `lookup()` find it): the stored row with the record's own `block_*` values in place of the legacy mirror, replacing any entry a refused create or fork of the same name had left there. An empty namespace-cache entry left by a refused fork is dropped, and the namespace is loaded; its first connection maker is created with the quarantine gate already in place, in a directory that until then held only the marker. The response is `APPLIED`. + +A replay of the same command returns the stored receipt and repeats step 5, which completes a creation whose commit was not acknowledged. A crash after step 2 leaves a marker with no rows: `UNKNOWN_UNAVAILABLE` (`incomplete_target_creation`) until the same command is replayed, which completes it with the incarnation id the marker announced. A crash after step 3 recovers `TARGET_QUARANTINED` from the metastore, and startup publishes its config. + +The target's replication `log_id` is not written back into the record: rewriting a record at the same revision would make a marker written before the rewrite look like `metastore_behind_marker` after a crash between the two. The log id is the namespace's own and is read live where it is needed (a later `AcquireSourceWriteFence` on the published target checks the caller's `expected_log_id` against it). ### 10.2 SealTargetImport @@ -559,6 +563,14 @@ The bulk import work consumes this internal Rust API, which does not depend on a // namespace::fence pub enum TargetState { Quarantined, ImportDraining, Validating, WriteFenced, Writable, Aborted } +/// CreateTargetQuarantined; the expectation is always ABSENT at revision 0. +pub struct CreateTargetRequest { + pub namespace: NamespaceName, + pub operation_id: Uuid, + pub command_id: Uuid, // idempotency key: a replay returns the stored result + pub config: TargetConfig, +} + pub struct MigrationCapability { // server-created, not constructible outside the module id: Uuid, namespace: NamespaceName, @@ -568,9 +580,10 @@ pub struct MigrationCapability { // server-created, not constructible outsi } impl NamespaceStore { - /// CreateTargetQuarantined, atomic with namespace creation. - pub async fn create_target_quarantined(&self, req: CreateTargetRequest) - -> Result; + /// CreateTargetQuarantined, atomic with namespace creation (section 10.1). Fence refusals + /// are `Error::NamespaceFence(FenceError)` with their stable outcome code. + pub async fn create_target_quarantined(&self, req: CreateTargetRequest, + server: ServerIdentity) -> crate::Result; /// Issue an import capability and a capability-bearing connection. Valid only in /// TARGET_QUARANTINED for the owning operation at `expected_revision`. @@ -720,7 +733,7 @@ Planned test names; the table is updated as tests land. | 8 | Evict and lazily reload a fenced namespace; identical admission | `tests::fence::lifecycle::evicted_namespace_reloads_same_gate`; landed at unit level: `namespace::store::fence_tests::evicted_namespace_reloads_with_the_same_controller`, `fence::registry::tests::seeded_from_load_fences_including_recovered_names` | | 9 | Filesystem recovery, `destroy_on_error`, undecodable records, missing target quarantine, metastore backup rollback fail closed with provenance | `meta_store::fence_tests::recovery::{fs_recovery_with_marker_unavailable, destroy_on_error_keeps_fenced_unavailable, undecodable_row_unavailable, incomplete_target_unavailable, metastore_rollback_detected_by_marker, lookup_never_creates, undecodable_name_with_fence_fails_startup, marker_in_invalid_directory_fails_startup}`; legacy behaviour kept: `destroy_on_error_without_fences_is_unchanged`, `undecodable_row_without_fences_is_skipped_as_before`; `meta_store::fence_tests::corrupt_fence_row_fails_closed` | | 10 | Wrong owner, stale revision, invalid role/state, replay, command-id reuse; replay before revision check | `fence::transition::tests::*` (exhaustive over states × commands) | -| 11 | Target creation raced with SQL, dump, replication, lifecycle never observable as writable or readable | `fence::target::tests::create_race_never_observable` | +| 11 | Target creation raced with SQL, dump, replication, lifecycle never observable as writable or readable | landed: `namespace::fence::target::tests::create_race_never_observable` (parked after the rows commit and before the config is published and the namespace loaded: SQL connections, stats, replication `hello` (never `UNAVAILABLE`), create, delete and fork of the name are denied or find nothing; afterwards the target is loaded behind the quarantine gate, SQL reads and WAL writes are refused, and lifecycle and replication are refused with `MIGRATION_TARGET_QUARANTINED`), `creating_gate_refuses_before_commit` (parked before the metastore transaction: the same attempts are refused and no database file is created), `create_replay_completes_interrupted_creation` (marker only, after a restart), `create_completes_when_the_caller_goes_away`, `indeterminate_create_is_completed_by_replay`, `create_rejects_existing_name` (a loaded or cold existing name, whose gate never moves, and another operation's target), `abort_keeps_traffic_denied` (also across a restart) | | 12 | Only the matching import capability writes a quarantined target; admin credentials and admin shell cannot | `fence::target::tests::import_requires_matching_capability`; `tests::fence::admin::admin_shell_cannot_write_quarantined` | | 13 | Seal enters `TARGET_IMPORT_DRAINING`, waits, reaches `TARGET_VALIDATING`, cannot resume import; only a durable validation receipt permits idempotent publication | `fence::target::tests::seal_waits_for_import_writers`, `sealed_target_rejects_import`, `publish_requires_validation_receipt`, `publish_is_idempotent` | | 14 | Enable writes idempotent, survives restart and response loss, irreversible | `fence::target::tests::enable_writes_idempotent_and_irreversible`, `enable_writes_survives_restart` | diff --git a/libsql-server/src/namespace/fence/controller.rs b/libsql-server/src/namespace/fence/controller.rs index ff946554c6..04149a66b4 100644 --- a/libsql-server/src/namespace/fence/controller.rs +++ b/libsql-server/src/namespace/fence/controller.rs @@ -58,6 +58,11 @@ pub struct GateSnapshot { /// allows. Never persisted, and cleared by every publication of a commit. It does not move /// the write generation: write admission is already closed wherever a read fence can be set. pub closing_reads: Option, + /// The in-memory gate of a `CreateTargetQuarantined` that is being persisted (section + /// 10.1): the name is becoming a quarantined target, so everything but maintenance and + /// observability is refused with `MIGRATION_TARGET_QUARANTINED`, and the namespace is not + /// set up. Never persisted; replaced by the record the command's commit publishes. + pub creating_target: Option, } impl GateSnapshot { @@ -68,6 +73,7 @@ impl GateSnapshot { indeterminate: None, installing: None, closing_reads: None, + creating_target: None, } } @@ -104,6 +110,20 @@ impl GateSnapshot { .with_detail(FenceDetail::IndeterminateCommit)); } } + if let Some((operation_id, command_id)) = self.creating_target { + if !matches!( + class, + OperationClass::Maintenance | OperationClass::Observability + ) { + return Err(FenceError::new( + FenceOutcome::MigrationTargetQuarantined, + format!( + "{class:?} is not permitted: fence command {command_id} of operation \ + {operation_id} is creating this namespace as a migration target" + ), + )); + } + } self.fence.permits(class)?; if let Some((operation_id, command_id)) = self.installing { if matches!( @@ -141,6 +161,11 @@ impl GateSnapshot { self.installing.is_some() } + /// Whether the namespace is being created as a quarantined target. + pub fn is_creating_target(&self) -> bool { + self.creating_target.is_some() + } + /// Normal write admission. pub fn write(&self) -> Admission { Admission::from( @@ -507,14 +532,31 @@ impl FenceController { } /// Publish a new gate. `fence: None` keeps the published fence. The write generation moves - /// whenever the state, the owning operation, the indeterminate flag or the installing gate - /// changes. Every publication removes the read-closing gate: the commit that follows it - /// either persists the read fence or proves that nothing changed. + /// whenever the state, the owning operation, the indeterminate flag, the installing gate or + /// the target-creation gate changes. Every publication removes the read-closing gate: the + /// commit that follows it either persists the read fence or proves that nothing changed. + /// A publication with a fence also removes the target-creation gate, which the committed + /// record replaces; one without keeps it. fn publish( &self, fence: Option, indeterminate: Option, installing: Option, + ) { + let creating_target = if fence.is_some() { + None + } else { + self.gate.borrow().creating_target + }; + self.publish_gate(fence, indeterminate, installing, creating_target); + } + + fn publish_gate( + &self, + fence: Option, + indeterminate: Option, + installing: Option, + creating_target: Option, ) { let mut generation_changed = false; self.gate.send_modify(|gate| { @@ -523,10 +565,12 @@ impl FenceController { || fence.record().map(|r| r.operation_id) != gate.fence.record().map(|r| r.operation_id) || indeterminate != gate.indeterminate - || installing != gate.installing; + || installing != gate.installing + || creating_target != gate.creating_target; gate.fence = fence; gate.indeterminate = indeterminate; gate.installing = installing; + gate.creating_target = creating_target; gate.closing_reads = None; if changed { gate.write_generation += 1; @@ -542,6 +586,7 @@ impl FenceController { write_generation = gate.write_generation, indeterminate = gate.indeterminate.is_some(), installing = gate.installing.is_some(), + creating_target = gate.creating_target.is_some(), "published namespace fence gate" ); } @@ -586,6 +631,33 @@ impl Transition { } } + /// Publish the in-memory target-creation gate for `CreateTargetQuarantined` `key` (section + /// 10.1): until the command's commit publishes the quarantined record, every class but + /// maintenance and observability is refused and the namespace is not set up. It is removed + /// with [`remove_creating_target`](Self::remove_creating_target) when the command is proven + /// not to have committed, and kept (with the indeterminate flag) when its outcome is + /// unknown. + pub fn install_creating_target(&mut self, key: CommandKey) { + let (indeterminate, installing) = { + let gate = self.controller.gate.borrow(); + (gate.indeterminate, gate.installing) + }; + self.controller + .publish_gate(None, indeterminate, installing, Some(key)); + } + + /// Remove the target-creation gate of a command that was proven not to have committed. + pub fn remove_creating_target(&mut self) { + let (indeterminate, installing, creating) = { + let gate = self.controller.gate.borrow(); + (gate.indeterminate, gate.installing, gate.creating_target) + }; + if creating.is_some() { + self.controller + .publish_gate(None, indeterminate, installing, None); + } + } + /// Publish the in-memory read-closing gate for `SetSourceReadFence` `key` (section 9, /// step 2): new SQL programs, dumps, replication calls and ATTACHes of the namespace are /// refused with `MIGRATION_READ_FENCED`. A read lease is only ever taken after checking the diff --git a/libsql-server/src/namespace/fence/mod.rs b/libsql-server/src/namespace/fence/mod.rs index 74f227cf23..fd8810e0db 100644 --- a/libsql-server/src/namespace/fence/mod.rs +++ b/libsql-server/src/namespace/fence/mod.rs @@ -10,8 +10,9 @@ //! ([`store`], driven by `MetaStore::apply_fence_command`), and the in-memory authority built //! on them: the per-namespace [`controller`] with its gate and read leases, the positive write //! [`drain`], the source [`read`] fence and its -//! [`stream`] leases for dump and replication, the [`registry`] that holds the controllers outside the -//! namespace cache, and the test [`hooks`] on their paths. +//! [`stream`] leases for dump and replication, quarantined migration [`target`]s, the +//! [`registry`] that holds the controllers outside the namespace cache, and the test [`hooks`] +//! on their paths. // The persistence, controller and protocol layers that consume these types land in the // following commits of this series; until then most of the module is unused by the rest of @@ -29,6 +30,7 @@ pub mod registry; pub mod state; pub mod store; pub mod stream; +pub mod target; pub mod transition; #[cfg(test)] diff --git a/libsql-server/src/namespace/fence/registry.rs b/libsql-server/src/namespace/fence/registry.rs index 610189677b..7c03102cd1 100644 --- a/libsql-server/src/namespace/fence/registry.rs +++ b/libsql-server/src/namespace/fence/registry.rs @@ -66,13 +66,13 @@ impl FenceRegistry { self.controllers.lock().remove(namespace) } - /// Refuse a namespace whose fence state is `UNKNOWN_UNAVAILABLE`, before any work is done - /// to serve it. + /// Refuse a namespace whose fence state is `UNKNOWN_UNAVAILABLE`, or that is being created + /// as a quarantined target, before any work is done to serve it. pub fn check_available(&self, namespace: &NamespaceName) -> Result<(), FenceError> { match self.get(namespace) { Some(controller) => { let gate = controller.gate(); - if gate.is_unavailable() { + if gate.is_unavailable() || gate.is_creating_target() { gate.permits(OperationClass::NormalRead) } else { Ok(()) diff --git a/libsql-server/src/namespace/fence/target.rs b/libsql-server/src/namespace/fence/target.rs new file mode 100644 index 0000000000..3b69b7efe5 --- /dev/null +++ b/libsql-server/src/namespace/fence/target.rs @@ -0,0 +1,546 @@ +//! Migration targets (`docs/NAMESPACE_FENCE.md` sections 10 and 11). +//! +//! A target namespace is created by the operation that will fill it, already in +//! `TARGET_QUARANTINED`, and it is quarantined from the first instant anything else could +//! observe it: the in-memory target-creation gate is installed before the metastore transaction +//! that writes its marker, config row, record and receipt; the committed record replaces that +//! gate; and only then is the config put where `exists()` and `lookup()` find it and the +//! namespace loaded, so its first connection maker is created behind the quarantine gate. +//! +//! [`CreateTargetRequest`] is the typed entry point that the admin route and bulk import both +//! use; `NamespaceStore::create_target_quarantined` runs it. `AbortQuarantinedTarget` needs +//! nothing of its own: it is an ordinary transition, and `TARGET_ABORTED` denies every +//! normal class exactly as the quarantine does. + +use uuid::Uuid; + +use crate::namespace::NamespaceName; + +use super::command::{FenceCommand, FenceRequest, TargetConfig}; +use super::outcome::{FenceDetail, FenceError, FenceOutcome}; +use super::state::FenceState; + +/// `CreateTargetQuarantined` for `namespace`, by `operation_id`. The expectation is always +/// `ABSENT` at revision 0, so it is not part of the request. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct CreateTargetRequest { + pub namespace: NamespaceName, + pub operation_id: Uuid, + /// Idempotency key: replaying the same command returns its stored result, and completes a + /// creation that was interrupted between its marker and its commit. + pub command_id: Uuid, + pub config: TargetConfig, +} + +impl From for FenceRequest { + fn from(req: CreateTargetRequest) -> Self { + FenceRequest { + namespace: req.namespace, + operation_id: req.operation_id, + command_id: req.command_id, + expected_state: FenceState::Absent, + expected_revision: 0, + command: FenceCommand::CreateTargetQuarantined { config: req.config }, + } + } +} + +/// The refusal of a target name that the server already knows, in memory or in the namespace +/// cache, although the metastore may not hold it yet (a create or fork in flight, or one the +/// fence refused after it had published its config in memory). +pub(crate) fn name_in_use(namespace: &NamespaceName) -> FenceError { + FenceError::new( + FenceOutcome::FencePreconditionFailed, + format!("namespace `{namespace}` already exists on this server"), + ) + .with_detail(FenceDetail::NamespaceExists) +} + +#[cfg(test)] +pub(crate) mod tests { + use std::sync::Arc; + + use libsql_replication::rpc::replication::replication_log_server::ReplicationLog; + use libsql_replication::rpc::replication::{HelloRequest, NAMESPACE_METADATA_KEY}; + use tempfile::{tempdir, TempDir}; + use tonic::metadata::BinaryMetadataValue; + + use super::*; + use crate::auth::Authenticated; + use crate::connection::config::DatabaseConfig; + use crate::connection::program::Program; + use crate::connection::{Connection as _, RequestContext}; + use crate::error::Error; + use crate::namespace::fence::command::FenceCommand; + use crate::namespace::fence::controller::FenceController; + use crate::namespace::fence::drain::tests::{raw, PROMPT}; + use crate::namespace::fence::hooks::{HookAction, HookPoint}; + use crate::namespace::fence::record::ServerIdentity; + use crate::namespace::fence::state::OperationClass; + use crate::namespace::meta_store::{metastore_connection_maker, FenceCommit, FenceCommitKind}; + use crate::namespace::store::fence_tests::open_store; + use crate::namespace::store::NamespaceStore; + use crate::namespace::RestoreOption; + use crate::query_result_builder::test::TestBuilder; + use crate::rpc::replication::replication_log::ReplicationLogService; + + pub(crate) const OP: Uuid = Uuid::from_u128(0xa); + const OTHER_OP: Uuid = Uuid::from_u128(0xb); + + pub(crate) fn server() -> ServerIdentity { + ServerIdentity { + build: "test".into(), + instance_id: Uuid::from_u128(0x99), + } + } + + pub(crate) fn create_request(ns: &'static str, command_id: u128) -> CreateTargetRequest { + CreateTargetRequest { + namespace: ns.into(), + operation_id: OP, + command_id: Uuid::from_u128(command_id), + config: TargetConfig { + max_db_size: Some(4096 * 1000), + ..Default::default() + }, + } + } + + /// Run `CreateTargetQuarantined` through the store on a task of its own. + pub(crate) fn create( + store: &NamespaceStore, + req: CreateTargetRequest, + ) -> tokio::task::JoinHandle> { + let store = store.clone(); + tokio::spawn(async move { store.create_target_quarantined(req, server()).await }) + } + + fn fence_error(e: &Error) -> &FenceError { + match e { + Error::NamespaceFence(f) => f, + other => panic!("expected a fence error, got {other:?}"), + } + } + + /// Denied by the fence, or not there at all: never served. + fn assert_not_served(what: &str, r: &crate::Result) { + match r { + Err(Error::NamespaceDoesntExist(_)) => (), + Err(Error::NamespaceFence(_)) => (), + other => panic!("{what}: expected a denial, got {other:?}"), + } + } + + fn assert_quarantined(fence: &FenceController) { + for class in [ + OperationClass::NormalRead, + OperationClass::NormalWrite, + OperationClass::Stream, + OperationClass::Lifecycle, + OperationClass::Vacuum, + ] { + let e = fence.permits(class).unwrap_err(); + assert_eq!( + e.outcome(), + FenceOutcome::MigrationTargetQuarantined, + "{class:?}" + ); + } + } + + async fn controller(store: &NamespaceStore, ns: &'static str) -> Arc { + store + .fence_gate(&ns.into()) + .await + .unwrap() + .expect("the target has a controller") + } + + /// The loaded target's controller, and a connection to it. + async fn loaded( + store: &NamespaceStore, + ns: &'static str, + ) -> (Arc, Arc) { + let (fence, maker) = store + .with(ns.into(), |ns| { + (ns.fence().clone(), ns.db.connection_maker()) + }) + .await + .unwrap(); + (fence, Arc::new(maker.create().await.unwrap())) + } + + async fn replication_hello(store: &NamespaceStore, ns: &'static str) -> tonic::Status { + let service = ReplicationLogService::new(store.clone(), None, None, false, false, true); + let mut req = tonic::Request::new(HelloRequest { + handshake_version: Some(1), + }); + req.metadata_mut().insert_bin( + NAMESPACE_METADATA_KEY, + BinaryMetadataValue::from_bytes(ns.as_bytes()), + ); + service.hello(req).await.unwrap_err() + } + + /// Every way of reaching `ns` other than the fence commands: none of them is served. + async fn attempt_everything(store: &NamespaceStore, ns: &'static str) { + let r = store + .with(ns.into(), |ns| ns.db.connection_maker()) + .await + .map(|_| ()); + assert_not_served("SQL connection", &r); + let r = store.stats(ns.into()).await.map(|_| ()); + assert_not_served("stats", &r); + // The dump route and the replication service reach the namespace the same way. + let hello = replication_hello(store, ns).await; + assert_ne!(hello.code(), tonic::Code::Ok); + assert_ne!(hello.code(), tonic::Code::Unavailable, "{hello:?}"); + let r = store + .create(ns.into(), RestoreOption::Latest, DatabaseConfig::default()) + .await; + assert_not_served("create", &r); + let r = store.destroy(ns.into(), false).await; + assert!(r.is_err(), "delete: {r:?}"); + let r = store + .fork("src".into(), ns.into(), DatabaseConfig::default(), None) + .await; + assert!(r.is_err(), "fork: {r:?}"); + } + + async fn store_with_source(dir: &TempDir) -> NamespaceStore { + let store = open_store(dir.path()).await; + store + .create("src".into(), RestoreOption::Latest, Default::default()) + .await + .unwrap(); + store + } + + /// Parked after its rows are committed and its quarantine gate published, and before its + /// config is published and the namespace loaded, the target is never observable: SQL, + /// dump/replication, create, delete and fork of the name are all denied or find nothing. + /// Afterwards it is loaded behind the quarantine gate. + #[tokio::test(flavor = "multi_thread")] + async fn create_race_never_observable() { + let dir = tempdir().unwrap(); + let store = store_with_source(&dir).await; + let fence = store.fence_controller(&"tgt".into()); + let paused = fence.hooks().pause_at(HookPoint::AfterTargetRowsCommitted); + let creating = create(&store, create_request("tgt", 1)); + paused.reached().await; + + assert_eq!(fence.gate().state(), FenceState::TargetQuarantined); + assert!(!store.exists(&"tgt".into()).await); + attempt_everything(&store, "tgt").await; + + paused.resume(); + let commit = creating.await.unwrap().unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + assert!(store.exists(&"tgt".into()).await); + + // Loaded behind the gate it was created with. + let (loaded_fence, conn) = loaded(&store, "tgt").await; + assert!(Arc::ptr_eq(&fence, &loaded_fence)); + assert_quarantined(&fence); + let ctx = RequestContext::new( + Authenticated::FullAccess, + "tgt".into(), + store.meta_store().clone(), + ); + let e = conn + .execute_program( + Program::seq(&["select 1"]), + ctx, + TestBuilder::default(), + None, + ) + .await + .map(|_| ()) + .unwrap_err(); + assert_eq!( + fence_error(&e).outcome(), + FenceOutcome::MigrationTargetQuarantined + ); + crate::namespace::fence::drain::tests::assert_fenced( + raw(&conn, "create table t (x)").await, + ); + // The in-memory config is the logical one; the stored row carries the legacy mirror. + let config = store.config_store("tgt".into()).await.unwrap().get(); + assert_eq!(config.max_db_pages, 1000); + assert!(!config.block_reads && !config.block_writes); + // Everything but the commands is still refused once it is loaded. + attempt_everything_loaded(&store, "tgt").await; + } + + /// Once loaded, the lifecycle paths are refused by the fence. + async fn attempt_everything_loaded(store: &NamespaceStore, ns: &'static str) { + let r = store + .create(ns.into(), RestoreOption::Latest, DatabaseConfig::default()) + .await; + assert!(r.is_err(), "create: {r:?}"); + let r = store.destroy(ns.into(), false).await; + assert_eq!( + fence_error(&r.unwrap_err()).outcome(), + FenceOutcome::MigrationTargetQuarantined + ); + let r = store + .fork("src".into(), ns.into(), DatabaseConfig::default(), None) + .await; + assert!(r.is_err(), "fork: {r:?}"); + let hello = replication_hello(store, ns).await; + assert_eq!( + FenceError::outcome_from_grpc_status(&hello), + Some(FenceOutcome::MigrationTargetQuarantined), + "{hello:?}" + ); + assert!(store.exists(&ns.into()).await); + } + + /// Before its commit, while the target-creation gate is in place, the name is refused + /// before any setup work. + #[tokio::test(flavor = "multi_thread")] + async fn creating_gate_refuses_before_commit() { + let dir = tempdir().unwrap(); + let store = store_with_source(&dir).await; + let fence = store.fence_controller(&"tgt".into()); + let paused = fence.hooks().pause_at(HookPoint::BeforeMetastoreCommit); + let creating = create(&store, create_request("tgt", 1)); + paused.reached().await; + + assert!(fence.gate().is_creating_target()); + assert_quarantined(&fence); + attempt_everything(&store, "tgt").await; + // No database was set up under the name. + assert!(!dir.path().join("dbs").join("tgt").join("data").exists()); + + paused.resume(); + creating.await.unwrap().unwrap(); + assert!(!fence.gate().is_creating_target()); + assert_eq!(fence.gate().state(), FenceState::TargetQuarantined); + } + + /// A creation interrupted between its marker and its commit leaves the name + /// `UNKNOWN_UNAVAILABLE` after a restart; only the same command completes it, and the + /// completed target is loaded quarantined. + #[tokio::test(flavor = "multi_thread")] + async fn create_replay_completes_interrupted_creation() { + let dir = tempdir().unwrap(); + { + let store = store_with_source(&dir).await; + create(&store, create_request("tgt", 1)) + .await + .unwrap() + .unwrap(); + store.shutdown().await.unwrap(); + } + // What a crash after the marker and before the commit leaves: the marker alone. + { + let (maker, _) = metastore_connection_maker(None, dir.path()).await.unwrap(); + let conn = maker().unwrap(); + for sql in [ + "DELETE FROM namespace_fence_receipts WHERE namespace = 'tgt'", + "DELETE FROM namespace_fences WHERE namespace = 'tgt'", + "DELETE FROM namespace_configs WHERE namespace = 'tgt'", + ] { + conn.execute(sql, ()).unwrap(); + } + } + std::fs::remove_file(dir.path().join("dbs").join("tgt").join("data")).ok(); + + let store = open_store(dir.path()).await; + let fence = store.fence_controller(&"tgt".into()); + assert!(fence.gate().is_unavailable()); + let r = store.with("tgt".into(), |_| ()).await; + assert_eq!( + fence_error(&r.unwrap_err()).detail(), + Some(FenceDetail::IncompleteTargetCreation) + ); + // Another command does not complete it. + let r = create(&store, create_request("tgt", 2)).await.unwrap(); + assert_eq!( + fence_error(&r.unwrap_err()).outcome(), + FenceOutcome::FenceStateUnavailable + ); + + let commit = create(&store, create_request("tgt", 1)) + .await + .unwrap() + .unwrap(); + assert_eq!(commit.kind, FenceCommitKind::Committed); + assert_eq!(fence.gate().state(), FenceState::TargetQuarantined); + let (loaded_fence, _) = loaded(&store, "tgt").await; + assert!(Arc::ptr_eq(&fence, &loaded_fence)); + assert_quarantined(&fence); + let config = store.config_store("tgt".into()).await.unwrap().get(); + assert!(!config.block_reads && !config.block_writes); + } + + /// The caller going away does not stop a creation: it is published and loaded, and a replay + /// returns the stored result. + #[tokio::test(flavor = "multi_thread")] + async fn create_completes_when_the_caller_goes_away() { + let dir = tempdir().unwrap(); + let store = store_with_source(&dir).await; + let fence = store.fence_controller(&"tgt".into()); + let paused = fence.hooks().pause_at(HookPoint::AfterTargetRowsCommitted); + let creating = create(&store, create_request("tgt", 1)); + paused.reached().await; + creating.abort(); + let _ = creating.await; + paused.resume(); + + // The creation carries on without its caller; the replay waits for it on the + // transition lock and then answers from the receipt. + let replay = tokio::time::timeout(PROMPT, create(&store, create_request("tgt", 1))) + .await + .unwrap() + .unwrap() + .unwrap(); + assert_eq!(replay.kind, FenceCommitKind::Replayed); + assert_eq!(replay.receipt.outcome, FenceOutcome::Applied); + assert!(store.exists(&"tgt".into()).await); + let (loaded_fence, _) = loaded(&store, "tgt").await; + assert!(Arc::ptr_eq(&fence, &loaded_fence)); + } + + /// A commit whose acknowledgement is lost keeps the name closed; the replay reconciles it + /// from the durable rows and publishes and loads the target. + #[tokio::test(flavor = "multi_thread")] + async fn indeterminate_create_is_completed_by_replay() { + let dir = tempdir().unwrap(); + let store = store_with_source(&dir).await; + let fence = store.fence_controller(&"tgt".into()); + fence + .hooks() + .arm(HookPoint::AfterMetastoreCommit, HookAction::Indeterminate); + let r = create(&store, create_request("tgt", 1)).await.unwrap(); + assert_eq!( + fence_error(&r.unwrap_err()).outcome(), + FenceOutcome::FenceCommitIndeterminate + ); + // Still refused before any setup, and not published. + assert!(fence.gate().is_creating_target()); + assert!(!store.exists(&"tgt".into()).await); + assert_not_served("SQL", &store.with("tgt".into(), |_| ()).await); + + let replay = create(&store, create_request("tgt", 1)) + .await + .unwrap() + .unwrap(); + assert_eq!(replay.kind, FenceCommitKind::Replayed); + assert!(!fence.gate().is_creating_target()); + assert!(fence.gate().indeterminate.is_none()); + assert_eq!(fence.gate().state(), FenceState::TargetQuarantined); + assert!(store.exists(&"tgt".into()).await); + loaded(&store, "tgt").await; + assert_quarantined(&fence); + } + + /// A name the server already has is refused, and its traffic is not disturbed; a name that + /// is refused keeps no creation gate. + #[tokio::test(flavor = "multi_thread")] + async fn create_rejects_existing_name() { + let dir = tempdir().unwrap(); + let store = store_with_source(&dir).await; + let src_conn = { + let maker = store + .with("src".into(), |ns| ns.db.connection_maker()) + .await + .unwrap(); + Arc::new(maker.create().await.unwrap()) + }; + raw(&src_conn, "create table t (x)").await.unwrap(); + let generation = store.fence_controller(&"src".into()).write_generation(); + + let r = create(&store, create_request("src", 1)).await.unwrap(); + let e = r.unwrap_err(); + assert_eq!( + fence_error(&e).outcome(), + FenceOutcome::FencePreconditionFailed + ); + assert_eq!(fence_error(&e).detail(), Some(FenceDetail::NamespaceExists)); + // The source's gate never moved, and it still takes writes. + let src_fence = store.fence_controller(&"src".into()); + assert_eq!(src_fence.write_generation(), generation); + assert!(!src_fence.gate().is_creating_target()); + raw(&src_conn, "insert into t values (1)").await.unwrap(); + + // A name that exists in the metastore but is not loaded is refused the same way. + store + .create("cold".into(), RestoreOption::Latest, Default::default()) + .await + .unwrap(); + let r = create(&store, create_request("cold", 2)).await.unwrap(); + assert_eq!( + fence_error(&r.unwrap_err()).detail(), + Some(FenceDetail::NamespaceExists) + ); + + // A second target of the same name by another operation is refused, and the first is + // untouched. + create(&store, create_request("tgt", 3)) + .await + .unwrap() + .unwrap(); + let mut other = create_request("tgt", 4); + other.operation_id = OTHER_OP; + let r = create(&store, other).await.unwrap(); + assert!(r.is_err()); + let fence = store.fence_controller(&"tgt".into()); + assert_eq!(fence.gate().operation_id(), Some(OP)); + assert!(!fence.gate().is_creating_target()); + } + + /// `AbortQuarantinedTarget` finishes the operation and keeps every normal class denied; + /// the target is not deletable by the generic lifecycle either. + #[tokio::test(flavor = "multi_thread")] + async fn abort_keeps_traffic_denied() { + let dir = tempdir().unwrap(); + let store = store_with_source(&dir).await; + create(&store, create_request("tgt", 1)) + .await + .unwrap() + .unwrap(); + let abort = FenceRequest { + namespace: "tgt".into(), + operation_id: OP, + command_id: Uuid::from_u128(2), + expected_state: FenceState::TargetQuarantined, + expected_revision: 1, + command: FenceCommand::AbortQuarantinedTarget, + }; + let commit = store + .execute_fence_command(abort.clone(), server()) + .await + .unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + let fence = controller(&store, "tgt").await; + assert_eq!(fence.gate().state(), FenceState::TargetAborted); + assert_quarantined(&fence); + let (_, conn) = loaded(&store, "tgt").await; + crate::namespace::fence::drain::tests::assert_fenced( + raw(&conn, "create table t (x)").await, + ); + let r = store.destroy("tgt".into(), false).await; + assert_eq!( + fence_error(&r.unwrap_err()).outcome(), + FenceOutcome::MigrationTargetQuarantined + ); + // A replay answers from the receipt; a new creation of the name is refused. + let replay = store.execute_fence_command(abort, server()).await.unwrap(); + assert_eq!(replay.kind, FenceCommitKind::Replayed); + let r = create(&store, create_request("tgt", 3)).await.unwrap(); + assert!(r.is_err()); + + // After a restart the aborted target is still denied. + store.shutdown().await.unwrap(); + let store = open_store(dir.path()).await; + let fence = controller(&store, "tgt").await; + assert_eq!(fence.gate().state(), FenceState::TargetAborted); + assert_quarantined(&fence); + let (_, conn) = loaded(&store, "tgt").await; + crate::namespace::fence::drain::tests::assert_fenced( + raw(&conn, "create table t (x)").await, + ); + } +} diff --git a/libsql-server/src/namespace/meta_store.rs b/libsql-server/src/namespace/meta_store.rs index ebd1ba2a64..17d23db89c 100644 --- a/libsql-server/src/namespace/meta_store.rs +++ b/libsql-server/src/namespace/meta_store.rs @@ -36,7 +36,7 @@ use super::fence::outcome::{FenceDetail, FenceError, FenceOutcome}; use super::fence::record::{ CommandReceipt, NamespaceFenceRecord, ServerIdentity, ValidationSnapshot, }; -use super::fence::state::OperationClass; +use super::fence::state::{OperationClass, Role}; use super::fence::store::{ self as fence_store, FenceStoreError, MarkerStatus, StoredFence, StoredReceipt, }; @@ -1379,6 +1379,56 @@ impl MetaStore { .map_err(fence_store_error) } + /// Make a migration target that the metastore holds visible in the in-memory config map, + /// which is what makes `exists()` and `lookup()` find it (section 10.1, step 5). The config + /// published is the stored row with the record's own `block_*` values in place of the + /// legacy mirror, as `restore_fences` does at startup. The caller has already installed + /// the target's gate. Returns whether the map changed; `false` also when the namespace is + /// not a target with a stored config. + pub async fn publish_target_config(&self, namespace: NamespaceName) -> Result { + let inner = self.inner.clone(); + tokio::task::spawn_blocking(move || -> std::result::Result { + // The connection lock first, as everywhere else that takes both. + let mut conn = inner.conn.blocking_lock(); + let tx = conn.transaction()?; + let (stored, _) = fence_store::read_fence(&tx, &inner.dbs_path, &namespace)?; + let record = match stored { + StoredFence::Record(r) if r.role == Role::Target => r, + _ => return Ok(false), + }; + let Some(row) = fence_store::read_config_row(&tx, &namespace)? else { + return Ok(false); + }; + drop(tx); + let config = Arc::new(fence_store::with_legacy_blocks(&row, &record.legacy_blocks)); + let mut configs = inner.configs.blocking_lock(); + match configs.get_mut(&namespace) { + Some(sender) + if metadata::DatabaseConfig::from(&*sender.borrow().config) + == metadata::DatabaseConfig::from(&*config) => + { + Ok(false) + } + // An entry that was put in the map by a create or fork of the same name that + // the fence then refused: the durable config replaces it. + Some(sender) => { + sender.send_modify(|c| { + c.version = c.version.wrapping_add(1); + c.config = config; + }); + Ok(true) + } + None => { + let (tx, _) = watch::channel(InnerConfig { version: 0, config }); + configs.insert(namespace, tx); + Ok(true) + } + } + }) + .await? + .map_err(fence_store_error) + } + /// Read a namespace's fence and all of its receipts (`InspectFence`). Never writes. pub async fn inspect_fence(&self, namespace: NamespaceName) -> Result { let inner = self.inner.clone(); diff --git a/libsql-server/src/namespace/store.rs b/libsql-server/src/namespace/store.rs index ca8e7031be..177f2cc471 100644 --- a/libsql-server/src/namespace/store.rs +++ b/libsql-server/src/namespace/store.rs @@ -22,9 +22,14 @@ use crate::stats::Stats; use super::broadcasters::{BroadcasterHandle, BroadcasterRegistry}; use super::configurator::{DynConfigurator, NamespaceConfigurators}; use super::fence::command::{FenceCommand, FenceRequest}; -use super::fence::controller::FenceController; +use super::fence::controller::{FenceController, Transition}; +use super::fence::hooks::HookPoint; +use super::fence::outcome::FenceOutcome; use super::fence::record::ServerIdentity; use super::fence::registry::FenceRegistry; +use super::fence::state::Role; +use super::fence::store::StoredFence; +use super::fence::target::{self, CreateTargetRequest}; use super::meta_store::{FenceCommit, FenceContext, MetaStore, MetaStoreHandle}; use super::schema_lock::SchemaLocksRegistry; use super::{Namespace, ResetCb, ResetOp, ResolveNamespacePathFn, RestoreOption}; @@ -237,6 +242,10 @@ impl NamespaceStore { return Err(Error::NamespaceStoreShutdown); } + // The destination is refused before anything is stored for it when it is being created + // as a migration target or its fence state is unknown. + self.inner.fences.check_available(&to)?; + // check that the source namespace exists if !self.inner.metadata.exists(&from).await { return Err(crate::error::Error::NamespaceDoesntExist(from.to_string())); @@ -450,6 +459,9 @@ impl NamespaceStore { restore_option: RestoreOption, db_config: DatabaseConfig, ) -> crate::Result<()> { + // A name that is being created as a migration target, or whose fence state is unknown, + // is refused before anything is stored for it. + self.inner.fences.check_available(&namespace)?; if let Some(shared_schema_name) = &db_config.shared_schema_name { // we hold a lock for the duration of the namespace creation let _lock = self @@ -550,6 +562,9 @@ impl NamespaceStore { server: ServerIdentity, ) -> crate::Result { let controller = match request.command { + FenceCommand::CreateTargetQuarantined { .. } => { + return self.run_create_target(request, server).await + } FenceCommand::AcquireSourceWriteFence { .. } => { self.with(request.namespace.clone(), |ns| ns.fence().clone()) .await? @@ -565,6 +580,148 @@ impl NamespaceStore { .await } + /// `CreateTargetQuarantined`, atomic with namespace creation (`docs/NAMESPACE_FENCE.md` + /// sections 10.1 and 11): the namespace is quarantined from the first instant it can be + /// observed, and is loaded behind the quarantine gate before this returns `APPLIED`. + /// + /// Fence refusals are [`Error::NamespaceFence`] with their stable outcome code. The work + /// runs on its own task under the namespace's transition lock, so a caller that goes away + /// does not interrupt it; replaying the same request returns the stored result, completes + /// the publication and load of a target whose commit was not acknowledged, and completes a + /// creation interrupted between its marker and its commit. + pub async fn create_target_quarantined( + &self, + request: CreateTargetRequest, + server: ServerIdentity, + ) -> crate::Result { + self.run_create_target(request.into(), server).await + } + + async fn run_create_target( + &self, + request: FenceRequest, + server: ServerIdentity, + ) -> crate::Result { + if self.inner.has_shutdown.load(Ordering::Relaxed) { + return Err(Error::NamespaceStoreShutdown); + } + let controller = self.inner.fences.controller(&request.namespace); + let this = self.clone(); + tokio::spawn(async move { + let mut transition = controller.begin_transition().await; + let ctx = FenceContext::now(server, None); + this.create_target_under(&mut transition, request, ctx) + .await + }) + .await? + } + + /// Section 10.1, steps 1 to 5, under `transition`. + async fn create_target_under( + &self, + transition: &mut Transition, + request: FenceRequest, + ctx: FenceContext, + ) -> crate::Result { + let controller = transition.controller().clone(); + let namespace = request.namespace.clone(); + let key = (request.operation_id, request.command_id); + + // A name with no fence state gets the target-creation gate before the metastore + // transaction writes the marker, so nothing can set the name up, serve it or store a + // config for it from before the commit to the load. A name that already has fence + // state (a replay, a creation interrupted after its marker, or a refusal) already has + // the gate its state implies. A name the server already knows is refused without + // touching its gate; it is checked again once the gate is in place, which closes the + // race with a create or fork that has not stored anything yet. + let fresh = { + let gate = controller.gate(); + matches!(gate.fence, StoredFence::None { .. }) && gate.indeterminate.is_none() + }; + if fresh { + if self.name_in_use(&namespace).await { + return Err(target::name_in_use(&namespace).into()); + } + transition.install_creating_target(key); + let _ = controller.hook(HookPoint::AfterInstallingGate).await; + if self.name_in_use(&namespace).await { + transition.remove_creating_target(); + return Err(target::name_in_use(&namespace).into()); + } + } + + // Steps 2 and 3: the marker, then the rows in one transaction. The commit publishes the + // quarantined record in place of the creation gate. A command proven not to have + // committed removes the creation gate; one whose commit is unknown keeps it, with the + // indeterminate flag, until the same command is replayed. + let commit = match transition.apply(&self.inner.metadata, request, ctx).await { + Ok(commit) => commit, + Err(e) => { + if controller.gate().indeterminate != Some(key) { + transition.remove_creating_target(); + } + return Err(e); + } + }; + let _ = controller.hook(HookPoint::AfterTargetRowsCommitted).await; + + // Steps 4 and 5: the gate is the target's; now make the config visible and load the + // namespace, whose first connection is created behind that gate. A replay does the same, + // which completes a creation whose commit was not acknowledged. + let is_target = commit + .record + .as_ref() + .is_some_and(|r| r.role == Role::Target && r.namespace == namespace); + if is_target { + self.inner + .metadata + .publish_target_config(namespace.clone()) + .await?; + self.clear_empty_entry(&namespace).await; + let loaded = self + .with(namespace.clone(), |ns| ns.fence().clone()) + .await?; + debug_assert!(Arc::ptr_eq(&loaded, &controller)); + } + if commit.receipt.outcome == FenceOutcome::Applied && commit.created_config.is_some() { + tracing::info!( + namespace = %namespace, + operation_id = %key.0, + command_id = %key.1, + "created namespace as a quarantined migration target" + ); + } + Ok(commit) + } + + /// Whether the server already knows `namespace`: its config is in memory, or the namespace + /// cache holds it (a loaded namespace, or a fork in flight, which holds its entry locked). + async fn name_in_use(&self, namespace: &NamespaceName) -> bool { + if self.inner.metadata.exists(namespace).await { + return true; + } + match self.inner.store.get(namespace).await { + Some(entry) => entry.read().await.is_some(), + None => false, + } + } + + /// Drop an empty namespace-cache entry for `namespace`, which a refused fork or a checkpoint + /// of a name that did not exist leaves behind, so that loading the namespace creates it. + async fn clear_empty_entry(&self, namespace: &NamespaceName) { + if let Some(entry) = self.inner.store.get(namespace).await { + if entry.read().await.is_none() { + self.inner.store.invalidate(namespace).await; + } + } + } + + /// The fence controller of `namespace`, creating an `UNFENCED` one if it has none. + #[cfg(test)] + pub(crate) fn fence_controller(&self, namespace: &NamespaceName) -> Arc { + self.inner.fences.controller(namespace) + } + /// The fence controller that admits reads of `namespace` without loading it: `None` when /// the namespace does not exist (and has no fence state). A namespace whose fence state is /// unavailable is refused. From 05e91c20e98fd08fc9d8974105a4cc5a2b84640f Mon Sep 17 00:00:00 2001 From: River Date: Tue, 29 Sep 2026 18:55:06 +0000 Subject: [PATCH 15/33] libsql-server: import capabilities and target seal drain Add the operation-owned import path into a quarantined migration target and the seal that ends it (docs/NAMESPACE_FENCE.md sections 7, 10.2 and 11). - MigrationCapability (server-issued, fields private) and CapabilityPurpose. The fence controller keeps the live capability set and a count of running import calls; every published transition drops capabilities whose state, owner or revision no longer match. - FenceConnState::with_capability: the WAL admits a capability connection's write transaction only while its capability matches the fence and is live; a validation connection never writes. - NamespaceStore::open_import_session and ImportSession::{with_raw, load_dump}: the only way to write into TARGET_QUARANTINED. The dump loader is split into load_dump_sql so that the loader used for namespaces created from a dump also runs under an import capability. - SealTargetImport closes import admission in memory, persists TARGET_IMPORT_DRAINING (invalidating every import capability), waits on release notifications for running import calls and for any import transaction holding the write slot (force_rollback rolls it back at the deadline), then persists TARGET_VALIDATING. A deadline leaves TARGET_IMPORT_DRAINING durable and closed until the owner's seal resumes it. Co-authored-by: Tomasz Szymczyszyn --- docs/NAMESPACE_FENCE.md | 62 +- libsql-server/src/connection/legacy.rs | 37 +- libsql-server/src/connection/mod.rs | 5 + .../src/namespace/configurator/helpers.rs | 81 +- .../src/namespace/configurator/mod.rs | 1 + .../src/namespace/fence/capability.rs | 212 +++++ .../src/namespace/fence/controller.rs | 267 +++++- libsql-server/src/namespace/fence/drain.rs | 5 +- libsql-server/src/namespace/fence/import.rs | 838 ++++++++++++++++++ libsql-server/src/namespace/fence/mod.rs | 6 +- libsql-server/src/namespace/store.rs | 69 +- 11 files changed, 1499 insertions(+), 84 deletions(-) create mode 100644 libsql-server/src/namespace/fence/capability.rs create mode 100644 libsql-server/src/namespace/fence/import.rs diff --git a/docs/NAMESPACE_FENCE.md b/docs/NAMESPACE_FENCE.md index ffe64a2e43..cf724e98b6 100644 --- a/docs/NAMESPACE_FENCE.md +++ b/docs/NAMESPACE_FENCE.md @@ -406,7 +406,7 @@ This is additive in proto3: older peers skip the unknown field; a newer replica Per namespace: - `transition_lock`: a `tokio::sync::Mutex` serialising commands on this namespace. A command holds it from its first check to its response (`FenceController::begin_transition` returns a `Transition` that owns the guard). -- `gate`: a `tokio::sync::watch` of `GateSnapshot { fence, write_generation, indeterminate, installing, closing_reads, creating_target }`, where `fence` is the durable fence as last published (a record, no record, or `UNKNOWN_UNAVAILABLE` with its detail) and `installing` is the in-memory `INSTALLING` gate of a closing command being persisted (section 8.3), which denies normal writes, vacuum, import writes and lifecycle work on top of `fence`. State, revision, owning operation and every admission (`permits(class)`, `write()`, `read()`) are derived from it through the permission matrix. The WAL wrapper, `CoreConnection`, dump, replication and lifecycle code read it without locks. Phase 5 adds the live capability set. +- `gate`: a `tokio::sync::watch` of `GateSnapshot { fence, write_generation, indeterminate, installing, closing_reads, creating_target }`, where `fence` is the durable fence as last published (a record, no record, or `UNKNOWN_UNAVAILABLE` with its detail) and `installing` is the in-memory `INSTALLING` gate of a closing command being persisted (section 8.3), which denies normal writes, vacuum, import writes and lifecycle work on top of `fence`. State, revision, owning operation and every admission (`permits(class)`, `write()`, `read()`) are derived from it through the permission matrix. The WAL wrapper, `CoreConnection`, dump, replication and lifecycle code read it without locks. The live capability set is kept beside it (`capabilities`, below). - `write_generation: u64`, in the snapshot, **incremented on every publication that changes the fence state, the owning operation, or the indeterminate flag**. That covers every transition that closes or opens write admission (acquire, release, create, seal, publish, enable, abort, adopt), and is conservative for the others. A replay that publishes the same durable state does not move it. - `indeterminate: Option<(operation_id, command_id)>`: set when a command's `COMMIT` failed (or the task running it died) so that whether it applied is unknown. While set, every class except `Maintenance` and `Observability` is denied with `FENCE_STATE_UNAVAILABLE` / `indeterminate_commit`, and every other command is refused with `FENCE_COMMIT_INDETERMINATE`. A replay of the same command is answered by the metastore from the durable row (replayed if it had committed, applied if it had not) and clears it (section 8.4). - the write-drain sources of the namespace's primary connection makers (`register_write_drain`): each maker's connection manager, held weakly, and its replication log id and last-committed-frame reader, which the write drain waits on and reads the boundary from (section 8.3). @@ -414,7 +414,7 @@ Per namespace: - `creating_target` (in the snapshot): the in-memory target-creation gate of a `CreateTargetQuarantined` being persisted (section 10.1), which denies every class except `Maintenance` and `Observability` with `MIGRATION_TARGET_QUARANTINED`, and makes `FenceRegistry::check_available` refuse the name, so nothing sets the namespace up, serves it or stores a config for it. Never persisted; it moves `write_generation`; the publication of the committed record replaces it, it is removed when the command is proven not to have committed, and it stays (with `indeterminate`) when the outcome is unknown. - `closing_reads` (in the snapshot): the in-memory read-closing gate of a `SetSourceReadFence` being persisted (section 9, step 2), which denies `NormalRead` and `Stream` with `MIGRATION_READ_FENCED` on top of `fence`. Never persisted; every publication clears it; it does not move `write_generation` (writes are already closed wherever a read fence can be set). - `read_leases`: the live read leases, each with its kind (`sql`, `dump`, `replication`), a cancel handle and a cancelled flag, and a `Notify` on every release. `FenceController::acquire_read_lease(class, kind, cancel)` checks the gate **under the lease lock** and registers the lease, so a lease is either refused by a read-closing gate published before it, or counted by a drain that closes admission after it; once admission is closed the set can only shrink. A `ReadLease` is released when dropped. -- `capabilities`: the live `MigrationCapability` set and an import-writer counter. +- `capabilities`: the live `MigrationCapability` set and an import-writer counter (the import calls running now, section 10.2), under a lock that may be taken before a gate borrow and never under one. `issue_capability` checks the gate under this lock and registers the capability; every publication of a transition drops the capabilities whose state, owner or revision no longer match; dropping a session revokes its capability. `begin_import_write` checks the capability against the gate and the live set under the same lock and counts the call, so a call is either refused by a seal that closed admission before it or counted by that seal. - in `cfg(test)` builds only, a `FenceTestHooks` (section 16). `FenceController::apply_command` runs the command on its own task: a caller that goes away after the commit (a lost response) does not prevent the publication. The metastore maps a failed `COMMIT` to `FENCE_COMMIT_INDETERMINATE`; any other error (a refusal by the transition function, a busy metastore, a failure before `COMMIT`) proves nothing was written and leaves the gate exactly as it was. @@ -428,8 +428,8 @@ Every write-transaction request at the WAL, every read lease and every lifecycle | `NormalWrite` | SQL over HTTP, Hrana, RPC, proxy; admin shell; schema migration; dump load outside the capability | normal-write column allows | | `Maintenance` | `TRUNCATE` checkpoint, the manager's checkpoint slot, storage monitor | always | | `Vacuum` | `vacuum_if_needed`, `Namespace::checkpoint`, snapshot at shutdown | normal-write column allows; otherwise skipped with a debug log. `CoreConnection::vacuum_if_needed_above` checks the gate first and also reports a WAL refusal of the `VACUUM` itself (the fence closing between the check and the statement) as skipped, not failed | -| `CapabilityImport` | import session writes | `TARGET_QUARANTINED`, matching `operation_id`, capability revision equal to the record's, capability not invalidated | -| `CapabilityValidate` | validation reads (never writes) | `TARGET_VALIDATING`, `TARGET_WRITE_FENCED` | +| `CapabilityImport` | import session writes | `TARGET_QUARANTINED`, matching `operation_id`, capability revision equal to the record's, capability live (issued by this server, not revoked, not invalidated by a transition); checked at the WAL in `admit_write` on every write transaction, and up front by every `ImportSession` call | +| `CapabilityValidate` | validation reads (never writes: the WAL refuses every write transaction of a validation connection) | `TARGET_VALIDATING`, `TARGET_WRITE_FENCED` | | `NormalRead` | SQL programs, Hrana cursors, `/beta/listen`, ATTACH of this namespace | normal-read column allows | | `Stream` | `/dump`, `hello`, `log_entries`, `batch_log_entries`, `snapshot` | dump/replication column allows | | `Observability` | stats, `/v1/jobs`, metrics | always; never counted as a read lease | @@ -439,6 +439,7 @@ Every write-transaction request at the WAL, every read lease and every lifecycle Every `LegacyConnection` receives a `FenceConnState` shared by its `ManagedConnectionWalWrapper` and its `CoreConnection`: the connection's class and capability (if any), `program_generation`, `txn_generation`, and a `denial` slot for the typed outcome of the last WAL refusal. - `begin_program()` records `program_generation` and clears the denial slot. It is called at the start of every `CoreConnection::run`, of every `with_raw` call (admin shell, schema migration, dump load, the configurators' own uses) and of `vacuum_if_needed`, so every way of running SQL on the connection is a program. +- A capability connection (an import or validation session, section 11) is built with `FenceConnState::with_capability`: its class is the capability's, and `admit_write` additionally requires the capability to match the fence (state its purpose admits, owner, revision) and to be live. - `begin_read_txn()` records `txn_generation`. The WAL wrapper calls it from `begin_read_txn` **before** the snapshot is taken, so a transition racing with it leaves the transaction with the older generation, which can only refuse a later upgrade. - `admit_write()` is the check of section 8.1 (2). A refusal is stored in the denial slot and returned. @@ -545,7 +546,14 @@ The target's replication `log_id` is not written back into the record: rewriting ### 10.2 SealTargetImport -CAS `TARGET_IMPORT_DRAINING` (revision + 1, so every issued import capability is invalidated and no new one can be issued); wait for the import-writer count and the manager's `CapabilityImport` holder to reach zero (same mechanism as section 8.3, with the drain policy from the request); CAS `TARGET_VALIDATING`. A timeout leaves `TARGET_IMPORT_DRAINING` durable and closed; only a replay of the same command resumes it. +`seal_target_import` (routed from `FenceController::execute`) runs under the transition lock: + +1. For a seal that can apply (the owner, at the current revision, of a `TARGET_QUARANTINED` target, with nothing indeterminate or installing), publish the in-memory `INSTALLING` gate: new import calls and import write transactions are refused, the write generation moves (so every transaction opened before it is stale) and queued import writers are woken and refused. A command that cannot apply does not touch the gate, so it cannot disturb a running import. +2. CAS `TARGET_IMPORT_DRAINING` (revision + 1). Its publication replaces the `INSTALLING` gate and drops every issued import capability: none can be issued again for the target. A command proven not to have committed removes the `INSTALLING` gate. +3. Wait, on release notifications and never on elapsed time, for the running import calls (the controller's import-writer counter) to end, then for every connection manager of the target to have no writer holding its write slot (an import transaction admitted before step 1, including one an idle session left open). With `on_deadline: force_rollback` the seal rolls back the transaction still holding the slot at the deadline (a running call ends its own; the rollback waits for the connection's lock) and waits again for the same deadline, at least 10 s. The request's drain policy applies; without one, `--namespace-fence-default-write-drain-ms`. A target without a loaded connection maker has no connection that could write. +4. CAS `TARGET_VALIDATING` (`complete_drain`, `DrainCompletion::TargetImport`). + +A deadline reached before step 4 answers `DRAINING` and leaves `TARGET_IMPORT_DRAINING` durable and closed, also across a restart. Only the owning operation resumes it: a replay of the same command, or a new seal of the owner, which joins the drain (section 5.3); another operation is refused. Import never resumes. ### 10.3 Validation and publication @@ -586,9 +594,11 @@ impl NamespaceStore { server: ServerIdentity) -> crate::Result; /// Issue an import capability and a capability-bearing connection. Valid only in - /// TARGET_QUARANTINED for the owning operation at `expected_revision`. - pub async fn open_import_session(&self, ns: NamespaceName, operation_id: Uuid, - expected_revision: u64) -> Result; + /// TARGET_QUARANTINED for the owning operation at `expected_revision`; loads the target. + /// Fence refusals are `Error::NamespaceFence(FenceError)`: OPERATION_CAPABILITY_REQUIRED in + /// any other state, FENCE_OWNED_BY_ANOTHER_OPERATION, FENCE_REVISION_MISMATCH. + pub async fn open_import_session(&self, namespace: NamespaceName, operation_id: Uuid, + expected_revision: u64) -> crate::Result; /// Read-only validation connection (`query_only`), TARGET_VALIDATING or TARGET_WRITE_FENCED. pub async fn open_validation_session(&self, ns: NamespaceName, operation_id: Uuid, @@ -600,17 +610,31 @@ impl NamespaceStore { pub async fn inspect_fence(&self, ns: NamespaceName) -> Result; } -pub struct ImportSession { /* capability, connection, import-writer guard */ } +impl MigrationCapability { // read-only accessors + pub fn id(&self) -> Uuid; + pub fn namespace(&self) -> &NamespaceName; + pub fn operation_id(&self) -> Uuid; + pub fn purpose(&self) -> CapabilityPurpose; + pub fn fence_revision(&self) -> u64; +} + +pub struct ImportSession { /* capability, controller, capability connection */ } impl ImportSession { pub fn capability(&self) -> &MigrationCapability; - /// Run a closure with the raw connection inside the capability; writes are admitted by the - /// WAL only while the capability is valid. + /// Run a closure with the raw connection inside the capability. Refused up front once the + /// capability is no longer valid; counted as an import writer until the closure returns; + /// writes are admitted by the WAL only while the capability is valid, and a write the WAL + /// refused is returned as that FenceError. pub async fn with_raw(&mut self, f: impl FnOnce(&mut rusqlite::Connection) -> R + Send + 'static) -> Result; + /// The server's dump loader (`load_dump_sql`, the loader a namespace created from a dump + /// uses) run inside `with_raw`. Dump errors are `Error::LoadDumpError`. + pub async fn load_dump(&mut self, dump: S) -> crate::Result<()> + where S: Stream> + Unpin; } ``` -`FenceError` carries a `FenceOutcome` (section 6) and converts into the server's `Error`, so a route built on top returns the same codes. Dropping an `ImportSession` decrements the import-writer count and wakes a waiting seal. The existing dump loader (`load_dump`) can run inside `ImportSession::with_raw`; this series includes a test that imports a small dump into a quarantined target that way. Streaming and memory bounds are not part of this series. +`FenceError` carries a `FenceOutcome` (section 6) and converts into the server's `Error`, so a route built on top returns the same codes. The import-writer count is of running calls: a session that is not running a call cannot start a write once the seal has moved the revision, and a transaction it left open holds the write slot, which the seal waits for as well (section 10.2). Dropping an `ImportSession` revokes its capability and closes its connection, which rolls back a transaction it left open and so releases the slot. The capability connection shares the target's write slot, WAL and replication log, and is not counted by the connection throttle. The loader holds the connection for the whole dump and re-renders each statement it runs (as it does for a namespace created from a dump), so the stored SQL text of schema objects differs from the source's in case and spacing. Streaming and memory bounds are not part of this series. ## 12. Incident adoption (Contract) @@ -685,7 +709,7 @@ How each path that can reach namespace data or lifecycle is covered. File refere | `rpc/replication/replication_log.rs` `hello`, `log_entries`, `batch_log_entries`, `snapshot` | Denied at request start (`FAILED_PRECONDITION` + `x-libsql-fence-code`, counted, rate-limited log); `FencedStream` replication leases for both streams and the batch; typed terminal status when the gate closes or at the deadline, lease released without the peer; `ReplicatedFence` in `hello`'s config (planned, section 6.2). | | `admin_shell.rs` | Writes denied at the WAL (no capability); reads checked against the gate, and a read lease held, per query (cancelled through the connection's interrupt handle). | | `schema/scheduler.rs`, `database/schema.rs` | Shared schema excluded from fencing; migration writes are WAL-gated; the scheduler's `block_writes` flag is not treated as drain evidence. | -| `namespace/configurator/helpers.rs` `load_dump`, `http/admin/mod.rs` `dump_stream_from_url` | Restore options and dump URLs are refused for fenced namespaces; import goes through `ImportSession`. | +| `namespace/configurator/helpers.rs` `load_dump`, `http/admin/mod.rs` `dump_stream_from_url` | Restore options and dump URLs are refused for fenced namespaces (planned, section 13.4 lifecycle work); outside a capability the loader's writes are refused at the WAL. Import goes through `ImportSession::load_dump`, which runs the same loader (`load_dump_sql`) on the capability connection. | | `connection/program.rs` ATTACH resolution | `check_program_auth` uses a non-creating lookup; the resolver returns the attached namespace's controller, the attachment is admitted as `NormalRead` of it with a read lease, and the connection keeps a lease on it for every later program until it is detached. | | `http/user/listen.rs` `/beta/listen` | `NormalRead`, read from the registry without loading the namespace; denied where reads are denied; the stream ends with an error event when reads are fenced. | | Raw internal connections (storage monitor, periodic checkpoint, shutdown checkpoint, replication logger, `checkpoint_db`, bottomless) | `Maintenance` / `Observability`; not leases; never blocked. | @@ -734,8 +758,8 @@ Planned test names; the table is updated as tests land. | 9 | Filesystem recovery, `destroy_on_error`, undecodable records, missing target quarantine, metastore backup rollback fail closed with provenance | `meta_store::fence_tests::recovery::{fs_recovery_with_marker_unavailable, destroy_on_error_keeps_fenced_unavailable, undecodable_row_unavailable, incomplete_target_unavailable, metastore_rollback_detected_by_marker, lookup_never_creates, undecodable_name_with_fence_fails_startup, marker_in_invalid_directory_fails_startup}`; legacy behaviour kept: `destroy_on_error_without_fences_is_unchanged`, `undecodable_row_without_fences_is_skipped_as_before`; `meta_store::fence_tests::corrupt_fence_row_fails_closed` | | 10 | Wrong owner, stale revision, invalid role/state, replay, command-id reuse; replay before revision check | `fence::transition::tests::*` (exhaustive over states × commands) | | 11 | Target creation raced with SQL, dump, replication, lifecycle never observable as writable or readable | landed: `namespace::fence::target::tests::create_race_never_observable` (parked after the rows commit and before the config is published and the namespace loaded: SQL connections, stats, replication `hello` (never `UNAVAILABLE`), create, delete and fork of the name are denied or find nothing; afterwards the target is loaded behind the quarantine gate, SQL reads and WAL writes are refused, and lifecycle and replication are refused with `MIGRATION_TARGET_QUARANTINED`), `creating_gate_refuses_before_commit` (parked before the metastore transaction: the same attempts are refused and no database file is created), `create_replay_completes_interrupted_creation` (marker only, after a restart), `create_completes_when_the_caller_goes_away`, `indeterminate_create_is_completed_by_replay`, `create_rejects_existing_name` (a loaded or cold existing name, whose gate never moves, and another operation's target), `abort_keeps_traffic_denied` (also across a restart) | -| 12 | Only the matching import capability writes a quarantined target; admin credentials and admin shell cannot | `fence::target::tests::import_requires_matching_capability`; `tests::fence::admin::admin_shell_cannot_write_quarantined` | -| 13 | Seal enters `TARGET_IMPORT_DRAINING`, waits, reaches `TARGET_VALIDATING`, cannot resume import; only a durable validation receipt permits idempotent publication | `fence::target::tests::seal_waits_for_import_writers`, `sealed_target_rejects_import`, `publish_requires_validation_receipt`, `publish_is_idempotent` | +| 12 | Only the matching import capability writes a quarantined target; admin credentials and admin shell cannot | landed: `namespace::fence::import::tests::import_requires_matching_capability` (plain connections with and without raw access, as the admin shell uses; issuing to another operation or at another revision; at the WAL, capabilities the server never issued, of another operation, at another revision, for validation, and revoked); `namespace::fence::target::tests::{create_race_never_observable, abort_keeps_traffic_denied}` (raw DDL refused); planned: `tests::fence::admin::admin_shell_cannot_write_quarantined` | +| 13 | Seal enters `TARGET_IMPORT_DRAINING`, waits, reaches `TARGET_VALIDATING`, cannot resume import; only a durable validation receipt permits idempotent publication | seal landed: `namespace::fence::import::tests::{seal_waits_for_import_writers (an import call parked inside its write transaction: import is closed at once, the transaction commits, and the seal reaches its completion commit only afterwards), seal_deadline_leaves_import_draining_until_replayed (an idle session's open transaction; `DRAINING`, also across a restart; another operation refused, the owner's new seal joins; the replay completes), seal_force_rollback_ends_open_import_transaction, sealed_target_rejects_import}`; planned: `fence::target::tests::{publish_requires_validation_receipt, publish_is_idempotent}` | | 14 | Enable writes idempotent, survives restart and response loss, irreversible | `fence::target::tests::enable_writes_idempotent_and_irreversible`, `enable_writes_survives_restart` | | 15 | Lost `EnableTargetWrites` response resolved from receipt/state | `fence::target::tests::enable_writes_response_loss_resolved` | | 16 | Read fence drains SQL, dump, `log_entries`, `snapshot`, including dead peers and forced termination | SQL landed: `namespace::fence::read::tests::{read_fence_waits_for_running_program, program_after_closing_gate_is_refused (parked after the read-closing gate, before the CAS), read_fence_cancels_at_deadline, unreleased_lease_answers_draining_and_replay_completes, idle_txn_fails_on_next_program, clear_read_fence_reopens_reads_not_writes, refused_read_fence_reopens_reads, attach_of_read_fenced_namespace_denied}`, `admin_shell::fence_tests::admin_shell_read_denied`; dump and replication landed: `namespace::fence::stream::tests::{dump_lease_released_on_cancel (a dump blocked mid-row on a peer that stopped reading is cancelled at the deadline, its lease released without the peer, the fence acknowledged, and the body ends with the fence error and no `COMMIT;`), read_fence_waits_for_dump, dump_refused_while_read_fenced, log_entries_stream_ends_typed, stream_lease_released_without_peer_read (a dead peer), snapshot_stream_ends_typed, replication_calls_denied_while_read_fenced, read_fence_forced_termination}`; planned: the HTTP-level check that an interrupted `/dump` response is aborted rather than completed, with `tests::fence::protocol::dump_codes` | @@ -745,7 +769,7 @@ Planned test names; the table is updated as tests land. | 20 | Metrics and audit logs | `tests::fence::observability::metrics_and_labels`; `fence::audit::tests::audit_event_fields` | | 21 | Capability discovery and mixed-version protection | `tests::fence::admin::capabilities`; `fence::store::tests::legacy_mirror_and_fk_guard` (bounded, see section 18) | | 22 | Adoption is two-person/audited, keeps admission closed, cannot reverse publication | `fence::tests::adopt_requires_key_and_two_approvers`, `adopt_keeps_gates_closed`, `adopt_cannot_touch_writable` | -| — | Import API usable by bulk import | `fence::target::tests::import_session_loads_dump_into_quarantined_target` | +| — | Import API usable by bulk import | landed: `namespace::fence::import::tests::import_session_loads_dump_into_quarantined_target` (a dump exported by the server from a source with tables, keys, a foreign key, an index, an autoincrement table, a trigger, a view and an FTS5 table loads through `ImportSession::load_dump`; after the seal the target's schema, rows, view and full-text results equal the source's) | ## 18. Limits @@ -775,10 +799,12 @@ libsql-server/src/namespace/fence/ store.rs metastore tables, fence CAS, marker file registry.rs FenceRegistry controller.rs FenceController, GateSnapshot, generations, leases - drain.rs write and import drains + drain.rs FenceController::execute, source write drain read.rs source read fence and its drain stream.rs stream leases: FencedStream for replication, dump cancel - target.rs target lifecycle, MigrationCapability, ImportSession, ValidationSession + target.rs quarantined target creation, ValidationSession (planned) + capability.rs MigrationCapability, CapabilityPurpose, ImportWriter + import.rs ImportSession, SealTargetImport drain audit.rs audit events and metrics hooks.rs cfg(test) FenceTestHooks libsql-server/src/http/admin/fence.rs diff --git a/libsql-server/src/connection/legacy.rs b/libsql-server/src/connection/legacy.rs index efc176cae8..3d441aaded 100644 --- a/libsql-server/src/connection/legacy.rs +++ b/libsql-server/src/connection/legacy.rs @@ -14,6 +14,7 @@ use tokio::time::Duration; use crate::error::Error; use crate::metrics::DESCRIBE_COUNT; use crate::namespace::broadcasters::BroadcasterHandle; +use crate::namespace::fence::capability::MigrationCapability; use crate::namespace::fence::controller::{FenceConnState, FenceController}; use crate::namespace::fence::state::OperationClass; use crate::namespace::meta_store::MetaStoreHandle; @@ -143,6 +144,33 @@ where #[tracing::instrument(skip(self))] pub(super) async fn make_connection(&self) -> Result> { + self.make_connection_with(FenceConnState::new( + self.fence.clone(), + OperationClass::NormalWrite, + )) + .await + } + + /// Open a connection that works under `capability` (an import or validation session, + /// `docs/NAMESPACE_FENCE.md` section 11). It shares the maker's write slot, WAL and + /// replication log with every other connection, and is admitted as the capability's class + /// only while the capability is valid. It is not counted by the connection throttle: it is + /// operation-owned work, and the fence, not the throttle, bounds how much of it runs. + pub(crate) async fn make_capability_connection( + &self, + capability: MigrationCapability, + ) -> Result> { + self.make_connection_with(FenceConnState::with_capability( + self.fence.clone(), + capability, + )) + .await + } + + async fn make_connection_with( + &self, + fence: Arc, + ) -> Result> { LegacyConnection::new( self.db_path.clone(), self.extensions.clone(), @@ -161,7 +189,7 @@ where self.resolve_attach_path.clone(), self.connection_manager.clone(), self.make_wal_manager.clone(), - FenceConnState::new(self.fence.clone(), OperationClass::NormalWrite), + fence, ) .await } @@ -213,6 +241,13 @@ impl LegacyConnection { } } +impl LegacyConnection { + /// The fence state shared by this connection's WAL wrapper and `CoreConnection`. + pub(crate) fn fence_state(&self) -> &Arc { + &self.fence + } +} + impl Clone for LegacyConnection { fn clone(&self) -> Self { Self { diff --git a/libsql-server/src/connection/mod.rs b/libsql-server/src/connection/mod.rs index 167a1a595c..ef8fe0002e 100644 --- a/libsql-server/src/connection/mod.rs +++ b/libsql-server/src/connection/mod.rs @@ -286,6 +286,11 @@ pub struct MakeThrottledConnection { } impl MakeThrottledConnection { + /// The connection maker this one throttles. + pub(crate) fn inner(&self) -> &F { + &self.connection_maker + } + fn new( semaphore: Arc, connection_maker: F, diff --git a/libsql-server/src/namespace/configurator/helpers.rs b/libsql-server/src/namespace/configurator/helpers.rs index a10ef89d6b..6fe8994663 100644 --- a/libsql-server/src/namespace/configurator/helpers.rs +++ b/libsql-server/src/namespace/configurator/helpers.rs @@ -301,6 +301,16 @@ async fn run_periodic_compactions(logger: Arc) -> anyhow::Res } async fn load_dump(dump: S, conn: PrimaryConnection) -> crate::Result<(), LoadDumpError> +where + S: Stream> + Unpin, +{ + let dump_content = read_dump(dump).await?; + tokio::task::spawn_blocking(move || conn.with_raw(|conn| load_dump_sql(&dump_content, conn))) + .await? +} + +/// Read a whole dump into memory. +pub(crate) async fn read_dump(dump: S) -> crate::Result where S: Stream> + Unpin, { @@ -310,13 +320,36 @@ where .read_to_string(&mut dump_content) .await .map_err(|e| LoadDumpError::Internal(format!("Failed to read dump content: {}", e)))?; + Ok(dump_content) +} +/// Parse `dump_content` and run its statements, one at a time, on `conn`. The dump must run +/// inside one transaction that it commits itself; `ATTACH` is refused. This is the loader both +/// for a namespace created from a dump and for an import session into a quarantined migration +/// target, which runs it under its capability (`docs/NAMESPACE_FENCE.md` section 11). +pub(crate) fn load_dump_sql( + dump_content: &str, + conn: &mut rusqlite::Connection, +) -> crate::Result<(), LoadDumpError> { if dump_content.to_lowercase().contains("attach") { return Err(LoadDumpError::InvalidSqlInput( "attach statements are not allowed in dumps".to_string(), )); } + conn.authorizer(Some(|auth: AuthContext<'_>| match auth.action { + AuthAction::Attach { filename: _ } => Authorization::Deny, + _ => Authorization::Allow, + })); + let result = run_dump_statements(dump_content, conn); + conn.authorizer(None::) -> Authorization>); + result +} + +fn run_dump_statements( + dump_content: &str, + conn: &mut rusqlite::Connection, +) -> crate::Result<(), LoadDumpError> { let mut parser = Box::new(Parser::new(dump_content.as_bytes())); let mut skipped_wasm_table = false; let mut n_stmt = 0; @@ -336,37 +369,20 @@ where } } - if n_stmt > 2 && conn.is_autocommit().await.unwrap() { + if n_stmt > 2 && conn.is_autocommit() { return Err(LoadDumpError::NoTxn); } let stmt_sql = cmd.to_string(); - tokio::task::spawn_blocking({ - let conn = conn.clone(); - move || -> crate::Result<(), LoadDumpError> { - conn.with_raw(|conn| { - conn.authorizer(Some(|auth: AuthContext<'_>| match auth.action { - AuthAction::Attach { filename: _ } => Authorization::Deny, - _ => Authorization::Allow, - })); - conn.execute(&stmt_sql, ()) - }) - .map_err(|e| match e { - rusqlite::Error::SqlInputError { - msg, sql, offset, .. - } => LoadDumpError::InvalidSqlInput(format!( - "msg: {}, sql: {}, offset: {}", - msg, sql, offset - )), - e => LoadDumpError::Internal(format!( - "statement: {}, error: {}", - n_stmt, e - )), - })?; - Ok(()) - } - }) - .await??; + conn.execute(&stmt_sql, ()).map_err(|e| match e { + rusqlite::Error::SqlInputError { + msg, sql, offset, .. + } => LoadDumpError::InvalidSqlInput(format!( + "msg: {}, sql: {}, offset: {}", + msg, sql, offset + )), + e => LoadDumpError::Internal(format!("statement: {}, error: {}", n_stmt, e)), + })?; } Ok(None) => break, Err(e) => { @@ -389,15 +405,8 @@ where } } - if !conn.is_autocommit().await.unwrap() { - tokio::task::spawn_blocking({ - let conn = conn.clone(); - move || -> crate::Result<(), LoadDumpError> { - conn.with_raw(|conn| conn.execute("rollback", ()))?; - Ok(()) - } - }) - .await??; + if !conn.is_autocommit() { + conn.execute("rollback", ())?; return Err(LoadDumpError::NoCommit); } diff --git a/libsql-server/src/namespace/configurator/mod.rs b/libsql-server/src/namespace/configurator/mod.rs index 029ab0b3ce..f8d11addd2 100644 --- a/libsql-server/src/namespace/configurator/mod.rs +++ b/libsql-server/src/namespace/configurator/mod.rs @@ -26,6 +26,7 @@ mod primary; mod replica; mod schema; +pub(crate) use helpers::{load_dump_sql, read_dump}; pub use primary::PrimaryConfigurator; pub use replica::ReplicaConfigurator; pub use schema::SchemaConfigurator; diff --git a/libsql-server/src/namespace/fence/capability.rs b/libsql-server/src/namespace/fence/capability.rs new file mode 100644 index 0000000000..46e492f6f0 --- /dev/null +++ b/libsql-server/src/namespace/fence/capability.rs @@ -0,0 +1,212 @@ +//! Migration capabilities (`docs/NAMESPACE_FENCE.md` sections 7.3 and 11). +//! +//! A [`MigrationCapability`] is the only thing that lets operation-owned work through a +//! target's quarantine. The server creates it ([`FenceController::issue_capability`]); its +//! fields are private and it cannot be built from outside this module, so holding one is proof +//! that the fence issued it. It names the namespace, the owning operation, what it is for and +//! the fence revision it was issued at, and it is valid only while the fence is still in the +//! state its purpose needs, at that revision, owned by that operation, and while the controller +//! still lists it as live. Every transition of the target moves the revision, so a capability +//! never outlives the state it was issued in. +//! +//! [`FenceController::issue_capability`]: super::controller::FenceController::issue_capability + +use std::collections::HashMap; +use std::sync::Arc; + +use uuid::Uuid; + +use crate::namespace::NamespaceName; + +use super::controller::FenceController; +use super::outcome::{FenceError, FenceOutcome}; +use super::state::{FenceState, OperationClass}; +use super::store::StoredFence; + +/// What a capability admits. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] +pub enum CapabilityPurpose { + /// Writes of an import session into a `TARGET_QUARANTINED` target. + Import, + /// Read-only validation of a `TARGET_VALIDATING` or `TARGET_WRITE_FENCED` target. + Validate, +} + +impl CapabilityPurpose { + /// The operation class work under a capability of this purpose is admitted as. + pub const fn class(self) -> OperationClass { + match self { + CapabilityPurpose::Import => OperationClass::CapabilityImport, + CapabilityPurpose::Validate => OperationClass::CapabilityValidate, + } + } + + pub const fn as_str(self) -> &'static str { + match self { + CapabilityPurpose::Import => "import", + CapabilityPurpose::Validate => "validate", + } + } + + /// Whether a capability of this purpose may be issued, and stays valid, in `state`. + pub const fn admits(self, state: FenceState) -> bool { + match self { + CapabilityPurpose::Import => matches!(state, FenceState::TargetQuarantined), + CapabilityPurpose::Validate => matches!( + state, + FenceState::TargetValidating | FenceState::TargetWriteFenced + ), + } + } +} + +/// A server-issued grant for operation-owned work on one namespace, valid at one fence +/// revision. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct MigrationCapability { + id: Uuid, + namespace: NamespaceName, + operation_id: Uuid, + purpose: CapabilityPurpose, + fence_revision: u64, +} + +impl MigrationCapability { + /// Only the controller issues capabilities. + pub(super) fn issue( + namespace: NamespaceName, + operation_id: Uuid, + purpose: CapabilityPurpose, + fence_revision: u64, + ) -> Self { + Self { + id: Uuid::new_v4(), + namespace, + operation_id, + purpose, + fence_revision, + } + } + + /// A capability the controller never issued, for tests that prove such a thing is refused. + #[cfg(test)] + pub(crate) fn forged( + namespace: NamespaceName, + operation_id: Uuid, + purpose: CapabilityPurpose, + fence_revision: u64, + ) -> Self { + Self::issue(namespace, operation_id, purpose, fence_revision) + } + + pub fn id(&self) -> Uuid { + self.id + } + + pub fn namespace(&self) -> &NamespaceName { + &self.namespace + } + + pub fn operation_id(&self) -> Uuid { + self.operation_id + } + + pub fn purpose(&self) -> CapabilityPurpose { + self.purpose + } + + pub fn fence_revision(&self) -> u64 { + self.fence_revision + } + + pub fn class(&self) -> OperationClass { + self.purpose.class() + } + + /// Whether `fence` is still the fence this capability was issued against: the state its + /// purpose needs, owned by its operation, at its revision. + pub(crate) fn matches(&self, fence: &StoredFence) -> bool { + self.check(fence).is_ok() + } + + /// [`matches`](Self::matches), with the refusal a holder gets when it does not. + pub(crate) fn check(&self, fence: &StoredFence) -> Result<(), FenceError> { + let Some(record) = fence.record() else { + return Err(stale( + self, + format!("namespace `{}` has no fence record", self.namespace), + )); + }; + if record.operation_id != self.operation_id { + return Err(FenceError::new( + FenceOutcome::FenceOwnedByAnotherOperation, + format!( + "the fence of namespace `{}` is owned by operation {}, not by operation {} \ + that holds this {} capability", + self.namespace, + record.operation_id, + self.operation_id, + self.purpose.as_str() + ), + )); + } + if !self.purpose.admits(record.state) { + return Err(stale( + self, + format!( + "namespace `{}` is in {}, which admits no {} capability", + self.namespace, + record.state, + self.purpose.as_str() + ), + )); + } + if record.revision != self.fence_revision { + return Err(stale( + self, + format!( + "the fence of namespace `{}` is at revision {}, and this {} capability was \ + issued at revision {}", + self.namespace, + record.revision, + self.purpose.as_str(), + self.fence_revision + ), + )); + } + Ok(()) + } +} + +fn stale(cap: &MigrationCapability, why: String) -> FenceError { + FenceError::new( + FenceOutcome::OperationCapabilityRequired, + format!( + "{why}: the {} capability {} is no longer valid", + cap.purpose.as_str(), + cap.id + ), + ) +} + +/// The capabilities a controller has issued and not yet revoked, and the import calls running +/// under them. +#[derive(Debug, Default)] +pub(super) struct CapabilitySet { + pub(super) live: HashMap, + /// Import calls running now ([`ImportWriter`]s). + pub(super) import_writers: usize, +} + +/// One running import call under a capability. The seal waits for every one of them to be +/// dropped (section 10.2). +#[derive(Debug)] +pub struct ImportWriter { + pub(super) controller: Arc, +} + +impl Drop for ImportWriter { + fn drop(&mut self) { + self.controller.end_import_write(); + } +} diff --git a/libsql-server/src/namespace/fence/controller.rs b/libsql-server/src/namespace/fence/controller.rs index 04149a66b4..1eeeffc804 100644 --- a/libsql-server/src/namespace/fence/controller.rs +++ b/libsql-server/src/namespace/fence/controller.rs @@ -24,6 +24,7 @@ use crate::namespace::meta_store::{FenceCommit, FenceContext, MetaStore}; use crate::namespace::NamespaceName; use crate::replication::FrameNo; +use super::capability::{CapabilityPurpose, CapabilitySet, ImportWriter, MigrationCapability}; use super::command::FenceRequest; #[cfg(test)] use super::hooks::FenceTestHooks; @@ -312,6 +313,12 @@ pub struct FenceController { read_leases: Mutex, /// Notified whenever a read lease is released. read_released: Notify, + /// The migration capabilities issued and not revoked, and the import calls running under + /// them (sections 7.2 and 10.2). Lock order: this lock may be taken before borrowing the + /// gate, never while a gate borrow is held. + capabilities: Mutex, + /// Notified whenever an import call ends. + import_released: Notify, #[cfg(test)] hooks: FenceTestHooks, } @@ -337,6 +344,8 @@ impl FenceController { write_drains: Mutex::new(Vec::new()), read_leases: Mutex::new(ReadLeaseSet::default()), read_released: Notify::new(), + capabilities: Mutex::new(CapabilitySet::default()), + import_released: Notify::new(), #[cfg(test)] hooks: FenceTestHooks::default(), }) @@ -484,6 +493,144 @@ impl FenceController { self.read_released.notify_waiters(); } + /// Issue a migration capability for `purpose` to `operation_id` (section 11). The fence + /// must be in a state the purpose admits, owned by `operation_id`, at `expected_revision`, + /// with no command being installed or reconciled. The capability stays valid until the + /// fence moves on (every transition moves the revision) or it is revoked. + pub fn issue_capability( + &self, + purpose: CapabilityPurpose, + operation_id: Uuid, + expected_revision: u64, + ) -> Result { + let mut caps = self.capabilities.lock(); + let gate = self.gate.borrow(); + gate.permits(purpose.class())?; + let Some(record) = gate.fence.record() else { + return Err(FenceError::new( + FenceOutcome::OperationCapabilityRequired, + format!("namespace `{}` has no fence record", self.namespace), + )); + }; + if record.operation_id != operation_id { + return Err(FenceError::new( + FenceOutcome::FenceOwnedByAnotherOperation, + format!( + "the fence of namespace `{}` is owned by operation {}, not by operation \ + {operation_id}", + self.namespace, record.operation_id + ), + )); + } + if record.revision != expected_revision { + return Err(FenceError::new( + FenceOutcome::FenceRevisionMismatch, + format!( + "the fence of namespace `{}` is at revision {}, not at the expected revision \ + {expected_revision}", + self.namespace, record.revision + ), + )); + } + if !purpose.admits(record.state) { + return Err(FenceError::new( + FenceOutcome::OperationCapabilityRequired, + format!( + "namespace `{}` is in {}, which admits no new {} capability", + self.namespace, + record.state, + purpose.as_str() + ), + )); + } + let cap = MigrationCapability::issue( + self.namespace.clone(), + operation_id, + purpose, + record.revision, + ); + drop(gate); + caps.live.insert(cap.id(), cap.clone()); + tracing::debug!( + namespace = %self.namespace, + %operation_id, + capability = %cap.id(), + purpose = purpose.as_str(), + revision = cap.fence_revision(), + "issued migration capability" + ); + Ok(cap) + } + + /// Revoke a capability: nothing is admitted under it any more. + pub fn revoke_capability(&self, id: Uuid) { + self.capabilities.lock().live.remove(&id); + } + + /// Whether `id` was issued by this controller and is neither revoked nor invalidated by a + /// transition. + pub fn capability_is_live(&self, id: Uuid) -> bool { + self.capabilities.lock().live.contains_key(&id) + } + + /// Admit one import call under `cap`, counted until the returned guard is dropped. The + /// capability is checked against the gate under the capability lock, so an import call is + /// either refused by a seal that closed admission before it, or counted by the seal, which + /// reads the count only after closing admission (section 10.2). + pub fn begin_import_write( + self: &Arc, + cap: &MigrationCapability, + ) -> Result { + if cap.namespace() != &self.namespace || cap.purpose() != CapabilityPurpose::Import { + return Err(FenceError::new( + FenceOutcome::OperationCapabilityRequired, + format!( + "a {} capability for namespace `{}` does not admit imports into `{}`", + cap.purpose().as_str(), + cap.namespace(), + self.namespace + ), + )); + } + let mut caps = self.capabilities.lock(); + { + let gate = self.gate.borrow(); + gate.permits(OperationClass::CapabilityImport)?; + cap.check(&gate.fence)?; + } + if !caps.live.contains_key(&cap.id()) { + return Err(revoked(cap)); + } + caps.import_writers += 1; + Ok(ImportWriter { + controller: self.clone(), + }) + } + + pub(super) fn end_import_write(&self) { + { + let mut caps = self.capabilities.lock(); + caps.import_writers = caps.import_writers.saturating_sub(1); + } + self.import_released.notify_waiters(); + } + + /// The import calls running now. + pub fn import_writers(&self) -> usize { + self.capabilities.lock().import_writers + } + + /// The capabilities issued and still live. + pub fn live_capabilities(&self) -> usize { + self.capabilities.lock().live.len() + } + + /// Notified whenever an import call ends. Enable the notification before checking + /// [`import_writers`](Self::import_writers), so an end in between is not missed. + pub(crate) fn import_released(&self) -> &Notify { + &self.import_released + } + /// Take the namespace's transition lock. Every fence command on the namespace runs while /// holding it, from its first check to its response. pub async fn begin_transition(self: &Arc) -> Transition { @@ -558,6 +705,7 @@ impl FenceController { installing: Option, creating_target: Option, ) { + let fence_published = fence.is_some(); let mut generation_changed = false; self.gate.send_modify(|gate| { let fence = fence.unwrap_or_else(|| gate.fence.clone()); @@ -590,6 +738,16 @@ impl FenceController { "published namespace fence gate" ); } + // A published transition invalidates every capability issued against an earlier state, + // owner or revision. Cloned first: the capability lock is never taken under a gate + // borrow. + if fence_published { + let fence = self.gate.borrow().fence.clone(); + self.capabilities + .lock() + .live + .retain(|_, cap| cap.matches(&fence)); + } // After the gate is published, so that every woken writer re-checks against it. if generation_changed { self.write_queues.lock().retain(|wake| wake()); @@ -805,6 +963,20 @@ fn indeterminate(key: CommandKey, why: &str) -> FenceError { .with_detail(FenceDetail::IndeterminateCommit) } +fn revoked(cap: &MigrationCapability) -> FenceError { + FenceError::new( + FenceOutcome::OperationCapabilityRequired, + format!( + "the {} capability {} of operation {} on namespace `{}` was revoked or was never \ + issued by this server", + cap.purpose().as_str(), + cap.id(), + cap.operation_id(), + cap.namespace() + ), + ) +} + fn pending_indeterminate((operation_id, command_id): CommandKey) -> FenceError { FenceError::new( FenceOutcome::FenceCommitIndeterminate, @@ -857,6 +1029,9 @@ impl Drop for ProgramReadLease { pub struct FenceConnState { controller: Arc, class: OperationClass, + /// The migration capability this connection works under, fixed at construction. Only a + /// connection opened for an import or validation session has one. + capability: Option, /// The write generation the current program was admitted under. program_generation: AtomicU64, /// The write generation the current read transaction was opened under. @@ -875,10 +1050,30 @@ pub struct FenceConnState { impl FenceConnState { pub fn new(controller: Arc, class: OperationClass) -> Arc { + Self::build(controller, class, None) + } + + /// The fence state of a connection that works under `capability` (an import or a + /// validation session): it is admitted as the capability's class, and only while the + /// capability is valid. + pub fn with_capability( + controller: Arc, + capability: MigrationCapability, + ) -> Arc { + let class = capability.class(); + Self::build(controller, class, Some(capability)) + } + + fn build( + controller: Arc, + class: OperationClass, + capability: Option, + ) -> Arc { let generation = controller.write_generation(); Arc::new(Self { controller, class, + capability, program_generation: AtomicU64::new(generation), txn_generation: AtomicU64::new(generation), denial: Mutex::new(None), @@ -974,6 +1169,10 @@ impl FenceConnState { self.class } + pub fn capability(&self) -> Option<&MigrationCapability> { + self.capability.as_ref() + } + pub fn program_generation(&self) -> u64 { self.program_generation.load(Ordering::Acquire) } @@ -1000,32 +1199,19 @@ impl FenceConnState { } /// The authoritative write admission (section 8.1, check 2): the live gate permits this - /// connection's class, and the program and its read transaction were both admitted under - /// the gate's current write generation. On refusal the typed outcome is left in the denial - /// slot and returned. + /// connection's class, the program and its read transaction were both admitted under the + /// gate's current write generation, and a capability connection's capability is still the + /// valid one (the fence's state, owner and revision are the ones it was issued at, and it + /// is live). A validation connection never writes. On refusal the typed outcome is left in + /// the denial slot and returned. pub fn admit_write(&self) -> Result<(), FenceError> { - let result = { - let gate = self.controller.gate.borrow(); - gate.permits(self.class).and_then(|()| { - let current = gate.write_generation; - let program = self.program_generation(); - let txn = self.txn_generation(); - if program == current && txn == current { - Ok(()) - } else { - Err(FenceError::new( - FenceOutcome::MigrationWriteFenced, - format!( - "the namespace fence changed after this transaction began \ - (program admitted at generation {program}, transaction opened at \ - generation {txn}, current generation {current}); roll back and \ - begin a new transaction" - ), - ) - .with_detail(FenceDetail::StaleTransaction)) - } - }) - }; + let result = self + .admit_write_under_gate() + .and_then(|()| match &self.capability { + // Outside the gate borrow: the capability lock is never taken under one. + Some(cap) if !self.controller.capability_is_live(cap.id()) => Err(revoked(cap)), + _ => Ok(()), + }); if let Err(e) = &result { tracing::debug!( namespace = %self.controller.namespace, @@ -1037,6 +1223,37 @@ impl FenceConnState { result } + fn admit_write_under_gate(&self) -> Result<(), FenceError> { + let gate = self.controller.gate.borrow(); + gate.permits(self.class)?; + let current = gate.write_generation; + let program = self.program_generation(); + let txn = self.txn_generation(); + if program != current || txn != current { + return Err(FenceError::new( + FenceOutcome::MigrationWriteFenced, + format!( + "the namespace fence changed after this transaction began (program admitted \ + at generation {program}, transaction opened at generation {txn}, current \ + generation {current}); roll back and begin a new transaction" + ), + ) + .with_detail(FenceDetail::StaleTransaction)); + } + match (self.class, &self.capability) { + (OperationClass::CapabilityValidate, _) => Err(FenceError::new( + FenceOutcome::OperationCapabilityRequired, + "a validation capability admits reads only", + )), + (_, Some(cap)) => cap.check(&gate.fence), + (OperationClass::CapabilityImport, None) => Err(FenceError::new( + FenceOutcome::OperationCapabilityRequired, + "an import write needs a migration capability", + )), + _ => Ok(()), + } + } + /// Take the typed outcome of the last refusal at the WAL, if any. pub fn take_denial(&self) -> Option { self.denial.lock().take() diff --git a/libsql-server/src/namespace/fence/drain.rs b/libsql-server/src/namespace/fence/drain.rs index 3e5c8ba5b8..f05924a92c 100644 --- a/libsql-server/src/namespace/fence/drain.rs +++ b/libsql-server/src/namespace/fence/drain.rs @@ -59,6 +59,9 @@ impl FenceController { FenceCommand::SetSourceReadFence { .. } => { super::read::set_source_read_fence(&mut transition, &meta, request, ctx).await } + FenceCommand::SealTargetImport { .. } => { + super::import::seal_target_import(&mut transition, &meta, request, ctx).await + } _ => transition.apply(&meta, request, ctx).await, } }) @@ -234,7 +237,7 @@ async fn drain_writers( /// /// With write admission closed a manager that has been seen without a writer stays without one /// (only checkpoints can take the slot), so the managers are waited for one after the other. -async fn wait_for_writers(sources: &[LiveWriteDrain], deadline: Instant) -> bool { +pub(super) async fn wait_for_writers(sources: &[LiveWriteDrain], deadline: Instant) -> bool { for source in sources { loop { let released = source.manager.released().notified(); diff --git a/libsql-server/src/namespace/fence/import.rs b/libsql-server/src/namespace/fence/import.rs new file mode 100644 index 0000000000..d4d36451c3 --- /dev/null +++ b/libsql-server/src/namespace/fence/import.rs @@ -0,0 +1,838 @@ +//! Import sessions into quarantined migration targets, and the seal that ends them +//! (`docs/NAMESPACE_FENCE.md` sections 10.2 and 11). +//! +//! An [`ImportSession`] is the only way to write into a `TARGET_QUARANTINED` target. It holds a +//! [`MigrationCapability`] issued to the operation that owns the target and a connection whose +//! fence state carries that capability, so the WAL admits its write transactions as +//! `CapabilityImport` for as long as the capability is valid, and refuses every other +//! connection's. Each call into the session counts as an import writer until it returns. +//! +//! `SealTargetImport` ends import for good. It closes import admission in memory, persists +//! `TARGET_IMPORT_DRAINING` (the revision moves, so every issued import capability is +//! invalidated and no new one can be issued), then waits for the running import calls and for +//! any import transaction still holding the write slot to end, and persists +//! `TARGET_VALIDATING`. Like the source write drain, it waits on release notifications and +//! never takes elapsed time as evidence that a writer has finished. + +use std::sync::Arc; +use std::time::Duration; + +use bytes::Bytes; +use futures::Stream; +use tokio::time::Instant; + +use crate::connection::legacy::LegacyConnection; +use crate::connection::Connection as _; +use crate::error::Error; +use crate::namespace::configurator::{load_dump_sql, read_dump}; +use crate::namespace::meta_store::{FenceCommit, FenceContext, MetaStore}; +use crate::namespace::replication_wal::ReplicationWalWrapper; + +use super::capability::MigrationCapability; +use super::command::{DrainPolicy, FenceCommand, FenceRequest, OnDeadline}; +use super::controller::{FenceController, LiveWriteDrain, Transition}; +use super::drain::{now_ms, wait_for_writers, FORCED_ROLLBACK_GRACE}; +use super::hooks::HookPoint; +use super::outcome::{FenceError, FenceOutcome}; +use super::state::FenceState; +use super::transition::DrainCompletion; + +/// An operation's write access to its quarantined target (section 11): a capability and a +/// connection that works under it. Dropping the session revokes the capability and closes the +/// connection, which rolls back a transaction the session left open. +pub struct ImportSession { + capability: MigrationCapability, + controller: Arc, + conn: LegacyConnection, +} + +impl std::fmt::Debug for ImportSession { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("ImportSession") + .field("capability", &self.capability) + .finish_non_exhaustive() + } +} + +impl ImportSession { + pub(crate) fn new( + capability: MigrationCapability, + controller: Arc, + conn: LegacyConnection, + ) -> Self { + Self { + capability, + controller, + conn, + } + } + + pub fn capability(&self) -> &MigrationCapability { + &self.capability + } + + /// Run `f` with the session's raw connection. The call is refused up front when the + /// capability is no longer valid (the target was sealed or aborted, or its fence moved on), + /// and counts as an import writer until `f` returns, so a seal waits for it. Inside `f` the + /// WAL admits a write transaction only while the capability is still valid: a write that + /// the fence refused there is reported as that refusal, whatever `f` made of the + /// `SQLITE_AUTH` it saw. + pub async fn with_raw( + &mut self, + f: impl FnOnce(&mut rusqlite::Connection) -> R + Send + 'static, + ) -> Result { + let writer = self.controller.begin_import_write(&self.capability)?; + let conn = self.conn.clone(); + let joined = tokio::task::spawn_blocking(move || { + let _writer = writer; + let result = conn.with_raw(f); + (result, conn.fence_state().take_denial()) + }) + .await; + match joined { + Ok((_, Some(denial))) => Err(denial), + Ok((result, None)) => Ok(result), + Err(e) if e.is_panic() => std::panic::resume_unwind(e.into_panic()), + Err(e) => Err(FenceError::new( + FenceOutcome::OperationCapabilityRequired, + format!("the import call did not complete: {e}"), + )), + } + } + + /// Load a SQL dump into the target with the server's dump loader, under the capability. + /// The dump must run in one transaction that it commits itself, as for a namespace + /// created from a dump. Fence refusals are [`Error::NamespaceFence`]; dump errors are + /// [`Error::LoadDumpError`]. + pub async fn load_dump(&mut self, dump: S) -> crate::Result<()> + where + S: Stream> + Unpin, + { + let content = read_dump(dump).await?; + self.with_raw(move |conn| load_dump_sql(&content, conn)) + .await??; + Ok(()) + } +} + +impl Drop for ImportSession { + fn drop(&mut self) { + self.controller.revoke_capability(self.capability.id()); + } +} + +/// `SealTargetImport` (section 10.2), under `transition`. +/// +/// Returns the `APPLIED` commit of `TARGET_VALIDATING` once no import call is running and no +/// import transaction holds the write slot, the `DRAINING` commit of `TARGET_IMPORT_DRAINING` +/// when the deadline passes first (import stays closed, and only a replay of the same command +/// resumes the drain), or the stored result of a replay. +pub async fn seal_target_import( + transition: &mut Transition, + meta: &MetaStore, + request: FenceRequest, + mut ctx: FenceContext, +) -> crate::Result { + let controller = transition.controller().clone(); + let key = (request.operation_id, request.command_id); + let policy = match &request.command { + FenceCommand::SealTargetImport { drain_policy } => { + drain_policy.unwrap_or_else(|| meta.fence_default_write_drain()) + } + _ => return transition.apply(meta, request, ctx).await, + }; + + // Close import admission in memory before persisting: the INSTALLING gate refuses new + // import calls and import write transactions, and moving the write generation makes every + // transaction opened before it stale and wakes the queued import writers. Only for a seal + // that can apply (the owner, at the current revision, of a quarantined target), so that a + // refused command does not disturb a running import. + let gate = controller.gate(); + let can_apply = gate.state() == FenceState::TargetQuarantined + && gate.indeterminate.is_none() + && !gate.is_installing() + && gate.operation_id() == Some(request.operation_id) + && gate.revision() == request.expected_revision; + if can_apply { + transition.install_closing_gate(key); + let _ = controller.hook(HookPoint::AfterInstallingGate).await; + } + + // Persist TARGET_IMPORT_DRAINING. Its publication replaces the INSTALLING gate and, the + // revision having moved, drops every issued import capability. + let commit = match transition.apply(meta, request, ctx.clone()).await { + Ok(commit) => commit, + Err(e) => { + transition.remove_installing(); + return Err(e); + } + }; + if commit.receipt.outcome != FenceOutcome::Draining { + return Ok(commit); + } + let drain_key = (commit.receipt.operation_id, commit.receipt.command_id); + + if !drain_import_writers(&controller, policy).await { + return Ok(commit); + } + + ctx.now_ms = now_ms(); + transition + .complete_drain(meta, drain_key, DrainCompletion::TargetImport, ctx) + .await +} + +/// Wait until no import call is running and no connection manager of the target has a writer +/// holding its write slot. At the deadline, `force_rollback` rolls back the import transactions +/// still holding the slot (an idle session's open transaction; a running call ends its own) +/// and waits again for the same deadline, but at least [`FORCED_ROLLBACK_GRACE`]. `false` when +/// the drain could not be proven within the policy. +async fn drain_import_writers(controller: &FenceController, policy: DrainPolicy) -> bool { + let namespace = controller.namespace().clone(); + let deadline_after = Duration::from_millis(policy.deadline_ms); + let mut deadline = Instant::now() + deadline_after; + let mut forced = false; + loop { + if wait_for_import_writers(controller, deadline).await { + // With import closed, a manager seen without a writer under its slot lock stays + // without one. A target with no loaded maker has no connection that could write. + let sources = controller.live_write_drains(); + if sources_have_no_writer(&sources) { + tracing::info!(%namespace, "import drain proven"); + return true; + } + continue; + } + match policy.on_deadline { + OnDeadline::ForceRollback if !forced => { + forced = true; + for source in controller.live_write_drains() { + let manager = source.manager.clone(); + // The rollback takes the connection's lock, which a running import call + // holds; the release it causes is what the drain keeps waiting for. + tokio::task::spawn_blocking(move || { + if let Some(id) = manager.abort_active() { + tracing::info!( + connection = id, + "import drain deadline passed; rolling back the active import \ + transaction" + ); + } + }); + } + deadline = Instant::now() + deadline_after.max(FORCED_ROLLBACK_GRACE); + } + _ => { + tracing::info!( + %namespace, + deadline_ms = policy.deadline_ms, + on_deadline = policy.on_deadline.as_str(), + forced, + import_writers = controller.import_writers(), + "import drain deadline passed with an import writer still active; \ + answering DRAINING" + ); + return false; + } + } + } +} + +/// Wait for the running import calls to end, then for every manager's write slot to be free of +/// writers. `false` when `deadline` passes first. +async fn wait_for_import_writers(controller: &FenceController, deadline: Instant) -> bool { + loop { + let released = controller.import_released().notified(); + tokio::pin!(released); + // Registered before the check, so an end in between is not missed. + released.as_mut().enable(); + if controller.import_writers() == 0 { + break; + } + tokio::select! { + _ = &mut released => {} + _ = tokio::time::sleep_until(deadline) => return false, + } + } + wait_for_writers(&controller.live_write_drains(), deadline).await +} + +fn sources_have_no_writer(sources: &[LiveWriteDrain]) -> bool { + sources + .iter() + .all(|source| source.manager.with_no_writer(|| ()).is_ok()) +} + +/// A refusal of `open_import_session` for a namespace that is not a primary. +pub(crate) fn not_importable(namespace: &crate::namespace::NamespaceName) -> Error { + FenceError::new( + FenceOutcome::FencePreconditionFailed, + format!("namespace `{namespace}` is not a primary database; it cannot be imported into"), + ) + .with_detail(super::outcome::FenceDetail::NotPrimary) + .into() +} + +#[cfg(test)] +mod tests { + use std::sync::Arc; + + use tempfile::{tempdir, TempDir}; + use uuid::Uuid; + + use super::*; + use crate::database::Connection; + use crate::namespace::fence::capability::CapabilityPurpose; + use crate::namespace::fence::drain::tests::{assert_fenced, raw, LONG, PROMPT}; + use crate::namespace::fence::state::OperationClass; + use crate::namespace::fence::target::tests::{create, create_request, server, OP}; + use crate::namespace::meta_store::FenceCommitKind; + use crate::namespace::store::fence_tests::open_store; + use crate::namespace::store::NamespaceStore; + use crate::namespace::{NamespaceName, RestoreOption}; + + const OTHER_OP: Uuid = Uuid::from_u128(0xb); + /// A deadline that has already passed. + const NOW: DrainPolicy = DrainPolicy { + deadline_ms: 0, + on_deadline: OnDeadline::Fail, + }; + + fn tgt() -> NamespaceName { + "tgt".into() + } + + /// A store holding the quarantined target `tgt` (revision 1), and its controller. + async fn target() -> (TempDir, NamespaceStore, Arc) { + let dir = tempdir().unwrap(); + let store = open_store(dir.path()).await; + create(&store, create_request("tgt", 1)) + .await + .unwrap() + .unwrap(); + let fence = store.with(tgt(), |ns| ns.fence().clone()).await.unwrap(); + assert_eq!(fence.gate().state(), FenceState::TargetQuarantined); + (dir, store, fence) + } + + fn seal_request(command_id: u128, expected_revision: u64, policy: DrainPolicy) -> FenceRequest { + FenceRequest { + namespace: tgt(), + operation_id: OP, + command_id: Uuid::from_u128(command_id), + expected_state: FenceState::TargetQuarantined, + expected_revision, + command: FenceCommand::SealTargetImport { + drain_policy: Some(policy), + }, + } + } + + /// Run `request` through the store, on a task of its own. + fn execute( + store: &NamespaceStore, + request: FenceRequest, + ) -> tokio::task::JoinHandle> { + let store = store.clone(); + tokio::spawn(async move { store.execute_fence_command(request, server()).await }) + } + + async fn until_state(fence: &FenceController, state: FenceState) { + let mut gate = fence.subscribe(); + tokio::time::timeout(PROMPT, gate.wait_for(|g| g.state() == state)) + .await + .expect("the fence reaches the state") + .unwrap(); + } + + /// A normal connection to the target (what the admin shell, a SQL request or a dump load + /// outside a capability would use). + async fn plain_conn(store: &NamespaceStore) -> Arc { + let maker = store + .with(tgt(), |ns| ns.db.connection_maker()) + .await + .unwrap(); + Arc::new(maker.create().await.unwrap()) + } + + /// Rows in `t` on the target, read through a raw connection. + async fn count(store: &NamespaceStore) -> i64 { + let conn = plain_conn(store).await; + tokio::task::spawn_blocking(move || { + conn.with_raw(|c| c.query_row("select count(*) from t", (), |r| r.get(0))) + }) + .await + .unwrap() + .unwrap() + } + + fn fence_err(r: crate::Result) -> FenceError { + match r { + Err(Error::NamespaceFence(e)) => e, + other => panic!("expected a fence error, got {other:?}"), + } + } + + /// Write through a connection that carries `capability`, and return the fence's refusal. + async fn refused_with(store: &NamespaceStore, capability: MigrationCapability) -> FenceError { + let conn = store + .capability_connection(&tgt(), capability) + .await + .unwrap(); + let r = conn.with_raw(|c| c.execute_batch("insert into t values (99)")); + assert_fenced(r); + conn.fence_state() + .take_denial() + .expect("the WAL gate records its refusal") + } + + /// Only the operation's own, server-issued capability at the current revision writes into + /// a quarantined target: plain connections (as an admin shell would use), capabilities of + /// another operation or revision, forged, revoked or validation capabilities are all + /// refused at the WAL, and issuing one is refused for another operation or revision. + #[tokio::test(flavor = "multi_thread")] + async fn import_requires_matching_capability() { + let (_dir, store, fence) = target().await; + let mut session = store.open_import_session(tgt(), OP, 1).await.unwrap(); + let cap = session.capability().clone(); + assert_eq!( + ( + cap.namespace(), + cap.operation_id(), + cap.purpose(), + cap.fence_revision() + ), + (&tgt(), OP, CapabilityPurpose::Import, 1) + ); + session + .with_raw(|c| c.execute_batch("create table t (x); insert into t values (1)")) + .await + .unwrap() + .unwrap(); + assert_eq!(fence.import_writers(), 0); + + // A normal connection, with or without raw access. + let conn = plain_conn(&store).await; + assert_fenced(raw(&conn, "insert into t values (2)").await); + + // Issuing: another operation, a stale or future revision. + let e = fence_err(store.open_import_session(tgt(), OTHER_OP, 1).await); + assert_eq!(e.outcome(), FenceOutcome::FenceOwnedByAnotherOperation); + for revision in [0, 2] { + let e = fence_err(store.open_import_session(tgt(), OP, revision).await); + assert_eq!(e.outcome(), FenceOutcome::FenceRevisionMismatch); + } + + // At the WAL: capabilities the server never issued, of this operation at this revision, + // of another operation, at another revision, or for validation. + let forged = + |op, purpose, revision| MigrationCapability::forged(tgt(), op, purpose, revision); + let cases = [ + ( + forged(OP, CapabilityPurpose::Import, 1), + FenceOutcome::OperationCapabilityRequired, + ), + ( + forged(OTHER_OP, CapabilityPurpose::Import, 1), + FenceOutcome::FenceOwnedByAnotherOperation, + ), + ( + forged(OP, CapabilityPurpose::Import, 2), + FenceOutcome::OperationCapabilityRequired, + ), + ( + forged(OP, CapabilityPurpose::Validate, 1), + FenceOutcome::OperationCapabilityRequired, + ), + ]; + for (capability, outcome) in cases { + let e = refused_with(&store, capability.clone()).await; + assert_eq!(e.outcome(), outcome, "{capability:?}: {e}"); + } + + // A capability that was issued, once its session is gone. + let other = store.open_import_session(tgt(), OP, 1).await.unwrap(); + let revoked = other.capability().clone(); + assert!(fence.capability_is_live(revoked.id())); + drop(other); + assert!(!fence.capability_is_live(revoked.id())); + let e = refused_with(&store, revoked).await; + assert_eq!(e.outcome(), FenceOutcome::OperationCapabilityRequired); + + // The live session still writes; nothing else did. + session + .with_raw(|c| c.execute_batch("insert into t values (3)")) + .await + .unwrap() + .unwrap(); + assert_eq!(count(&store).await, 2); + } + + /// The seal closes import at once and waits, on release notifications, for an import call + /// that was already running: its write transaction, which held the slot before the seal, + /// commits, and only then is `TARGET_VALIDATING` persisted. + #[tokio::test(flavor = "multi_thread")] + async fn seal_waits_for_import_writers() { + let (_dir, store, fence) = target().await; + let mut session = store.open_import_session(tgt(), OP, 1).await.unwrap(); + session + .with_raw(|c| c.execute_batch("create table t (x)")) + .await + .unwrap() + .unwrap(); + let mut idle = store.open_import_session(tgt(), OP, 1).await.unwrap(); + + let (entered_tx, entered) = tokio::sync::oneshot::channel(); + let (resume, resume_rx) = std::sync::mpsc::channel::<()>(); + let committed = Arc::new(std::sync::atomic::AtomicBool::new(false)); + let running = tokio::spawn({ + let committed = committed.clone(); + async move { + let r = session + .with_raw(move |c| { + c.execute_batch("begin immediate; insert into t values (1)")?; + entered_tx.send(()).unwrap(); + resume_rx.recv().unwrap(); + c.execute_batch("commit")?; + committed.store(true, std::sync::atomic::Ordering::SeqCst); + Ok::<_, rusqlite::Error>(()) + }) + .await; + (session, r) + } + }); + entered.await.unwrap(); + assert_eq!(fence.import_writers(), 1); + + let sealing = execute(&store, seal_request(10, 1, LONG)); + until_state(&fence, FenceState::TargetImportDraining).await; + assert_eq!(fence.gate().revision(), 2); + + // Import is closed for good: no new capability, and no new call on a live session. + let e = fence_err(store.open_import_session(tgt(), OP, 2).await); + assert_eq!(e.outcome(), FenceOutcome::OperationCapabilityRequired); + let e = idle + .with_raw(|c| c.execute_batch("insert into t values (2)")) + .await + .unwrap_err(); + assert_eq!(e.outcome(), FenceOutcome::OperationCapabilityRequired); + assert!(!sealing.is_finished()); + assert_eq!(fence.gate().state(), FenceState::TargetImportDraining); + + // The running import finishes; its transaction commits, and only then does the seal + // reach the commit of TARGET_VALIDATING. + let completing = fence.hooks().pause_at(HookPoint::BeforeMetastoreCommit); + resume.send(()).unwrap(); + tokio::time::timeout(PROMPT, completing.reached()) + .await + .expect("the seal completes once the import writer is gone"); + assert!( + committed.load(std::sync::atomic::Ordering::SeqCst), + "the seal proceeded while the import transaction was still running" + ); + completing.resume(); + let (mut session, r) = running.await.unwrap(); + r.unwrap().unwrap(); + let commit = sealing.await.unwrap().unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + assert_eq!( + (fence.gate().state(), fence.gate().revision()), + (FenceState::TargetValidating, 3) + ); + // The publication of TARGET_IMPORT_DRAINING dropped both sessions' capabilities. + assert_eq!(fence.live_capabilities(), 0); + assert_eq!(count(&store).await, 1); + let e = session + .with_raw(|c| c.execute_batch("insert into t values (3)")) + .await + .unwrap_err(); + assert_eq!(e.outcome(), FenceOutcome::OperationCapabilityRequired); + } + + /// An import transaction left open by an idle session holds the write slot: the seal does + /// not complete past its deadline (`on_deadline: fail`), `TARGET_IMPORT_DRAINING` stays + /// durable and closed (also across a restart), another operation cannot touch it, and only + /// the owner's seal resumes it: a replay of the same seal completes once the session is + /// gone (its transaction rolled back). + #[tokio::test(flavor = "multi_thread")] + async fn seal_deadline_leaves_import_draining_until_replayed() { + let (dir, store, fence) = target().await; + let mut session = store.open_import_session(tgt(), OP, 1).await.unwrap(); + session + .with_raw(|c| c.execute_batch("create table t (x)")) + .await + .unwrap() + .unwrap(); + session + .with_raw(|c| c.execute_batch("begin immediate; insert into t values (1)")) + .await + .unwrap() + .unwrap(); + assert_eq!(fence.import_writers(), 0); + + let commit = execute(&store, seal_request(10, 1, NOW)) + .await + .unwrap() + .unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Draining); + assert_eq!( + (fence.gate().state(), fence.gate().revision()), + (FenceState::TargetImportDraining, 2) + ); + + // Another operation's seal is refused; a new seal of the owner joins the drain, which + // still cannot complete. + let mut foreign = seal_request(11, 2, NOW); + foreign.operation_id = OTHER_OP; + foreign.expected_state = FenceState::TargetImportDraining; + let e = fence_err(execute(&store, foreign).await.unwrap()); + assert_eq!(e.outcome(), FenceOutcome::FenceOwnedByAnotherOperation); + let mut join = seal_request(12, 2, NOW); + join.expected_state = FenceState::TargetImportDraining; + let joined = execute(&store, join).await.unwrap().unwrap(); + assert_eq!(joined.receipt.outcome, FenceOutcome::Draining); + assert_eq!( + (fence.gate().state(), fence.gate().revision()), + (FenceState::TargetImportDraining, 2) + ); + + // Still draining after a restart, which drops the open transaction. + drop(session); + store.shutdown().await.unwrap(); + let store = open_store(dir.path()).await; + let fence = store.fence_controller(&tgt()); + assert_eq!(fence.gate().state(), FenceState::TargetImportDraining); + assert!(fence.permits(OperationClass::CapabilityImport).is_ok()); + let e = fence_err(store.open_import_session(tgt(), OP, 2).await); + assert_eq!(e.outcome(), FenceOutcome::OperationCapabilityRequired); + + let replay = execute(&store, seal_request(10, 1, NOW)) + .await + .unwrap() + .unwrap(); + assert_eq!(replay.receipt.outcome, FenceOutcome::Applied); + assert_eq!( + (fence.gate().state(), fence.gate().revision()), + (FenceState::TargetValidating, 3) + ); + assert_eq!(count(&store).await, 0); + let replay = execute(&store, seal_request(10, 1, NOW)) + .await + .unwrap() + .unwrap(); + assert_eq!(replay.kind, FenceCommitKind::Replayed); + } + + /// With `on_deadline: force_rollback` the seal rolls back an import transaction that an + /// idle session left holding the write slot, waits for the release, and completes. + #[tokio::test(flavor = "multi_thread")] + async fn seal_force_rollback_ends_open_import_transaction() { + let (_dir, store, fence) = target().await; + let mut session = store.open_import_session(tgt(), OP, 1).await.unwrap(); + session + .with_raw(|c| c.execute_batch("create table t (x)")) + .await + .unwrap() + .unwrap(); + session + .with_raw(|c| c.execute_batch("begin immediate; insert into t values (1)")) + .await + .unwrap() + .unwrap(); + let policy = DrainPolicy { + deadline_ms: 0, + on_deadline: OnDeadline::ForceRollback, + }; + let commit = execute(&store, seal_request(10, 1, policy)) + .await + .unwrap() + .unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + assert_eq!(fence.gate().state(), FenceState::TargetValidating); + drop(session); + assert_eq!(count(&store).await, 0); + } + + /// Once sealed, nothing imports: no capability is issued at the new revision, an older + /// session's calls are refused, and even a capability naming the current revision is + /// refused at the WAL. + #[tokio::test(flavor = "multi_thread")] + async fn sealed_target_rejects_import() { + let (_dir, store, fence) = target().await; + let mut session = store.open_import_session(tgt(), OP, 1).await.unwrap(); + session + .with_raw(|c| c.execute_batch("create table t (x)")) + .await + .unwrap() + .unwrap(); + let commit = execute(&store, seal_request(10, 1, LONG)) + .await + .unwrap() + .unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + assert_eq!( + (fence.gate().state(), fence.gate().revision()), + (FenceState::TargetValidating, 3) + ); + + for revision in [1, 3] { + let e = fence_err(store.open_import_session(tgt(), OP, revision).await); + assert_eq!(e.outcome(), FenceOutcome::OperationCapabilityRequired); + } + let e = session + .with_raw(|c| c.execute_batch("insert into t values (1)")) + .await + .unwrap_err(); + assert_eq!(e.outcome(), FenceOutcome::OperationCapabilityRequired); + let e = refused_with( + &store, + MigrationCapability::forged(tgt(), OP, CapabilityPurpose::Import, 3), + ) + .await; + assert_eq!(e.outcome(), FenceOutcome::OperationCapabilityRequired); + assert_fenced(raw(&plain_conn(&store).await, "insert into t values (1)").await); + assert_eq!(count(&store).await, 0); + } + + /// A synthetic representative schema: tables with keys and a foreign key, indexes, a + /// trigger, a view and an FTS5 table. + const SCHEMA: &str = " + create table users (id integer primary key, email text not null unique, name text); + create table orders ( + id integer primary key, + user_id integer not null references users(id), + total real not null, + note text + ); + create index orders_by_user on orders(user_id, total); + create table audit (id integer primary key autoincrement, what text); + create trigger orders_audit after insert on orders begin + insert into audit (what) values ('order ' || new.id); + end; + create view order_totals as + select u.email, sum(o.total) as total from users u join orders o on o.user_id = u.id + group by u.email; + create virtual table docs using fts5(title, body); + insert into users (email, name) values ('a@example.com', 'A'), ('b@example.com', 'B'); + insert into orders (user_id, total, note) values (1, 10.5, 'first'), (1, 2, null), + (2, 7.25, 'it''s quoted'); + insert into docs (title, body) values ('fence', 'operation owned namespace fence'), + ('import', 'quarantined target import'); + "; + + /// What a target must reproduce of the source: the schema, the rows, and the derived data + /// (the view, the full-text index). + fn contents(c: &rusqlite::Connection) -> Vec { + let mut out = Vec::new(); + let mut q = |sql: &str| { + use rusqlite::types::ValueRef; + let mut stmt = c.prepare(sql).unwrap(); + let n = stmt.column_count(); + let rows = stmt + .query_map((), |r| { + (0..n) + .map(|i| { + r.get_ref(i).map(|v| match v { + ValueRef::Text(t) => String::from_utf8_lossy(t).into_owned(), + other => format!("{other:?}"), + }) + }) + .collect::>>() + }) + .unwrap(); + for row in rows { + out.push(format!("{sql}: {}", row.unwrap().join(", "))); + } + }; + // The loader re-renders every statement it runs, so the stored SQL of an object differs + // from the source's in case and spacing only. + q("select type, name, tbl_name, \ + lower(replace(replace(replace(sql, ' ', ''), char(10), ''), '\"', '')) \ + from sqlite_schema order by type, name"); + q("select * from users order by id"); + q("select * from orders order by id"); + q("select * from audit order by id"); + q("select * from order_totals order by email"); + q("select title from docs where docs match 'quarantined' order by rowid"); + q("select count(*) from docs"); + out + } + + /// The server's own dump loader runs inside an import session: a dump exported from a + /// source with a representative schema loads into the quarantined target, which then holds + /// exactly what the source held, while nothing else could write to it. + #[tokio::test(flavor = "multi_thread")] + async fn import_session_loads_dump_into_quarantined_target() { + let (_dir, store, fence) = target().await; + store + .create("src".into(), RestoreOption::Latest, Default::default()) + .await + .unwrap(); + let src = { + let maker = store + .with("src".into(), |ns| ns.db.connection_maker()) + .await + .unwrap(); + Arc::new(maker.create().await.unwrap()) + }; + let (dump, expected) = { + let src = src.clone(); + tokio::task::spawn_blocking(move || { + src.with_raw(|c| { + c.execute_batch(SCHEMA).unwrap(); + let mut dump = Vec::new(); + crate::connection::dump::exporter::export_dump(c, &mut dump, false).unwrap(); + (dump, contents(c)) + }) + }) + .await + .unwrap() + }; + let text = String::from_utf8(dump.clone()).unwrap(); + assert!(text.contains("CREATE VIRTUAL TABLE"), "{text}"); + + let mut session = store.open_import_session(tgt(), OP, 1).await.unwrap(); + let stream = futures::stream::iter( + dump.chunks(64) + .map(|c| Ok(Bytes::copy_from_slice(c))) + .collect::>(), + ); + session.load_dump(stream).await.unwrap(); + assert_eq!(fence.gate().state(), FenceState::TargetQuarantined); + // Nothing but the session could have written it. + assert_fenced( + raw( + &plain_conn(&store).await, + "insert into audit (what) values ('x')", + ) + .await, + ); + + // A second load fails as the loader always does on a dump that is not in a + // transaction, without the fence being involved. + let r = session + .load_dump(futures::stream::iter(vec![Ok(Bytes::from_static( + b"savepoint a; release a; savepoint b;", + ))])) + .await; + assert!( + matches!( + r, + Err(Error::LoadDumpError(crate::error::LoadDumpError::NoTxn)) + ), + "{r:?}" + ); + drop(session); + + let commit = execute(&store, seal_request(10, 1, LONG)) + .await + .unwrap() + .unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + let conn = plain_conn(&store).await; + let imported = tokio::task::spawn_blocking(move || conn.with_raw(|c| contents(c))) + .await + .unwrap(); + assert_eq!(imported, expected); + } +} diff --git a/libsql-server/src/namespace/fence/mod.rs b/libsql-server/src/namespace/fence/mod.rs index fd8810e0db..a8d80da59c 100644 --- a/libsql-server/src/namespace/fence/mod.rs +++ b/libsql-server/src/namespace/fence/mod.rs @@ -10,8 +10,8 @@ //! ([`store`], driven by `MetaStore::apply_fence_command`), and the in-memory authority built //! on them: the per-namespace [`controller`] with its gate and read leases, the positive write //! [`drain`], the source [`read`] fence and its -//! [`stream`] leases for dump and replication, quarantined migration [`target`]s, the -//! [`registry`] that holds the controllers outside the namespace cache, and the test [`hooks`] +//! [`stream`] leases for dump and replication, quarantined migration [`target`]s with their +//! [`capability`]-scoped [`import`] sessions and seal drain, the [`registry`] that holds the controllers outside the namespace cache, and the test [`hooks`] //! on their paths. // The persistence, controller and protocol layers that consume these types land in the @@ -19,10 +19,12 @@ // the crate. This attribute is removed once they are wired. #![allow(dead_code)] +pub mod capability; pub mod command; pub mod controller; pub mod drain; pub mod hooks; +pub mod import; pub mod outcome; pub mod read; pub mod record; diff --git a/libsql-server/src/namespace/store.rs b/libsql-server/src/namespace/store.rs index 177f2cc471..43414bc738 100644 --- a/libsql-server/src/namespace/store.rs +++ b/libsql-server/src/namespace/store.rs @@ -13,7 +13,7 @@ use tokio_stream::wrappers::BroadcastStream; use crate::auth::Authenticated; use crate::broadcaster::BroadcastMsg; use crate::connection::config::DatabaseConfig; -use crate::database::DatabaseKind; +use crate::database::{Database, DatabaseKind, PrimaryConnectionMaker}; use crate::error::Error; use crate::metrics::NAMESPACE_LOAD_LATENCY; use crate::namespace::{NamespaceBottomlessDbId, NamespaceBottomlessDbIdInit, NamespaceName}; @@ -21,9 +21,11 @@ use crate::stats::Stats; use super::broadcasters::{BroadcasterHandle, BroadcasterRegistry}; use super::configurator::{DynConfigurator, NamespaceConfigurators}; +use super::fence::capability::CapabilityPurpose; use super::fence::command::{FenceCommand, FenceRequest}; use super::fence::controller::{FenceController, Transition}; use super::fence::hooks::HookPoint; +use super::fence::import::{self, ImportSession}; use super::fence::outcome::FenceOutcome; use super::fence::record::ServerIdentity; use super::fence::registry::FenceRegistry; @@ -694,6 +696,71 @@ impl NamespaceStore { Ok(commit) } + /// Issue an import capability to `operation_id` and open a connection that works under it + /// (`docs/NAMESPACE_FENCE.md` section 11): the only way to write into a quarantined target. + /// Valid only while the target is `TARGET_QUARANTINED`, owned by `operation_id`, at + /// `expected_revision`; the target is loaded if it is not. Fence refusals are + /// [`Error::NamespaceFence`] with their stable outcome code (`OPERATION_CAPABILITY_REQUIRED` + /// in any other state, `FENCE_OWNED_BY_ANOTHER_OPERATION`, `FENCE_REVISION_MISMATCH`). + pub async fn open_import_session( + &self, + namespace: NamespaceName, + operation_id: uuid::Uuid, + expected_revision: u64, + ) -> crate::Result { + let (controller, maker) = self.primary_maker(&namespace).await?; + let capability = controller.issue_capability( + CapabilityPurpose::Import, + operation_id, + expected_revision, + )?; + match maker + .inner() + .make_capability_connection(capability.clone()) + .await + { + Ok(conn) => Ok(ImportSession::new(capability, controller, conn)), + Err(e) => { + controller.revoke_capability(capability.id()); + Err(e) + } + } + } + + /// The fence controller and connection maker of the primary `namespace`, loading it. + async fn primary_maker( + &self, + namespace: &NamespaceName, + ) -> crate::Result<(Arc, Arc)> { + let (controller, maker) = self + .with(namespace.clone(), |ns| { + let maker = match &ns.db { + Database::Primary(p) => Some(p.connection_maker()), + _ => None, + }; + (ns.fence().clone(), maker) + }) + .await?; + match maker { + Some(maker) => Ok((controller, maker)), + None => Err(import::not_importable(namespace)), + } + } + + /// A connection to `namespace` that works under `capability`, whether or not this server + /// issued it: for tests that prove the WAL refuses a capability that is not the valid one. + #[cfg(test)] + pub(crate) async fn capability_connection( + &self, + namespace: &NamespaceName, + capability: super::fence::capability::MigrationCapability, + ) -> crate::Result< + crate::connection::legacy::LegacyConnection, + > { + let (_, maker) = self.primary_maker(namespace).await?; + maker.inner().make_capability_connection(capability).await + } + /// Whether the server already knows `namespace`: its config is in memory, or the namespace /// cache holds it (a loaded namespace, or a fork in flight, which holds its entry locked). async fn name_in_use(&self, namespace: &NamespaceName) -> bool { From 928ab7d4069d27b4ac720dc29d54a9f7041d4131 Mon Sep 17 00:00:00 2001 From: River Date: Tue, 29 Sep 2026 19:24:41 +0000 Subject: [PATCH 16/33] libsql-server: validate, publish and enable writes on migration targets Co-authored-by: Tomasz Szymczyszyn --- docs/NAMESPACE_FENCE.md | 29 +- .../src/namespace/fence/controller.rs | 32 + libsql-server/src/namespace/fence/target.rs | 566 +++++++++++++++++- libsql-server/src/namespace/store.rs | 93 ++- 4 files changed, 700 insertions(+), 20 deletions(-) diff --git a/docs/NAMESPACE_FENCE.md b/docs/NAMESPACE_FENCE.md index cf724e98b6..718a67fe5f 100644 --- a/docs/NAMESPACE_FENCE.md +++ b/docs/NAMESPACE_FENCE.md @@ -557,9 +557,11 @@ A deadline reached before step 4 answers `DRAINING` and leaves `TARGET_IMPORT_DR ### 10.3 Validation and publication -In `TARGET_VALIDATING` the operation reads through the validation capability (`validation-query`, or the internal API). `RecordTargetValidation` stores the result in the record and receipt. `PublishTargetReadableWriteFenced` requires that the most recent `RecordTargetValidation` of the owning operation has `result: ok`; otherwise `FENCE_PRECONDITION_FAILED`, `validation_receipt_required`. Publication clears the legacy `block_reads` mirror. +`NamespaceStore::open_validation_session` issues a server-owned `Validate` capability and opens a capability connection for the owning operation at the current revision. The connection has SQLite `query_only` enabled; `ValidationSession::with_raw` re-checks the capability before every call, and the WAL independently refuses every write transaction of class `CapabilityValidate`. Dropping the session revokes its capability. A session is valid in `TARGET_VALIDATING` or `TARGET_WRITE_FENCED`; every transition that moves the revision invalidates it, so the operation opens a new session at the returned revision when it needs more validation reads. -`EnableTargetWrites` is CAS from `TARGET_WRITE_FENCED`; commit, publish the gate with a new generation, respond. The legacy mirror is restored to the values given at creation. Replays return the stored result; a new command asking for the same thing returns `ALREADY_APPLIED`. Every other command on a `TARGET_WRITABLE` record from the same operation is `INVALID_FENCE_TRANSITION`. +For a new `RecordTargetValidation`, `NamespaceStore::execute_fence_command` observes the target's current replication-log id and frame and its SQLite page count through a validation session, and stores that snapshot with the caller's result and bounded summary in the record. The durable command key is checked before collecting the snapshot and again if a concurrent copy invalidates the capability: exact replay and command-id conflict always reach the metastore's replay/fingerprint check without requiring a fresh capability or observation. `PublishTargetReadableWriteFenced` requires that the most recent `RecordTargetValidation` of the owning operation has `result: ok`; otherwise `FENCE_PRECONDITION_FAILED`, `validation_receipt_required`. Publication makes normal reads visible, keeps writes fenced and clears the durable legacy `block_reads` mirror. + +`EnableTargetWrites` is CAS from `TARGET_WRITE_FENCED`; commit, publish the gate with a new generation, respond. The durable legacy mirror is restored to the values given at creation. Replays return the stored result; a new command asking for the same thing returns `ALREADY_APPLIED`. Every other command on a `TARGET_WRITABLE` record from the same operation is `INVALID_FENCE_TRANSITION`. A transaction opened under the prior generation cannot upgrade to a write after publication; a fresh transaction can. `AbortQuarantinedTarget` moves to `TARGET_ABORTED`; all normal traffic stays denied. @@ -600,9 +602,11 @@ impl NamespaceStore { pub async fn open_import_session(&self, namespace: NamespaceName, operation_id: Uuid, expected_revision: u64) -> crate::Result; - /// Read-only validation connection (`query_only`), TARGET_VALIDATING or TARGET_WRITE_FENCED. + /// Issue a read-only validation capability and a `query_only` capability connection. Valid + /// in TARGET_VALIDATING or TARGET_WRITE_FENCED for the owner at `expected_revision`; loads + /// the target. Fence refusals use the same `Error::NamespaceFence` shape as import. pub async fn open_validation_session(&self, ns: NamespaceName, operation_id: Uuid, - expected_revision: u64) -> Result; + expected_revision: u64) -> crate::Result; pub async fn apply_fence_command(&self, ns: NamespaceName, cmd: FenceCommand) -> Result; @@ -631,10 +635,17 @@ impl ImportSession { /// uses) run inside `with_raw`. Dump errors are `Error::LoadDumpError`. pub async fn load_dump(&mut self, dump: S) -> crate::Result<()> where S: Stream> + Unpin; +pub struct ValidationSession { /* capability, controller, query_only capability connection */ } +impl ValidationSession { + pub fn capability(&self) -> &MigrationCapability; + /// Re-check the live validation capability, then run one raw read call. The WAL still + /// refuses a write if the closure disables `query_only`. + pub async fn with_raw(&mut self, + f: impl FnOnce(&mut rusqlite::Connection) -> R + Send + 'static) -> Result; } ``` -`FenceError` carries a `FenceOutcome` (section 6) and converts into the server's `Error`, so a route built on top returns the same codes. The import-writer count is of running calls: a session that is not running a call cannot start a write once the seal has moved the revision, and a transaction it left open holds the write slot, which the seal waits for as well (section 10.2). Dropping an `ImportSession` revokes its capability and closes its connection, which rolls back a transaction it left open and so releases the slot. The capability connection shares the target's write slot, WAL and replication log, and is not counted by the connection throttle. The loader holds the connection for the whole dump and re-renders each statement it runs (as it does for a namespace created from a dump), so the stored SQL text of schema objects differs from the source's in case and spacing. Streaming and memory bounds are not part of this series. +`FenceError` carries a `FenceOutcome` (section 6) and converts into the server's `Error`, so a route built on top returns the same codes. The import-writer count is of running calls: a session that is not running a call cannot start a write once the seal has moved the revision, and a transaction it left open holds the write slot, which the seal waits for as well (section 10.2). Dropping either session revokes its capability and closes its connection; dropping an `ImportSession` therefore rolls back a transaction it left open and releases the slot. Capability connections share the target's WAL and replication log and are not counted by the connection throttle. The loader holds the connection for the whole dump and re-renders each statement it runs (as it does for a namespace created from a dump), so the stored SQL text of schema objects differs from the source's in case and spacing. Streaming and memory bounds are not part of this series. ## 12. Incident adoption (Contract) @@ -751,7 +762,7 @@ Planned test names; the table is updated as tests land. | 2 | Active writer commits or is rolled back before freeze acknowledgement; nothing commits after | landed: `namespace::fence::drain::tests::{active_writer_commits_before_ack, forced_rollback_before_ack, no_commit_after_ack}` (the boundary equals the last committed replication frame and no frame follows it; autocommit, `BEGIN IMMEDIATE`, DDL and a pre-fence read transaction upgrading are refused), `installing_gate_closes_writes_before_persisting`, `refused_acquire_reopens_writes`, `release_reopens_with_new_generation` | | 3 | Autocommit, explicit transactions, queued writers, batches, DDL, schema jobs, old WebSockets, read-to-write upgrades cannot bypass | landed: `connection::connection_manager::fence_tests::{fence_rejects_read_to_write_upgrade, fence_rejects_ddl_and_pragma, fence_rejects_raw_with_raw_write}` (autocommit, explicit transactions, DDL, header-writing pragma, `BEGIN IMMEDIATE`, `VACUUM`, `with_raw` users); `connection::connection_manager::fence_tests::fence_rejects_queued_writer` (a writer parked in the queue behind an open transaction leaves it with `MIGRATION_WRITE_FENCED` when the fence changes, and the holder keeps the slot); maintenance and vacuum under a fence: `queued_checkpoint_survives_fence_wake`, `checkpoint_allowed_while_fenced`, `vacuum_skipped_while_fenced`; drain primitives: `abort_active_tolerates_closed_connection`, `release_notifies_drain_waiters`, `fence::controller::tests::write_queues_are_woken_on_every_generation_change`; planned: `tests::fence::protocol::old_ws_session_cannot_write`, `batch_denied_mid_batch`; `fence::tests::acquire_rejects_shared_schema` | | 4 | Program that captured config before the fence is rejected at the WAL | landed: `connection::connection_manager::fence_tests::wal_gate_rejects_program_admitted_before_fence` (a SQL function parks the program between admission and its write while the fence is acquired and released) | -| 5 | Pre-fence transactions cannot write after release or publication | landed: `connection::connection_manager::fence_tests::stale_generation_cannot_write_after_release`; unfenced behaviour unchanged: `unfenced_namespace_is_unchanged`; planned: `stale_generation_cannot_write_after_enable_writes` | +| 5 | Pre-fence transactions cannot write after release or publication | landed: `connection::connection_manager::fence_tests::stale_generation_cannot_write_after_release`, `namespace::fence::target::tests::stale_generation_cannot_write_after_enable_writes`; unfenced behaviour unchanged: `unfenced_namespace_is_unchanged` | | 6 | Acquisition timeout returns `DRAINING`, admission stays closed | landed: `namespace::fence::drain::tests::{deadline_returns_draining_and_stays_closed, replay_of_draining_resumes_and_completes}` | | 7 | Restart at every persistence boundary; indeterminate persistence keeps the gate closed until same-command reconciliation | landed: `namespace::fence::tests::restart_at_each_boundary` (a crash at each of 17 boundaries of `AcquireSourceWriteFence` and `ReleaseSourceWriteFence`: after the `INSTALLING` gate, before, at and after each metastore commit, with and without a lagging marker, before publication, before boundary capture and before the response; the restart recovers the prior or the committed state, admits writes only if an opening transition committed, repairs the marker, and a replay finishes the command), `restart_in_draining_waits_for_the_same_command` (a writer active at the crash; nothing advances until the same command is replayed, which completes at once), `indeterminate_commit_keeps_gate_closed` (through the drain path, both when the commit happened and when it did not), `acquire_response_loss_resolved_by_replay_and_inspect`; `fence::controller::tests::{indeterminate_commit_keeps_writes_closed_until_replayed, indeterminate_commit_that_did_not_apply_is_retried_by_replay, failed_before_commit_leaves_gate_unchanged, publication_happens_before_the_response, committed_command_is_published_when_the_caller_goes_away}`, `namespace::store::fence_tests::restart_installs_the_durable_gate_before_serving` | | 8 | Evict and lazily reload a fenced namespace; identical admission | `tests::fence::lifecycle::evicted_namespace_reloads_same_gate`; landed at unit level: `namespace::store::fence_tests::evicted_namespace_reloads_with_the_same_controller`, `fence::registry::tests::seeded_from_load_fences_including_recovered_names` | @@ -759,9 +770,9 @@ Planned test names; the table is updated as tests land. | 10 | Wrong owner, stale revision, invalid role/state, replay, command-id reuse; replay before revision check | `fence::transition::tests::*` (exhaustive over states × commands) | | 11 | Target creation raced with SQL, dump, replication, lifecycle never observable as writable or readable | landed: `namespace::fence::target::tests::create_race_never_observable` (parked after the rows commit and before the config is published and the namespace loaded: SQL connections, stats, replication `hello` (never `UNAVAILABLE`), create, delete and fork of the name are denied or find nothing; afterwards the target is loaded behind the quarantine gate, SQL reads and WAL writes are refused, and lifecycle and replication are refused with `MIGRATION_TARGET_QUARANTINED`), `creating_gate_refuses_before_commit` (parked before the metastore transaction: the same attempts are refused and no database file is created), `create_replay_completes_interrupted_creation` (marker only, after a restart), `create_completes_when_the_caller_goes_away`, `indeterminate_create_is_completed_by_replay`, `create_rejects_existing_name` (a loaded or cold existing name, whose gate never moves, and another operation's target), `abort_keeps_traffic_denied` (also across a restart) | | 12 | Only the matching import capability writes a quarantined target; admin credentials and admin shell cannot | landed: `namespace::fence::import::tests::import_requires_matching_capability` (plain connections with and without raw access, as the admin shell uses; issuing to another operation or at another revision; at the WAL, capabilities the server never issued, of another operation, at another revision, for validation, and revoked); `namespace::fence::target::tests::{create_race_never_observable, abort_keeps_traffic_denied}` (raw DDL refused); planned: `tests::fence::admin::admin_shell_cannot_write_quarantined` | -| 13 | Seal enters `TARGET_IMPORT_DRAINING`, waits, reaches `TARGET_VALIDATING`, cannot resume import; only a durable validation receipt permits idempotent publication | seal landed: `namespace::fence::import::tests::{seal_waits_for_import_writers (an import call parked inside its write transaction: import is closed at once, the transaction commits, and the seal reaches its completion commit only afterwards), seal_deadline_leaves_import_draining_until_replayed (an idle session's open transaction; `DRAINING`, also across a restart; another operation refused, the owner's new seal joins; the replay completes), seal_force_rollback_ends_open_import_transaction, sealed_target_rejects_import}`; planned: `fence::target::tests::{publish_requires_validation_receipt, publish_is_idempotent}` | -| 14 | Enable writes idempotent, survives restart and response loss, irreversible | `fence::target::tests::enable_writes_idempotent_and_irreversible`, `enable_writes_survives_restart` | -| 15 | Lost `EnableTargetWrites` response resolved from receipt/state | `fence::target::tests::enable_writes_response_loss_resolved` | +| 13 | Seal enters `TARGET_IMPORT_DRAINING`, waits, reaches `TARGET_VALIDATING`, cannot resume import; only a durable validation receipt permits idempotent publication | seal landed: `namespace::fence::import::tests::{seal_waits_for_import_writers (an import call parked inside its write transaction: import is closed at once, the transaction commits, and the seal reaches its completion commit only afterwards), seal_deadline_leaves_import_draining_until_replayed (an idle session's open transaction; `DRAINING`, also across a restart; another operation refused, the owner's new seal joins; the replay completes), seal_force_rollback_ends_open_import_transaction, sealed_target_rejects_import}`; validation/publication landed: `namespace::fence::target::tests::{validation_session_is_read_only (query-only plus WAL defence, live capability checks, server snapshot and a concurrent exact replay while the receipt is committed but not published), publish_requires_validation_receipt, publish_is_idempotent}` | +| 14 | Enable writes idempotent, survives restart and response loss, irreversible | landed: `namespace::fence::target::tests::{enable_writes_idempotent_and_irreversible, enable_writes_survives_restart}` | +| 15 | Lost `EnableTargetWrites` response resolved from receipt/state | landed: `namespace::fence::target::tests::enable_writes_response_loss_resolved` | | 16 | Read fence drains SQL, dump, `log_entries`, `snapshot`, including dead peers and forced termination | SQL landed: `namespace::fence::read::tests::{read_fence_waits_for_running_program, program_after_closing_gate_is_refused (parked after the read-closing gate, before the CAS), read_fence_cancels_at_deadline, unreleased_lease_answers_draining_and_replay_completes, idle_txn_fails_on_next_program, clear_read_fence_reopens_reads_not_writes, refused_read_fence_reopens_reads, attach_of_read_fenced_namespace_denied}`, `admin_shell::fence_tests::admin_shell_read_denied`; dump and replication landed: `namespace::fence::stream::tests::{dump_lease_released_on_cancel (a dump blocked mid-row on a peer that stopped reading is cancelled at the deadline, its lease released without the peer, the fence acknowledged, and the body ends with the fence error and no `COMMIT;`), read_fence_waits_for_dump, dump_refused_while_read_fenced, log_entries_stream_ends_typed, stream_lease_released_without_peer_read (a dead peer), snapshot_stream_ends_typed, replication_calls_denied_while_read_fenced, read_fence_forced_termination}`; planned: the HTTP-level check that an interrupted `/dump` response is aborted rather than completed, with `tests::fence::protocol::dump_codes` | | 17 | Delete, reset, fork, restore, config, schema mutation rejected | `tests::fence::lifecycle::lifecycle_rejected_while_fenced` | | 18 | Codes through HTTP, Hrana, RPC, dump, replication, replica write proxy; distinguishable from auth/timeout/not-found; old peers compatible; no retry loops | `tests::fence::protocol::{http_codes, hrana_http_codes, hrana_ws_codes, rpc_codes, dump_codes, replication_codes, replica_proxy_preserves_code, auth_and_not_found_distinct, denial_not_retried}`; `libsql-replication` `proxy_error_stable_code_is_additive` | diff --git a/libsql-server/src/namespace/fence/controller.rs b/libsql-server/src/namespace/fence/controller.rs index 1eeeffc804..4dfd45743b 100644 --- a/libsql-server/src/namespace/fence/controller.rs +++ b/libsql-server/src/namespace/fence/controller.rs @@ -573,6 +573,38 @@ impl FenceController { self.capabilities.lock().live.contains_key(&id) } + /// Check that `cap` is the live server-issued capability of `purpose` for this namespace + /// and still matches the published fence. Validation sessions use this before every call; + /// import sessions perform the same checks while also incrementing their writer count. + pub(crate) fn check_capability( + &self, + cap: &MigrationCapability, + purpose: CapabilityPurpose, + ) -> Result<(), FenceError> { + if cap.namespace() != &self.namespace || cap.purpose() != purpose { + return Err(FenceError::new( + FenceOutcome::OperationCapabilityRequired, + format!( + "a {} capability for namespace `{}` does not admit {} work on `{}`", + cap.purpose().as_str(), + cap.namespace(), + purpose.as_str(), + self.namespace + ), + )); + } + let caps = self.capabilities.lock(); + { + let gate = self.gate.borrow(); + gate.permits(purpose.class())?; + cap.check(&gate.fence)?; + } + if !caps.live.contains_key(&cap.id()) { + return Err(revoked(cap)); + } + Ok(()) + } + /// Admit one import call under `cap`, counted until the returned guard is dropped. The /// capability is checked against the gate under the capability lock, so an import call is /// either refused by a seal that closed admission before it, or counted by the seal, which diff --git a/libsql-server/src/namespace/fence/target.rs b/libsql-server/src/namespace/fence/target.rs index 3b69b7efe5..38729dd191 100644 --- a/libsql-server/src/namespace/fence/target.rs +++ b/libsql-server/src/namespace/fence/target.rs @@ -14,10 +14,16 @@ use uuid::Uuid; +use crate::connection::legacy::LegacyConnection; +use crate::connection::Connection as _; +use crate::namespace::replication_wal::ReplicationWalWrapper; use crate::namespace::NamespaceName; +use super::capability::{CapabilityPurpose, MigrationCapability}; use super::command::{FenceCommand, FenceRequest, TargetConfig}; +use super::controller::FenceController; use super::outcome::{FenceDetail, FenceError, FenceOutcome}; +use super::record::ValidationSnapshot; use super::state::FenceState; /// `CreateTargetQuarantined` for `namespace`, by `operation_id`. The expectation is always @@ -45,6 +51,103 @@ impl From for FenceRequest { } } +/// Read-only access used by the owning operation to validate a sealed target. The connection +/// carries a server-issued validation capability and has SQLite's `query_only` mode enabled; +/// every call checks that the capability still matches the target's owner, state and revision. +pub struct ValidationSession { + capability: MigrationCapability, + controller: std::sync::Arc, + conn: LegacyConnection, +} + +impl std::fmt::Debug for ValidationSession { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("ValidationSession") + .field("capability", &self.capability) + .finish_non_exhaustive() + } +} + +impl ValidationSession { + pub(crate) async fn new( + capability: MigrationCapability, + controller: std::sync::Arc, + conn: LegacyConnection, + ) -> crate::Result { + let mut this = Self { + capability, + controller, + conn, + }; + this.with_raw(|conn| conn.pragma_update(None, "query_only", true)) + .await??; + Ok(this) + } + + pub fn capability(&self) -> &MigrationCapability { + &self.capability + } + + /// Run a read-only operation on the capability connection. The capability is checked before + /// the call, and a write refused at the WAL is returned as its typed fence outcome even if + /// the closure swallowed SQLite's `SQLITE_AUTH`. + pub async fn with_raw( + &mut self, + f: impl FnOnce(&mut rusqlite::Connection) -> R + Send + 'static, + ) -> Result { + self.controller + .check_capability(&self.capability, CapabilityPurpose::Validate)?; + let conn = self.conn.clone(); + let joined = tokio::task::spawn_blocking(move || { + let result = conn.with_raw(f); + (result, conn.fence_state().take_denial()) + }) + .await; + match joined { + Ok((_, Some(denial))) => Err(denial), + Ok((result, None)) => Ok(result), + Err(e) if e.is_panic() => std::panic::resume_unwind(e.into_panic()), + Err(e) => Err(FenceError::new( + FenceOutcome::OperationCapabilityRequired, + format!("the validation call did not complete: {e}"), + )), + } + } + + /// What the server records beside `RecordTargetValidation`: the target's current + /// replication-log identity and frame, and SQLite page count. Target writes have already + /// been positively drained, so these observations cannot race a mutation. + pub(crate) async fn snapshot(&mut self) -> crate::Result { + let sources = self.controller.live_write_drains(); + let Some(latest) = sources.last() else { + return Err(FenceError::new( + FenceOutcome::FenceStateUnavailable, + format!( + "validation target `{}` has no live primary replication log", + self.capability.namespace() + ), + ) + .into()); + }; + let log_id = latest.log_id; + let frame_no = (latest.current_frame_no)().unwrap_or(0); + let page_count = self + .with_raw(|conn| conn.query_row("PRAGMA page_count", (), |row| row.get::<_, u64>(0))) + .await??; + Ok(ValidationSnapshot { + log_id, + frame_no, + page_count, + }) + } +} + +impl Drop for ValidationSession { + fn drop(&mut self) { + self.controller.revoke_capability(self.capability.id()); + } +} + /// The refusal of a target name that the server already knows, in memory or in the namespace /// cache, although the metastore may not hold it yet (a create or fork in flight, or one the /// fence refused after it had published its config in memory). @@ -69,9 +172,9 @@ pub(crate) mod tests { use crate::auth::Authenticated; use crate::connection::config::DatabaseConfig; use crate::connection::program::Program; - use crate::connection::{Connection as _, RequestContext}; + use crate::connection::RequestContext; use crate::error::Error; - use crate::namespace::fence::command::FenceCommand; + use crate::namespace::fence::command::{FenceCommand, ValidationResult}; use crate::namespace::fence::controller::FenceController; use crate::namespace::fence::drain::tests::{raw, PROMPT}; use crate::namespace::fence::hooks::{HookAction, HookPoint}; @@ -82,6 +185,7 @@ pub(crate) mod tests { use crate::namespace::store::NamespaceStore; use crate::namespace::RestoreOption; use crate::query_result_builder::test::TestBuilder; + use crate::query_result_builder::QueryResultBuilder as _; use crate::rpc::replication::replication_log::ReplicationLogService; pub(crate) const OP: Uuid = Uuid::from_u128(0xa); @@ -115,6 +219,148 @@ pub(crate) mod tests { tokio::spawn(async move { store.create_target_quarantined(req, server()).await }) } + fn target_command( + command_id: u128, + expected_state: FenceState, + expected_revision: u64, + command: FenceCommand, + ) -> FenceRequest { + FenceRequest { + namespace: "tgt".into(), + operation_id: OP, + command_id: Uuid::from_u128(command_id), + expected_state, + expected_revision, + command, + } + } + + fn execute( + store: &NamespaceStore, + request: FenceRequest, + ) -> tokio::task::JoinHandle> { + let store = store.clone(); + tokio::spawn(async move { store.execute_fence_command(request, server()).await }) + } + + /// A target with a small table, sealed at TARGET_VALIDATING revision 3. + async fn validating_target() -> (TempDir, NamespaceStore, Arc) { + let dir = tempdir().unwrap(); + let store = open_store(dir.path()).await; + create(&store, create_request("tgt", 1)) + .await + .unwrap() + .unwrap(); + let mut session = store + .open_import_session("tgt".into(), OP, 1) + .await + .unwrap(); + session + .with_raw(|conn| { + conn.execute_batch("create table t (x); insert into t values (1), (2)") + }) + .await + .unwrap() + .unwrap(); + drop(session); + let seal = target_command( + 10, + FenceState::TargetQuarantined, + 1, + FenceCommand::SealTargetImport { drain_policy: None }, + ); + let commit = execute(&store, seal).await.unwrap().unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + let fence = controller(&store, "tgt").await; + assert_eq!( + (fence.gate().state(), fence.gate().revision()), + (FenceState::TargetValidating, 3) + ); + (dir, store, fence) + } + + async fn record_validation( + store: &NamespaceStore, + command_id: u128, + expected_revision: u64, + result: ValidationResult, + ) -> FenceCommit { + execute( + store, + target_command( + command_id, + FenceState::TargetValidating, + expected_revision, + FenceCommand::RecordTargetValidation { + result, + summary: format!("validation {result:?}"), + }, + ), + ) + .await + .unwrap() + .unwrap() + } + + /// A target with successful validation, published readable and write-fenced at revision 5. + async fn write_fenced_target() -> (TempDir, NamespaceStore, Arc) { + let (dir, store, fence) = validating_target().await; + record_validation(&store, 20, 3, ValidationResult::Ok).await; + let publish = target_command( + 21, + FenceState::TargetValidating, + 4, + FenceCommand::PublishTargetReadableWriteFenced, + ); + execute(&store, publish).await.unwrap().unwrap(); + assert_eq!( + (fence.gate().state(), fence.gate().revision()), + (FenceState::TargetWriteFenced, 5) + ); + (dir, store, fence) + } + + fn enable_request(command_id: u128) -> FenceRequest { + target_command( + command_id, + FenceState::TargetWriteFenced, + 5, + FenceCommand::EnableTargetWrites, + ) + } + + async fn count_rows(store: &NamespaceStore) -> i64 { + let (_, conn) = loaded(store, "tgt").await; + tokio::task::spawn_blocking(move || { + conn.with_raw(|c| c.query_row("select count(*) from t", (), |row| row.get(0))) + }) + .await + .unwrap() + .unwrap() + } + + /// Run one normal SQL program, including its legacy config checks, and require every step to + /// succeed. Raw access is intentionally not used for restart mirror assertions. + async fn program( + store: &NamespaceStore, + conn: &Arc, + sql: &'static str, + ) { + let ctx = RequestContext::new( + Authenticated::FullAccess, + "tgt".into(), + store.meta_store().clone(), + ); + let steps = conn + .execute_program(Program::seq(&[sql]), ctx, TestBuilder::default(), None) + .await + .unwrap() + .into_ret(); + for (i, step) in steps.iter().enumerate() { + assert!(step.is_ok(), "step {i} failed: {step:?}"); + } + } + fn fence_error(e: &Error) -> &FenceError { match e { Error::NamespaceFence(f) => f, @@ -491,6 +737,322 @@ pub(crate) mod tests { assert!(!fence.gate().is_creating_target()); } + /// A validation session carries the owner's current capability, admits reads through the + /// quarantine, is `query_only`, and is invalidated by the validation receipt's revision. + /// The receipt records the target snapshot observed by the server. + #[tokio::test(flavor = "multi_thread")] + async fn validation_session_is_read_only() { + let (_dir, store, fence) = validating_target().await; + let mut session = store + .open_validation_session("tgt".into(), OP, 3) + .await + .unwrap(); + assert_eq!(session.capability().purpose(), CapabilityPurpose::Validate); + let (query_only, count) = session + .with_raw(|conn| { + let query_only = + conn.query_row("PRAGMA query_only", (), |row| row.get::<_, i64>(0)); + let count = + conn.query_row("select count(*) from t", (), |row| row.get::<_, i64>(0)); + (query_only, count) + }) + .await + .unwrap(); + assert_eq!(query_only.unwrap(), 1); + assert_eq!(count.unwrap(), 2); + match session + .with_raw(|conn| conn.execute_batch("insert into t values (3)")) + .await + .unwrap() + { + Err(rusqlite::Error::SqliteFailure(e, _)) => { + assert_eq!(e.code, rusqlite::ErrorCode::ReadOnly) + } + other => panic!("query_only validation connection accepted a write: {other:?}"), + } + let e = session + .with_raw(|conn| { + conn.pragma_update(None, "query_only", false).unwrap(); + conn.execute_batch("insert into t values (3)") + }) + .await + .unwrap_err(); + assert_eq!(e.outcome(), FenceOutcome::OperationCapabilityRequired); + + let error = store + .open_validation_session("tgt".into(), OTHER_OP, 3) + .await + .unwrap_err(); + assert_eq!( + fence_error(&error).outcome(), + FenceOutcome::FenceOwnedByAnotherOperation + ); + let error = store + .open_validation_session("tgt".into(), OP, 2) + .await + .unwrap_err(); + assert_eq!( + fence_error(&error).outcome(), + FenceOutcome::FenceRevisionMismatch + ); + + let request = target_command( + 20, + FenceState::TargetValidating, + 3, + FenceCommand::RecordTargetValidation { + result: ValidationResult::Ok, + summary: "validation Ok".into(), + }, + ); + let after_commit = fence.hooks().pause_at(HookPoint::AfterMetastoreCommit); + let first = execute(&store, request.clone()); + after_commit.reached().await; + // The metastore has the receipt but the live gate still has revision 3. The concurrent + // replay must not demand another validation snapshot or capability before it waits for + // the first command to publish. + let replay = execute(&store, request); + tokio::task::yield_now().await; + after_commit.resume(); + let commit = first.await.unwrap().unwrap(); + let replay = replay.await.unwrap().unwrap(); + assert_eq!(replay.kind, FenceCommitKind::Replayed); + let validation = commit.record.as_ref().unwrap().validation.as_ref().unwrap(); + let snapshot = validation.snapshot.expect("the server records a snapshot"); + let (log_id, frame_no) = store + .with("tgt".into(), |ns| { + let logger = ns.db.logger().unwrap(); + let frame_no = *logger.new_frame_notifier.borrow(); + (logger.log_id(), frame_no) + }) + .await + .unwrap(); + assert_eq!(snapshot.log_id, log_id); + assert_eq!(snapshot.frame_no, frame_no.unwrap_or(0)); + assert!(snapshot.page_count > 0); + assert_eq!(fence.gate().revision(), 4); + + let e = session + .with_raw(|conn| conn.query_row("select 1", (), |row| row.get::<_, i64>(0))) + .await + .unwrap_err(); + assert_eq!(e.outcome(), FenceOutcome::OperationCapabilityRequired); + let mut current = store + .open_validation_session("tgt".into(), OP, 4) + .await + .unwrap(); + assert_eq!( + current + .with_raw( + |conn| conn.query_row("select count(*) from t", (), |row| row.get::<_, i64>(0)) + ) + .await + .unwrap() + .unwrap(), + 2 + ); + } + + /// Publication cannot make a target readable until the latest durable validation result of + /// the owning operation is successful. + #[tokio::test(flavor = "multi_thread")] + async fn publish_requires_validation_receipt() { + let (_dir, store, fence) = validating_target().await; + let publish = |command_id, revision| { + target_command( + command_id, + FenceState::TargetValidating, + revision, + FenceCommand::PublishTargetReadableWriteFenced, + ) + }; + let e = execute(&store, publish(20, 3)).await.unwrap().unwrap_err(); + assert_eq!( + fence_error(&e).detail(), + Some(FenceDetail::ValidationReceiptRequired) + ); + record_validation(&store, 21, 3, ValidationResult::Failed).await; + let e = execute(&store, publish(22, 4)).await.unwrap().unwrap_err(); + assert_eq!( + fence_error(&e).detail(), + Some(FenceDetail::ValidationReceiptRequired) + ); + assert_eq!( + (fence.gate().state(), fence.gate().revision()), + (FenceState::TargetValidating, 4) + ); + assert!(fence.permits(OperationClass::NormalRead).is_err()); + } + + /// A successful validation makes publication possible once; exact replay returns the stored + /// result and a new command with the same goal answers ALREADY_APPLIED without moving the + /// revision. + #[tokio::test(flavor = "multi_thread")] + async fn publish_is_idempotent() { + let (dir, store, fence) = validating_target().await; + record_validation(&store, 20, 3, ValidationResult::Ok).await; + let publish = target_command( + 21, + FenceState::TargetValidating, + 4, + FenceCommand::PublishTargetReadableWriteFenced, + ); + let commit = execute(&store, publish.clone()).await.unwrap().unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + assert_eq!( + (fence.gate().state(), fence.gate().revision()), + (FenceState::TargetWriteFenced, 5) + ); + assert!(fence.permits(OperationClass::NormalRead).is_ok()); + assert!(fence.permits(OperationClass::NormalWrite).is_err()); + assert_eq!(count_rows(&store).await, 2); + + let replay = execute(&store, publish).await.unwrap().unwrap(); + assert_eq!(replay.kind, FenceCommitKind::Replayed); + assert_eq!(replay.receipt.outcome, FenceOutcome::Applied); + let again = execute( + &store, + target_command( + 22, + FenceState::TargetValidating, + 4, + FenceCommand::PublishTargetReadableWriteFenced, + ), + ) + .await + .unwrap() + .unwrap(); + assert_eq!(again.receipt.outcome, FenceOutcome::AlreadyApplied); + assert_eq!( + (again.receipt.revision_before, again.receipt.revision_after), + (5, 5) + ); + assert_eq!(fence.gate().revision(), 5); + + store.shutdown().await.unwrap(); + let store = open_store(dir.path()).await; + let fence = controller(&store, "tgt").await; + assert_eq!(fence.gate().state(), FenceState::TargetWriteFenced); + let (_, conn) = loaded(&store, "tgt").await; + program(&store, &conn, "select count(*) from t").await; + } + + /// Enabling writes is idempotent but irreversible: it opens normal writes exactly after the + /// commit is published; replay and a new same-goal command are safe, while no target command + /// can close or abort it again. + #[tokio::test(flavor = "multi_thread")] + async fn enable_writes_idempotent_and_irreversible() { + let (_dir, store, fence) = write_fenced_target().await; + let request = enable_request(30); + let commit = execute(&store, request.clone()).await.unwrap().unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + assert_eq!( + (fence.gate().state(), fence.gate().revision()), + (FenceState::TargetWritable, 6) + ); + assert!(fence.permits(OperationClass::NormalWrite).is_ok()); + + let replay = execute(&store, request).await.unwrap().unwrap(); + assert_eq!(replay.kind, FenceCommitKind::Replayed); + assert_eq!(replay.receipt.outcome, FenceOutcome::Applied); + let again = execute(&store, enable_request(31)).await.unwrap().unwrap(); + assert_eq!(again.receipt.outcome, FenceOutcome::AlreadyApplied); + assert_eq!(fence.gate().revision(), 6); + + let abort = target_command( + 32, + FenceState::TargetWritable, + 6, + FenceCommand::AbortQuarantinedTarget, + ); + let e = execute(&store, abort).await.unwrap().unwrap_err(); + assert_eq!( + fence_error(&e).outcome(), + FenceOutcome::InvalidFenceTransition + ); + assert_eq!(fence.gate().state(), FenceState::TargetWritable); + let (_, conn) = loaded(&store, "tgt").await; + raw(&conn, "insert into t values (3)").await.unwrap(); + assert_eq!(count_rows(&store).await, 3); + } + + /// The committed writable state is installed before a restarted server can expose the + /// target, and the legacy config mirror no longer blocks its writes. + #[tokio::test(flavor = "multi_thread")] + async fn enable_writes_survives_restart() { + let (dir, store, _fence) = write_fenced_target().await; + execute(&store, enable_request(30)).await.unwrap().unwrap(); + store.shutdown().await.unwrap(); + + let store = open_store(dir.path()).await; + let fence = controller(&store, "tgt").await; + assert_eq!( + (fence.gate().state(), fence.gate().revision()), + (FenceState::TargetWritable, 6) + ); + let (_, conn) = loaded(&store, "tgt").await; + program(&store, &conn, "insert into t values (3)").await; + assert_eq!(count_rows(&store).await, 3); + } + + /// Losing the response after EnableTargetWrites commits is resolved by inspection and exact + /// replay; the detached command still publishes the writable gate before it releases the + /// transition lock. + #[tokio::test(flavor = "multi_thread")] + async fn enable_writes_response_loss_resolved() { + let (_dir, store, fence) = write_fenced_target().await; + let request = enable_request(30); + let after_commit = fence.hooks().pause_at(HookPoint::AfterMetastoreCommit); + let lost = execute(&store, request.clone()); + after_commit.reached().await; + let before_response = fence.hooks().pause_at(HookPoint::BeforeResponse); + lost.abort(); + assert!(lost.await.unwrap_err().is_cancelled()); + after_commit.resume(); + before_response.reached().await; + + assert_eq!(fence.gate().state(), FenceState::TargetWritable); + let inspected = store + .meta_store() + .inspect_fence("tgt".into()) + .await + .unwrap(); + assert_eq!(inspected.fence.state(), FenceState::TargetWritable); + assert!(inspected.receipts.iter().any(|stored| { + matches!( + &stored.receipt, + Ok(receipt) + if receipt.operation_id == OP + && receipt.command_id == Uuid::from_u128(30) + && receipt.outcome == FenceOutcome::Applied + ) + })); + before_response.resume(); + + let replay = execute(&store, request).await.unwrap().unwrap(); + assert_eq!(replay.kind, FenceCommitKind::Replayed); + assert_eq!(replay.receipt.outcome, FenceOutcome::Applied); + assert_eq!(fence.gate().state(), FenceState::TargetWritable); + } + + /// A read transaction opened before target write authority is published cannot upgrade to + /// a write afterwards; rolling it back and starting a fresh program succeeds. + #[tokio::test(flavor = "multi_thread")] + async fn stale_generation_cannot_write_after_enable_writes() { + let (_dir, store, fence) = write_fenced_target().await; + let (_, conn) = loaded(&store, "tgt").await; + raw(&conn, "begin; select count(*) from t").await.unwrap(); + execute(&store, enable_request(30)).await.unwrap().unwrap(); + assert_eq!(fence.gate().state(), FenceState::TargetWritable); + + crate::namespace::fence::drain::tests::assert_fenced( + raw(&conn, "insert into t values (3)").await, + ); + raw(&conn, "commit").await.unwrap(); + raw(&conn, "insert into t values (4)").await.unwrap(); + assert_eq!(count_rows(&store).await, 3); + } + /// `AbortQuarantinedTarget` finishes the operation and keeps every normal class denied; /// the target is not deletable by the generic lifecycle either. #[tokio::test(flavor = "multi_thread")] diff --git a/libsql-server/src/namespace/store.rs b/libsql-server/src/namespace/store.rs index 43414bc738..889370ad79 100644 --- a/libsql-server/src/namespace/store.rs +++ b/libsql-server/src/namespace/store.rs @@ -29,9 +29,9 @@ use super::fence::import::{self, ImportSession}; use super::fence::outcome::FenceOutcome; use super::fence::record::ServerIdentity; use super::fence::registry::FenceRegistry; -use super::fence::state::Role; +use super::fence::state::{FenceState, Role}; use super::fence::store::StoredFence; -use super::fence::target::{self, CreateTargetRequest}; +use super::fence::target::{self, CreateTargetRequest, ValidationSession}; use super::meta_store::{FenceCommit, FenceContext, MetaStore, MetaStoreHandle}; use super::schema_lock::SchemaLocksRegistry; use super::{Namespace, ResetCb, ResetOp, ResolveNamespacePathFn, RestoreOption}; @@ -573,13 +573,58 @@ impl NamespaceStore { } _ => self.inner.fences.controller(&request.namespace), }; - controller - .execute( - &self.inner.metadata, - request, - FenceContext::now(server, None), - ) - .await + let mut ctx = FenceContext::now(server, None); + // A new validation receipt records what this server observes of the sealed target. Do + // this only when the live gate exactly matches the request: an exact replay after the + // revision or state has advanced must reach the metastore's replay check without first + // trying to issue a now-invalid validation capability. + if matches!( + &request.command, + FenceCommand::RecordTargetValidation { .. } + ) { + let needs_snapshot = { + let gate = controller.gate(); + gate.state() == FenceState::TargetValidating + && gate.operation_id() == Some(request.operation_id) + && gate.revision() == request.expected_revision + }; + if needs_snapshot && !self.validation_command_recorded(&request).await? { + let snapshot = async { + let mut session = self + .open_validation_session( + request.namespace.clone(), + request.operation_id, + request.expected_revision, + ) + .await?; + session.snapshot().await + } + .await; + match snapshot { + Ok(snapshot) => ctx.validation_snapshot = Some(snapshot), + // A concurrent copy of this command can commit between the gate check and + // the capability call. Once its receipt exists, let execute take the + // transition lock and perform the authoritative replay/fingerprint check. + Err(_) if self.validation_command_recorded(&request).await? => {} + Err(e) => return Err(e), + } + } + } + controller.execute(&self.inner.metadata, request, ctx).await + } + + async fn validation_command_recorded(&self, request: &FenceRequest) -> crate::Result { + let operation_id = request.operation_id.to_string(); + let command_id = request.command_id.to_string(); + let inspected = self + .inner + .metadata + .inspect_fence(request.namespace.clone()) + .await?; + Ok(inspected + .receipts + .iter() + .any(|stored| stored.operation_id == operation_id && stored.command_id == command_id)) } /// `CreateTargetQuarantined`, atomic with namespace creation (`docs/NAMESPACE_FENCE.md` @@ -727,6 +772,36 @@ impl NamespaceStore { } } + /// Issue a read-only validation capability to `operation_id` and open a `query_only` + /// connection under it (`docs/NAMESPACE_FENCE.md` sections 10.3 and 11). Valid only while + /// the target is `TARGET_VALIDATING` or `TARGET_WRITE_FENCED`, owned by the operation at + /// `expected_revision`; the target is loaded if it is not. + pub async fn open_validation_session( + &self, + namespace: NamespaceName, + operation_id: uuid::Uuid, + expected_revision: u64, + ) -> crate::Result { + let (controller, maker) = self.primary_maker(&namespace).await?; + let capability = controller.issue_capability( + CapabilityPurpose::Validate, + operation_id, + expected_revision, + )?; + let conn = match maker + .inner() + .make_capability_connection(capability.clone()) + .await + { + Ok(conn) => conn, + Err(e) => { + controller.revoke_capability(capability.id()); + return Err(e); + } + }; + ValidationSession::new(capability, controller, conn).await + } + /// The fence controller and connection maker of the primary `namespace`, loading it. async fn primary_maker( &self, From ba230cc78777a55d9a9013eea7616e27d847e9d8 Mon Sep 17 00:00:00 2001 From: River Date: Tue, 29 Sep 2026 19:48:01 +0000 Subject: [PATCH 17/33] libsql-server: make the fenced snapshot stream test independent of compaction timing `snapshot_stream_ends_typed` polled `get_snapshot_file(1)` and unwrapped its result while the compactor was still writing the first snapshot and the snapshot merger could be replacing merged files. The lookup lists the snapshot directory and then opens the chosen file, so under load it could fail with `NotFound` (directory not created yet, or a file removed by a merge between the listing and the open) and the test panicked. Open the stream through the `snapshot` RPC itself in a bounded loop that treats only "snapshot not found" and the vanished-file error as "not yet", and asserts that no read lease is held after a failed attempt. An opened stream holds its file open, so a later merge cannot affect it. Co-authored-by: Tomasz Szymczyszyn --- libsql-server/src/namespace/fence/stream.rs | 46 +++++++++++++++++---- 1 file changed, 37 insertions(+), 9 deletions(-) diff --git a/libsql-server/src/namespace/fence/stream.rs b/libsql-server/src/namespace/fence/stream.rs index e57ba2f181..2a043d89fa 100644 --- a/libsql-server/src/namespace/fence/stream.rs +++ b/libsql-server/src/namespace/fence/stream.rs @@ -221,6 +221,7 @@ pub fn fence_status(error: FenceError) -> tonic::Status { #[cfg(test)] mod tests { use bytes::Bytes; + use futures::stream::BoxStream; use futures::{Stream, StreamExt}; use libsql_replication::rpc::replication::replication_log_server::ReplicationLog; use libsql_replication::rpc::replication::{ @@ -513,19 +514,11 @@ mod tests { .await .unwrap() .unwrap(); - tokio::time::timeout(PROMPT, async { - // The snapshot is written by the compactor's own thread. - while s.logger.get_snapshot_file(1).await.unwrap().is_none() { - tokio::time::sleep(std::time::Duration::from_millis(10)).await; - } - }) - .await - .expect("the snapshot was never written"); let acquired = s.execute(s.acquire(OP, 1, LONG)).await.unwrap(); assert_eq!(fence_outcome(&acquired), FenceOutcome::Applied); let r = Replication::new(&s).await; - let stream = r.service.snapshot(r.offset(1)).await.unwrap().into_inner(); + let stream = open_snapshot_stream(&s, &r).await; assert_eq!(s.fence.read_lease_counts().replication, 1); let fenced = tokio::time::timeout(PROMPT, s.execute(read_fence(&s, 2, LONG))) @@ -540,6 +533,41 @@ mod tests { assert!(next_frame(&mut stream).await.is_none()); } + /// Opens a `snapshot` stream from frame 1 once the compactor has written a snapshot. + /// + /// The snapshot is written, and possibly merged with the previous one, by the compactor's + /// own tasks, concurrently with this call. The server looks a snapshot up by listing the + /// snapshot directory and then opening the file it chose, so while the directory does not + /// exist yet, or while a merge removes the files it replaced, the call can fail with + /// "snapshot not found" or with the `NotFound` of the vanished file. Those two failures, + /// and only those, mean "not yet": any other status (a fence refusal in particular) fails + /// the test. A stream that was opened holds its file open, so a later merge cannot affect it. + async fn open_snapshot_stream( + s: &Source, + r: &Replication, + ) -> BoxStream<'static, Result> { + tokio::time::timeout(PROMPT, async { + loop { + match r.service.snapshot(r.offset(1)).await { + Ok(stream) => break stream.into_inner(), + Err(status) => { + let not_yet = match status.code() { + tonic::Code::Unavailable => status.message() == "snapshot not found", + tonic::Code::Internal => status.message().contains("os error 2"), + _ => false, + }; + assert!(not_yet, "unexpected snapshot status: {status:?}"); + assert_eq!(s.fence.read_lease_counts().total(), 0); + // A polling interval, not evidence: the loop ends on the condition. + tokio::time::sleep(std::time::Duration::from_millis(10)).await; + } + } + } + }) + .await + .expect("the snapshot was never written") + } + /// While reads are fenced every replication call is refused at its start with the typed /// status (never `UNAVAILABLE`), and no lease is taken. #[tokio::test] From bede5dd6d03198677d9a66553195529d45b64d21 Mon Sep 17 00:00:00 2001 From: River Date: Wed, 30 Sep 2026 08:21:03 +0000 Subject: [PATCH 18/33] libsql-server: namespace fence admin API and capability discovery Serve the fence contract of docs/NAMESPACE_FENCE.md section 4 on the admin listener (new http/admin/fence.rs): - GET /v1/fence/capabilities, always served: protocol version, whether fences are enabled, served commands, states, proxy stable_code support, server build and instance id, and the number of active fences counted from the fence registry. - GET /v1/namespaces/:ns/fence (InspectFence): the durable record, the owning operation's receipts (or ?receipts=all), live admission, drain counters and the live replication log id; read-only, and also served while fence tables exist with the flag off. - One POST route per source and target command, all through NamespaceStore::execute_fence_command, and validation-query, which runs one read-only program under the operation's validation capability with a 10 000-row bound. Command routes answer 404 unless --enable-namespace-fence is on, refuse to run without an admin auth key (admin_auth_required) or on a replica (not_primary), parse bodies strictly (invalid_argument), and refuse restore options on target creation (restore_not_allowed). Responses carry the outcome, replay flag, fence view, receipt and drain counters with the outcome's admin status. Co-authored-by: Tomasz Szymczyszyn --- docs/NAMESPACE_FENCE.md | 19 +- libsql-server/src/http/admin/fence.rs | 1024 +++++++++++++++++ libsql-server/src/http/admin/mod.rs | 6 + .../src/namespace/fence/controller.rs | 31 + libsql-server/src/namespace/fence/mod.rs | 17 + libsql-server/src/namespace/fence/registry.rs | 16 + libsql-server/src/namespace/store.rs | 34 +- libsql-server/tests/common/http.rs | 14 + libsql-server/tests/fence/admin.rs | 731 ++++++++++++ libsql-server/tests/fence/mod.rs | 188 +++ libsql-server/tests/tests.rs | 1 + 11 files changed, 2075 insertions(+), 6 deletions(-) create mode 100644 libsql-server/src/http/admin/fence.rs create mode 100644 libsql-server/tests/fence/admin.rs create mode 100644 libsql-server/tests/fence/mod.rs diff --git a/docs/NAMESPACE_FENCE.md b/docs/NAMESPACE_FENCE.md index 718a67fe5f..74a4fae029 100644 --- a/docs/NAMESPACE_FENCE.md +++ b/docs/NAMESPACE_FENCE.md @@ -257,6 +257,18 @@ POST /v1/namespaces/:namespace/fence/adopt Import itself is not an admin route in this series. It is an internal API (section 11) that the bulk import work exposes over its own route. +### 4.5 Implementation notes + +The routes are in `libsql-server/src/http/admin/fence.rs`, behind the admin listener's authentication middleware. + +- **Availability.** `GET /v1/fence/capabilities` is always served. Command routes and `validation-query` answer `404` unless `--enable-namespace-fence` is on. `InspectFence` is also served while the metastore holds fence tables with the flag off (fences are enforced either way, section 13.1), and answers `404` on a server that never used fences. +- **Order of checks.** Admin authentication (middleware, `401`), then the `404` above, then `not_primary` on a replica-kind server, then `admin_auth_required` for command routes and `validation-query` when no admin auth key is configured (`InspectFence` is read-only and does not need one), then the request body. Every command runs through `NamespaceStore::execute_fence_command` (which routes `CreateTargetQuarantined` to the atomic creation and adds the server's validation snapshot to `RecordTargetValidation`), so replay handling precedes every other check as section 5.3 requires. +- **Request bodies** are strict: an unparsable body, a malformed id, an unknown `expected_state`, a missing required field or an unknown field is `FENCE_PRECONDITION_FAILED` with detail `invalid_argument`. `drain_policy.on_deadline` defaults to `fail`. `create-quarantined` refuses `dump_url`, `restore`, `restore_option`, `timestamp` and `from_backup` with `restore_not_allowed`, and `shared_schema` / `shared_schema_name` with `shared_schema_unsupported`; `max_db_size` takes the same byte-size values as `/v1/namespaces/:namespace/create`, and a `jwt_key` is parsed before anything is written. +- **Fence view.** Beyond the fields of section 4.3 it reports `incarnation.current_log_id` (the replication log id of the namespace as loaded now, `null` when it is not loaded; a caller can take `expected_namespace_identity.log_id` from it), `admission.indeterminate`, `drain_started_at`, the owning operation's latest `validation` (with the server snapshot), `last_command_id`, `written_by`, `adoptions`, and for `UNKNOWN_UNAVAILABLE` the `detail`, `reason` and the record the marker holds (`marker_record`). A namespace being created as a target reports `TARGET_QUARANTINED` with the durable revision it has so far. `provenance.metastore_restored_from_backup` is `false` until restore provenance is tracked. Error responses carry the live view from the namespace's controller when it has one, otherwise what the metastore holds, or `null` when the namespace name itself is invalid. +- **`InspectFence`** reads the metastore and the live controller; it neither loads the namespace nor creates a controller, so `drain` counters are zero for a namespace that is not loaded. Without `?receipts=all` only the owning operation's receipts are listed; `?receipts` with any other value is `invalid_argument`. +- **`validation-query`** opens a `ValidationSession` (section 10.3) for the request and closes it afterwards. `stmts` follow the `/v1/execute` statement shape (`sql`, positional `args` or `named_args`, `want_rows`; `sql_id` is not supported). The response is `{"results": [...], "fence": ..., "drain": ...}`, one result per statement with `cols` and `rows` in the Hrana value encoding. A request whose statements return more than 10 000 rows in total is refused with `invalid_argument` rather than truncated, so a validation never looks at part of a result. A statement that tries to write is `OPERATION_CAPABILITY_REQUIRED` (`403`); other SQLite errors are `invalid_argument`. `expected_state`, when given, must match the live state (`FENCE_REVISION_MISMATCH` otherwise). +- **Capability discovery** lists in `commands` the commands this server serves (`AdoptFence` is added with its route, section 12), every state in `states`, and counts `active_fences` from the fence registry, which is seeded from the metastore at startup: records in any state but `RELEASED` or `TARGET_WRITABLE`, unavailable names, targets being created and indeterminate commits. `proxy_stable_code` reports whether the proxy's `stable_code` (section 6.1) is supported. `metastore` reports `restored_from_backup: false` until restore provenance is tracked. + ## 5. Durable state (Design, with contract points marked) ### 5.1 Metastore schema @@ -716,6 +728,7 @@ How each path that can reach namespace data or lifecycle is covered. File refere | `namespace/store.rs` `with`, `load_namespace`, `make_namespace`, eviction | Registry check before setup; controller passed into setup; eviction keeps the registry entry. | | `store.rs` `create`, `destroy`, `reset`, `fork`, `checkpoint`, restore options | Create refuses names with a record; `CreateTargetQuarantined` is the atomic quarantined create; destroy, reset, fork (either side) and any restore are denied while lifecycle is denied; checkpoint uses a non-creating lookup and skips vacuum. | | `http/admin/mod.rs` config, create, fork, delete, checkpoint, stats | Config POST, create, fork, delete follow the lifecycle column; config GET, stats and checkpoint are allowed. Fence routes live in `http/admin/fence.rs`. | +| `http/admin/fence.rs` | Fence routes (section 4.5): capability discovery, `InspectFence`, one route per command through `NamespaceStore::execute_fence_command`, and `validation-query` through `open_validation_session`; admin auth key and primary required for every route that changes state or uses a capability. | | `http/user/dump.rs`, `connection/dump/exporter.rs` | Gate check (`Stream`) before connection creation (typed, no panic on create error); `Dump` lease held by the export; cancel checked before every row and by the pipe's pending write; the body stream ends with the fence error (aborted response) on cancellation. | | `rpc/replication/replication_log.rs` `hello`, `log_entries`, `batch_log_entries`, `snapshot` | Denied at request start (`FAILED_PRECONDITION` + `x-libsql-fence-code`, counted, rate-limited log); `FencedStream` replication leases for both streams and the batch; typed terminal status when the gate closes or at the deadline, lease released without the peer; `ReplicatedFence` in `hello`'s config (planned, section 6.2). | | `admin_shell.rs` | Writes denied at the WAL (no capability); reads checked against the gate, and a read lease held, per query (cancelled through the connection's interrupt handle). | @@ -758,7 +771,7 @@ Planned test names; the table is updated as tests land. | # | Requirement | Planned tests | |---|---|---| -| 1 | Concurrent acquisition by two operations: one owner, typed conflict for the loser | landed: `namespace::fence::drain::tests::acquire_race_single_owner` (the first acquisition is parked after closing admission while the second waits on the transition lock); planned: `tests::fence::admin::concurrent_acquire_one_owner` | +| 1 | Concurrent acquisition by two operations: one owner, typed conflict for the loser | landed: `namespace::fence::drain::tests::acquire_race_single_owner` (the first acquisition is parked after closing admission while the second waits on the transition lock); over the admin API: `tests::fence::admin::concurrent_acquire_one_owner` (two operations acquire at once over HTTP: one `200 APPLIED`, the other `409 FENCE_OWNED_BY_ANOTHER_OPERATION` with the winner in its fence view); admin walks: `tests::fence::admin::{source_walk_over_http, target_walk_over_http, inspect_reports_drain_counters, mutating_routes_require_admin_key}` | | 2 | Active writer commits or is rolled back before freeze acknowledgement; nothing commits after | landed: `namespace::fence::drain::tests::{active_writer_commits_before_ack, forced_rollback_before_ack, no_commit_after_ack}` (the boundary equals the last committed replication frame and no frame follows it; autocommit, `BEGIN IMMEDIATE`, DDL and a pre-fence read transaction upgrading are refused), `installing_gate_closes_writes_before_persisting`, `refused_acquire_reopens_writes`, `release_reopens_with_new_generation` | | 3 | Autocommit, explicit transactions, queued writers, batches, DDL, schema jobs, old WebSockets, read-to-write upgrades cannot bypass | landed: `connection::connection_manager::fence_tests::{fence_rejects_read_to_write_upgrade, fence_rejects_ddl_and_pragma, fence_rejects_raw_with_raw_write}` (autocommit, explicit transactions, DDL, header-writing pragma, `BEGIN IMMEDIATE`, `VACUUM`, `with_raw` users); `connection::connection_manager::fence_tests::fence_rejects_queued_writer` (a writer parked in the queue behind an open transaction leaves it with `MIGRATION_WRITE_FENCED` when the fence changes, and the holder keeps the slot); maintenance and vacuum under a fence: `queued_checkpoint_survives_fence_wake`, `checkpoint_allowed_while_fenced`, `vacuum_skipped_while_fenced`; drain primitives: `abort_active_tolerates_closed_connection`, `release_notifies_drain_waiters`, `fence::controller::tests::write_queues_are_woken_on_every_generation_change`; planned: `tests::fence::protocol::old_ws_session_cannot_write`, `batch_denied_mid_batch`; `fence::tests::acquire_rejects_shared_schema` | | 4 | Program that captured config before the fence is rejected at the WAL | landed: `connection::connection_manager::fence_tests::wal_gate_rejects_program_admitted_before_fence` (a SQL function parks the program between admission and its write while the fence is acquired and released) | @@ -778,7 +791,7 @@ Planned test names; the table is updated as tests land. | 18 | Codes through HTTP, Hrana, RPC, dump, replication, replica write proxy; distinguishable from auth/timeout/not-found; old peers compatible; no retry loops | `tests::fence::protocol::{http_codes, hrana_http_codes, hrana_ws_codes, rpc_codes, dump_codes, replication_codes, replica_proxy_preserves_code, auth_and_not_found_distinct, denial_not_retried}`; `libsql-replication` `proxy_error_stable_code_is_additive` | | 19 | Corrupt or unknown durable fence state fails closed | `fence::store::tests::corrupt_payload_fails_closed`, `unknown_format_version_fails_closed` | | 20 | Metrics and audit logs | `tests::fence::observability::metrics_and_labels`; `fence::audit::tests::audit_event_fields` | -| 21 | Capability discovery and mixed-version protection | `tests::fence::admin::capabilities`; `fence::store::tests::legacy_mirror_and_fk_guard` (bounded, see section 18) | +| 21 | Capability discovery and mixed-version protection | capability discovery landed: `tests::fence::admin::{capabilities, capabilities_when_disabled}`; planned: `fence::store::tests::legacy_mirror_and_fk_guard` (bounded, see section 18) | | 22 | Adoption is two-person/audited, keeps admission closed, cannot reverse publication | `fence::tests::adopt_requires_key_and_two_approvers`, `adopt_keeps_gates_closed`, `adopt_cannot_touch_writable` | | — | Import API usable by bulk import | landed: `namespace::fence::import::tests::import_session_loads_dump_into_quarantined_target` (a dump exported by the server from a source with tables, keys, a foreign key, an index, an autoincrement table, a trigger, a view and an FTS5 table loads through `ImportSession::load_dump`; after the seal the target's schema, rows, view and full-text results equal the source's) | @@ -813,7 +826,7 @@ libsql-server/src/namespace/fence/ drain.rs FenceController::execute, source write drain read.rs source read fence and its drain stream.rs stream leases: FencedStream for replication, dump cancel - target.rs quarantined target creation, ValidationSession (planned) + target.rs quarantined target creation, ValidationSession capability.rs MigrationCapability, CapabilityPurpose, ImportWriter import.rs ImportSession, SealTargetImport drain audit.rs audit events and metrics diff --git a/libsql-server/src/http/admin/fence.rs b/libsql-server/src/http/admin/fence.rs new file mode 100644 index 0000000000..ca0955706a --- /dev/null +++ b/libsql-server/src/http/admin/fence.rs @@ -0,0 +1,1024 @@ +//! The namespace fence admin API (`docs/NAMESPACE_FENCE.md` section 4). +//! +//! `GET /v1/fence/capabilities` is always served. The other routes answer `404` unless the +//! server was started with `--enable-namespace-fence`, except `InspectFence`, which is also +//! served while fence state exists in the metastore with the flag off (fences are enforced +//! either way, section 13.1). Every mutating route runs through +//! [`NamespaceStore::execute_fence_command`], which owns replay, the drains and target creation. + +use std::sync::Arc; + +use axum::extract::{Path, Query, State}; +use axum::response::{IntoResponse, Response}; +use axum::routing::{get, post}; +use axum::Json; +use bytes::Bytes; +use hyper::StatusCode; +use serde::de::DeserializeOwned; +use serde::Deserialize; +use serde_json::{json, Map, Value}; +use uuid::Uuid; + +use crate::auth::parse_jwt_keys; +use crate::error::Error; +use crate::hrana::proto; +use crate::namespace::fence::command::{ + CommandKind, DrainPolicy, FenceCommand, FenceRequest, OnDeadline, TargetConfig, + ValidationResult, +}; +use crate::namespace::fence::controller::{DrainCounters, FenceController}; +use crate::namespace::fence::outcome::{FenceDetail, FenceError, FenceOutcome}; +use crate::namespace::fence::record::{CommandReceipt, NamespaceFenceRecord, ServerIdentity}; +use crate::namespace::fence::state::FenceState; +use crate::namespace::fence::store::{StoredFence, StoredReceipt}; +use crate::namespace::fence::{server_identity, FENCE_PROTOCOL_VERSION, PROXY_STABLE_CODE}; +use crate::namespace::meta_store::FenceCommit; +use crate::namespace::NamespaceName; +use crate::net::Connector; + +use super::AppState; + +/// The most rows one `validation-query` request returns, over all of its statements. A query +/// that would return more is refused rather than truncated, so a validation never silently +/// looks at part of a result. +pub const MAX_VALIDATION_QUERY_ROWS: usize = 10_000; + +/// The commands this server serves over the admin API, reported by capability discovery. +/// Adoption is served once its route exists. +const SERVED_COMMANDS: [&str; 11] = [ + "InspectFence", + CommandKind::AcquireSourceWriteFence.as_str(), + CommandKind::SetSourceReadFence.as_str(), + CommandKind::ClearSourceReadFence.as_str(), + CommandKind::ReleaseSourceWriteFence.as_str(), + CommandKind::CreateTargetQuarantined.as_str(), + CommandKind::SealTargetImport.as_str(), + CommandKind::RecordTargetValidation.as_str(), + CommandKind::PublishTargetReadableWriteFenced.as_str(), + CommandKind::EnableTargetWrites.as_str(), + CommandKind::AbortQuarantinedTarget.as_str(), +]; + +/// The fence routes, added to the admin router. +pub(super) fn routes() -> axum::Router>> { + let command = |kind: CommandKind| { + post( + move |State(state): State>>, + Path(namespace): Path, + body: Bytes| async move { + handle_command(state, namespace, kind, body).await + }, + ) + }; + axum::Router::new() + .route("/v1/fence/capabilities", get(handle_capabilities)) + .route("/v1/namespaces/:namespace/fence", get(handle_inspect)) + .route( + "/v1/namespaces/:namespace/fence/source/acquire-write-fence", + command(CommandKind::AcquireSourceWriteFence), + ) + .route( + "/v1/namespaces/:namespace/fence/source/set-read-fence", + command(CommandKind::SetSourceReadFence), + ) + .route( + "/v1/namespaces/:namespace/fence/source/clear-read-fence", + command(CommandKind::ClearSourceReadFence), + ) + .route( + "/v1/namespaces/:namespace/fence/source/release-write-fence", + command(CommandKind::ReleaseSourceWriteFence), + ) + .route( + "/v1/namespaces/:namespace/fence/target/create-quarantined", + command(CommandKind::CreateTargetQuarantined), + ) + .route( + "/v1/namespaces/:namespace/fence/target/seal-import", + command(CommandKind::SealTargetImport), + ) + .route( + "/v1/namespaces/:namespace/fence/target/validation-receipt", + command(CommandKind::RecordTargetValidation), + ) + .route( + "/v1/namespaces/:namespace/fence/target/publish-readable", + command(CommandKind::PublishTargetReadableWriteFenced), + ) + .route( + "/v1/namespaces/:namespace/fence/target/enable-writes", + command(CommandKind::EnableTargetWrites), + ) + .route( + "/v1/namespaces/:namespace/fence/target/abort", + command(CommandKind::AbortQuarantinedTarget), + ) + .route( + "/v1/namespaces/:namespace/fence/target/validation-query", + post(handle_validation_query), + ) +} + +// --------------------------------------------------------------------------------------------- +// Handlers + +async fn handle_capabilities(State(state): State>>) -> Json { + let meta = state.namespaces.meta_store(); + Json(json!({ + "fence_protocol_version": FENCE_PROTOCOL_VERSION, + "enabled": meta.fence_enabled(), + "commands": SERVED_COMMANDS, + "states": FenceState::ALL.iter().map(|s| s.as_str()).collect::>(), + "proxy_stable_code": PROXY_STABLE_CODE, + "server": server_json(&server_identity()), + "active_fences": state.namespaces.active_fences(), + // Metastore restore provenance is not tracked yet. + "metastore": { "restored_from_backup": false, "restored_generation": null }, + })) +} + +#[derive(Debug, Default, Deserialize)] +struct InspectQuery { + #[serde(default)] + receipts: Option, +} + +async fn handle_inspect( + State(state): State>>, + Path(namespace): Path, + Query(query): Query, +) -> Response { + let meta = state.namespaces.meta_store(); + if !meta.fence_enabled() && !meta.fence_enforced() { + return StatusCode::NOT_FOUND.into_response(); + } + let namespace = match NamespaceName::from_string(namespace) { + Ok(ns) => ns, + Err(e) => return invalid_argument(e.to_string()).into_response(), + }; + if !state.namespaces.is_primary() { + return ErrorReply::new(not_primary()).into_response(); + } + let all = match query.receipts.as_deref() { + None => false, + Some("all") => true, + Some(other) => { + return invalid_argument(format!("unknown `receipts` value `{other}`")).into_response() + } + }; + let (inspection, controller) = match state.namespaces.inspect_fence(&namespace).await { + Ok(found) => found, + Err(e) => return fence_or_error(&state, &namespace, e).await, + }; + if matches!( + inspection.fence, + StoredFence::None { + namespace_exists: false + } + ) && controller.as_ref().map_or(true, |c| { + let gate = c.gate(); + matches!( + gate.fence, + StoredFence::None { + namespace_exists: false + } + ) && gate.creating_target.is_none() + }) { + return ( + StatusCode::NOT_FOUND, + Json(json!({ "error": format!("namespace `{namespace}` does not exist") })), + ) + .into_response(); + } + + let owner = inspection + .fence + .record() + .map(|r| r.operation_id.to_string()); + let receipts: Vec = inspection + .receipts + .iter() + .filter(|r| all || owner.as_deref() == Some(r.operation_id.as_str())) + .map(stored_receipt_json) + .collect(); + let body = json!({ + "outcome": FenceOutcome::Applied.as_str(), + "replayed": false, + "fence": fence_json(&namespace, &inspection.fence, controller.as_deref()), + "receipts": receipts, + "drain": drain_json(controller.as_deref()), + }); + (StatusCode::OK, Json(body)).into_response() +} + +async fn handle_command( + state: Arc>, + namespace: String, + kind: CommandKind, + body: Bytes, +) -> Response { + if !state.namespaces.meta_store().fence_enabled() { + return StatusCode::NOT_FOUND.into_response(); + } + let namespace = match NamespaceName::from_string(namespace) { + Ok(ns) => ns, + Err(e) => return invalid_argument(e.to_string()).into_response(), + }; + if let Err(e) = mutating_preconditions(&state) { + return error_reply(&state, &namespace, e).await; + } + let request = match parse_command(namespace.clone(), kind, &body) { + Ok(request) => request, + Err(e) => return error_reply(&state, &namespace, e).await, + }; + match state + .namespaces + .execute_fence_command(request, server_identity()) + .await + { + Ok(commit) => success_reply(&state, &namespace, commit), + Err(e) => fence_or_error(&state, &namespace, e).await, + } +} + +async fn handle_validation_query( + State(state): State>>, + Path(namespace): Path, + body: Bytes, +) -> Response { + if !state.namespaces.meta_store().fence_enabled() { + return StatusCode::NOT_FOUND.into_response(); + } + let namespace = match NamespaceName::from_string(namespace) { + Ok(ns) => ns, + Err(e) => return invalid_argument(e.to_string()).into_response(), + }; + if let Err(e) = mutating_preconditions(&state) { + return error_reply(&state, &namespace, e).await; + } + let query = match parse_validation_query(&body) { + Ok(query) => query, + Err(e) => return error_reply(&state, &namespace, e).await, + }; + if let Some(expected) = query.expected_state { + let current = state + .namespaces + .existing_fence_controller(&namespace) + .map(|c| c.gate().state()); + if let Some(current) = current.filter(|s| *s != expected) { + let e = FenceError::new( + FenceOutcome::FenceRevisionMismatch, + format!("the namespace is {current}, not {expected}"), + ); + return error_reply(&state, &namespace, e).await; + } + } + let mut session = match state + .namespaces + .open_validation_session( + namespace.clone(), + query.operation_id, + query.expected_revision, + ) + .await + { + Ok(session) => session, + Err(e) => return fence_or_error(&state, &namespace, e).await, + }; + + let mut results = Vec::with_capacity(query.stmts.len()); + let mut remaining = MAX_VALIDATION_QUERY_ROWS; + for (index, stmt) in query.stmts.into_iter().enumerate() { + let budget = remaining; + let ran = session + .with_raw(move |conn| run_validation_stmt(conn, &stmt, budget)) + .await; + match ran { + Ok(Ok(result)) => { + remaining -= result.rows.len(); + results.push(result); + } + Ok(Err(e)) => { + return error_reply(&state, &namespace, e.into_fence_error(index)).await; + } + Err(e) => return error_reply(&state, &namespace, e).await, + } + } + drop(session); + + let controller = state.namespaces.existing_fence_controller(&namespace); + let fence = controller + .as_ref() + .map(|c| fence_json(&namespace, &c.gate().fence, Some(c))) + .unwrap_or(Value::Null); + let body = json!({ + "results": results, + "fence": fence, + "drain": drain_json(controller.as_deref()), + }); + (StatusCode::OK, Json(body)).into_response() +} + +// --------------------------------------------------------------------------------------------- +// Preconditions and replies + +/// Section 4.1, for every route that changes fence state or works under a capability. The admin +/// authentication itself is the admin router's middleware and has already run. +fn mutating_preconditions(state: &AppState) -> Result<(), FenceError> { + if !state.namespaces.is_primary() { + return Err(not_primary()); + } + if !state.admin_auth_configured { + return Err(FenceError::new( + FenceOutcome::FencePreconditionFailed, + "namespace fence commands need an admin auth key: without one the admin API is \ + unauthenticated", + ) + .with_detail(FenceDetail::AdminAuthRequired)); + } + Ok(()) +} + +fn not_primary() -> FenceError { + FenceError::new( + FenceOutcome::FencePreconditionFailed, + "namespace fences live on the primary; this server is a replica", + ) + .with_detail(FenceDetail::NotPrimary) +} + +fn invalid_argument(message: impl Into) -> ErrorReply { + ErrorReply::new( + FenceError::new(FenceOutcome::FencePreconditionFailed, message) + .with_detail(FenceDetail::InvalidArgument), + ) +} + +/// An error reply without a fence view (the namespace name itself could not be used). +struct ErrorReply { + error: FenceError, + fence: Value, + drain: Value, +} + +impl ErrorReply { + fn new(error: FenceError) -> Self { + Self { + error, + fence: Value::Null, + drain: Value::Null, + } + } +} + +impl IntoResponse for ErrorReply { + fn into_response(self) -> Response { + let outcome = self.error.outcome(); + let mut body = json!({ + "outcome": outcome.as_str(), + "replayed": false, + "error": self.error.message(), + "fence": self.fence, + "drain": self.drain, + }); + if let Some(detail) = self.error.detail() { + body["detail"] = json!(detail.as_str()); + } + (outcome.admin_http_status(), Json(body)).into_response() + } +} + +/// An error reply carrying the namespace's current fence view: the live gate if the namespace +/// has a controller, otherwise what the metastore holds. +async fn error_reply( + state: &AppState, + namespace: &NamespaceName, + error: FenceError, +) -> Response { + let mut reply = ErrorReply::new(error); + match state.namespaces.existing_fence_controller(namespace) { + Some(controller) => { + reply.fence = fence_json(namespace, &controller.gate().fence, Some(&controller)); + reply.drain = drain_json(Some(&controller)); + } + None => { + if let Ok((inspection, _)) = state.namespaces.inspect_fence(namespace).await { + reply.fence = fence_json(namespace, &inspection.fence, None); + } + } + } + reply.into_response() +} + +/// A fence refusal in the fence response shape; any other error as the admin API reports it. +async fn fence_or_error(state: &AppState, namespace: &NamespaceName, e: Error) -> Response { + match e { + Error::NamespaceFence(e) => error_reply(state, namespace, e).await, + e => e.into_response(), + } +} + +fn success_reply( + state: &AppState, + namespace: &NamespaceName, + commit: FenceCommit, +) -> Response { + let controller = state.namespaces.existing_fence_controller(namespace); + let fence = match (&commit.record, &controller) { + (Some(record), _) => fence_json( + namespace, + &StoredFence::Record(record.clone()), + controller.as_deref(), + ), + (None, Some(c)) => fence_json(namespace, &c.gate().fence, Some(c)), + (None, None) => Value::Null, + }; + let outcome = commit.receipt.outcome; + let body = json!({ + "outcome": outcome.as_str(), + "replayed": commit.kind == crate::namespace::meta_store::FenceCommitKind::Replayed, + "fence": fence, + "receipt": receipt_json(&commit.receipt), + "drain": drain_json(controller.as_deref()), + }); + (outcome.admin_http_status(), Json(body)).into_response() +} + +// --------------------------------------------------------------------------------------------- +// Request parsing + +/// A JSON object request body whose fields are taken one by one; whatever is left at the end is +/// an unknown field and refused. +struct Body(Map); + +impl Body { + fn parse(bytes: &[u8]) -> Result { + if bytes.iter().all(u8::is_ascii_whitespace) { + return Err(invalid("the request body must be a JSON object")); + } + match serde_json::from_slice::(bytes) { + Ok(Value::Object(map)) => Ok(Self(map)), + Ok(_) => Err(invalid("the request body must be a JSON object")), + Err(e) => Err(invalid(format!("the request body is not valid JSON: {e}"))), + } + } + + fn has(&self, key: &str) -> bool { + self.0.contains_key(key) + } + + fn opt(&mut self, key: &str) -> Result, FenceError> { + match self.0.remove(key) { + None | Some(Value::Null) => Ok(None), + // Through text rather than `from_value`: some protocol types (the Hrana values of + // a statement) only deserialize from borrowed strings. + Some(v) => serde_json::from_str(&v.to_string()) + .map(Some) + .map_err(|e| invalid(format!("invalid `{key}`: {e}"))), + } + } + + fn req(&mut self, key: &str) -> Result { + self.opt(key)? + .ok_or_else(|| invalid(format!("missing `{key}`"))) + } + + fn uuid(&mut self, key: &str) -> Result { + let s: String = self.req(key)?; + Uuid::parse_str(&s).map_err(|e| invalid(format!("invalid `{key}`: {e}"))) + } + + fn state(&mut self, key: &str) -> Result { + let s: String = self.req(key)?; + s.parse() + .map_err(|e: crate::namespace::fence::state::UnknownFenceState| invalid(e.to_string())) + } + + fn finish(self) -> Result<(), FenceError> { + match self.0.keys().next() { + None => Ok(()), + Some(key) => Err(invalid(format!("unknown field `{key}`"))), + } + } +} + +fn invalid(message: impl Into) -> FenceError { + FenceError::new(FenceOutcome::FencePreconditionFailed, message) + .with_detail(FenceDetail::InvalidArgument) +} + +#[derive(Debug, Deserialize)] +#[serde(deny_unknown_fields)] +struct DrainPolicyBody { + deadline_ms: u64, + #[serde(default)] + on_deadline: Option, +} + +fn drain_policy(body: &mut Body) -> Result, FenceError> { + let Some(p) = body.opt::("drain_policy")? else { + return Ok(None); + }; + let on_deadline = match p.on_deadline.as_deref() { + None | Some("fail") => OnDeadline::Fail, + Some("force_rollback") => OnDeadline::ForceRollback, + Some(other) => { + return Err(invalid(format!( + "invalid `drain_policy.on_deadline` `{other}`: expected `fail` or `force_rollback`" + ))) + } + }; + Ok(Some(DrainPolicy { + deadline_ms: p.deadline_ms, + on_deadline, + })) +} + +/// Restore and dump options a target creation refuses: import goes through the migration +/// capability (section 4.4). +const RESTORE_FIELDS: [&str; 5] = [ + "dump_url", + "restore", + "restore_option", + "timestamp", + "from_backup", +]; + +fn parse_command( + namespace: NamespaceName, + kind: CommandKind, + bytes: &[u8], +) -> Result { + let mut body = Body::parse(bytes)?; + if kind == CommandKind::CreateTargetQuarantined { + if let Some(field) = RESTORE_FIELDS.iter().find(|f| body.has(f)) { + return Err(FenceError::new( + FenceOutcome::FencePreconditionFailed, + format!( + "`{field}` is not accepted: a migration target is created empty and filled \ + through the operation's import capability" + ), + ) + .with_detail(FenceDetail::RestoreNotAllowed)); + } + if body.has("shared_schema") || body.has("shared_schema_name") { + return Err(FenceError::new( + FenceOutcome::FencePreconditionFailed, + "namespace fences do not support shared schemas", + ) + .with_detail(FenceDetail::SharedSchemaUnsupported)); + } + } + + let operation_id = body.uuid("operation_id")?; + let command_id = body.uuid("command_id")?; + let expected_state = body.state("expected_state")?; + let expected_revision: u64 = body.req("expected_revision")?; + + let command = match kind { + CommandKind::AcquireSourceWriteFence => { + #[derive(Deserialize)] + #[serde(deny_unknown_fields)] + struct Identity { + log_id: String, + } + let identity: Identity = body.req("expected_namespace_identity")?; + let expected_log_id = Uuid::parse_str(&identity.log_id).map_err(|e| { + invalid(format!("invalid `expected_namespace_identity.log_id`: {e}")) + })?; + FenceCommand::AcquireSourceWriteFence { + expected_log_id, + drain_policy: drain_policy(&mut body)?, + } + } + CommandKind::SetSourceReadFence => FenceCommand::SetSourceReadFence { + drain_policy: drain_policy(&mut body)?, + }, + CommandKind::ClearSourceReadFence => FenceCommand::ClearSourceReadFence, + CommandKind::ReleaseSourceWriteFence => FenceCommand::ReleaseSourceWriteFence, + CommandKind::CreateTargetQuarantined => { + let jwt_key: Option = body.opt("jwt_key")?; + if let Some(key) = jwt_key.as_deref() { + parse_jwt_keys(key).map_err(|e| invalid(format!("invalid `jwt_key`: {e}")))?; + } + let max_db_size: Option = body.opt("max_db_size")?; + FenceCommand::CreateTargetQuarantined { + config: TargetConfig { + max_db_size: max_db_size.map(|s| s.as_u64()), + jwt_key, + txn_timeout_s: body.opt("txn_timeout_s")?, + allow_attach: body.opt("allow_attach")?.unwrap_or(false), + durability_mode: body.opt("durability_mode")?, + bottomless_db_id: body.opt("bottomless_db_id")?, + }, + } + } + CommandKind::SealTargetImport => FenceCommand::SealTargetImport { + drain_policy: drain_policy(&mut body)?, + }, + CommandKind::RecordTargetValidation => { + let result: String = body.req("result")?; + let result = match result.as_str() { + "ok" => ValidationResult::Ok, + "failed" => ValidationResult::Failed, + other => { + return Err(invalid(format!( + "invalid `result` `{other}`: expected `ok` or `failed`" + ))) + } + }; + FenceCommand::RecordTargetValidation { + result, + summary: body.opt("summary")?.unwrap_or_default(), + } + } + CommandKind::PublishTargetReadableWriteFenced => { + FenceCommand::PublishTargetReadableWriteFenced + } + CommandKind::EnableTargetWrites => FenceCommand::EnableTargetWrites, + CommandKind::AbortQuarantinedTarget => FenceCommand::AbortQuarantinedTarget, + CommandKind::AdoptFence => { + return Err(invalid("adoption is not served by this route")); + } + }; + body.finish()?; + Ok(FenceRequest { + namespace, + operation_id, + command_id, + expected_state, + expected_revision, + command, + }) +} + +struct ValidationQuery { + operation_id: Uuid, + expected_state: Option, + expected_revision: u64, + stmts: Vec, +} + +fn parse_validation_query(bytes: &[u8]) -> Result { + let mut body = Body::parse(bytes)?; + let operation_id = body.uuid("operation_id")?; + let expected_state = if body.has("expected_state") { + Some(body.state("expected_state")?) + } else { + None + }; + let expected_revision = body.req("expected_revision")?; + let stmts: Vec = body.req("stmts")?; + body.finish()?; + if stmts.is_empty() { + return Err(invalid("`stmts` is empty")); + } + Ok(ValidationQuery { + operation_id, + expected_state, + expected_revision, + stmts, + }) +} + +// --------------------------------------------------------------------------------------------- +// Validation queries + +enum StmtFailure { + Invalid(String), + Sqlite(rusqlite::Error), + TooManyRows, +} + +impl StmtFailure { + fn into_fence_error(self, index: usize) -> FenceError { + match self { + StmtFailure::Invalid(message) => invalid(format!("statement {index}: {message}")), + StmtFailure::Sqlite(rusqlite::Error::SqliteFailure(e, message)) + if e.code == rusqlite::ErrorCode::ReadOnly => + { + FenceError::new( + FenceOutcome::OperationCapabilityRequired, + format!( + "statement {index}: the validation capability is read-only: {}", + message.unwrap_or_else(|| e.to_string()) + ), + ) + } + StmtFailure::Sqlite(e) => invalid(format!("statement {index}: {e}")), + StmtFailure::TooManyRows => invalid(format!( + "statement {index}: the request returns more than {MAX_VALIDATION_QUERY_ROWS} rows" + )), + } + } +} + +impl From for StmtFailure { + fn from(e: rusqlite::Error) -> Self { + StmtFailure::Sqlite(e) + } +} + +fn to_sql_value(value: &proto::Value) -> Result { + use rusqlite::types::Value as V; + Ok(match value { + proto::Value::None => return Err(StmtFailure::Invalid("an argument has no value".into())), + proto::Value::Null => V::Null, + proto::Value::Integer { value } => V::Integer(*value), + proto::Value::Float { value } => V::Real(*value), + proto::Value::Text { value } => V::Text(value.to_string()), + proto::Value::Blob { value } => V::Blob(value.to_vec()), + }) +} + +fn from_sql_value(value: rusqlite::types::ValueRef<'_>) -> proto::Value { + use rusqlite::types::ValueRef as V; + match value { + V::Null => proto::Value::Null, + V::Integer(value) => proto::Value::Integer { value }, + V::Real(value) => proto::Value::Float { value }, + V::Text(bytes) => proto::Value::Text { + value: String::from_utf8_lossy(bytes).into(), + }, + V::Blob(bytes) => proto::Value::Blob { + value: Bytes::copy_from_slice(bytes), + }, + } +} + +/// Run one statement of a `validation-query` and collect at most `budget` rows. +fn run_validation_stmt( + conn: &mut rusqlite::Connection, + stmt: &proto::Stmt, + budget: usize, +) -> Result { + let sql = stmt.sql.as_deref().ok_or_else(|| { + StmtFailure::Invalid("`sql` is required (`sql_id` is not supported)".into()) + })?; + let mut prepared = conn.prepare(sql)?; + if !stmt.args.is_empty() && !stmt.named_args.is_empty() { + return Err(StmtFailure::Invalid( + "`args` and `named_args` cannot be combined".into(), + )); + } + for (i, arg) in stmt.args.iter().enumerate() { + prepared.raw_bind_parameter(i + 1, to_sql_value(arg)?)?; + } + for arg in &stmt.named_args { + let index = prepared + .parameter_index(&arg.name)? + .ok_or_else(|| StmtFailure::Invalid(format!("unknown parameter `{}`", arg.name)))?; + prepared.raw_bind_parameter(index, to_sql_value(&arg.value)?)?; + } + let cols: Vec = prepared + .columns() + .iter() + .map(|c| proto::Col { + name: Some(c.name().to_string()), + decltype: c.decl_type().map(str::to_string), + }) + .collect(); + let want_rows = stmt.want_rows.unwrap_or(true); + let column_count = cols.len(); + let mut rows = Vec::new(); + let mut raw = prepared.raw_query(); + while let Some(row) = raw.next()? { + if !want_rows { + continue; + } + if rows.len() == budget { + return Err(StmtFailure::TooManyRows); + } + let mut values = Vec::with_capacity(column_count); + for i in 0..column_count { + values.push(from_sql_value(row.get_ref(i)?)); + } + rows.push(proto::Row { values }); + } + Ok(proto::StmtResult { + cols, + rows, + ..Default::default() + }) +} + +// --------------------------------------------------------------------------------------------- +// Response views + +fn timestamp(ms: i64) -> Value { + chrono::DateTime::::from_timestamp_millis(ms) + .map(|t| json!(t.to_rfc3339_opts(chrono::SecondsFormat::Millis, true))) + .unwrap_or(Value::Null) +} + +fn server_json(server: &ServerIdentity) -> Value { + json!({ "build": server.build, "instance_id": server.instance_id.to_string() }) +} + +fn drain_json(controller: Option<&FenceController>) -> Value { + let counters = controller.map(|c| c.drain_counters()).unwrap_or_default(); + let DrainCounters { + active_writers, + read_leases, + import_writers, + } = counters; + json!({ + "active_writers": active_writers, + "read_leases": { + "sql": read_leases.sql, + "dump": read_leases.dump, + "replication": read_leases.replication, + }, + "import_writers": import_writers, + }) +} + +fn record_fields(record: &NamespaceFenceRecord, out: &mut Map) { + out.insert("role".into(), json!(record.role.as_str())); + out.insert("revision".into(), json!(record.revision)); + out.insert( + "operation_id".into(), + json!(record.operation_id.to_string()), + ); + out.insert( + "frozen_boundary".into(), + record + .frozen_boundary + .map(|b| json!({ "log_id": b.log_id.to_string(), "frame_no": b.frame_no })) + .unwrap_or(Value::Null), + ); + out.insert( + "drain_policy".into(), + record + .drain_policy + .map(|p| json!({ "deadline_ms": p.deadline_ms, "on_deadline": p.on_deadline.as_str() })) + .unwrap_or(Value::Null), + ); + out.insert( + "drain_started_at".into(), + record + .drain_started_at_ms + .map(timestamp) + .unwrap_or(Value::Null), + ); + out.insert( + "validation".into(), + record + .validation + .as_ref() + .map(|v| { + json!({ + "operation_id": v.operation_id.to_string(), + "command_id": v.command_id.to_string(), + "result": v.result.as_str(), + "summary": v.summary, + "snapshot": v.snapshot.map(|s| json!({ + "log_id": s.log_id.to_string(), + "frame_no": s.frame_no, + "page_count": s.page_count, + })), + "recorded_at": timestamp(v.recorded_at_ms), + }) + }) + .unwrap_or(Value::Null), + ); + out.insert("created_at".into(), timestamp(record.created_at_ms)); + out.insert( + "last_transition_at".into(), + timestamp(record.last_transition_at_ms), + ); + out.insert( + "last_command_id".into(), + json!(record.last_command_id.to_string()), + ); + out.insert("written_by".into(), server_json(&record.written_by)); + out.insert( + "adoptions".into(), + Value::Array( + record + .adoptions + .iter() + .map(|a| { + json!({ + "previous_operation_id": a.previous_operation_id.to_string(), + "new_operation_id": a.new_operation_id.to_string(), + "command_id": a.command_id.to_string(), + "approvers": a.approvers, + "incident_ref": a.incident_ref, + "reason": a.reason, + "at": timestamp(a.at_ms), + "revision": a.revision, + }) + }) + .collect(), + ), + ); +} + +/// The fence view of section 4.3. `fence` is the durable state being reported; `controller`, +/// when the namespace has one, supplies the live admission and the live log id. +fn fence_json( + namespace: &NamespaceName, + fence: &StoredFence, + controller: Option<&FenceController>, +) -> Value { + let gate = controller.map(|c| c.gate()); + let mut out = Map::new(); + out.insert("namespace".into(), json!(namespace.as_str())); + let state = match &gate { + Some(g) if g.is_creating_target() => FenceState::TargetQuarantined, + _ => fence.state(), + }; + out.insert("state".into(), json!(state.as_str())); + out.insert("role".into(), Value::Null); + out.insert("revision".into(), json!(fence.revision())); + out.insert("operation_id".into(), Value::Null); + let current_log_id = controller.and_then(|c| c.current_log_id()); + let (log_id, incarnation_id) = fence + .record() + .map(|r| (r.identity.log_id, r.identity.target_incarnation_id)) + .unwrap_or((None, None)); + out.insert( + "incarnation".into(), + json!({ + "log_id": log_id.map(|id| id.to_string()), + "target_incarnation_id": incarnation_id.map(|id| id.to_string()), + "current_log_id": current_log_id.map(|id| id.to_string()), + }), + ); + let (write, read, generation) = match &gate { + Some(g) => (g.write(), g.read(), g.write_generation), + None => (state.write_admission(), state.read_admission(), 0), + }; + out.insert( + "admission".into(), + json!({ + "write": write.as_str(), + "read": read.as_str(), + "generation": generation, + "indeterminate": gate.as_ref().is_some_and(|g| g.indeterminate.is_some()), + }), + ); + let marker = match fence { + StoredFence::None { .. } => Value::Null, + StoredFence::Record(record) => { + record_fields(record, &mut out); + json!("consistent") + } + StoredFence::Unavailable { + detail, + reason, + marker, + } => { + out.insert("detail".into(), json!(detail.as_str())); + out.insert("reason".into(), json!(reason)); + out.insert( + "marker_record".into(), + marker + .as_ref() + .map(|m| { + let mut inner = Map::new(); + inner.insert("state".into(), json!(m.state.as_str())); + record_fields(m, &mut inner); + Value::Object(inner) + }) + .unwrap_or(Value::Null), + ); + json!(detail.as_str()) + } + }; + out.insert("server".into(), server_json(&server_identity())); + out.insert( + "provenance".into(), + json!({ "metastore_restored_from_backup": false, "marker": marker }), + ); + Value::Object(out) +} + +fn receipt_json(receipt: &CommandReceipt) -> Value { + json!({ + "operation_id": receipt.operation_id.to_string(), + "command_id": receipt.command_id.to_string(), + "command": receipt.command.as_str(), + "fingerprint": receipt.fingerprint.to_string(), + "outcome": receipt.outcome.as_str(), + "revision_before": receipt.revision_before, + "revision_after": receipt.revision_after, + "state_after": receipt.state_after.as_str(), + "applied_at": timestamp(receipt.applied_at_ms), + "instance_id": receipt.instance_id.to_string(), + }) +} + +fn stored_receipt_json(stored: &StoredReceipt) -> Value { + match &stored.receipt { + Ok(receipt) => receipt_json(receipt), + Err(e) => json!({ + "operation_id": stored.operation_id, + "command_id": stored.command_id, + "revision_after": stored.revision_after, + "applied_at": timestamp(stored.applied_at_ms), + "error": e.to_string(), + }), + } +} diff --git a/libsql-server/src/http/admin/mod.rs b/libsql-server/src/http/admin/mod.rs index 6953696312..e9a25b33af 100644 --- a/libsql-server/src/http/admin/mod.rs +++ b/libsql-server/src/http/admin/mod.rs @@ -30,6 +30,7 @@ use crate::namespace::{DumpStream, NamespaceName, NamespaceStore, RestoreOption} use crate::net::Connector; use crate::LIBSQL_PAGE_SIZE; +pub mod fence; pub mod stats; #[derive(Clone)] @@ -49,6 +50,9 @@ struct AppState { connector: C, metrics: Metrics, set_env_filter: Option anyhow::Result<()> + Sync + Send + 'static>>, + /// Whether an admin auth key is configured. Namespace fence commands refuse to run without + /// one (`docs/NAMESPACE_FENCE.md` section 4.1). + admin_auth_configured: bool, } impl FromRef>> for Metrics { @@ -170,12 +174,14 @@ where .route("/profile/heap/disable/:id", post(disable_profile_heap)) .route("/profile/heap/:id", delete(delete_profile_heap)) .route("/log-filter", post(handle_set_log_filter)) + .merge(fence::routes()) .with_state(Arc::new(AppState { namespaces: namespaces.clone(), connector, user_http_server, metrics, set_env_filter, + admin_auth_configured: auth.is_some(), })) .layer( tower_http::trace::TraceLayer::new_for_http() diff --git a/libsql-server/src/namespace/fence/controller.rs b/libsql-server/src/namespace/fence/controller.rs index 4dfd45743b..621576bacb 100644 --- a/libsql-server/src/namespace/fence/controller.rs +++ b/libsql-server/src/namespace/fence/controller.rs @@ -233,6 +233,14 @@ pub enum LeaseKind { Replication, } +/// The live drain counters of a namespace (see [`FenceController::drain_counters`]). +#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)] +pub struct DrainCounters { + pub active_writers: usize, + pub read_leases: ReadLeaseCounts, + pub import_writers: usize, +} + /// The number of read leases held on a namespace, by kind. #[derive(Debug, Clone, Copy, Default, PartialEq, Eq)] pub struct ReadLeaseCounts { @@ -652,6 +660,29 @@ impl FenceController { self.capabilities.lock().import_writers } + /// The live drain counters reported by `InspectFence` and every admin response + /// (`docs/NAMESPACE_FENCE.md` section 4.3): connections holding a write slot for a write + /// transaction, read leases by kind, and running import calls. A snapshot; never waits. + pub fn drain_counters(&self) -> DrainCounters { + let active_writers = self + .live_write_drains() + .iter() + .filter(|source| source.manager.has_writer()) + .count(); + DrainCounters { + active_writers, + read_leases: self.read_lease_counts(), + import_writers: self.import_writers(), + } + } + + /// The replication log id of the namespace as it is loaded now, if it is loaded on this + /// server as a primary. After a dirty restart this can differ from the log id a source was + /// acquired on (section 8.5). + pub fn current_log_id(&self) -> Option { + self.live_write_drains().last().map(|source| source.log_id) + } + /// The capabilities issued and still live. pub fn live_capabilities(&self) -> usize { self.capabilities.lock().live.len() diff --git a/libsql-server/src/namespace/fence/mod.rs b/libsql-server/src/namespace/fence/mod.rs index a8d80da59c..af5b76182e 100644 --- a/libsql-server/src/namespace/fence/mod.rs +++ b/libsql-server/src/namespace/fence/mod.rs @@ -45,3 +45,20 @@ pub(crate) mod proto { /// Version of the fence admin protocol reported by capability discovery. pub const FENCE_PROTOCOL_VERSION: u32 = 1; + +/// Whether this server fills the proxy protocol's additive `Error.stable_code` field and maps +/// it on the replica side (`docs/NAMESPACE_FENCE.md` section 6.1). Reported by capability +/// discovery so that deployment tooling can check every server before fences are used. +pub const PROXY_STABLE_CODE: bool = false; + +/// The identity of this server process: its build and an id generated once per process. It is +/// written into records and receipts, and reported by the admin API. +pub fn server_identity() -> record::ServerIdentity { + static IDENTITY: std::sync::OnceLock = std::sync::OnceLock::new(); + IDENTITY + .get_or_init(|| record::ServerIdentity { + build: crate::version::version(), + instance_id: uuid::Uuid::new_v4(), + }) + .clone() +} diff --git a/libsql-server/src/namespace/fence/registry.rs b/libsql-server/src/namespace/fence/registry.rs index 7c03102cd1..176cfddd4e 100644 --- a/libsql-server/src/namespace/fence/registry.rs +++ b/libsql-server/src/namespace/fence/registry.rs @@ -82,6 +82,22 @@ impl FenceRegistry { } } + /// How many namespaces have an active fence (`docs/NAMESPACE_FENCE.md` section 4.4, + /// `active_fences`): a record in any state but `RELEASED` or `TARGET_WRITABLE`, an + /// unavailable state, a target being created, or a commit whose outcome is not known yet. + pub fn active_count(&self) -> usize { + let controllers: Vec<_> = self.controllers.lock().values().cloned().collect(); + controllers + .iter() + .filter(|controller| { + let gate = controller.gate(); + gate.state().is_active() + || gate.indeterminate.is_some() + || gate.is_creating_target() + }) + .count() + } + pub fn len(&self) -> usize { self.controllers.lock().len() } diff --git a/libsql-server/src/namespace/store.rs b/libsql-server/src/namespace/store.rs index 889370ad79..09d27a5695 100644 --- a/libsql-server/src/namespace/store.rs +++ b/libsql-server/src/namespace/store.rs @@ -32,7 +32,7 @@ use super::fence::registry::FenceRegistry; use super::fence::state::{FenceState, Role}; use super::fence::store::StoredFence; use super::fence::target::{self, CreateTargetRequest, ValidationSession}; -use super::meta_store::{FenceCommit, FenceContext, MetaStore, MetaStoreHandle}; +use super::meta_store::{FenceCommit, FenceContext, FenceInspection, MetaStore, MetaStoreHandle}; use super::schema_lock::SchemaLocksRegistry; use super::{Namespace, ResetCb, ResetOp, ResolveNamespacePathFn, RestoreOption}; @@ -556,8 +556,6 @@ impl NamespaceStore { /// (`docs/NAMESPACE_FENCE.md` sections 5.3 and 8). `AcquireSourceWriteFence` loads the /// namespace first, so that its connection manager and replication log are registered with /// the namespace's controller before the drain needs them. - // The admin routes that call this are not part of the server yet. - #[cfg_attr(not(test), allow(dead_code))] pub(crate) async fn execute_fence_command( &self, request: FenceRequest, @@ -881,6 +879,36 @@ impl NamespaceStore { Ok(None) } + /// Whether this store serves primary namespaces. Fences live on the primary that owns the + /// WAL; a replica-kind server refuses every fence route (`docs/NAMESPACE_FENCE.md` 4.1). + pub(crate) fn is_primary(&self) -> bool { + !self.inner.db_kind.is_replica() + } + + /// The fence controller `namespace` already has, without creating one. + pub(crate) fn existing_fence_controller( + &self, + namespace: &NamespaceName, + ) -> Option> { + self.inner.fences.get(namespace) + } + + /// `InspectFence`: the durable fence and receipts of `namespace` as the metastore holds + /// them, and the namespace's controller if it has one (for the live gate and drain + /// counters). Read-only: it neither loads the namespace nor creates a controller. + pub(crate) async fn inspect_fence( + &self, + namespace: &NamespaceName, + ) -> crate::Result<(FenceInspection, Option>)> { + let inspection = self.inner.metadata.inspect_fence(namespace.clone()).await?; + Ok((inspection, self.inner.fences.get(namespace))) + } + + /// How many namespaces on this server have an active fence (capability discovery). + pub(crate) fn active_fences(&self) -> usize { + self.inner.fences.active_count() + } + pub(crate) fn schema_locks(&self) -> &SchemaLocksRegistry { &self.inner.schema_locks } diff --git a/libsql-server/tests/common/http.rs b/libsql-server/tests/common/http.rs index 8716a60503..a08a928478 100644 --- a/libsql-server/tests/common/http.rs +++ b/libsql-server/tests/common/http.rs @@ -41,6 +41,20 @@ impl Client { Ok(Response(self.0.get(s.parse()?).await?)) } + pub(crate) async fn get_with_headers( + &self, + url: &str, + headers: &[(HeaderName, &str)], + ) -> anyhow::Result { + let mut request = hyper::Request::get(url).body(Body::empty())?; + for (key, val) in headers { + request + .headers_mut() + .insert(key.clone(), val.parse().unwrap()); + } + Ok(Response(self.0.request(request).await?)) + } + pub(crate) async fn post(&self, url: &str, body: T) -> anyhow::Result { self.post_with_headers(url, &[], body).await } diff --git a/libsql-server/tests/fence/admin.rs b/libsql-server/tests/fence/admin.rs new file mode 100644 index 0000000000..d1ac8413c4 --- /dev/null +++ b/libsql-server/tests/fence/admin.rs @@ -0,0 +1,731 @@ +//! The fence admin API over HTTP (`docs/NAMESPACE_FENCE.md` section 4). + +use hyper::StatusCode; +use serde_json::{json, Value}; +use tempfile::tempdir; +use uuid::Uuid; + +use super::{command_body, connect, make_primary, sim, state_of, Admin, Primary, ADMIN_KEY}; + +fn uuid(n: u128) -> Uuid { + Uuid::from_u128(n) +} + +/// Load `ns` on the server with one write, and return the replication log id the server +/// reports for it. +async fn load_and_log_id(admin: &Admin, ns: &str) -> anyhow::Result { + let conn = connect(ns)?; + conn.execute("create table if not exists t (x)", ()).await?; + conn.execute("insert into t values (1)", ()).await?; + let (status, body) = admin.inspect(ns).await?; + assert_eq!(status, StatusCode::OK, "{body}"); + assert_eq!(state_of(&body), ("UNFENCED", 0), "{body}"); + Ok(body["fence"]["incarnation"]["current_log_id"] + .as_str() + .unwrap_or_else(|| panic!("no current_log_id: {body}")) + .to_string()) +} + +fn acquire_body(op: Uuid, cmd: Uuid, log_id: &str) -> Value { + command_body( + op, + cmd, + "UNFENCED", + 0, + json!({ + "expected_namespace_identity": { "log_id": log_id }, + "drain_policy": { "deadline_ms": 5000, "on_deadline": "fail" }, + }), + ) +} + +#[test] +fn capabilities() { + let mut sim = sim(); + let tmp = tempdir().unwrap(); + make_primary(&mut sim, tmp.path().to_path_buf(), Primary::default()); + sim.client("client", async { + let admin = Admin::new(Some(ADMIN_KEY)); + let (status, body) = admin.get("/v1/fence/capabilities").await?; + assert_eq!(status, StatusCode::OK, "{body}"); + assert_eq!(body["fence_protocol_version"], 1); + assert_eq!(body["enabled"], true); + assert_eq!(body["active_fences"], 0); + assert_eq!(body["proxy_stable_code"], false); + let commands: Vec<&str> = body["commands"] + .as_array() + .unwrap() + .iter() + .map(|c| c.as_str().unwrap()) + .collect(); + for command in [ + "InspectFence", + "AcquireSourceWriteFence", + "SetSourceReadFence", + "ClearSourceReadFence", + "ReleaseSourceWriteFence", + "CreateTargetQuarantined", + "SealTargetImport", + "RecordTargetValidation", + "PublishTargetReadableWriteFenced", + "EnableTargetWrites", + "AbortQuarantinedTarget", + ] { + assert!(commands.contains(&command), "{command} missing: {body}"); + } + let states = body["states"].as_array().unwrap(); + assert!(states.contains(&json!("SOURCE_WRITE_FENCED")), "{body}"); + assert!(states.contains(&json!("UNKNOWN_UNAVAILABLE")), "{body}"); + assert!(body["server"]["build"] + .as_str() + .unwrap() + .starts_with("sqld ")); + Uuid::parse_str(body["server"]["instance_id"].as_str().unwrap())?; + assert_eq!(body["metastore"]["restored_from_backup"], false); + + // An active fence is counted. + admin.create_namespace("src").await?; + let log_id = load_and_log_id(&admin, "src").await?; + let (status, body) = admin + .command( + "src", + "source/acquire-write-fence", + acquire_body(uuid(1), uuid(2), &log_id), + ) + .await?; + assert_eq!(status, StatusCode::OK, "{body}"); + let (_, body) = admin.get("/v1/fence/capabilities").await?; + assert_eq!(body["active_fences"], 1, "{body}"); + + // The admin API's own authentication still applies. + let (status, _) = Admin::new(None).get("/v1/fence/capabilities").await?; + assert_eq!(status, StatusCode::UNAUTHORIZED); + Ok(()) + }); + sim.run().unwrap(); +} + +#[test] +fn capabilities_when_disabled() { + let mut sim = sim(); + let tmp = tempdir().unwrap(); + make_primary( + &mut sim, + tmp.path().to_path_buf(), + Primary { + fence_enabled: false, + ..Default::default() + }, + ); + sim.client("client", async { + let admin = Admin::new(Some(ADMIN_KEY)); + let (status, body) = admin.get("/v1/fence/capabilities").await?; + assert_eq!(status, StatusCode::OK, "{body}"); + assert_eq!(body["enabled"], false); + assert_eq!(body["fence_protocol_version"], 1); + + admin.create_namespace("src").await?; + let (status, _) = admin.inspect("src").await?; + assert_eq!(status, StatusCode::NOT_FOUND); + for route in [ + "source/acquire-write-fence", + "source/release-write-fence", + "target/create-quarantined", + "target/validation-query", + ] { + let (status, body) = admin + .command( + "src", + route, + acquire_body(uuid(1), uuid(2), &uuid(3).to_string()), + ) + .await?; + assert_eq!(status, StatusCode::NOT_FOUND, "{route}: {body}"); + } + // The namespace is untouched. + let conn = connect("src")?; + conn.execute("create table t (x)", ()).await?; + Ok(()) + }); + sim.run().unwrap(); +} + +#[test] +fn mutating_routes_require_admin_key() { + let mut sim = sim(); + let tmp = tempdir().unwrap(); + make_primary( + &mut sim, + tmp.path().to_path_buf(), + Primary { + admin_key: None, + ..Default::default() + }, + ); + sim.client("client", async { + let admin = Admin::new(None); + admin.create_namespace("src").await?; + let log_id = load_and_log_id(&admin, "src").await?; + + let (status, body) = admin + .command( + "src", + "source/acquire-write-fence", + acquire_body(uuid(1), uuid(2), &log_id), + ) + .await?; + assert_eq!(status, StatusCode::PRECONDITION_FAILED, "{body}"); + assert_eq!(body["outcome"], "FENCE_PRECONDITION_FAILED"); + assert_eq!(body["detail"], "admin_auth_required"); + assert_eq!(state_of(&body), ("UNFENCED", 0), "{body}"); + + let (status, body) = admin + .command( + "tgt", + "target/create-quarantined", + command_body(uuid(1), uuid(3), "ABSENT", 0, json!({})), + ) + .await?; + assert_eq!(status, StatusCode::PRECONDITION_FAILED, "{body}"); + assert_eq!(body["detail"], "admin_auth_required"); + + let (status, body) = admin + .command( + "tgt", + "target/validation-query", + json!({ + "operation_id": uuid(1).to_string(), + "expected_revision": 1, + "stmts": [{ "sql": "select 1" }], + }), + ) + .await?; + assert_eq!(status, StatusCode::PRECONDITION_FAILED, "{body}"); + assert_eq!(body["detail"], "admin_auth_required"); + + // Nothing was fenced or created, and reading the state is still possible. + let (status, body) = admin.inspect("src").await?; + assert_eq!(status, StatusCode::OK, "{body}"); + assert_eq!(state_of(&body), ("UNFENCED", 0)); + let (status, _) = admin.inspect("tgt").await?; + assert_eq!(status, StatusCode::NOT_FOUND); + connect("src")? + .execute("insert into t values (2)", ()) + .await?; + Ok(()) + }); + sim.run().unwrap(); +} + +/// Acceptance test: two operations race to acquire the same source; exactly one owns it. +#[test] +fn concurrent_acquire_one_owner() { + let mut sim = sim(); + let tmp = tempdir().unwrap(); + make_primary(&mut sim, tmp.path().to_path_buf(), Primary::default()); + sim.client("client", async { + let admin = Admin::new(Some(ADMIN_KEY)); + admin.create_namespace("src").await?; + let log_id = load_and_log_id(&admin, "src").await?; + + let other = Admin::new(Some(ADMIN_KEY)); + let (a, b) = tokio::join!( + admin.command( + "src", + "source/acquire-write-fence", + acquire_body(uuid(0xa), uuid(1), &log_id), + ), + other.command( + "src", + "source/acquire-write-fence", + acquire_body(uuid(0xb), uuid(2), &log_id), + ), + ); + let (a, b) = (a?, b?); + let mut results = [a, b]; + results.sort_by_key(|(status, _)| status.as_u16()); + let [(won_status, won), (lost_status, lost)] = results; + assert_eq!(won_status, StatusCode::OK, "{won}"); + assert_eq!(won["outcome"], "APPLIED"); + assert_eq!(state_of(&won).0, "SOURCE_WRITE_FENCED"); + assert_eq!(lost_status, StatusCode::CONFLICT, "{lost}"); + assert_eq!(lost["outcome"], "FENCE_OWNED_BY_ANOTHER_OPERATION"); + // The loser is shown who owns the namespace. + assert_eq!(lost["fence"]["operation_id"], won["fence"]["operation_id"]); + + let (_, body) = admin.inspect("src").await?; + assert_eq!(body["fence"]["operation_id"], won["fence"]["operation_id"]); + assert!(connect("src")? + .execute("insert into t values (2)", ()) + .await + .is_err()); + Ok(()) + }); + sim.run().unwrap(); +} + +#[test] +fn source_walk_over_http() { + let mut sim = sim(); + let tmp = tempdir().unwrap(); + make_primary(&mut sim, tmp.path().to_path_buf(), Primary::default()); + sim.client("client", async { + let admin = Admin::new(Some(ADMIN_KEY)); + admin.create_namespace("src").await?; + let log_id = load_and_log_id(&admin, "src").await?; + let op = uuid(0xa); + let conn = connect("src")?; + + // A wrong identity is refused before anything changes. + let (status, body) = admin + .command( + "src", + "source/acquire-write-fence", + acquire_body(op, uuid(1), &uuid(0xdead).to_string()), + ) + .await?; + assert_eq!(status, StatusCode::PRECONDITION_FAILED, "{body}"); + assert_eq!(body["detail"], "namespace_identity_mismatch"); + + let (status, acquired) = admin + .command( + "src", + "source/acquire-write-fence", + acquire_body(op, uuid(2), &log_id), + ) + .await?; + assert_eq!(status, StatusCode::OK, "{acquired}"); + assert_eq!(acquired["outcome"], "APPLIED"); + assert_eq!(acquired["replayed"], false); + let (state, rev) = state_of(&acquired); + assert_eq!(state, "SOURCE_WRITE_FENCED"); + assert_eq!(acquired["fence"]["role"], "SOURCE"); + assert_eq!(acquired["fence"]["admission"]["write"], "closed"); + assert_eq!(acquired["fence"]["admission"]["read"], "open"); + assert_eq!( + acquired["fence"]["frozen_boundary"]["log_id"], + log_id.as_str() + ); + assert_eq!(acquired["receipt"]["command"], "AcquireSourceWriteFence"); + assert_eq!(acquired["drain"]["active_writers"], 0); + assert!(conn.execute("insert into t values (2)", ()).await.is_err()); + conn.query("select * from t", ()).await?; + + // Replay returns the stored receipt. + let (status, replay) = admin + .command( + "src", + "source/acquire-write-fence", + acquire_body(op, uuid(2), &log_id), + ) + .await?; + assert_eq!(status, StatusCode::OK, "{replay}"); + assert_eq!(replay["replayed"], true); + assert_eq!(replay["receipt"], acquired["receipt"]); + + // The same command id with a different request is a conflict. + let (status, body) = admin + .command( + "src", + "source/acquire-write-fence", + command_body( + op, + uuid(2), + "UNFENCED", + 0, + json!({ "expected_namespace_identity": { "log_id": log_id } }), + ), + ) + .await?; + assert_eq!(status, StatusCode::CONFLICT, "{body}"); + assert_eq!(body["outcome"], "FENCE_COMMAND_CONFLICT"); + + // A stale revision is refused. + let (status, body) = admin + .command( + "src", + "source/set-read-fence", + command_body(op, uuid(3), "SOURCE_WRITE_FENCED", rev - 1, json!({})), + ) + .await?; + assert_eq!(status, StatusCode::CONFLICT, "{body}"); + assert_eq!(body["outcome"], "FENCE_REVISION_MISMATCH"); + assert_eq!(state_of(&body), ("SOURCE_WRITE_FENCED", rev)); + + // Read fence, then clear it. + let (status, body) = admin + .command( + "src", + "source/set-read-fence", + command_body( + op, + uuid(4), + "SOURCE_WRITE_FENCED", + rev, + json!({ "drain_policy": { "deadline_ms": 5000 } }), + ), + ) + .await?; + assert_eq!(status, StatusCode::OK, "{body}"); + let (state, rev) = state_of(&body); + assert_eq!(state, "SOURCE_READ_FENCED"); + assert_eq!(body["fence"]["admission"]["read"], "closed"); + assert!(conn.query("select * from t", ()).await.is_err()); + + let (status, body) = admin + .command( + "src", + "source/clear-read-fence", + command_body(op, uuid(5), "SOURCE_READ_FENCED", rev, json!({})), + ) + .await?; + assert_eq!(status, StatusCode::OK, "{body}"); + let (state, rev) = state_of(&body); + assert_eq!(state, "SOURCE_WRITE_FENCED"); + connect("src")?.query("select * from t", ()).await?; + + // Release reopens writes. + let release = command_body(op, uuid(6), "SOURCE_WRITE_FENCED", rev, json!({})); + let (status, released) = admin + .command("src", "source/release-write-fence", release.clone()) + .await?; + assert_eq!(status, StatusCode::OK, "{released}"); + assert_eq!(state_of(&released).0, "RELEASED"); + assert_eq!(released["fence"]["admission"]["write"], "open"); + connect("src")? + .execute("insert into t values (3)", ()) + .await?; + + let (status, replay) = admin + .command("src", "source/release-write-fence", release) + .await?; + assert_eq!(status, StatusCode::OK, "{replay}"); + assert_eq!(replay["replayed"], true); + assert_eq!(replay["receipt"], released["receipt"]); + + // Inspect shows the operation's receipts, and all of them on request. + let (status, body) = admin.inspect("src").await?; + assert_eq!(status, StatusCode::OK, "{body}"); + assert_eq!(state_of(&body).0, "RELEASED"); + let receipts = body["receipts"].as_array().unwrap(); + assert!(receipts.len() >= 4, "{body}"); + assert!(receipts + .iter() + .all(|r| r["operation_id"] == op.to_string().as_str())); + let (_, all) = admin.get("/v1/namespaces/src/fence?receipts=all").await?; + assert!(all["receipts"].as_array().unwrap().len() >= receipts.len()); + + // Malformed requests are typed refusals. + let (status, body) = admin + .command( + "src", + "source/release-write-fence", + json!({ "operation_id": "x" }), + ) + .await?; + assert_eq!(status, StatusCode::PRECONDITION_FAILED, "{body}"); + assert_eq!(body["detail"], "invalid_argument"); + let (status, body) = admin + .command( + "src", + "source/release-write-fence", + command_body(op, uuid(7), "RELEASED", 0, json!({ "surprise": 1 })), + ) + .await?; + assert_eq!(status, StatusCode::PRECONDITION_FAILED, "{body}"); + assert_eq!(body["detail"], "invalid_argument"); + Ok(()) + }); + sim.run().unwrap(); +} + +#[test] +fn target_walk_over_http() { + let mut sim = sim(); + let tmp = tempdir().unwrap(); + make_primary(&mut sim, tmp.path().to_path_buf(), Primary::default()); + sim.client("client", async { + let admin = Admin::new(Some(ADMIN_KEY)); + let op = uuid(0xa); + + // Restore options are refused, and nothing is created. + let (status, body) = admin + .command( + "tgt", + "target/create-quarantined", + command_body( + op, + uuid(1), + "ABSENT", + 0, + json!({ "dump_url": "file:///tmp/dump.sql" }), + ), + ) + .await?; + assert_eq!(status, StatusCode::PRECONDITION_FAILED, "{body}"); + assert_eq!(body["detail"], "restore_not_allowed"); + assert_eq!(admin.inspect("tgt").await?.0, StatusCode::NOT_FOUND); + + let create = command_body( + op, + uuid(2), + "ABSENT", + 0, + json!({ "max_db_size": 10_000_000, "durability_mode": "strong" }), + ); + let (status, created) = admin + .command("tgt", "target/create-quarantined", create.clone()) + .await?; + assert_eq!(status, StatusCode::OK, "{created}"); + assert_eq!(created["outcome"], "APPLIED"); + let (state, rev) = state_of(&created); + assert_eq!(state, "TARGET_QUARANTINED"); + assert_eq!(created["fence"]["role"], "TARGET"); + assert!(created["fence"]["incarnation"]["target_incarnation_id"].is_string()); + let (status, replay) = admin + .command("tgt", "target/create-quarantined", create) + .await?; + assert_eq!(status, StatusCode::OK, "{replay}"); + assert_eq!(replay["replayed"], true); + + // Normal SQL is refused while the target is quarantined. + assert!(connect("tgt")?.query("select 1", ()).await.is_err()); + let (status, body) = admin + .post("/v1/namespaces/tgt/create", json!({})) + .await?; + assert!(!status.is_success(), "{status} {body}"); + + // Validation queries are refused before the import is sealed. + let (status, body) = admin + .command( + "tgt", + "target/validation-query", + json!({ + "operation_id": op.to_string(), + "expected_revision": rev, + "stmts": [{ "sql": "select 1" }], + }), + ) + .await?; + assert_eq!(status, StatusCode::FORBIDDEN, "{body}"); + assert_eq!(body["outcome"], "OPERATION_CAPABILITY_REQUIRED"); + + let (status, body) = admin + .command( + "tgt", + "target/seal-import", + command_body(op, uuid(3), "TARGET_QUARANTINED", rev, json!({})), + ) + .await?; + assert_eq!(status, StatusCode::OK, "{body}"); + let (state, rev) = state_of(&body); + assert_eq!(state, "TARGET_VALIDATING"); + + let (status, body) = admin + .command( + "tgt", + "target/validation-query", + json!({ + "operation_id": op.to_string(), + "expected_state": "TARGET_VALIDATING", + "expected_revision": rev, + "stmts": [ + { "sql": "select count(*) as n from sqlite_master" }, + { "sql": "select ? + 1 as v", "args": [{ "type": "integer", "value": "41" }] }, + ], + }), + ) + .await?; + assert_eq!(status, StatusCode::OK, "{body}"); + assert_eq!(body["results"][0]["cols"][0]["name"], "n"); + assert_eq!( + body["results"][1]["rows"][0][0], + json!({ "type": "integer", "value": "42" }) + ); + + // A validation query cannot write. + let (status, body) = admin + .command( + "tgt", + "target/validation-query", + json!({ + "operation_id": op.to_string(), + "expected_revision": rev, + "stmts": [{ "sql": "create table sneaky (x)" }], + }), + ) + .await?; + assert_eq!(status, StatusCode::FORBIDDEN, "{body}"); + assert_eq!(body["outcome"], "OPERATION_CAPABILITY_REQUIRED"); + + // Another operation cannot validate. + let (status, body) = admin + .command( + "tgt", + "target/validation-query", + json!({ + "operation_id": uuid(0xb).to_string(), + "expected_revision": rev, + "stmts": [{ "sql": "select 1" }], + }), + ) + .await?; + assert_eq!(status, StatusCode::CONFLICT, "{body}"); + assert_eq!(body["outcome"], "FENCE_OWNED_BY_ANOTHER_OPERATION"); + + // Publication needs a successful validation receipt. + let (status, body) = admin + .command( + "tgt", + "target/publish-readable", + command_body(op, uuid(4), "TARGET_VALIDATING", rev, json!({})), + ) + .await?; + assert_eq!(status, StatusCode::PRECONDITION_FAILED, "{body}"); + assert_eq!(body["detail"], "validation_receipt_required"); + + let (status, body) = admin + .command( + "tgt", + "target/validation-receipt", + command_body( + op, + uuid(5), + "TARGET_VALIDATING", + rev, + json!({ "result": "ok", "summary": "row counts match" }), + ), + ) + .await?; + assert_eq!(status, StatusCode::OK, "{body}"); + let (state, rev) = state_of(&body); + assert_eq!(state, "TARGET_VALIDATING"); + assert_eq!(body["fence"]["validation"]["result"], "ok"); + assert!(body["fence"]["validation"]["snapshot"]["page_count"].is_u64()); + + let (status, body) = admin + .command( + "tgt", + "target/publish-readable", + command_body(op, uuid(6), "TARGET_VALIDATING", rev, json!({})), + ) + .await?; + assert_eq!(status, StatusCode::OK, "{body}"); + let (state, rev) = state_of(&body); + assert_eq!(state, "TARGET_WRITE_FENCED"); + let conn = connect("tgt")?; + conn.query("select 1", ()).await?; + assert!(conn.execute("create table t (x)", ()).await.is_err()); + + let enable = command_body(op, uuid(7), "TARGET_WRITE_FENCED", rev, json!({})); + let (status, body) = admin + .command("tgt", "target/enable-writes", enable.clone()) + .await?; + assert_eq!(status, StatusCode::OK, "{body}"); + assert_eq!(body["outcome"], "APPLIED"); + let (state, rev_after) = state_of(&body); + assert_eq!(state, "TARGET_WRITABLE"); + connect("tgt")?.execute("create table t (x)", ()).await?; + + let (status, body) = admin + .command("tgt", "target/enable-writes", enable) + .await?; + assert_eq!(status, StatusCode::OK, "{body}"); + assert_eq!(body["replayed"], true); + let (status, body) = admin + .command( + "tgt", + "target/enable-writes", + command_body(op, uuid(8), "TARGET_WRITE_FENCED", rev, json!({})), + ) + .await?; + assert_eq!(status, StatusCode::OK, "{body}"); + assert_eq!(body["outcome"], "ALREADY_APPLIED"); + assert_eq!(state_of(&body), ("TARGET_WRITABLE", rev_after)); + + // Abort is not possible once writes are enabled. + let (status, body) = admin + .command( + "tgt", + "target/abort", + command_body(op, uuid(9), "TARGET_WRITABLE", rev_after, json!({})), + ) + .await?; + assert_eq!(status, StatusCode::CONFLICT, "{body}"); + assert_eq!(body["outcome"], "INVALID_FENCE_TRANSITION"); + Ok(()) + }); + sim.run().unwrap(); +} + +/// A write transaction open when the fence is requested holds the drain: `InspectFence` counts +/// it, the acquisition answers `DRAINING` (202) at its deadline, and replaying the command once +/// the transaction has committed completes the fence. +#[test] +fn inspect_reports_drain_counters() { + let mut sim = sim(); + let tmp = tempdir().unwrap(); + make_primary(&mut sim, tmp.path().to_path_buf(), Primary::default()); + sim.client("client", async { + let admin = Admin::new(Some(ADMIN_KEY)); + admin.create_namespace("src").await?; + let log_id = load_and_log_id(&admin, "src").await?; + + let (_, body) = admin.inspect("src").await?; + assert_eq!( + body["drain"], + json!({ + "active_writers": 0, + "read_leases": { "sql": 0, "dump": 0, "replication": 0 }, + "import_writers": 0, + }) + ); + + let conn = connect("src")?; + let tx = conn.transaction().await?; + tx.execute("insert into t values (2)", ()).await?; + + let (_, body) = admin.inspect("src").await?; + assert_eq!(body["drain"]["active_writers"], 1, "{body}"); + + let op = uuid(0xa); + let acquire = command_body( + op, + uuid(1), + "UNFENCED", + 0, + json!({ + "expected_namespace_identity": { "log_id": log_id }, + "drain_policy": { "deadline_ms": 100, "on_deadline": "fail" }, + }), + ); + let (status, body) = admin + .command("src", "source/acquire-write-fence", acquire.clone()) + .await?; + assert_eq!(status, StatusCode::ACCEPTED, "{body}"); + assert_eq!(body["outcome"], "DRAINING"); + assert_eq!(state_of(&body).0, "SOURCE_DRAINING"); + assert_eq!(body["fence"]["admission"]["write"], "closed"); + assert_eq!(body["drain"]["active_writers"], 1, "{body}"); + + // The transaction admitted before the fence commits; new writes are refused. + tx.commit().await?; + assert!(connect("src")? + .execute("insert into t values (3)", ()) + .await + .is_err()); + + let (status, body) = admin + .command("src", "source/acquire-write-fence", acquire) + .await?; + assert_eq!(status, StatusCode::OK, "{body}"); + assert_eq!(body["outcome"], "APPLIED"); + assert_eq!(state_of(&body).0, "SOURCE_WRITE_FENCED"); + assert_eq!(body["drain"]["active_writers"], 0); + let mut rows = connect("src")?.query("select count(*) from t", ()).await?; + let n: i64 = rows.next().await?.unwrap().get(0)?; + assert_eq!(n, 2); + Ok(()) + }); + sim.run().unwrap(); +} diff --git a/libsql-server/tests/fence/mod.rs b/libsql-server/tests/fence/mod.rs new file mode 100644 index 0000000000..a7a149907f --- /dev/null +++ b/libsql-server/tests/fence/mod.rs @@ -0,0 +1,188 @@ +#![allow(deprecated)] + +//! Namespace fence integration tests (`docs/NAMESPACE_FENCE.md`), driven over the admin API. + +mod admin; + +use std::path::PathBuf; +use std::time::Duration; + +use hyper::StatusCode; +use libsql_server::config::{AdminApiConfig, MetaStoreConfig, RpcServerConfig, UserApiConfig}; +use s3s::header::AUTHORIZATION; +use serde_json::{json, Value}; +use turmoil::{Builder, Sim}; +use uuid::Uuid; + +use crate::common::http::Client; +use crate::common::net::{ + init_tracing, SimServer as _, TestServer, TurmoilAcceptor, TurmoilConnector, +}; + +pub const ADMIN_KEY: &str = "fence-admin-key"; + +pub struct Primary { + /// `None` starts the admin API without an auth key. + pub admin_key: Option<&'static str>, + pub fence_enabled: bool, +} + +impl Default for Primary { + fn default() -> Self { + Self { + admin_key: Some(ADMIN_KEY), + fence_enabled: true, + } + } +} + +pub fn sim() -> Sim<'static> { + Builder::new() + .simulation_duration(Duration::from_secs(1000)) + .build() +} + +/// A primary on host `primary`: user API on 8080, admin API on 9090. +pub fn make_primary(sim: &mut Sim, path: PathBuf, primary: Primary) { + init_tracing(); + let Primary { + admin_key, + fence_enabled, + } = primary; + sim.host("primary", move || { + let path = path.clone(); + async move { + let server = TestServer { + path: path.into(), + user_api_config: UserApiConfig::default(), + admin_api_config: Some(AdminApiConfig { + acceptor: TurmoilAcceptor::bind(([0, 0, 0, 0], 9090)).await?, + connector: TurmoilConnector, + disable_metrics: true, + auth_key: admin_key.map(Into::into), + }), + rpc_server_config: Some(RpcServerConfig { + acceptor: TurmoilAcceptor::bind(([0, 0, 0, 0], 4567)).await?, + tls_config: None, + }), + meta_store_config: MetaStoreConfig { + namespace_fence: fence_enabled, + ..Default::default() + }, + disable_namespaces: false, + disable_default_namespace: true, + ..Default::default() + }; + server.start_sim(8080).await?; + Ok(()) + } + }); +} + +/// The admin API of `primary`, authenticating with `key` when there is one. +pub struct Admin { + client: Client, + key: Option, +} + +impl Admin { + pub fn new(key: Option<&str>) -> Self { + Self { + client: Client::new(), + key: key.map(|k| format!("basic {k}")), + } + } + + fn headers(&self) -> Vec<(hyper::header::HeaderName, &str)> { + self.key + .as_deref() + .map(|k| vec![(AUTHORIZATION, k)]) + .unwrap_or_default() + } + + async fn json(resp: crate::common::http::Response) -> anyhow::Result<(StatusCode, Value)> { + let status = resp.status(); + let body = resp.body_string().await?; + let value = if body.trim().is_empty() { + Value::Null + } else { + serde_json::from_str(&body).unwrap_or(Value::String(body)) + }; + Ok((status, value)) + } + + pub async fn get(&self, path: &str) -> anyhow::Result<(StatusCode, Value)> { + let url = format!("http://primary:9090{path}"); + Self::json(self.client.get_with_headers(&url, &self.headers()).await?).await + } + + pub async fn post(&self, path: &str, body: Value) -> anyhow::Result<(StatusCode, Value)> { + let url = format!("http://primary:9090{path}"); + Self::json( + self.client + .post_with_headers(&url, &self.headers(), body) + .await?, + ) + .await + } + + pub async fn create_namespace(&self, ns: &str) -> anyhow::Result<()> { + let (status, body) = self + .post(&format!("/v1/namespaces/{ns}/create"), json!({})) + .await?; + anyhow::ensure!(status.is_success(), "create {ns}: {status} {body}"); + Ok(()) + } + + pub async fn inspect(&self, ns: &str) -> anyhow::Result<(StatusCode, Value)> { + self.get(&format!("/v1/namespaces/{ns}/fence")).await + } + + /// A fence command: `route` is the part after `/fence/`. + pub async fn command( + &self, + ns: &str, + route: &str, + body: Value, + ) -> anyhow::Result<(StatusCode, Value)> { + self.post(&format!("/v1/namespaces/{ns}/fence/{route}"), body) + .await + } +} + +/// The common fields of a fence command, with `extra` merged in. +pub fn command_body( + operation_id: Uuid, + command_id: Uuid, + expected_state: &str, + expected_revision: u64, + extra: Value, +) -> Value { + let mut body = json!({ + "operation_id": operation_id.to_string(), + "command_id": command_id.to_string(), + "expected_state": expected_state, + "expected_revision": expected_revision, + }); + if let Value::Object(extra) = extra { + body.as_object_mut().unwrap().extend(extra); + } + body +} + +pub fn state_of(body: &Value) -> (&str, u64) { + ( + body["fence"]["state"].as_str().unwrap_or(""), + body["fence"]["revision"].as_u64().unwrap_or(u64::MAX), + ) +} + +/// A connection to namespace `ns` over the user API. +pub fn connect(ns: &str) -> anyhow::Result { + let db = libsql::Database::open_remote_with_connector( + format!("http://{ns}.primary:8080"), + "", + TurmoilConnector, + )?; + Ok(db.connect()?) +} diff --git a/libsql-server/tests/tests.rs b/libsql-server/tests/tests.rs index ab475df546..55a88d2dab 100644 --- a/libsql-server/tests/tests.rs +++ b/libsql-server/tests/tests.rs @@ -6,6 +6,7 @@ mod common; mod auth; mod cluster; mod embedded_replica; +mod fence; mod hrana; mod namespaces; mod standalone; From 2edd3b65afb8180ee7a04861e38f446d1cb158c2 Mon Sep 17 00:00:00 2001 From: River Date: Wed, 30 Sep 2026 09:13:46 +0000 Subject: [PATCH 19/33] libsql-server: deny lifecycle operations on fenced namespaces Refuse config changes, delete, reset, fork (as source or destination), create over an existing record (with or without a dump URL), linking to a shared schema and schema migration while a namespace's fence denies lifecycle work. A single check reads the fence registry, so it also sees in-memory gates (a closing transition, a target being created, an indeterminate commit) and never loads the namespace. Paths that persist through the metastore are refused again inside its transaction. A fork reads the source's log without a read lease, so it now holds the source's transition lock for its whole run and checks the gate under it: a write or read fence command either completes first and the fork is refused, or waits for the fork. Reset writes nothing to the metastore and is checked under the same lock. The fork destination is checked before anything is stored and again with its namespace entry held; previously a fork onto an existing, unloaded namespace would publish its config in memory and remove its directory. A dump URL is no longer fetched for a create that is refused. Registering a schema migration job checks the schema and every linked namespace, so a link written by a binary that does not know fences cannot lead to a migration step being refused halfway through a job. Co-authored-by: Tomasz Szymczyszyn --- docs/NAMESPACE_FENCE.md | 30 +- libsql-server/src/http/admin/mod.rs | 13 +- libsql-server/src/namespace/fence/registry.rs | 11 + libsql-server/src/namespace/store.rs | 268 +++++++++++++++++- libsql-server/src/schema/db.rs | 17 ++ libsql-server/src/schema/error.rs | 3 + libsql-server/src/schema/scheduler.rs | 258 +++++++++++++++++ libsql-server/tests/common/http.rs | 16 +- libsql-server/tests/fence/admin.rs | 35 +-- libsql-server/tests/fence/lifecycle.rs | 201 +++++++++++++ libsql-server/tests/fence/mod.rs | 40 +++ 11 files changed, 841 insertions(+), 51 deletions(-) create mode 100644 libsql-server/tests/fence/lifecycle.rs diff --git a/docs/NAMESPACE_FENCE.md b/docs/NAMESPACE_FENCE.md index 74a4fae029..3841e1d769 100644 --- a/docs/NAMESPACE_FENCE.md +++ b/docs/NAMESPACE_FENCE.md @@ -712,6 +712,24 @@ A namespace that startup cannot recover is registered `UNKNOWN_UNAVAILABLE` in m v1 does not fence shared-schema databases or namespaces linked to one, because schema migration fan-out would have to obey the fence on every linked namespace. Acquisition and target creation reject them with `FENCE_PRECONDITION_FAILED`, `shared_schema_unsupported`. While a fence is active, config mutation (which includes linking to a shared schema) is denied. +The two sets therefore never meet: acquisition reads the config row in the same metastore transaction that writes the record, and linking writes the config row in a transaction that checks the fence. Registering a schema migration job still checks the schema and every namespace linked to it (under the schema's exclusive lock) and registers nothing if any of them denies lifecycle work, so a link made by a binary that does not know fences cannot lead to a migration step being refused at a fenced namespace's WAL halfway through a job. + +### 13.5 Lifecycle interlocks (implementation) + +`NamespaceStore::check_lifecycle` is the one check: the namespace's gate in the registry must permit `Lifecycle` (section 3.3). It reads the registry only, so it never loads or creates the namespace, and it sees the in-memory gates as well as the durable state: a closing transition being installed, a target being created, an indeterminate commit and an unavailable state all refuse lifecycle work. A name without a controller has no fence state and is left to the existing checks. + +| Operation | Where it is refused | +|---|---| +| Config `POST` | Before the namespace is loaded (`http/admin/mod.rs`), and in the metastore transaction that would store it (`try_process`). A config written without a flush is only a fork destination's, which is checked below. | +| Delete | In the metastore transaction of `MetaStore::remove`. | +| Create over an existing record, with or without `dump_url`, and linking to a shared schema at creation | Before a dump is fetched (`http/admin/mod.rs`), first in `NamespaceStore::create`, and in the metastore transaction that stores the config. | +| Fork, as source | Under the source's transition lock, held for the whole fork. A fork reads the source's log without a read lease; holding the lock means a write or read fence command on the source either finished before the check (and the fork is refused) or starts after the fork has finished. | +| Fork, as destination | Before anything is stored, and again with the destination's namespace entry held, which no fence command can load past. Without it, a fork onto an existing fenced namespace that is not loaded would publish its config in memory and replace its directory. | +| Reset (the replica's reset callback) | Under the namespace's transition lock, before its entry is taken and its data destroyed. Reset writes nothing to the metastore, so this is its only check; a refused reset is logged. | +| Schema migration | When the job is registered (section 13.4). | + +Lock order is transition lock, then namespace entry, everywhere: fence commands that load a namespace (`CreateTargetQuarantined`) take them in that order, `AcquireSourceWriteFence` releases the entry before it takes the transition lock, and fork and reset take the transition lock first. + ## 14. Code-path coverage How each path that can reach namespace data or lifecycle is covered. File references are to `libsql-server/src/`. @@ -726,14 +744,14 @@ How each path that can reach namespace data or lifecycle is covered. File refere | `rpc/proxy.rs`, `rpc/streaming_exec.rs`, `rpc/replica_proxy.rs`, `connection/write_proxy.rs` | `stable_code` set on the primary and carried back on the replica; connection-creation denials are `FAILED_PRECONDITION`, not `UNAVAILABLE`. | | `namespace/meta_store.rs` `handle()`, `restore()`, `maybe_recover_from_fs`, `destroy_on_error`, `process`/`try_process`, `remove`, bottomless metastore restore | Non-creating lookups; fail-closed decoding; marker-aware recovery; rename-aside; publish only after commit; fence check inside the config and remove transactions; restore provenance surfaced. | | `namespace/store.rs` `with`, `load_namespace`, `make_namespace`, eviction | Registry check before setup; controller passed into setup; eviction keeps the registry entry. | -| `store.rs` `create`, `destroy`, `reset`, `fork`, `checkpoint`, restore options | Create refuses names with a record; `CreateTargetQuarantined` is the atomic quarantined create; destroy, reset, fork (either side) and any restore are denied while lifecycle is denied; checkpoint uses a non-creating lookup and skips vacuum. | -| `http/admin/mod.rs` config, create, fork, delete, checkpoint, stats | Config POST, create, fork, delete follow the lifecycle column; config GET, stats and checkpoint are allowed. Fence routes live in `http/admin/fence.rs`. | +| `store.rs` `create`, `destroy`, `reset`, `fork`, `checkpoint`, restore options | `check_lifecycle` (section 13.5): create refuses a name whose fence denies lifecycle work; `CreateTargetQuarantined` is the atomic quarantined create; destroy is refused in the metastore transaction; reset and fork as source are checked under the namespace's transition lock; fork's destination is checked before anything is stored and again under its entry lock; every restore (create with a dump, reset, fork to a point in time) is one of these paths and is refused there; checkpoint uses a non-creating lookup and skips vacuum. | +| `http/admin/mod.rs` config, create, fork, delete, checkpoint, stats | Config POST and create (before a `dump_url` is fetched) are checked before the namespace is loaded; fork and delete are refused by the store; all of them return the fence error with `423`. Config GET, stats and checkpoint are allowed. Fence routes live in `http/admin/fence.rs`. | | `http/admin/fence.rs` | Fence routes (section 4.5): capability discovery, `InspectFence`, one route per command through `NamespaceStore::execute_fence_command`, and `validation-query` through `open_validation_session`; admin auth key and primary required for every route that changes state or uses a capability. | | `http/user/dump.rs`, `connection/dump/exporter.rs` | Gate check (`Stream`) before connection creation (typed, no panic on create error); `Dump` lease held by the export; cancel checked before every row and by the pipe's pending write; the body stream ends with the fence error (aborted response) on cancellation. | | `rpc/replication/replication_log.rs` `hello`, `log_entries`, `batch_log_entries`, `snapshot` | Denied at request start (`FAILED_PRECONDITION` + `x-libsql-fence-code`, counted, rate-limited log); `FencedStream` replication leases for both streams and the batch; typed terminal status when the gate closes or at the deadline, lease released without the peer; `ReplicatedFence` in `hello`'s config (planned, section 6.2). | | `admin_shell.rs` | Writes denied at the WAL (no capability); reads checked against the gate, and a read lease held, per query (cancelled through the connection's interrupt handle). | -| `schema/scheduler.rs`, `database/schema.rs` | Shared schema excluded from fencing; migration writes are WAL-gated; the scheduler's `block_writes` flag is not treated as drain evidence. | -| `namespace/configurator/helpers.rs` `load_dump`, `http/admin/mod.rs` `dump_stream_from_url` | Restore options and dump URLs are refused for fenced namespaces (planned, section 13.4 lifecycle work); outside a capability the loader's writes are refused at the WAL. Import goes through `ImportSession::load_dump`, which runs the same loader (`load_dump_sql`) on the capability connection. | +| `schema/scheduler.rs`, `database/schema.rs` | Shared schema excluded from fencing; registering a migration job is refused while the schema or a linked namespace denies lifecycle work (section 13.4); migration writes are WAL-gated; the scheduler's `block_writes` flag is not treated as drain evidence. | +| `namespace/configurator/helpers.rs` `load_dump`, `http/admin/mod.rs` `dump_stream_from_url` | Restore options and dump URLs are refused for fenced namespaces before the dump is fetched (section 13.5); outside a capability the loader's writes are refused at the WAL. Import goes through `ImportSession::load_dump`, which runs the same loader (`load_dump_sql`) on the capability connection. | | `connection/program.rs` ATTACH resolution | `check_program_auth` uses a non-creating lookup; the resolver returns the attached namespace's controller, the attachment is admitted as `NormalRead` of it with a read lease, and the connection keeps a lease on it for every later program until it is detached. | | `http/user/listen.rs` `/beta/listen` | `NormalRead`, read from the registry without loading the namespace; denied where reads are denied; the stream ends with an error event when reads are fenced. | | Raw internal connections (storage monitor, periodic checkpoint, shutdown checkpoint, replication logger, `checkpoint_db`, bottomless) | `Maintenance` / `Observability`; not leases; never blocked. | @@ -773,7 +791,7 @@ Planned test names; the table is updated as tests land. |---|---|---| | 1 | Concurrent acquisition by two operations: one owner, typed conflict for the loser | landed: `namespace::fence::drain::tests::acquire_race_single_owner` (the first acquisition is parked after closing admission while the second waits on the transition lock); over the admin API: `tests::fence::admin::concurrent_acquire_one_owner` (two operations acquire at once over HTTP: one `200 APPLIED`, the other `409 FENCE_OWNED_BY_ANOTHER_OPERATION` with the winner in its fence view); admin walks: `tests::fence::admin::{source_walk_over_http, target_walk_over_http, inspect_reports_drain_counters, mutating_routes_require_admin_key}` | | 2 | Active writer commits or is rolled back before freeze acknowledgement; nothing commits after | landed: `namespace::fence::drain::tests::{active_writer_commits_before_ack, forced_rollback_before_ack, no_commit_after_ack}` (the boundary equals the last committed replication frame and no frame follows it; autocommit, `BEGIN IMMEDIATE`, DDL and a pre-fence read transaction upgrading are refused), `installing_gate_closes_writes_before_persisting`, `refused_acquire_reopens_writes`, `release_reopens_with_new_generation` | -| 3 | Autocommit, explicit transactions, queued writers, batches, DDL, schema jobs, old WebSockets, read-to-write upgrades cannot bypass | landed: `connection::connection_manager::fence_tests::{fence_rejects_read_to_write_upgrade, fence_rejects_ddl_and_pragma, fence_rejects_raw_with_raw_write}` (autocommit, explicit transactions, DDL, header-writing pragma, `BEGIN IMMEDIATE`, `VACUUM`, `with_raw` users); `connection::connection_manager::fence_tests::fence_rejects_queued_writer` (a writer parked in the queue behind an open transaction leaves it with `MIGRATION_WRITE_FENCED` when the fence changes, and the holder keeps the slot); maintenance and vacuum under a fence: `queued_checkpoint_survives_fence_wake`, `checkpoint_allowed_while_fenced`, `vacuum_skipped_while_fenced`; drain primitives: `abort_active_tolerates_closed_connection`, `release_notifies_drain_waiters`, `fence::controller::tests::write_queues_are_woken_on_every_generation_change`; planned: `tests::fence::protocol::old_ws_session_cannot_write`, `batch_denied_mid_batch`; `fence::tests::acquire_rejects_shared_schema` | +| 3 | Autocommit, explicit transactions, queued writers, batches, DDL, schema jobs, old WebSockets, read-to-write upgrades cannot bypass | landed: `connection::connection_manager::fence_tests::{fence_rejects_read_to_write_upgrade, fence_rejects_ddl_and_pragma, fence_rejects_raw_with_raw_write}` (autocommit, explicit transactions, DDL, header-writing pragma, `BEGIN IMMEDIATE`, `VACUUM`, `with_raw` users); `connection::connection_manager::fence_tests::fence_rejects_queued_writer` (a writer parked in the queue behind an open transaction leaves it with `MIGRATION_WRITE_FENCED` when the fence changes, and the holder keeps the slot); maintenance and vacuum under a fence: `queued_checkpoint_survives_fence_wake`, `checkpoint_allowed_while_fenced`, `vacuum_skipped_while_fenced`; drain primitives: `abort_active_tolerates_closed_connection`, `release_notifies_drain_waiters`, `fence::controller::tests::write_queues_are_woken_on_every_generation_change`; schema jobs: `schema::scheduler::test::fence::{acquire_rejects_shared_schema (a shared schema and a linked namespace cannot be fenced; a fenced namespace cannot be linked by create or config), migration_not_registered_while_linked_namespace_fenced}`; planned: `tests::fence::protocol::old_ws_session_cannot_write`, `batch_denied_mid_batch` | | 4 | Program that captured config before the fence is rejected at the WAL | landed: `connection::connection_manager::fence_tests::wal_gate_rejects_program_admitted_before_fence` (a SQL function parks the program between admission and its write while the fence is acquired and released) | | 5 | Pre-fence transactions cannot write after release or publication | landed: `connection::connection_manager::fence_tests::stale_generation_cannot_write_after_release`, `namespace::fence::target::tests::stale_generation_cannot_write_after_enable_writes`; unfenced behaviour unchanged: `unfenced_namespace_is_unchanged` | | 6 | Acquisition timeout returns `DRAINING`, admission stays closed | landed: `namespace::fence::drain::tests::{deadline_returns_draining_and_stays_closed, replay_of_draining_resumes_and_completes}` | @@ -787,7 +805,7 @@ Planned test names; the table is updated as tests land. | 14 | Enable writes idempotent, survives restart and response loss, irreversible | landed: `namespace::fence::target::tests::{enable_writes_idempotent_and_irreversible, enable_writes_survives_restart}` | | 15 | Lost `EnableTargetWrites` response resolved from receipt/state | landed: `namespace::fence::target::tests::enable_writes_response_loss_resolved` | | 16 | Read fence drains SQL, dump, `log_entries`, `snapshot`, including dead peers and forced termination | SQL landed: `namespace::fence::read::tests::{read_fence_waits_for_running_program, program_after_closing_gate_is_refused (parked after the read-closing gate, before the CAS), read_fence_cancels_at_deadline, unreleased_lease_answers_draining_and_replay_completes, idle_txn_fails_on_next_program, clear_read_fence_reopens_reads_not_writes, refused_read_fence_reopens_reads, attach_of_read_fenced_namespace_denied}`, `admin_shell::fence_tests::admin_shell_read_denied`; dump and replication landed: `namespace::fence::stream::tests::{dump_lease_released_on_cancel (a dump blocked mid-row on a peer that stopped reading is cancelled at the deadline, its lease released without the peer, the fence acknowledged, and the body ends with the fence error and no `COMMIT;`), read_fence_waits_for_dump, dump_refused_while_read_fenced, log_entries_stream_ends_typed, stream_lease_released_without_peer_read (a dead peer), snapshot_stream_ends_typed, replication_calls_denied_while_read_fenced, read_fence_forced_termination}`; planned: the HTTP-level check that an interrupted `/dump` response is aborted rather than completed, with `tests::fence::protocol::dump_codes` | -| 17 | Delete, reset, fork, restore, config, schema mutation rejected | `tests::fence::lifecycle::lifecycle_rejected_while_fenced` | +| 17 | Delete, reset, fork, restore, config, schema mutation rejected | landed: `tests::fence::lifecycle::lifecycle_rejected_while_fenced` (over the admin API, for a write-fenced source and a quarantined target: delete, fork as source and as destination, create with a `dump_url` whose file does not exist, create over the record, linking to a shared schema at creation and config `POST` are all `423` with the fence code; state, revision and data are unchanged and no copy exists; a target cannot be created with a shared schema; after release the source takes writes and config, fork and delete work again); `namespace::store::fence_tests::{reset_refused_while_fenced (called directly and as the replicator's reset callback; after release reset works and wipes the data), lifecycle_refused_while_fenced (fork either side, create over, delete, config and shared-schema link in the metastore transaction)}`; `schema::scheduler::test::fence::{acquire_rejects_shared_schema, migration_not_registered_while_linked_namespace_fenced}` | | 18 | Codes through HTTP, Hrana, RPC, dump, replication, replica write proxy; distinguishable from auth/timeout/not-found; old peers compatible; no retry loops | `tests::fence::protocol::{http_codes, hrana_http_codes, hrana_ws_codes, rpc_codes, dump_codes, replication_codes, replica_proxy_preserves_code, auth_and_not_found_distinct, denial_not_retried}`; `libsql-replication` `proxy_error_stable_code_is_additive` | | 19 | Corrupt or unknown durable fence state fails closed | `fence::store::tests::corrupt_payload_fails_closed`, `unknown_format_version_fails_closed` | | 20 | Metrics and audit logs | `tests::fence::observability::metrics_and_labels`; `fence::audit::tests::audit_event_fields` | diff --git a/libsql-server/src/http/admin/mod.rs b/libsql-server/src/http/admin/mod.rs index e9a25b33af..c54461184f 100644 --- a/libsql-server/src/http/admin/mod.rs +++ b/libsql-server/src/http/admin/mod.rs @@ -328,10 +328,11 @@ async fn handle_post_config( // Check that the jwt keys are correct parse_jwt_keys(jwt_key)?; } - let store = app_state - .namespaces - .config_store(NamespaceName::from_string(namespace.clone())?) - .await?; + let namespace_name = NamespaceName::from_string(namespace.clone())?; + // Config mutation is lifecycle work: refused while a fence denies it, before the namespace + // is loaded (and again in the metastore transaction that would store it). + app_state.namespaces.check_lifecycle(&namespace_name)?; + let store = app_state.namespaces.config_store(namespace_name).await?; let original = (*store.get()).clone(); let mut updated = original.clone(); updated.block_reads = req.block_reads; @@ -398,6 +399,10 @@ async fn handle_create_namespace( ) -> crate::Result<()> { let mut config = DatabaseConfig::default(); + // Creating over a name whose fence denies lifecycle work is refused before a dump is + // fetched or anything is stored. + app_state.namespaces.check_lifecycle(&namespace)?; + if let Some(jwt_key) = req.jwt_key { // Check that the jwt keys are correct parse_jwt_keys(&jwt_key)?; diff --git a/libsql-server/src/namespace/fence/registry.rs b/libsql-server/src/namespace/fence/registry.rs index 176cfddd4e..24dd0ea23e 100644 --- a/libsql-server/src/namespace/fence/registry.rs +++ b/libsql-server/src/namespace/fence/registry.rs @@ -82,6 +82,17 @@ impl FenceRegistry { } } + /// Refuse generic lifecycle and configuration work on `namespace` while its gate denies it + /// (section 3.3, the lifecycle column): an active fence, a closing transition being + /// installed, a target being created, an indeterminate commit or an unavailable state. A + /// name without a controller has no fence state and is not refused here. + pub fn check_lifecycle(&self, namespace: &NamespaceName) -> Result<(), FenceError> { + match self.get(namespace) { + Some(controller) => controller.gate().permits(OperationClass::Lifecycle), + None => Ok(()), + } + } + /// How many namespaces have an active fence (`docs/NAMESPACE_FENCE.md` section 4.4, /// `active_fences`): a record in any state but `RELEASED` or `TARGET_WRITABLE`, an /// unavailable state, a target being created, or a commit whose outcome is not known yet. diff --git a/libsql-server/src/namespace/store.rs b/libsql-server/src/namespace/store.rs index 09d27a5695..9a83252b53 100644 --- a/libsql-server/src/namespace/store.rs +++ b/libsql-server/src/namespace/store.rs @@ -182,6 +182,17 @@ impl NamespaceStore { namespace: NamespaceName, restore_option: RestoreOption, ) -> anyhow::Result<()> { + // Reset destroys the namespace's data and writes nothing to the metastore, so the fence + // check is made here, under the namespace's transition lock: a fence command either + // finished before this check or starts after the reset (section 3.3, lifecycle). + let _transition = self + .inner + .fences + .controller(&namespace) + .begin_transition() + .await; + self.check_lifecycle(&namespace)?; + // The process for reseting is as follow: // - get a lock on the namespace entry, if the entry exists, then it's a lock on the entry, // if it doesn't exist, insert an empty entry and take a lock on it @@ -221,18 +232,26 @@ impl NamespaceStore { Box::new(move |op| { let this = this.clone(); tokio::spawn(async move { - match op { - ResetOp::Reset(ns) => { - tracing::info!("received reset signal for: {ns}"); - if let Err(e) = this.reset(ns.clone(), RestoreOption::Latest).await { - tracing::error!("error resetting namespace `{ns}`: {e}"); - } - } - } + let _ = this.handle_reset_op(op).await; }); }) } + /// A reset requested by a replica's replicator. A namespace whose fence denies lifecycle + /// work is not reset: the refusal is logged and returned. + async fn handle_reset_op(&self, op: ResetOp) -> anyhow::Result<()> { + match op { + ResetOp::Reset(ns) => { + tracing::info!("received reset signal for: {ns}"); + let result = self.reset(ns.clone(), RestoreOption::Latest).await; + if let Err(e) = &result { + tracing::error!("error resetting namespace `{ns}`: {e}"); + } + result + } + } + } + pub async fn fork( &self, from: NamespaceName, @@ -245,14 +264,23 @@ impl NamespaceStore { } // The destination is refused before anything is stored for it when it is being created - // as a migration target or its fence state is unknown. + // as a migration target, its fence state is unknown, or its fence denies lifecycle work + // (an existing fenced namespace, whose directory the fork would otherwise replace). self.inner.fences.check_available(&to)?; + self.check_lifecycle(&to)?; // check that the source namespace exists if !self.inner.metadata.exists(&from).await { return Err(crate::error::Error::NamespaceDoesntExist(from.to_string())); } + // A fork reads the source's data without a read lease, so it runs under the source's + // transition lock and checks the source's gate under it: a fence command on the source + // (a write or read fence) either finished before this check, and the fork is refused, + // or waits for the fork to finish (section 3.3, fork as source). + let _from_transition = self.inner.fences.controller(&from).begin_transition().await; + self.check_lifecycle(&from)?; + let to_entry = self .inner .store @@ -262,6 +290,9 @@ impl NamespaceStore { if to_lock.is_some() { return Err(crate::error::Error::NamespaceAlreadyExist(to.to_string())); } + // With the destination's entry held, a fence command cannot load the destination, so + // the check cannot go stale before the fork has stored and flushed its config. + self.check_lifecycle(&to)?; // FIXME: we could potentially delete the namespace while trying to fork it if !self.inner.metadata.exists(&from).await { @@ -462,8 +493,10 @@ impl NamespaceStore { db_config: DatabaseConfig, ) -> crate::Result<()> { // A name that is being created as a migration target, or whose fence state is unknown, - // is refused before anything is stored for it. + // is refused before anything is stored for it; so is a name whose fence denies lifecycle + // work (creating over an existing record, with or without a restore). self.inner.fences.check_available(&namespace)?; + self.check_lifecycle(&namespace)?; if let Some(shared_schema_name) = &db_config.shared_schema_name { // we hold a lock for the duration of the namespace creation let _lock = self @@ -552,6 +585,16 @@ impl NamespaceStore { &self.inner.metadata } + /// Refuse generic lifecycle and configuration work on `namespace` (config mutation, + /// delete, reset, fork on either side, create over an existing record, restore, dump load, + /// shared-schema linking, schema migration) while its fence denies it + /// (`docs/NAMESPACE_FENCE.md` section 3.3), without loading the namespace. A name without + /// fence state is not refused here: the existing checks apply to it. Paths that persist + /// through the metastore are refused again inside its transaction. + pub(crate) fn check_lifecycle(&self, namespace: &NamespaceName) -> crate::Result<()> { + Ok(self.inner.fences.check_lifecycle(namespace)?) + } + /// Run one fence command on its namespace, including the drain it starts /// (`docs/NAMESPACE_FENCE.md` sections 5.3 and 8). `AcquireSourceWriteFence` loads the /// namespace first, so that its connection manager and replication log are registered with @@ -947,6 +990,7 @@ pub(crate) mod fence_tests { use super::*; use crate::config::MetaStoreConfig; + use crate::connection::Connection as _; use crate::namespace::configurator::{BaseNamespaceConfig, PrimaryConfig, PrimaryConfigurator}; use crate::namespace::fence::command::{FenceCommand, FenceRequest}; use crate::namespace::fence::outcome::{FenceDetail, FenceOutcome}; @@ -1148,4 +1192,208 @@ pub(crate) mod fence_tests { store.destroy("ns".into(), false).await.unwrap(); assert!(store.inner.fences.get(&"ns".into()).is_none()); } + + fn release(ns: &'static str, command_id: u128) -> FenceRequest { + FenceRequest { + namespace: ns.into(), + operation_id: OP, + command_id: Uuid::from_u128(command_id), + expected_state: FenceState::SourceDraining, + expected_revision: 1, + command: FenceCommand::ReleaseSourceWriteFence, + } + } + + /// Create `ns` holding a table `t` with one row. + async fn create_with_row(store: &NamespaceStore, ns: &'static str) { + store + .create(ns.into(), RestoreOption::Latest, Default::default()) + .await + .unwrap(); + let conn = store + .with(ns.into(), |ns| ns.db.connection_maker()) + .await + .unwrap() + .create() + .await + .unwrap(); + tokio::task::spawn_blocking(move || { + conn.with_raw(|c| c.execute_batch("create table t (x); insert into t values (1);")) + }) + .await + .unwrap() + .unwrap(); + } + + /// The number of rows in `ns`'s table `t`, or the error reading it. + async fn rows(store: &NamespaceStore, ns: &'static str) -> rusqlite::Result { + let conn = store + .with(ns.into(), |ns| ns.db.connection_maker()) + .await + .unwrap() + .create() + .await + .unwrap(); + tokio::task::spawn_blocking(move || { + conn.with_raw(|c| c.query_row("select count(*) from t", (), |r| r.get(0))) + }) + .await + .unwrap() + } + + #[track_caller] + fn assert_fenced(result: crate::Result<()>, outcome: FenceOutcome) { + match result { + Err(Error::NamespaceFence(e)) => assert_eq!(e.outcome(), outcome, "{e}"), + other => panic!("expected {outcome}, got {other:?}"), + } + } + + #[track_caller] + fn assert_fenced_anyhow(result: anyhow::Result<()>, outcome: FenceOutcome) { + match result { + Err(e) => match e.downcast_ref::() { + Some(Error::NamespaceFence(e)) => assert_eq!(e.outcome(), outcome, "{e}"), + _ => panic!("expected {outcome}, got {e:?}"), + }, + Ok(()) => panic!("expected {outcome}, got Ok"), + } + } + + /// Reset, which destroys the namespace's data and recreates it, is refused while the + /// namespace is fenced, both called directly and as the replicator's reset callback does. + #[tokio::test(flavor = "multi_thread")] + async fn reset_refused_while_fenced() { + let tmp = tempdir().unwrap(); + let store = open_store(tmp.path()).await; + create_with_row(&store, "ns").await; + let fence = store.inner.fences.controller(&"ns".into()); + fence + .apply_command(store.meta_store(), acquire("ns"), ctx()) + .await + .unwrap(); + + assert_fenced_anyhow( + store.reset("ns".into(), RestoreOption::Latest).await, + FenceOutcome::MigrationWriteFenced, + ); + assert_fenced_anyhow( + store.handle_reset_op(ResetOp::Reset("ns".into())).await, + FenceOutcome::MigrationWriteFenced, + ); + // The namespace was not touched and still serves reads. + assert_eq!(rows(&store, "ns").await.unwrap(), 1); + + // Once the fence is released, reset works as before, and its data is gone. + fence + .apply_command(store.meta_store(), release("ns", 2), ctx()) + .await + .unwrap(); + assert_eq!(fence.gate().state(), FenceState::Released); + store + .handle_reset_op(ResetOp::Reset("ns".into())) + .await + .unwrap(); + assert!(rows(&store, "ns").await.is_err()); + } + + /// Fork is lifecycle work on both sides: a fenced source is not copied, and a fenced + /// destination (whose directory a fork would replace) is not overwritten. Create over a + /// fenced name, delete and config mutation, including linking the namespace to a shared + /// schema, are refused too. + #[tokio::test(flavor = "multi_thread")] + async fn lifecycle_refused_while_fenced() { + let tmp = tempdir().unwrap(); + let store = open_store(tmp.path()).await; + create_with_row(&store, "src").await; + create_with_row(&store, "other").await; + let fence = store.inner.fences.controller(&"src".into()); + fence + .apply_command(store.meta_store(), acquire("src"), ctx()) + .await + .unwrap(); + let write_fenced = FenceOutcome::MigrationWriteFenced; + + // Fork with the fenced namespace as the source: nothing is created. + assert_fenced( + store + .fork("src".into(), "copy".into(), Default::default(), None) + .await, + write_fenced, + ); + assert!(!store.exists(&"copy".into()).await); + assert!(!tmp.path().join("dbs").join("copy").exists()); + // Fork onto the fenced namespace: its data and config are untouched. + let config_before = store.config_store("src".into()).await.unwrap().get(); + assert_fenced( + store + .fork( + "other".into(), + "src".into(), + DatabaseConfig { + block_reason: Some("fork".into()), + ..Default::default() + }, + None, + ) + .await, + write_fenced, + ); + assert_eq!(rows(&store, "src").await.unwrap(), 1); + let config_after = store.config_store("src".into()).await.unwrap().get(); + assert_eq!(config_after.block_reason, config_before.block_reason); + assert_eq!(config_after.block_writes, config_before.block_writes); + + // Create over it, with or without a restore, and delete. + assert_fenced( + store + .create("src".into(), RestoreOption::Latest, Default::default()) + .await, + write_fenced, + ); + assert_fenced(store.destroy("src".into(), false).await, write_fenced); + + // Config mutation, including linking the namespace to a shared schema, is refused in the + // metastore transaction that would store it. + let handle = store.config_store("src".into()).await.unwrap(); + assert_fenced( + handle + .store(DatabaseConfig { + block_reason: Some("changed".into()), + ..Default::default() + }) + .await, + write_fenced, + ); + assert_fenced( + handle + .store(DatabaseConfig { + shared_schema_name: Some("other".into()), + ..Default::default() + }) + .await, + write_fenced, + ); + assert_eq!(handle.get().block_reason, config_before.block_reason); + assert!(handle.get().shared_schema_name.is_none()); + assert_eq!(rows(&store, "src").await.unwrap(), 1); + + // After release the same operations follow the existing policy again. + fence + .apply_command(store.meta_store(), release("src", 2), ctx()) + .await + .unwrap(); + store + .fork("src".into(), "copy".into(), Default::default(), None) + .await + .unwrap(); + assert_eq!(rows(&store, "copy").await.unwrap(), 1); + assert!(matches!( + store + .create("src".into(), RestoreOption::Latest, Default::default()) + .await, + Err(Error::NamespaceAlreadyExist(_)) + )); + store.destroy("src".into(), false).await.unwrap(); + } } diff --git a/libsql-server/src/schema/db.rs b/libsql-server/src/schema/db.rs index d0bce10128..aef39b7be0 100644 --- a/libsql-server/src/schema/db.rs +++ b/libsql-server/src/schema/db.rs @@ -110,6 +110,23 @@ pub(crate) fn schema_has_linked_dbs( Ok(has_linked) } +/// The namespaces linked to `schema`. +pub(crate) fn linked_namespaces( + conn: &rusqlite::Connection, + schema: &NamespaceName, +) -> Result, Error> { + let mut stmt = + conn.prepare("SELECT namespace FROM shared_schema_links WHERE shared_schema_name = ?")?; + let names = stmt + .query_map([schema.as_str()], |row| row.get::<_, String>(0))? + .collect::>>()?; + // A link whose name does not decode cannot name a fenced namespace. + Ok(names + .into_iter() + .filter_map(|name| NamespaceName::from_string(name).ok()) + .collect()) +} + /// Create a migration job, and returns the job_id pub(super) fn register_schema_migration_job( conn: &mut rusqlite::Connection, diff --git a/libsql-server/src/schema/error.rs b/libsql-server/src/schema/error.rs index 13f21f3c15..528251a2b3 100644 --- a/libsql-server/src/schema/error.rs +++ b/libsql-server/src/schema/error.rs @@ -45,6 +45,8 @@ pub enum Error { InteractiveTxnNotAllowed, #[error("Connection left in transaction state")] ConnectionInTxnState, + #[error("{0}")] + NamespaceFence(#[from] crate::namespace::fence::outcome::FenceError), } impl ResponseError for Error {} @@ -58,6 +60,7 @@ impl IntoResponse for &Error { self.format_err(StatusCode::BAD_REQUEST) } Error::MigrationExecuteError(e) => e.as_ref().into_response(), + Error::NamespaceFence(e) => self.format_err(e.outcome().admin_http_status()), _ => self.format_err(StatusCode::INTERNAL_SERVER_ERROR), } } diff --git a/libsql-server/src/schema/scheduler.rs b/libsql-server/src/schema/scheduler.rs index d9431b2d86..b9844b7373 100644 --- a/libsql-server/src/schema/scheduler.rs +++ b/libsql-server/src/schema/scheduler.rs @@ -410,6 +410,24 @@ impl Scheduler { .schema_locks() .acquire_exlusive(schema.clone()) .await; + // Schema migration is lifecycle work on the schema and on every namespace linked to it. + // Fences refuse shared schemas and linked namespaces, and linking a fenced namespace is + // refused, so this finds nothing in normal operation; if it does (a link made by a binary + // that does not know fences), no job is registered rather than a migration step being + // refused at a fenced namespace's WAL. + self.namespace_store + .check_lifecycle(&schema) + .map_err(fence_error)?; + let linked = with_conn_async(self.migration_db.clone(), { + let schema = schema.clone(); + move |conn| super::db::linked_namespaces(conn, &schema) + }) + .await?; + for namespace in &linked { + self.namespace_store + .check_lifecycle(namespace) + .map_err(fence_error)?; + } with_conn_async(self.migration_db.clone(), move |conn| { register_schema_migration_job(conn, &schema, &migration) }) @@ -427,6 +445,14 @@ impl Scheduler { } } +/// The schema error for a fence refusal returned by `NamespaceStore::check_lifecycle`. +fn fence_error(e: crate::Error) -> Error { + match e { + crate::Error::NamespaceFence(e) => Error::NamespaceFence(e), + e => Error::Registration(e.into()), + } +} + async fn try_step_task( _permit: OwnedSemaphorePermit, namespace_store: NamespaceStore, @@ -1229,4 +1255,236 @@ mod test { .is_err()); } } + + /// Namespace fences and shared schemas (`docs/NAMESPACE_FENCE.md` section 13.4). + mod fence { + use uuid::Uuid; + + use super::*; + use crate::config::MetaStoreConfig; + use crate::namespace::fence::command::{FenceCommand, FenceRequest}; + use crate::namespace::fence::outcome::{FenceDetail, FenceOutcome}; + use crate::namespace::fence::record::ServerIdentity; + use crate::namespace::fence::state::FenceState; + use crate::namespace::meta_store::FenceContext; + + const LOG: Uuid = Uuid::from_u128(0x10); + const OP: Uuid = Uuid::from_u128(0xa); + + fn server() -> ServerIdentity { + ServerIdentity { + build: "test".into(), + instance_id: Uuid::from_u128(0x99), + } + } + + fn acquire(ns: &'static str, command_id: u128) -> FenceRequest { + FenceRequest { + namespace: ns.into(), + operation_id: OP, + command_id: Uuid::from_u128(command_id), + expected_state: FenceState::Unfenced, + expected_revision: 0, + command: FenceCommand::AcquireSourceWriteFence { + expected_log_id: LOG, + drain_policy: None, + }, + } + } + + fn release(ns: &'static str, command_id: u128) -> FenceRequest { + FenceRequest { + namespace: ns.into(), + operation_id: OP, + command_id: Uuid::from_u128(command_id), + expected_state: FenceState::SourceDraining, + expected_revision: 1, + command: FenceCommand::ReleaseSourceWriteFence, + } + } + + /// A primary store with fences enabled and a shared schema `schema` with one linked + /// namespace `linked`. + async fn setup( + path: &Path, + ) -> (NamespaceStore, Scheduler, mpsc::Receiver) { + let (maker, manager) = metastore_connection_maker(None, path).await.unwrap(); + let meta_store = MetaStore::new( + MetaStoreConfig { + namespace_fence: true, + ..Default::default() + }, + path, + maker().unwrap(), + manager, + DatabaseKind::Primary, + ) + .await + .unwrap(); + let (sender, receiver) = mpsc::channel(100); + let config = make_config(sender.into(), path); + let store = + NamespaceStore::new(false, false, 10, meta_store, config, DatabaseKind::Primary) + .await + .unwrap(); + let scheduler = Scheduler::new(store.clone(), maker().unwrap()) + .await + .unwrap(); + store + .create( + "schema".into(), + RestoreOption::Latest, + DatabaseConfig { + is_shared_schema: true, + ..Default::default() + }, + ) + .await + .unwrap(); + store + .create( + "linked".into(), + RestoreOption::Latest, + DatabaseConfig { + shared_schema_name: Some("schema".into()), + ..Default::default() + }, + ) + .await + .unwrap(); + (store, scheduler, receiver) + } + + /// Fence `ns` at the metastore (`SOURCE_DRAINING`, write admission closed). + async fn fence(store: &NamespaceStore, ns: &'static str) { + store + .fence_controller(&ns.into()) + .apply_command( + store.meta_store(), + acquire(ns, 1), + FenceContext::now(server(), Some(LOG)), + ) + .await + .unwrap(); + } + + #[track_caller] + fn assert_fence_error(result: crate::Result<()>, outcome: FenceOutcome) { + match result { + Err(crate::Error::NamespaceFence(e)) => assert_eq!(e.outcome(), outcome, "{e}"), + other => panic!("expected {outcome}, got {other:?}"), + } + } + + /// A shared schema and a namespace linked to one cannot be fenced, and a fenced + /// namespace cannot be linked to a shared schema. + #[tokio::test(flavor = "multi_thread")] + async fn acquire_rejects_shared_schema() { + let tmp = tempdir().unwrap(); + let (store, scheduler, _receiver) = setup(tmp.path()).await; + + for ns in ["schema", "linked"] { + match store.execute_fence_command(acquire(ns, 1), server()).await { + Err(crate::Error::NamespaceFence(e)) => { + assert_eq!(e.outcome(), FenceOutcome::FencePreconditionFailed, "{e}"); + assert_eq!(e.detail(), Some(FenceDetail::SharedSchemaUnsupported)); + } + other => panic!("{ns}: expected shared_schema_unsupported, got {other:?}"), + } + // The refused acquisition left the namespace unfenced and writable. + let gate = store.fence_controller(&ns.into()).gate(); + assert_eq!(gate.state(), FenceState::Unfenced); + assert!(gate.write().is_open()); + } + + // A fenced namespace is not linked to the schema, whether by creating it with a + // shared schema or by changing its config. + store + .create("plain".into(), RestoreOption::Latest, Default::default()) + .await + .unwrap(); + fence(&store, "plain").await; + let linked_config = || DatabaseConfig { + shared_schema_name: Some("schema".into()), + ..Default::default() + }; + assert_fence_error( + store + .create("plain".into(), RestoreOption::Latest, linked_config()) + .await, + FenceOutcome::MigrationWriteFenced, + ); + let handle = store.config_store("plain".into()).await.unwrap(); + assert_fence_error( + handle.store(linked_config()).await, + FenceOutcome::MigrationWriteFenced, + ); + assert!(handle.get().shared_schema_name.is_none()); + let links = super::super::super::db::linked_namespaces( + &scheduler.migration_db.lock(), + &"schema".into(), + ) + .unwrap(); + assert_eq!(links, vec![NamespaceName::from("linked")]); + } + + /// A schema migration is lifecycle work on every linked namespace: if a fenced namespace + /// is linked to the schema (here by writing the link directly, as a binary that does not + /// know fences could), no migration job is registered until the fence is released. + #[tokio::test(flavor = "multi_thread")] + async fn migration_not_registered_while_linked_namespace_fenced() { + let tmp = tempdir().unwrap(); + let (store, scheduler, _receiver) = setup(tmp.path()).await; + store + .create("plain".into(), RestoreOption::Latest, Default::default()) + .await + .unwrap(); + fence(&store, "plain").await; + scheduler + .migration_db + .lock() + .execute( + "INSERT INTO shared_schema_links (shared_schema_name, namespace) \ + VALUES ('schema', 'plain')", + (), + ) + .unwrap(); + + let migration = || Program::seq(&["create table test (c)"]).into(); + match scheduler + .register_migration_job("schema".into(), migration()) + .await + { + Err(Error::NamespaceFence(e)) => { + assert_eq!(e.outcome(), FenceOutcome::MigrationWriteFenced, "{e}") + } + other => panic!("expected MIGRATION_WRITE_FENCED, got {other:?}"), + } + assert!(!super::super::super::db::has_pending_migration_jobs( + &scheduler.migration_db.lock(), + &"schema".into(), + ) + .unwrap()); + + // Released, the namespace is ordinary again and the migration is registered. + store + .fence_controller(&"plain".into()) + .apply_command( + store.meta_store(), + release("plain", 2), + FenceContext::now(server(), Some(LOG)), + ) + .await + .unwrap(); + scheduler + .register_migration_job("schema".into(), migration()) + .await + .unwrap(); + assert!(super::super::super::db::has_pending_migration_jobs( + &scheduler.migration_db.lock(), + &"schema".into(), + ) + .unwrap()); + } + } } diff --git a/libsql-server/tests/common/http.rs b/libsql-server/tests/common/http.rs index a08a928478..ccaad17171 100644 --- a/libsql-server/tests/common/http.rs +++ b/libsql-server/tests/common/http.rs @@ -90,12 +90,26 @@ impl Client { &self, url: &str, body: T, + ) -> anyhow::Result { + self.delete_with_headers(url, &[], body).await + } + + pub(crate) async fn delete_with_headers( + &self, + url: &str, + headers: &[(HeaderName, &str)], + body: T, ) -> anyhow::Result { let bytes: Bytes = serde_json::to_vec(&body)?.into(); let body = Body::from(bytes); - let request = hyper::Request::delete(url) + let mut request = hyper::Request::delete(url) .header("Content-Type", "application/json") .body(body)?; + for (key, val) in headers { + request + .headers_mut() + .insert(key.clone(), val.parse().unwrap()); + } let resp = self.0.request(request).await?; Ok(Response(resp)) diff --git a/libsql-server/tests/fence/admin.rs b/libsql-server/tests/fence/admin.rs index d1ac8413c4..2c64935c46 100644 --- a/libsql-server/tests/fence/admin.rs +++ b/libsql-server/tests/fence/admin.rs @@ -1,44 +1,19 @@ //! The fence admin API over HTTP (`docs/NAMESPACE_FENCE.md` section 4). use hyper::StatusCode; -use serde_json::{json, Value}; +use serde_json::json; use tempfile::tempdir; use uuid::Uuid; -use super::{command_body, connect, make_primary, sim, state_of, Admin, Primary, ADMIN_KEY}; +use super::{ + acquire_body, command_body, connect, load_and_log_id, make_primary, sim, state_of, Admin, + Primary, ADMIN_KEY, +}; fn uuid(n: u128) -> Uuid { Uuid::from_u128(n) } -/// Load `ns` on the server with one write, and return the replication log id the server -/// reports for it. -async fn load_and_log_id(admin: &Admin, ns: &str) -> anyhow::Result { - let conn = connect(ns)?; - conn.execute("create table if not exists t (x)", ()).await?; - conn.execute("insert into t values (1)", ()).await?; - let (status, body) = admin.inspect(ns).await?; - assert_eq!(status, StatusCode::OK, "{body}"); - assert_eq!(state_of(&body), ("UNFENCED", 0), "{body}"); - Ok(body["fence"]["incarnation"]["current_log_id"] - .as_str() - .unwrap_or_else(|| panic!("no current_log_id: {body}")) - .to_string()) -} - -fn acquire_body(op: Uuid, cmd: Uuid, log_id: &str) -> Value { - command_body( - op, - cmd, - "UNFENCED", - 0, - json!({ - "expected_namespace_identity": { "log_id": log_id }, - "drain_policy": { "deadline_ms": 5000, "on_deadline": "fail" }, - }), - ) -} - #[test] fn capabilities() { let mut sim = sim(); diff --git a/libsql-server/tests/fence/lifecycle.rs b/libsql-server/tests/fence/lifecycle.rs new file mode 100644 index 0000000000..90ac6d9749 --- /dev/null +++ b/libsql-server/tests/fence/lifecycle.rs @@ -0,0 +1,201 @@ +//! Lifecycle and configuration operations on fenced namespaces, over the admin API +//! (`docs/NAMESPACE_FENCE.md` section 3.3, the lifecycle column; section 17 row 17). + +use hyper::StatusCode; +use libsql::Value as SqlValue; +use serde_json::{json, Value}; +use tempfile::tempdir; +use uuid::Uuid; + +use super::{ + acquire_body, command_body, connect, load_and_log_id, make_primary, sim, state_of, Admin, + Primary, ADMIN_KEY, +}; + +fn uuid(n: u128) -> Uuid { + Uuid::from_u128(n) +} + +/// Every generic lifecycle and configuration route, attempted on `ns`, with a name for the +/// failure message. `ns` must be refused by each of them. +async fn lifecycle_attempts( + admin: &Admin, + ns: &str, +) -> anyhow::Result> { + let mut results = Vec::new(); + let (status, body) = admin.delete(&format!("/v1/namespaces/{ns}")).await?; + results.push(("delete", status, body)); + let (status, body) = admin + .post(&format!("/v1/namespaces/{ns}/fork/{ns}-copy"), json!({})) + .await?; + results.push(("fork as source", status, body)); + let (status, body) = admin + .post(&format!("/v1/namespaces/other/fork/{ns}"), json!({})) + .await?; + results.push(("fork as destination", status, body)); + // The dump file does not exist: a refusal made before the dump is fetched is the fence's. + let (status, body) = admin + .post( + &format!("/v1/namespaces/{ns}/create"), + json!({ "dump_url": "file:///nonexistent/dump.sql" }), + ) + .await?; + results.push(("create with dump_url", status, body)); + let (status, body) = admin + .post(&format!("/v1/namespaces/{ns}/create"), json!({})) + .await?; + results.push(("create over the record", status, body)); + let (status, body) = admin + .post( + &format!("/v1/namespaces/{ns}/create"), + json!({ "shared_schema_name": "schema" }), + ) + .await?; + results.push(("link to a shared schema", status, body)); + let (status, body) = admin + .post( + &format!("/v1/namespaces/{ns}/config"), + json!({ "block_reads": false, "block_writes": false, "block_reason": null }), + ) + .await?; + results.push(("config", status, body)); + Ok(results) +} + +async fn count_rows(ns: &str) -> anyhow::Result { + let mut rows = connect(ns)?.query("select count(*) from t", ()).await?; + let row = rows.next().await?.expect("one row"); + match row.get_value(0)? { + SqlValue::Integer(n) => Ok(n), + other => anyhow::bail!("unexpected count {other:?}"), + } +} + +#[test] +fn lifecycle_rejected_while_fenced() { + let mut sim = sim(); + let tmp = tempdir().unwrap(); + make_primary(&mut sim, tmp.path().to_path_buf(), Primary::default()); + sim.client("client", async { + let admin = Admin::new(Some(ADMIN_KEY)); + admin.create_namespace("other").await?; + let (status, body) = admin + .post( + "/v1/namespaces/schema/create", + json!({ "shared_schema": true }), + ) + .await?; + assert!(status.is_success(), "{status} {body}"); + + // A write-fenced source. + admin.create_namespace("src").await?; + let log_id = load_and_log_id(&admin, "src").await?; + let source_op = uuid(0x100); + let (status, body) = admin + .command( + "src", + "source/acquire-write-fence", + acquire_body(source_op, uuid(1), &log_id), + ) + .await?; + assert_eq!(status, StatusCode::OK, "{body}"); + // SOURCE_DRAINING at revision 1, then SOURCE_WRITE_FENCED once the drain is proven. + assert_eq!(state_of(&body), ("SOURCE_WRITE_FENCED", 2), "{body}"); + + // A quarantined target. A target cannot be created with a shared schema. + let target_op = uuid(0x200); + let (status, body) = admin + .command( + "tgt", + "target/create-quarantined", + command_body( + target_op, + uuid(2), + "ABSENT", + 0, + json!({ "shared_schema_name": "schema" }), + ), + ) + .await?; + assert_eq!(status, StatusCode::PRECONDITION_FAILED, "{body}"); + assert_eq!(body["detail"], "shared_schema_unsupported", "{body}"); + let (status, body) = admin + .command( + "tgt", + "target/create-quarantined", + command_body(target_op, uuid(3), "ABSENT", 0, json!({})), + ) + .await?; + assert_eq!(status, StatusCode::OK, "{body}"); + assert_eq!(state_of(&body), ("TARGET_QUARANTINED", 1), "{body}"); + + for (ns, state, revision, code) in [ + ("src", "SOURCE_WRITE_FENCED", 2, "MIGRATION_WRITE_FENCED"), + ( + "tgt", + "TARGET_QUARANTINED", + 1, + "MIGRATION_TARGET_QUARANTINED", + ), + ] { + for (what, status, body) in lifecycle_attempts(&admin, ns).await? { + assert_eq!(status, StatusCode::LOCKED, "{what} on {ns}: {body}"); + let message = body["error"].as_str().unwrap_or_default(); + assert!( + message.starts_with(code), + "{what} on {ns}: expected {code}, got {body}" + ); + } + // Nothing moved: same state and revision, and no copy was created. + let (status, body) = admin.inspect(ns).await?; + assert_eq!(status, StatusCode::OK, "{body}"); + assert_eq!(state_of(&body), (state, revision), "{body}"); + let (status, body) = admin + .get(&format!("/v1/namespaces/{ns}-copy/config")) + .await?; + assert_eq!(status, StatusCode::NOT_FOUND, "{body}"); + } + // The source's data is intact and still served to readers. + assert_eq!(count_rows("src").await?, 1); + // Reading config and stats is still allowed. + let (status, body) = admin.get("/v1/namespaces/src/config").await?; + assert_eq!(status, StatusCode::OK, "{body}"); + let (status, body) = admin.get("/v1/namespaces/src/stats").await?; + assert_eq!(status, StatusCode::OK, "{body}"); + + // Released, the source is an ordinary namespace again: it takes writes and lifecycle + // operations follow the existing policy. + let (status, body) = admin + .command( + "src", + "source/release-write-fence", + command_body(source_op, uuid(4), "SOURCE_WRITE_FENCED", 2, json!({})), + ) + .await?; + assert_eq!(status, StatusCode::OK, "{body}"); + assert_eq!(state_of(&body).0, "RELEASED", "{body}"); + connect("src")? + .execute("insert into t values (2)", ()) + .await?; + assert_eq!(count_rows("src").await?, 2); + let (status, body) = admin + .post( + "/v1/namespaces/src/config", + json!({ "block_reads": false, "block_writes": false }), + ) + .await?; + assert_eq!(status, StatusCode::OK, "{body}"); + let (status, body) = admin + .post("/v1/namespaces/src/fork/src-copy", json!({})) + .await?; + assert_eq!(status, StatusCode::OK, "{body}"); + assert_eq!(count_rows("src-copy").await?, 2); + let (status, body) = admin.delete("/v1/namespaces/src-copy").await?; + assert_eq!(status, StatusCode::OK, "{body}"); + + // The target stays quarantined: it serves no SQL. + assert!(count_rows("tgt").await.is_err()); + Ok(()) + }); + sim.run().unwrap(); +} diff --git a/libsql-server/tests/fence/mod.rs b/libsql-server/tests/fence/mod.rs index a7a149907f..c8be3cde94 100644 --- a/libsql-server/tests/fence/mod.rs +++ b/libsql-server/tests/fence/mod.rs @@ -3,6 +3,7 @@ //! Namespace fence integration tests (`docs/NAMESPACE_FENCE.md`), driven over the admin API. mod admin; +mod lifecycle; use std::path::PathBuf; use std::time::Duration; @@ -126,6 +127,16 @@ impl Admin { .await } + pub async fn delete(&self, path: &str) -> anyhow::Result<(StatusCode, Value)> { + let url = format!("http://primary:9090{path}"); + Self::json( + self.client + .delete_with_headers(&url, &self.headers(), json!({})) + .await?, + ) + .await + } + pub async fn create_namespace(&self, ns: &str) -> anyhow::Result<()> { let (status, body) = self .post(&format!("/v1/namespaces/{ns}/create"), json!({})) @@ -186,3 +197,32 @@ pub fn connect(ns: &str) -> anyhow::Result { )?; Ok(db.connect()?) } + +/// Load `ns` on the server with one write, and return the replication log id the server +/// reports for it. +pub async fn load_and_log_id(admin: &Admin, ns: &str) -> anyhow::Result { + let conn = connect(ns)?; + conn.execute("create table if not exists t (x)", ()).await?; + conn.execute("insert into t values (1)", ()).await?; + let (status, body) = admin.inspect(ns).await?; + assert_eq!(status, StatusCode::OK, "{body}"); + assert_eq!(state_of(&body), ("UNFENCED", 0), "{body}"); + Ok(body["fence"]["incarnation"]["current_log_id"] + .as_str() + .unwrap_or_else(|| panic!("no current_log_id: {body}")) + .to_string()) +} + +/// An `AcquireSourceWriteFence` body for a namespace in `UNFENCED` at revision 0. +pub fn acquire_body(op: Uuid, cmd: Uuid, log_id: &str) -> Value { + command_body( + op, + cmd, + "UNFENCED", + 0, + json!({ + "expected_namespace_identity": { "log_id": log_id }, + "drain_policy": { "deadline_ms": 5000, "on_deadline": "fail" }, + }), + ) +} From 5043c2a8d65499a190a4417cbad669cda6a66cb1 Mon Sep 17 00:00:00 2001 From: River Date: Wed, 30 Sep 2026 09:26:10 +0000 Subject: [PATCH 20/33] libsql-replication: add stable error code and replicated fence to protocols Two additive proto3 fields for the namespace fence: - `proxy.Error.stable_code` (tag 4): a stable machine-readable outcome such as `MIGRATION_WRITE_FENCED`, so a replica can return the same typed outcome the primary would have. Absent means "no typed outcome"; older peers skip it. - `metadata.DatabaseConfig.fence` (tag 14, `ReplicatedFence { state, revision }`): the live fence as the primary's replication `hello` sees it. It is filled only by `hello` and only while a fence is active, and is never part of a stored configuration. The primary now fills `DatabaseConfig.fence` in `hello` from the namespace's fence gate. Filling `stable_code` on fence denials and mapping it on the replica side come in a later change; until then no server sets it, and capability discovery keeps reporting `proxy_stable_code: false`. Tests cover wire compatibility in both directions (an absent field encodes exactly as before) and that `hello` carries the fence only while one is active. Co-authored-by: Tomasz Szymczyszyn --- docs/NAMESPACE_FENCE.md | 17 ++- libsql-replication/proto/metadata.proto | 13 ++ libsql-replication/proto/proxy.proto | 5 + libsql-replication/src/generated/metadata.rs | 17 +++ libsql-replication/src/generated/proxy.rs | 6 + libsql-replication/src/rpc.rs | 115 ++++++++++++++++++ libsql-server/src/connection/config.rs | 3 + .../src/namespace/fence/controller.rs | 13 ++ libsql-server/src/namespace/fence/stream.rs | 79 ++++++++++++ libsql-server/src/rpc/proxy.rs | 1 + .../src/rpc/replication/replication_log.rs | 11 +- 11 files changed, 274 insertions(+), 6 deletions(-) diff --git a/docs/NAMESPACE_FENCE.md b/docs/NAMESPACE_FENCE.md index 3841e1d769..614c1c9559 100644 --- a/docs/NAMESPACE_FENCE.md +++ b/docs/NAMESPACE_FENCE.md @@ -401,11 +401,19 @@ message Error { } ``` -This is additive in proto3: older peers skip the unknown field; a newer replica treats an absent field as "no typed outcome". For fence denials the primary sets `code = SQL_ERROR` and `stable_code`. The replica threads `stable_code` through `Error::RpcQueryError` into the same user HTTP status and Hrana code the primary would have returned. A fence denial at proxy connection creation is returned as `FAILED_PRECONDITION` with the metadata above, never `UNAVAILABLE`, so the replica's write-proxy reconnect loop (which retries `UNAVAILABLE` without bound) does not spin on it. +This is additive in proto3: older peers skip the unknown field; a newer replica treats an absent field as "no typed outcome". An error without a stable code encodes exactly as before. For fence denials the primary sets `code = SQL_ERROR` and `stable_code`. The replica threads `stable_code` through `Error::RpcQueryError` into the same user HTTP status and Hrana code the primary would have returned. A fence denial at proxy connection creation is returned as `FAILED_PRECONDITION` with the metadata above, never `UNAVAILABLE`, so the replica's write-proxy reconnect loop (which retries `UNAVAILABLE` without bound) does not spin on it. + +Capability discovery reports `proxy_stable_code: true` only once a server both fills the field on fence denials and maps it on the replica side; a server that has the field in its protocol but does neither reports `false`. ### 6.2 Replicated configuration addition -`libsql-replication/proto/metadata.proto` `DatabaseConfig` gains `optional ReplicatedFence fence = 14;` with `state` (string) and `revision` (uint64). The primary fills it from the gate in `hello`; a newer replica server uses it to deny local reads in states whose normal-read column is "deny". Older replicas ignore it and still see the legacy `block_*` mirror. +`libsql-replication/proto/metadata.proto` `DatabaseConfig` gains `optional ReplicatedFence fence = 14;` with `state` (the state name, e.g. `SOURCE_WRITE_FENCED`) and `revision` (the record's revision). A configuration without it encodes exactly as before. + +- **Filled only by `hello`, only while a fence is active.** The primary's `hello` sets it from the published gate (`GateSnapshot::replicated`): the state and revision of any state other than `UNFENCED`, `RELEASED` and `TARGET_WRITABLE`, and nothing otherwise. It is never part of a stored configuration: converting a `DatabaseConfig` for the metastore leaves it empty, and converting a received one drops it. +- **The rest of the configuration is the namespace's own.** `hello` sends the logical configuration unchanged; the legacy `block_*` mirror of section 13.2 exists only in the stored config row, for an older binary reading the metastore. +- **When a replica sees it.** The dump/replication column of the permission matrix equals the normal-read column, so `hello` is answered only in states whose normal reads are allowed; in every state that denies reads, `hello` is refused and open streams end with the typed terminal status of section 9. A replica therefore learns of a read fence from that status, and of the fence in general (state and revision) from `hello`. A newer replica server uses both to deny its own local reads in states whose normal-read column is "deny". +- **A fence change does not move the configuration version**, so it does not by itself invalidate a replica's session token or make it call `hello` again; the read fence reaches a replica by ending its streams. +- **Older replica servers** skip the field. The read fence still stops their replication (refused `hello`, ended streams), but not their local reads of data they already hold (section 18). ## 7. FenceController (Design) @@ -748,7 +756,7 @@ How each path that can reach namespace data or lifecycle is covered. File refere | `http/admin/mod.rs` config, create, fork, delete, checkpoint, stats | Config POST and create (before a `dump_url` is fetched) are checked before the namespace is loaded; fork and delete are refused by the store; all of them return the fence error with `423`. Config GET, stats and checkpoint are allowed. Fence routes live in `http/admin/fence.rs`. | | `http/admin/fence.rs` | Fence routes (section 4.5): capability discovery, `InspectFence`, one route per command through `NamespaceStore::execute_fence_command`, and `validation-query` through `open_validation_session`; admin auth key and primary required for every route that changes state or uses a capability. | | `http/user/dump.rs`, `connection/dump/exporter.rs` | Gate check (`Stream`) before connection creation (typed, no panic on create error); `Dump` lease held by the export; cancel checked before every row and by the pipe's pending write; the body stream ends with the fence error (aborted response) on cancellation. | -| `rpc/replication/replication_log.rs` `hello`, `log_entries`, `batch_log_entries`, `snapshot` | Denied at request start (`FAILED_PRECONDITION` + `x-libsql-fence-code`, counted, rate-limited log); `FencedStream` replication leases for both streams and the batch; typed terminal status when the gate closes or at the deadline, lease released without the peer; `ReplicatedFence` in `hello`'s config (planned, section 6.2). | +| `rpc/replication/replication_log.rs` `hello`, `log_entries`, `batch_log_entries`, `snapshot` | Denied at request start (`FAILED_PRECONDITION` + `x-libsql-fence-code`, counted, rate-limited log); `FencedStream` replication leases for both streams and the batch; typed terminal status when the gate closes or at the deadline, lease released without the peer; `ReplicatedFence` in `hello`'s config while a fence is active (section 6.2; the gate is read without a lease, since `hello` streams no data). | | `admin_shell.rs` | Writes denied at the WAL (no capability); reads checked against the gate, and a read lease held, per query (cancelled through the connection's interrupt handle). | | `schema/scheduler.rs`, `database/schema.rs` | Shared schema excluded from fencing; registering a migration job is refused while the schema or a linked namespace denies lifecycle work (section 13.4); migration writes are WAL-gated; the scheduler's `block_writes` flag is not treated as drain evidence. | | `namespace/configurator/helpers.rs` `load_dump`, `http/admin/mod.rs` `dump_stream_from_url` | Restore options and dump URLs are refused for fenced namespaces before the dump is fetched (section 13.5); outside a capability the loader's writes are refused at the WAL. Import goes through `ImportSession::load_dump`, which runs the same loader (`load_dump_sql`) on the capability connection. | @@ -806,7 +814,7 @@ Planned test names; the table is updated as tests land. | 15 | Lost `EnableTargetWrites` response resolved from receipt/state | landed: `namespace::fence::target::tests::enable_writes_response_loss_resolved` | | 16 | Read fence drains SQL, dump, `log_entries`, `snapshot`, including dead peers and forced termination | SQL landed: `namespace::fence::read::tests::{read_fence_waits_for_running_program, program_after_closing_gate_is_refused (parked after the read-closing gate, before the CAS), read_fence_cancels_at_deadline, unreleased_lease_answers_draining_and_replay_completes, idle_txn_fails_on_next_program, clear_read_fence_reopens_reads_not_writes, refused_read_fence_reopens_reads, attach_of_read_fenced_namespace_denied}`, `admin_shell::fence_tests::admin_shell_read_denied`; dump and replication landed: `namespace::fence::stream::tests::{dump_lease_released_on_cancel (a dump blocked mid-row on a peer that stopped reading is cancelled at the deadline, its lease released without the peer, the fence acknowledged, and the body ends with the fence error and no `COMMIT;`), read_fence_waits_for_dump, dump_refused_while_read_fenced, log_entries_stream_ends_typed, stream_lease_released_without_peer_read (a dead peer), snapshot_stream_ends_typed, replication_calls_denied_while_read_fenced, read_fence_forced_termination}`; planned: the HTTP-level check that an interrupted `/dump` response is aborted rather than completed, with `tests::fence::protocol::dump_codes` | | 17 | Delete, reset, fork, restore, config, schema mutation rejected | landed: `tests::fence::lifecycle::lifecycle_rejected_while_fenced` (over the admin API, for a write-fenced source and a quarantined target: delete, fork as source and as destination, create with a `dump_url` whose file does not exist, create over the record, linking to a shared schema at creation and config `POST` are all `423` with the fence code; state, revision and data are unchanged and no copy exists; a target cannot be created with a shared schema; after release the source takes writes and config, fork and delete work again); `namespace::store::fence_tests::{reset_refused_while_fenced (called directly and as the replicator's reset callback; after release reset works and wipes the data), lifecycle_refused_while_fenced (fork either side, create over, delete, config and shared-schema link in the metastore transaction)}`; `schema::scheduler::test::fence::{acquire_rejects_shared_schema, migration_not_registered_while_linked_namespace_fenced}` | -| 18 | Codes through HTTP, Hrana, RPC, dump, replication, replica write proxy; distinguishable from auth/timeout/not-found; old peers compatible; no retry loops | `tests::fence::protocol::{http_codes, hrana_http_codes, hrana_ws_codes, rpc_codes, dump_codes, replication_codes, replica_proxy_preserves_code, auth_and_not_found_distinct, denial_not_retried}`; `libsql-replication` `proxy_error_stable_code_is_additive` | +| 18 | Codes through HTTP, Hrana, RPC, dump, replication, replica write proxy; distinguishable from auth/timeout/not-found; old peers compatible; no retry loops | `tests::fence::protocol::{http_codes, hrana_http_codes, hrana_ws_codes, rpc_codes, dump_codes, replication_codes, replica_proxy_preserves_code, auth_and_not_found_distinct, denial_not_retried}`; landed: `libsql-replication` `rpc::test::{proxy_error_stable_code_is_additive, replicated_fence_is_additive}` (each new field is skipped by a peer that does not know it, absent from an older peer's message, and absent fields encode exactly as before), `namespace::fence::stream::tests::hello_carries_replicated_fence` (no fence before acquisition and after release; state and revision while write-fenced; the stored configuration never carries it) | | 19 | Corrupt or unknown durable fence state fails closed | `fence::store::tests::corrupt_payload_fails_closed`, `unknown_format_version_fails_closed` | | 20 | Metrics and audit logs | `tests::fence::observability::metrics_and_labels`; `fence::audit::tests::audit_event_fields` | | 21 | Capability discovery and mixed-version protection | capability discovery landed: `tests::fence::admin::{capabilities, capabilities_when_disabled}`; planned: `fence::store::tests::legacy_mirror_and_fk_guard` (bounded, see section 18) | @@ -823,6 +831,7 @@ What this design and its tests do not prove: - **"Commit unknown"** is the caller's classification when neither replay nor inspection answers; the server's part is that replay and inspection always answer when the server is reachable. - **Two-person adoption** is a separate secret plus a recorded two-approver request, not verified identities. - **Delivered data** cannot be recalled: bytes already sent, frames held by embedded replicas, and a replica server partitioned from the primary are outside the server's reach. +- **Older replica servers** stop replicating under a read fence but keep serving local reads of what they already hold; only a replica server that understands the replicated fence (section 6.2) denies them. - **Metastore rollback detection** relies on the namespace directory's marker. If both the metastore and the namespace directory are lost or restored from backup together, the server cannot detect that a newer fence existed; the caller's durable intent is authoritative then. - **Release and pinning** of a server build that contains the fence are outside this change. diff --git a/libsql-replication/proto/metadata.proto b/libsql-replication/proto/metadata.proto index 797e9da7f8..374e78ec3a 100644 --- a/libsql-replication/proto/metadata.proto +++ b/libsql-replication/proto/metadata.proto @@ -27,4 +27,17 @@ message DatabaseConfig { optional bool shared_schema = 11; optional string shared_schema_name = 12; optional DurabilityMode durability_mode = 13; + // The namespace fence as seen by the primary when it answered. Only ever filled by the + // primary's replication `Hello`, and only while a fence is active; it is never part of a + // stored configuration. Absent from older primaries; older replicas ignore it and still + // see the legacy `block_*` fields above. + optional ReplicatedFence fence = 14; +} + +// The part of a namespace fence a replica needs to apply the primary's read admission. +message ReplicatedFence { + // The fence state name, e.g. "SOURCE_READ_FENCED". + string state = 1; + // The revision of the fence record the state belongs to. + uint64 revision = 2; } diff --git a/libsql-replication/proto/proxy.proto b/libsql-replication/proto/proxy.proto index 5949cb078d..7d1f3897e8 100644 --- a/libsql-replication/proto/proxy.proto +++ b/libsql-replication/proto/proxy.proto @@ -43,6 +43,11 @@ message Error { ErrorCode code = 1; string message = 2; int32 extended_code = 3; + // Stable machine-readable outcome of the error, e.g. "MIGRATION_WRITE_FENCED" for a + // request refused by a namespace fence. Absent when the error has no typed outcome, and + // always absent from older servers: a receiver treats an absent field as "no typed + // outcome" and falls back to `code`. + optional string stable_code = 4; } message ResultRows { diff --git a/libsql-replication/src/generated/metadata.rs b/libsql-replication/src/generated/metadata.rs index dac5a62511..8dbe2f7266 100644 --- a/libsql-replication/src/generated/metadata.rs +++ b/libsql-replication/src/generated/metadata.rs @@ -32,6 +32,23 @@ pub struct DatabaseConfig { pub shared_schema_name: ::core::option::Option<::prost::alloc::string::String>, #[prost(enumeration = "DurabilityMode", optional, tag = "13")] pub durability_mode: ::core::option::Option, + /// The namespace fence as seen by the primary when it answered. Only ever filled by the + /// primary's replication `Hello`, and only while a fence is active; it is never part of a + /// stored configuration. Absent from older primaries; older replicas ignore it and still + /// see the legacy `block_*` fields above. + #[prost(message, optional, tag = "14")] + pub fence: ::core::option::Option, +} +/// The part of a namespace fence a replica needs to apply the primary's read admission. +#[allow(clippy::derive_partial_eq_without_eq)] +#[derive(Clone, PartialEq, ::prost::Message)] +pub struct ReplicatedFence { + /// The fence state name, e.g. "SOURCE_READ_FENCED". + #[prost(string, tag = "1")] + pub state: ::prost::alloc::string::String, + /// The revision of the fence record the state belongs to. + #[prost(uint64, tag = "2")] + pub revision: u64, } #[derive(Clone, Copy, Debug, PartialEq, Eq, Hash, PartialOrd, Ord, ::prost::Enumeration)] #[repr(i32)] diff --git a/libsql-replication/src/generated/proxy.rs b/libsql-replication/src/generated/proxy.rs index a76bb9bf65..cb63ce6ea2 100644 --- a/libsql-replication/src/generated/proxy.rs +++ b/libsql-replication/src/generated/proxy.rs @@ -77,6 +77,12 @@ pub struct Error { pub message: ::prost::alloc::string::String, #[prost(int32, tag = "3")] pub extended_code: i32, + /// Stable machine-readable outcome of the error, e.g. "MIGRATION_WRITE_FENCED" for a + /// request refused by a namespace fence. Absent when the error has no typed outcome, and + /// always absent from older servers: a receiver treats an absent field as "no typed + /// outcome" and falls back to `code`. + #[prost(string, optional, tag = "4")] + pub stable_code: ::core::option::Option<::prost::alloc::string::String>, } /// Nested message and enum types in `Error`. pub mod error { diff --git a/libsql-replication/src/rpc.rs b/libsql-replication/src/rpc.rs index 57f22ffa1a..242ae083d7 100644 --- a/libsql-replication/src/rpc.rs +++ b/libsql-replication/src/rpc.rs @@ -105,3 +105,118 @@ pub mod metadata { #![allow(clippy::all)] include!("generated/metadata.rs"); } + +#[cfg(test)] +mod test { + use prost::Message; + + use super::metadata::{DatabaseConfig, ReplicatedFence}; + use super::proxy::{error::ErrorCode, Error}; + + /// `proxy.Error` as a peer built before `stable_code` existed knows it. + #[derive(Clone, PartialEq, ::prost::Message)] + struct ErrorWithoutStableCode { + #[prost(enumeration = "ErrorCode", tag = "1")] + code: i32, + #[prost(string, tag = "2")] + message: String, + #[prost(int32, tag = "3")] + extended_code: i32, + } + + /// The legacy part of `metadata.DatabaseConfig`, as a peer built before `fence` existed + /// knows it (the fields in between are skipped the same way as `fence` is). + #[derive(Clone, PartialEq, ::prost::Message)] + struct DatabaseConfigWithoutFence { + #[prost(bool, tag = "1")] + block_reads: bool, + #[prost(bool, tag = "2")] + block_writes: bool, + #[prost(string, optional, tag = "3")] + block_reason: Option, + #[prost(uint64, tag = "4")] + max_db_pages: u64, + } + + fn error(stable_code: Option<&str>) -> Error { + Error { + code: ErrorCode::SqlError as i32, + message: "writes are fenced".into(), + extended_code: 23, + stable_code: stable_code.map(Into::into), + } + } + + #[test] + fn proxy_error_stable_code_is_additive() { + // A newer server's error, read by an older replica: the known fields are intact and + // the stable code is skipped. + let new = error(Some("MIGRATION_WRITE_FENCED")); + let old = ErrorWithoutStableCode::decode(&new.encode_to_vec()[..]).unwrap(); + assert_eq!( + old, + ErrorWithoutStableCode { + code: ErrorCode::SqlError as i32, + message: "writes are fenced".into(), + extended_code: 23, + } + ); + + // An older server's error, read by a newer replica: no typed outcome. + let decoded = Error::decode(&old.encode_to_vec()[..]).unwrap(); + assert_eq!(decoded, error(None)); + + // Without a stable code the encoding is exactly the older one, so an error that has no + // typed outcome is unchanged on the wire. + assert_eq!(error(None).encode_to_vec(), old.encode_to_vec()); + + // And the field round-trips between newer peers. + assert_eq!(Error::decode(&new.encode_to_vec()[..]).unwrap(), new); + } + + fn config(fence: Option) -> DatabaseConfig { + DatabaseConfig { + block_reads: true, + block_writes: true, + block_reason: Some("namespace fence".into()), + max_db_pages: 1024, + fence, + ..Default::default() + } + } + + #[test] + fn replicated_fence_is_additive() { + let fence = ReplicatedFence { + state: "SOURCE_READ_FENCED".into(), + revision: 3, + }; + + // A newer primary's config, read by an older replica: the legacy block fields are + // intact, so the older replica still applies the legacy mirror of the fence. + let new = config(Some(fence.clone())); + let old = DatabaseConfigWithoutFence::decode(&new.encode_to_vec()[..]).unwrap(); + assert_eq!( + old, + DatabaseConfigWithoutFence { + block_reads: true, + block_writes: true, + block_reason: Some("namespace fence".into()), + max_db_pages: 1024, + } + ); + + // An older primary's config, read by a newer replica: no fence. + let decoded = DatabaseConfig::decode(&old.encode_to_vec()[..]).unwrap(); + assert_eq!(decoded.fence, None); + assert_eq!(decoded, config(None)); + + // Without a fence the encoding is exactly the older one: a stored configuration, which + // never carries a fence, is unchanged. + assert_eq!(config(None).encode_to_vec(), old.encode_to_vec()); + + // And the fence round-trips between newer peers. + let round_trip = DatabaseConfig::decode(&new.encode_to_vec()[..]).unwrap(); + assert_eq!(round_trip.fence, Some(fence)); + } +} diff --git a/libsql-server/src/connection/config.rs b/libsql-server/src/connection/config.rs index 970014415d..1e835d395f 100644 --- a/libsql-server/src/connection/config.rs +++ b/libsql-server/src/connection/config.rs @@ -108,6 +108,9 @@ impl From<&DatabaseConfig> for metadata::DatabaseConfig { shared_schema: Some(value.is_shared_schema), shared_schema_name: value.shared_schema_name.as_ref().map(|s| s.to_string()), durability_mode: Some(metadata::DurabilityMode::from(value.durability_mode).into()), + // Never part of a configuration: the replication `hello` fills it from the live + // fence gate, and it is not stored. + fence: None, } } } diff --git a/libsql-server/src/namespace/fence/controller.rs b/libsql-server/src/namespace/fence/controller.rs index 621576bacb..e61421f341 100644 --- a/libsql-server/src/namespace/fence/controller.rs +++ b/libsql-server/src/namespace/fence/controller.rs @@ -18,6 +18,8 @@ use parking_lot::Mutex; use tokio::sync::{watch, Notify, OwnedMutexGuard}; use uuid::Uuid; +use libsql_replication::rpc::metadata::ReplicatedFence; + use crate::connection::connection_manager::{ConnectionManager, WeakConnectionManager}; use crate::error::Error; use crate::namespace::meta_store::{FenceCommit, FenceContext, MetaStore}; @@ -157,6 +159,17 @@ impl GateSnapshot { Ok(()) } + /// The fence as a replica needs it (`docs/NAMESPACE_FENCE.md` section 6.2): the state and + /// revision of an active fence, `None` while no fence is active. The primary fills it into + /// the configuration its replication `hello` returns; it is never stored. + pub fn replicated(&self) -> Option { + let state = self.state(); + state.is_active().then(|| ReplicatedFence { + state: state.as_str().into(), + revision: self.revision(), + }) + } + /// Whether a closing transition is being installed. pub fn is_installing(&self) -> bool { self.installing.is_some() diff --git a/libsql-server/src/namespace/fence/stream.rs b/libsql-server/src/namespace/fence/stream.rs index 2a043d89fa..bd9ae56097 100644 --- a/libsql-server/src/namespace/fence/stream.rs +++ b/libsql-server/src/namespace/fence/stream.rs @@ -223,15 +223,18 @@ mod tests { use bytes::Bytes; use futures::stream::BoxStream; use futures::{Stream, StreamExt}; + use libsql_replication::rpc::metadata::{self, ReplicatedFence}; use libsql_replication::rpc::replication::replication_log_server::ReplicationLog; use libsql_replication::rpc::replication::{ Frame, HelloRequest, LogOffset, NAMESPACE_METADATA_KEY, SESSION_TOKEN_KEY, }; use tonic::metadata::{AsciiMetadataValue, BinaryMetadataValue}; + use uuid::Uuid; use super::*; use crate::error::Error; use crate::http::user::dump::dump_stream; + use crate::namespace::fence::command::{FenceCommand, FenceRequest}; use crate::namespace::fence::drain::tests::{fence_outcome, raw, Source, LONG, OP, PROMPT}; use crate::namespace::fence::read::tests::{fenced_source, read_fence, NOW}; use crate::namespace::fence::state::FenceState; @@ -413,6 +416,82 @@ mod tests { } } + /// The configuration a replica's `hello` receives from `s`. + async fn hello_config(s: &Source) -> metadata::DatabaseConfig { + let replication = Replication { + service: ReplicationLogService::new(s.store.clone(), None, None, false, false, true), + token: None, + }; + replication + .service + .hello(replication.request(HelloRequest { + handshake_version: Some(1), + })) + .await + .unwrap() + .into_inner() + .config + .expect("hello always carries the configuration") + } + + /// `hello` carries the live fence while one is active (`docs/NAMESPACE_FENCE.md` section + /// 6.2), and nothing before the fence and once it is released. The stored configuration + /// never carries it. + #[tokio::test] + async fn hello_carries_replicated_fence() { + let s = Source::new().await; + let unfenced = hello_config(&s).await; + assert_eq!(unfenced.fence, None); + assert!(!unfenced.block_writes); + + let acquired = s.execute(s.acquire(OP, 1, LONG)).await.unwrap(); + assert_eq!(fence_outcome(&acquired), FenceOutcome::Applied); + let gate = s.fence.gate(); + assert_eq!(gate.state(), FenceState::SourceWriteFenced); + let fenced = hello_config(&s).await; + assert_eq!( + fenced.fence, + Some(ReplicatedFence { + state: "SOURCE_WRITE_FENCED".into(), + revision: gate.revision(), + }) + ); + // Apart from the fence, `hello` sends the namespace's logical configuration unchanged: + // the legacy `block_*` mirror lives only in the stored config row (section 13.2). + let stored = s + .store + .with("ns".into(), |ns| { + metadata::DatabaseConfig::from(ns.config().as_ref()) + }) + .await + .unwrap(); + assert_eq!(stored.fence, None); + assert_eq!( + metadata::DatabaseConfig { + fence: None, + ..fenced + }, + stored + ); + + let released = s + .execute(FenceRequest { + namespace: "ns".into(), + operation_id: OP, + command_id: Uuid::from_u128(2), + expected_state: gate.state(), + expected_revision: gate.revision(), + command: FenceCommand::ReleaseSourceWriteFence, + }) + .await + .unwrap(); + assert_eq!(fence_outcome(&released), FenceOutcome::Applied); + assert_eq!(s.fence.gate().state(), FenceState::Released); + let after = hello_config(&s).await; + assert_eq!(after.fence, None); + assert!(!after.block_writes); + } + fn assert_read_fenced_status(status: &tonic::Status) { assert_eq!(status.code(), tonic::Code::FailedPrecondition, "{status:?}"); assert_eq!( diff --git a/libsql-server/src/rpc/proxy.rs b/libsql-server/src/rpc/proxy.rs index 1ab5d1d19a..32bc6aeb96 100644 --- a/libsql-server/src/rpc/proxy.rs +++ b/libsql-server/src/rpc/proxy.rs @@ -57,6 +57,7 @@ pub mod rpc { message: other.to_string(), code: code as i32, extended_code, + stable_code: None, } } } diff --git a/libsql-server/src/rpc/replication/replication_log.rs b/libsql-server/src/rpc/replication/replication_log.rs index 4457c49e7a..7b6c5d51d7 100644 --- a/libsql-server/src/rpc/replication/replication_log.rs +++ b/libsql-server/src/rpc/replication/replication_log.rs @@ -8,6 +8,7 @@ use bytes::Bytes; use chrono::{DateTime, Utc}; use futures::stream::BoxStream; use futures_core::Future; +use libsql_replication::rpc::metadata; pub use libsql_replication::rpc::replication as rpc; use libsql_replication::rpc::replication::log_offset::WalFlavor; use libsql_replication::rpc::replication::replication_log_server::ReplicationLog; @@ -431,7 +432,7 @@ impl ReplicationLog for ReplicationLogService { guard.insert((replica_addr, namespace.clone())); } } - let (logger, config, version, _, _, _) = self + let (logger, config, version, _, _, fence) = self .logger_from_namespace(namespace, "hello", &req, false) .await?; @@ -451,7 +452,13 @@ impl ReplicationLog for ReplicationLogService { generation_id: self.generation_id.to_string(), generation_start_index: 0, current_replication_index: *logger.new_frame_notifier.borrow(), - config: Some(config.as_ref().into()), + config: Some(metadata::DatabaseConfig { + // The live fence, for a replica that applies it (section 6.2 of + // `docs/NAMESPACE_FENCE.md`); older replicas skip it and see the legacy + // `block_*` mirror. The stored configuration never carries it. + fence: fence.gate().replicated(), + ..config.as_ref().into() + }), }; Ok(tonic::Response::new(response)) From 979f567ce9a4aa5bc06cec5e154c06e4625ccd3f Mon Sep 17 00:00:00 2001 From: River Date: Wed, 30 Sep 2026 09:48:16 +0000 Subject: [PATCH 21/33] libsql-server: typed fence outcomes across HTTP, Hrana and dump A fence denial reaching the user-facing protocols is now a typed answer everywhere instead of a generic or fatal error: - HTTP error bodies for fence errors gain an additive `code` field (and `detail` when there is one), with `423` for data-plane denials, through every wrapper the error arrives in. The legacy `/` API answers a batch with a fenced step as a whole with `423`. - Hrana gains `StmtError::Fence` and `BatchError::Fence`, whose code is the stable fence code, so step denials and whole-request denials (a read under a read fence, a quarantined target) are Hrana errors on `/v1`, `/v2`, `/v3`, cursors and WebSockets, and the stream stays usable. - `/v1/execute` and `/v1/batch` answer whole-request denials with `423` and the code. Integration tests cover each protocol, an old WebSocket transaction, a batch denied mid-way, `/dump`, and that `401`/`404` stay distinct; a lib test shows a dump cancelled by the read drain fails its response body. Co-authored-by: Tomasz Szymczyszyn --- docs/NAMESPACE_FENCE.md | 25 +- libsql-server/src/error.rs | 85 +- libsql-server/src/hrana/batch.rs | 6 + libsql-server/src/hrana/stmt.rs | 6 + .../src/http/user/hrana_over_http_1.rs | 17 + libsql-server/src/http/user/mod.rs | 5 +- libsql-server/src/http/user/result_builder.rs | 13 + libsql-server/src/namespace/fence/outcome.rs | 21 + libsql-server/src/namespace/fence/stream.rs | 42 + libsql-server/src/schema/error.rs | 2 +- libsql-server/tests/fence/lifecycle.rs | 1 + libsql-server/tests/fence/mod.rs | 15 +- libsql-server/tests/fence/protocol.rs | 853 ++++++++++++++++++ 13 files changed, 1083 insertions(+), 8 deletions(-) create mode 100644 libsql-server/tests/fence/protocol.rs diff --git a/docs/NAMESPACE_FENCE.md b/docs/NAMESPACE_FENCE.md index 614c1c9559..90cd8b6517 100644 --- a/docs/NAMESPACE_FENCE.md +++ b/docs/NAMESPACE_FENCE.md @@ -386,6 +386,23 @@ Notes: - gRPC statuses carry the code in the `x-libsql-fence-code` metadata entry and as the message prefix `": "`. - Authentication (`401`), missing namespace (`404`), timeouts and transport errors are distinct from all of the above. +### 6.0 How each user protocol reports a denial (implementation) + +A fence denial reaches the protocol layer as `Error::NamespaceFence` in one of two places: as the error of one **step** (a write refused by the early check or at the WAL, section 8.1), or as the error of the whole **program** (a read refused at program start, section 9, or a namespace refused before a connection exists: quarantined creation, unknown fence state). Each protocol maps both: + +| Entry point | Step denial | Whole-request denial | +|---|---|---| +| Legacy `/` | The batch is answered as a whole: `423` with `{"error", "code", "detail"?}`. The legacy API has no per-step codes, and the statements after the refused one did not run (an open transaction was rolled back). | `423` with the same body. | +| `/v1/execute` | `423` with the Hrana 1 error body `{"message", "code"}`. | Same. | +| `/v1/batch` | `200`; the step's entry in `step_errors` carries the code, like any step error. | `423` with `{"message", "code"}`. | +| Hrana `/v2`, `/v3` pipeline, `/dev/.../pipeline` | The request's result is `{"type": "error", "error": {"message", "code"}}`; in a batch, the step's `step_errors` entry. The stream (baton) stays usable. | The request's result is an error with the code (the stream stays usable). A namespace refused before a connection exists is `423` with `{"error", "code", "detail"?}`. | +| Hrana `/v3/cursor` | A `step_error` entry with the code. | An `error` entry with the code. | +| Hrana WebSocket | `response_error` with the code; in a batch, the step's `step_errors` entry. The connection and the stream stay usable. | `response_error` with the code. | +| `/dump` | — | Refused before the export starts: `423` with `{"error", "code"}`. Cancelled by the read drain while running: the response body fails, so the client sees an aborted response and never a trailing `COMMIT;` (section 9). | +| Admin routes (config, create, fork, delete) | — | `423` with `{"error", "code", "detail"?}`. | + +`Error::fence_error()` finds the denial through the wrappers it can arrive in (`Ref`, `Anyhow`, a schema `Migration` error); `FenceError::http_status` and `FenceError::http_error_body` are the single mapping both HTTP APIs use. Hrana's `StmtError::Fence` and `BatchError::Fence` carry the error, and their `code()` is the stable code. Authentication failures (`401`) and missing namespaces (`404`) are unchanged and carry no fence code. + ### 6.1 Proxy protocol addition `libsql-replication/proto/proxy.proto`: @@ -748,7 +765,7 @@ How each path that can reach namespace data or lifecycle is covered. File refere | `connection_core.rs` `checkpoint`, `vacuum_if_needed`, `force_rollback` | Checkpoint is `Maintenance`; vacuum is `Vacuum` and skipped while writes are fenced; `force_rollback` is the drain's abort. | | `connection/connection_manager.rs` | Authoritative gate in `begin_write_txn` before `acquire()`, generation check, non-`BUSY` refusal; classed queue entries; queue wake on generation change; active-writer query, release notification, `abort_active()`. | | `connection/legacy.rs` | `FenceConnState` wired into every `LegacyConnection`; the controller is passed to `MakeLegacyConnection::new` before the first connection. `with_raw` users are covered by the WAL gate. | -| HTTP `/`, `/v1/execute`, `/v1/batch`, Hrana `/v2`, `/v3`, cursors, WebSocket, dev route | Core checks and WAL gate; typed status and `code` field; Hrana codes. | +| HTTP `/`, `/v1/execute`, `/v1/batch`, Hrana `/v2`, `/v3`, cursors, WebSocket, dev route | Core checks and WAL gate; typed status and `code` field; Hrana codes, for step and whole-request denials alike, with the stream left usable (section 6.0). | | `rpc/proxy.rs`, `rpc/streaming_exec.rs`, `rpc/replica_proxy.rs`, `connection/write_proxy.rs` | `stable_code` set on the primary and carried back on the replica; connection-creation denials are `FAILED_PRECONDITION`, not `UNAVAILABLE`. | | `namespace/meta_store.rs` `handle()`, `restore()`, `maybe_recover_from_fs`, `destroy_on_error`, `process`/`try_process`, `remove`, bottomless metastore restore | Non-creating lookups; fail-closed decoding; marker-aware recovery; rename-aside; publish only after commit; fence check inside the config and remove transactions; restore provenance surfaced. | | `namespace/store.rs` `with`, `load_namespace`, `make_namespace`, eviction | Registry check before setup; controller passed into setup; eviction keeps the registry entry. | @@ -799,7 +816,7 @@ Planned test names; the table is updated as tests land. |---|---|---| | 1 | Concurrent acquisition by two operations: one owner, typed conflict for the loser | landed: `namespace::fence::drain::tests::acquire_race_single_owner` (the first acquisition is parked after closing admission while the second waits on the transition lock); over the admin API: `tests::fence::admin::concurrent_acquire_one_owner` (two operations acquire at once over HTTP: one `200 APPLIED`, the other `409 FENCE_OWNED_BY_ANOTHER_OPERATION` with the winner in its fence view); admin walks: `tests::fence::admin::{source_walk_over_http, target_walk_over_http, inspect_reports_drain_counters, mutating_routes_require_admin_key}` | | 2 | Active writer commits or is rolled back before freeze acknowledgement; nothing commits after | landed: `namespace::fence::drain::tests::{active_writer_commits_before_ack, forced_rollback_before_ack, no_commit_after_ack}` (the boundary equals the last committed replication frame and no frame follows it; autocommit, `BEGIN IMMEDIATE`, DDL and a pre-fence read transaction upgrading are refused), `installing_gate_closes_writes_before_persisting`, `refused_acquire_reopens_writes`, `release_reopens_with_new_generation` | -| 3 | Autocommit, explicit transactions, queued writers, batches, DDL, schema jobs, old WebSockets, read-to-write upgrades cannot bypass | landed: `connection::connection_manager::fence_tests::{fence_rejects_read_to_write_upgrade, fence_rejects_ddl_and_pragma, fence_rejects_raw_with_raw_write}` (autocommit, explicit transactions, DDL, header-writing pragma, `BEGIN IMMEDIATE`, `VACUUM`, `with_raw` users); `connection::connection_manager::fence_tests::fence_rejects_queued_writer` (a writer parked in the queue behind an open transaction leaves it with `MIGRATION_WRITE_FENCED` when the fence changes, and the holder keeps the slot); maintenance and vacuum under a fence: `queued_checkpoint_survives_fence_wake`, `checkpoint_allowed_while_fenced`, `vacuum_skipped_while_fenced`; drain primitives: `abort_active_tolerates_closed_connection`, `release_notifies_drain_waiters`, `fence::controller::tests::write_queues_are_woken_on_every_generation_change`; schema jobs: `schema::scheduler::test::fence::{acquire_rejects_shared_schema (a shared schema and a linked namespace cannot be fenced; a fenced namespace cannot be linked by create or config), migration_not_registered_while_linked_namespace_fenced}`; planned: `tests::fence::protocol::old_ws_session_cannot_write`, `batch_denied_mid_batch` | +| 3 | Autocommit, explicit transactions, queued writers, batches, DDL, schema jobs, old WebSockets, read-to-write upgrades cannot bypass | landed: `connection::connection_manager::fence_tests::{fence_rejects_read_to_write_upgrade, fence_rejects_ddl_and_pragma, fence_rejects_raw_with_raw_write}` (autocommit, explicit transactions, DDL, header-writing pragma, `BEGIN IMMEDIATE`, `VACUUM`, `with_raw` users); `connection::connection_manager::fence_tests::fence_rejects_queued_writer` (a writer parked in the queue behind an open transaction leaves it with `MIGRATION_WRITE_FENCED` when the fence changes, and the holder keeps the slot); maintenance and vacuum under a fence: `queued_checkpoint_survives_fence_wake`, `checkpoint_allowed_while_fenced`, `vacuum_skipped_while_fenced`; drain primitives: `abort_active_tolerates_closed_connection`, `release_notifies_drain_waiters`, `fence::controller::tests::write_queues_are_woken_on_every_generation_change`; schema jobs: `schema::scheduler::test::fence::{acquire_rejects_shared_schema (a shared schema and a linked namespace cannot be fenced; a fenced namespace cannot be linked by create or config), migration_not_registered_while_linked_namespace_fenced}`; old WebSockets and batches over the protocols: `tests::fence::protocol::{old_ws_session_cannot_write (a WebSocket session whose transaction began before the fence cannot write while fenced nor after the release; after a rollback the same session writes), batch_denied_mid_batch (the read before the write runs, the write step gets the code, a step conditional on it is skipped and one conditional on its failure runs)}` | | 4 | Program that captured config before the fence is rejected at the WAL | landed: `connection::connection_manager::fence_tests::wal_gate_rejects_program_admitted_before_fence` (a SQL function parks the program between admission and its write while the fence is acquired and released) | | 5 | Pre-fence transactions cannot write after release or publication | landed: `connection::connection_manager::fence_tests::stale_generation_cannot_write_after_release`, `namespace::fence::target::tests::stale_generation_cannot_write_after_enable_writes`; unfenced behaviour unchanged: `unfenced_namespace_is_unchanged` | | 6 | Acquisition timeout returns `DRAINING`, admission stays closed | landed: `namespace::fence::drain::tests::{deadline_returns_draining_and_stays_closed, replay_of_draining_resumes_and_completes}` | @@ -812,9 +829,9 @@ Planned test names; the table is updated as tests land. | 13 | Seal enters `TARGET_IMPORT_DRAINING`, waits, reaches `TARGET_VALIDATING`, cannot resume import; only a durable validation receipt permits idempotent publication | seal landed: `namespace::fence::import::tests::{seal_waits_for_import_writers (an import call parked inside its write transaction: import is closed at once, the transaction commits, and the seal reaches its completion commit only afterwards), seal_deadline_leaves_import_draining_until_replayed (an idle session's open transaction; `DRAINING`, also across a restart; another operation refused, the owner's new seal joins; the replay completes), seal_force_rollback_ends_open_import_transaction, sealed_target_rejects_import}`; validation/publication landed: `namespace::fence::target::tests::{validation_session_is_read_only (query-only plus WAL defence, live capability checks, server snapshot and a concurrent exact replay while the receipt is committed but not published), publish_requires_validation_receipt, publish_is_idempotent}` | | 14 | Enable writes idempotent, survives restart and response loss, irreversible | landed: `namespace::fence::target::tests::{enable_writes_idempotent_and_irreversible, enable_writes_survives_restart}` | | 15 | Lost `EnableTargetWrites` response resolved from receipt/state | landed: `namespace::fence::target::tests::enable_writes_response_loss_resolved` | -| 16 | Read fence drains SQL, dump, `log_entries`, `snapshot`, including dead peers and forced termination | SQL landed: `namespace::fence::read::tests::{read_fence_waits_for_running_program, program_after_closing_gate_is_refused (parked after the read-closing gate, before the CAS), read_fence_cancels_at_deadline, unreleased_lease_answers_draining_and_replay_completes, idle_txn_fails_on_next_program, clear_read_fence_reopens_reads_not_writes, refused_read_fence_reopens_reads, attach_of_read_fenced_namespace_denied}`, `admin_shell::fence_tests::admin_shell_read_denied`; dump and replication landed: `namespace::fence::stream::tests::{dump_lease_released_on_cancel (a dump blocked mid-row on a peer that stopped reading is cancelled at the deadline, its lease released without the peer, the fence acknowledged, and the body ends with the fence error and no `COMMIT;`), read_fence_waits_for_dump, dump_refused_while_read_fenced, log_entries_stream_ends_typed, stream_lease_released_without_peer_read (a dead peer), snapshot_stream_ends_typed, replication_calls_denied_while_read_fenced, read_fence_forced_termination}`; planned: the HTTP-level check that an interrupted `/dump` response is aborted rather than completed, with `tests::fence::protocol::dump_codes` | +| 16 | Read fence drains SQL, dump, `log_entries`, `snapshot`, including dead peers and forced termination | SQL landed: `namespace::fence::read::tests::{read_fence_waits_for_running_program, program_after_closing_gate_is_refused (parked after the read-closing gate, before the CAS), read_fence_cancels_at_deadline, unreleased_lease_answers_draining_and_replay_completes, idle_txn_fails_on_next_program, clear_read_fence_reopens_reads_not_writes, refused_read_fence_reopens_reads, attach_of_read_fenced_namespace_denied}`, `admin_shell::fence_tests::admin_shell_read_denied`; dump and replication landed: `namespace::fence::stream::tests::{dump_lease_released_on_cancel (a dump blocked mid-row on a peer that stopped reading is cancelled at the deadline, its lease released without the peer, the fence acknowledged, and the body ends with the fence error and no `COMMIT;`), read_fence_waits_for_dump, dump_refused_while_read_fenced, log_entries_stream_ends_typed, stream_lease_released_without_peer_read (a dead peer), snapshot_stream_ends_typed, replication_calls_denied_while_read_fenced, read_fence_forced_termination}`; the response `/dump` returns: `namespace::fence::stream::tests::dump_response_aborted_on_cancel` (a dump cancelled by the read drain fails its response body, with the fence error and no `COMMIT;`); over HTTP: `tests::fence::protocol::dump_codes` (complete under a write fence; `423` + code under a read fence and on a quarantined target) | | 17 | Delete, reset, fork, restore, config, schema mutation rejected | landed: `tests::fence::lifecycle::lifecycle_rejected_while_fenced` (over the admin API, for a write-fenced source and a quarantined target: delete, fork as source and as destination, create with a `dump_url` whose file does not exist, create over the record, linking to a shared schema at creation and config `POST` are all `423` with the fence code; state, revision and data are unchanged and no copy exists; a target cannot be created with a shared schema; after release the source takes writes and config, fork and delete work again); `namespace::store::fence_tests::{reset_refused_while_fenced (called directly and as the replicator's reset callback; after release reset works and wipes the data), lifecycle_refused_while_fenced (fork either side, create over, delete, config and shared-schema link in the metastore transaction)}`; `schema::scheduler::test::fence::{acquire_rejects_shared_schema, migration_not_registered_while_linked_namespace_fenced}` | -| 18 | Codes through HTTP, Hrana, RPC, dump, replication, replica write proxy; distinguishable from auth/timeout/not-found; old peers compatible; no retry loops | `tests::fence::protocol::{http_codes, hrana_http_codes, hrana_ws_codes, rpc_codes, dump_codes, replication_codes, replica_proxy_preserves_code, auth_and_not_found_distinct, denial_not_retried}`; landed: `libsql-replication` `rpc::test::{proxy_error_stable_code_is_additive, replicated_fence_is_additive}` (each new field is skipped by a peer that does not know it, absent from an older peer's message, and absent fields encode exactly as before), `namespace::fence::stream::tests::hello_carries_replicated_fence` (no fence before acquisition and after release; state and revision while write-fenced; the stored configuration never carries it) | +| 18 | Codes through HTTP, Hrana, RPC, dump, replication, replica write proxy; distinguishable from auth/timeout/not-found; old peers compatible; no retry loops | user protocols landed: `tests::fence::protocol::{http_codes (legacy `/`, `/v1/execute`, `/v1/batch` for a write-fenced write, a read-fenced read, a quarantined target and a namespace whose marker cannot be decoded: `423` with `code`, and `detail` where there is one), hrana_http_codes (`/v2`, `/v3` pipelines and `/v3/cursor`: step and whole-request errors carry the code and the baton stays usable), hrana_ws_codes (the same over a WebSocket, whose stream reads again after the read fence is cleared), dump_codes, auth_and_not_found_distinct (`401` without or with a wrong credential and `404` for a missing namespace, with no fence code, on the same fenced server)}`, `error::fence_tests::fence_errors_carry_code` (the body through every wrapper; `block_*`'s `Blocked` keeps its mapping); planned: `tests::fence::protocol::{rpc_codes, replication_codes, replica_proxy_preserves_code, denial_not_retried}`; landed: `libsql-replication` `rpc::test::{proxy_error_stable_code_is_additive, replicated_fence_is_additive}` (each new field is skipped by a peer that does not know it, absent from an older peer's message, and absent fields encode exactly as before), `namespace::fence::stream::tests::hello_carries_replicated_fence` (no fence before acquisition and after release; state and revision while write-fenced; the stored configuration never carries it) | | 19 | Corrupt or unknown durable fence state fails closed | `fence::store::tests::corrupt_payload_fails_closed`, `unknown_format_version_fails_closed` | | 20 | Metrics and audit logs | `tests::fence::observability::metrics_and_labels`; `fence::audit::tests::audit_event_fields` | | 21 | Capability discovery and mixed-version protection | capability discovery landed: `tests::fence::admin::{capabilities, capabilities_when_disabled}`; planned: `fence::store::tests::legacy_mirror_and_fk_guard` (bounded, see section 18) | diff --git a/libsql-server/src/error.rs b/libsql-server/src/error.rs index f0cb631769..1408c67432 100644 --- a/libsql-server/src/error.rs +++ b/libsql-server/src/error.rs @@ -151,6 +151,29 @@ pub trait ResponseError: std::error::Error { impl ResponseError for Error {} +impl Error { + /// The fence denial this error carries, looking through the wrappers it can arrive in. + pub(crate) fn fence_error(&self) -> Option<&crate::namespace::fence::outcome::FenceError> { + match self { + Error::NamespaceFence(e) => Some(e), + Error::Migration(crate::schema::Error::NamespaceFence(e)) => Some(e), + Error::Ref(this) => this.fence_error(), + Error::Anyhow(e) => e.downcast_ref::().and_then(Error::fence_error), + _ => None, + } + } +} + +/// The HTTP response for a fence denial (`docs/NAMESPACE_FENCE.md` section 6): the fence status +/// and the JSON error body with the additive `code` (and `detail`) fields. +pub(crate) fn fence_error_response( + e: &crate::namespace::fence::outcome::FenceError, +) -> axum::response::Response { + let status = e.http_status(); + tracing::debug!("HTTP API: {status}, {e}"); + (status, axum::Json(e.http_error_body())).into_response() +} + impl IntoResponse for Error { fn into_response(self) -> axum::response::Response { (&self).into_response() @@ -226,7 +249,7 @@ impl IntoResponse for &Error { AttachInMigration => self.format_err(StatusCode::BAD_REQUEST), RuntimeTaskJoinError(_) => self.format_err(StatusCode::INTERNAL_SERVER_ERROR), NotAPrimary => self.format_err(StatusCode::BAD_REQUEST), - NamespaceFence(e) => self.format_err(e.outcome().admin_http_status()), + NamespaceFence(e) => fence_error_response(e), } } } @@ -338,3 +361,63 @@ impl IntoResponse for &ForkError { } } } + +#[cfg(test)] +mod fence_tests { + use super::*; + use crate::namespace::fence::outcome::{FenceDetail, FenceError, FenceOutcome}; + + async fn response(e: &Error) -> (StatusCode, serde_json::Value) { + let response = e.into_response(); + let status = response.status(); + let body = hyper::body::to_bytes(response.into_body()).await.unwrap(); + (status, serde_json::from_slice(&body).unwrap()) + } + + /// Section 6: a fence denial is `423` with the additive `code` (and `detail`) field, found + /// through every wrapper the error can arrive in; other errors keep their shape. + #[tokio::test] + async fn fence_errors_carry_code() { + for outcome in [ + FenceOutcome::MigrationWriteFenced, + FenceOutcome::MigrationReadFenced, + FenceOutcome::MigrationTargetQuarantined, + FenceOutcome::FenceStateUnavailable, + ] { + let fence = FenceError::new(outcome, "denied"); + let wrapped = [ + Error::NamespaceFence(fence.clone()), + Error::Ref(std::sync::Arc::new(Error::NamespaceFence(fence.clone()))), + Error::Anyhow(anyhow::anyhow!(Error::NamespaceFence(fence.clone()))), + Error::Migration(crate::schema::Error::NamespaceFence(fence.clone())), + ]; + for e in &wrapped { + assert_eq!(e.fence_error(), Some(&fence), "{e:?}"); + let (status, body) = response(e).await; + assert_eq!(status, StatusCode::LOCKED, "{e:?}"); + assert_eq!(body["code"], outcome.as_str(), "{e:?}"); + assert_eq!(body["error"], fence.to_string(), "{e:?}"); + assert!(body.get("detail").is_none(), "{body}"); + } + } + + let unavailable = FenceError::new(FenceOutcome::FenceStateUnavailable, "corrupt") + .with_detail(FenceDetail::CorruptRecord); + let (_, body) = response(&Error::NamespaceFence(unavailable)).await; + assert_eq!(body["detail"], "corrupt_record"); + + let precondition = FenceError::new(FenceOutcome::FencePreconditionFailed, "no") + .with_detail(FenceDetail::NotPrimary); + let (status, body) = response(&Error::NamespaceFence(precondition)).await; + assert_eq!(status, StatusCode::PRECONDITION_FAILED); + assert_eq!(body["code"], "FENCE_PRECONDITION_FAILED"); + assert_eq!(body["detail"], "not_primary"); + + // The legacy `block_*` refusal keeps its mapping and has no code. + let blocked = Error::Blocked(Some("maintenance".into())); + assert_eq!(blocked.fence_error(), None); + let (status, body) = response(&blocked).await; + assert_eq!(status, StatusCode::INTERNAL_SERVER_ERROR); + assert!(body.get("code").is_none(), "{body}"); + } +} diff --git a/libsql-server/src/hrana/batch.rs b/libsql-server/src/hrana/batch.rs index 9b73551774..4292660088 100644 --- a/libsql-server/src/hrana/batch.rs +++ b/libsql-server/src/hrana/batch.rs @@ -27,6 +27,10 @@ pub enum BatchError { ResponseTooLarge, #[error("Schema migration error: {message}")] SchemaError { message: String }, + /// The whole batch was refused by the namespace fence (`docs/NAMESPACE_FENCE.md` + /// section 6), for example a read under a read fence. + #[error(transparent)] + Fence(crate::namespace::fence::outcome::FenceError), } fn proto_cond_to_cond( @@ -183,6 +187,7 @@ pub fn batch_error_from_sqld_error(sqld_error: SqldError) -> Result { BatchError::ResponseTooLarge } + SqldError::NamespaceFence(e) => BatchError::Fence(e), sqld_error => return Err(sqld_error), }) } @@ -201,6 +206,7 @@ impl BatchError { Self::TransactionBusy => "TRANSACTION_BUSY", Self::ResponseTooLarge => "RESPONSE_TOO_LARGE", Self::SchemaError { message: _ } => "SCHEMA_MIGRATION_ERROR", + Self::Fence(e) => e.outcome().as_str(), } } } diff --git a/libsql-server/src/hrana/stmt.rs b/libsql-server/src/hrana/stmt.rs index 4414c72e1d..49bb65d2f7 100644 --- a/libsql-server/src/hrana/stmt.rs +++ b/libsql-server/src/hrana/stmt.rs @@ -49,6 +49,10 @@ pub enum StmtError { ResponseTooLarge, #[error("error executing a request on the primary: {0}")] Proxy(String), + /// A denial by the namespace fence (`docs/NAMESPACE_FENCE.md` section 6). Its Hrana code is + /// the fence's stable code. + #[error(transparent)] + Fence(crate::namespace::fence::outcome::FenceError), } pub async fn execute_stmt( @@ -216,6 +220,7 @@ pub fn stmt_error_from_sqld_error(sqld_error: SqldError) -> Result Ok(StmtError::Blocked { reason }), SqldError::RpcQueryError(e) => Ok(StmtError::Proxy(e.message)), + SqldError::NamespaceFence(e) => Ok(StmtError::Fence(e)), SqldError::RusqliteError(rusqlite_error) | SqldError::RusqliteErrorExtended(rusqlite_error, _) => match rusqlite_error { rusqlite::Error::SqliteFailure(sqlite_error, Some(message)) => { @@ -271,6 +276,7 @@ impl StmtError { Self::Blocked { .. } => "BLOCKED", Self::ResponseTooLarge => "RESPONSE_TOO_LARGE", Self::Proxy(_) => "PROXY_ERROR", + Self::Fence(e) => e.outcome().as_str(), } } } diff --git a/libsql-server/src/http/user/hrana_over_http_1.rs b/libsql-server/src/http/user/hrana_over_http_1.rs index 57ad971d91..e3809e7211 100644 --- a/libsql-server/src/http/user/hrana_over_http_1.rs +++ b/libsql-server/src/http/user/hrana_over_http_1.rs @@ -13,6 +13,9 @@ use super::db_factory::MakeConnectionExtractor; enum ResponseError { #[error(transparent)] Stmt(hrana::stmt::StmtError), + /// A whole batch refused by the namespace fence. + #[error(transparent)] + Fence(crate::namespace::fence::outcome::FenceError), } pub async fn handle_index() -> hyper::Response { @@ -81,6 +84,7 @@ pub(crate) async fn handle_batch( hrana::batch::execute_batch(&db, ctx, pgm, req_body.batch.replication_index) .await .map(|result| RespBody { result }) + .map_err(catch_batch_fence_error) .context("Could not execute batch") }) .await?; @@ -131,6 +135,7 @@ where fn response_error_response(err: ResponseError) -> hyper::Response { use hrana::stmt::StmtError; let status = match &err { + ResponseError::Fence(err) => err.http_status(), ResponseError::Stmt(err) => match err { StmtError::SqlParse { .. } | StmtError::SqlNoStmt @@ -145,6 +150,7 @@ fn response_error_response(err: ResponseError) -> hyper::Response { hyper::StatusCode::SERVICE_UNAVAILABLE } StmtError::SqliteError { .. } => hyper::StatusCode::INTERNAL_SERVER_ERROR, + StmtError::Fence(err) => err.http_status(), }, }; @@ -184,10 +190,21 @@ fn catch_stmt_error(err: anyhow::Error) -> anyhow::Error { } } +/// A batch the fence refused as a whole is answered like a refused statement, with the fence's +/// status and code, rather than as an internal error. +fn catch_batch_fence_error(err: anyhow::Error) -> anyhow::Error { + match err.downcast::() { + Ok(hrana::batch::BatchError::Fence(e)) => anyhow!(ResponseError::Fence(e)), + Ok(batch_err) => anyhow!(batch_err), + Err(err) => err, + } +} + impl ResponseError { pub fn code(&self) -> &'static str { match self { Self::Stmt(err) => err.code(), + Self::Fence(err) => err.outcome().as_str(), } } } diff --git a/libsql-server/src/http/user/mod.rs b/libsql-server/src/http/user/mod.rs index 7d46dce8cd..294c378e01 100644 --- a/libsql-server/src/http/user/mod.rs +++ b/libsql-server/src/http/user/mod.rs @@ -143,9 +143,12 @@ async fn handle_query( let db = connection_maker.create().await?; let builder = JsonHttpPayloadBuilder::new(); - let builder = db + let mut builder = db .execute_batch_or_rollback(batch, ctx, builder, query.replication_index) .await?; + if let Some(denial) = builder.take_fence_denial() { + return Err(Error::NamespaceFence(denial)); + } let res = ( [(header::CONTENT_TYPE, "application/json")], diff --git a/libsql-server/src/http/user/result_builder.rs b/libsql-server/src/http/user/result_builder.rs index 1364b5d956..cd341ba6cb 100644 --- a/libsql-server/src/http/user/result_builder.rs +++ b/libsql-server/src/http/user/result_builder.rs @@ -7,6 +7,7 @@ use serde::{Serialize, Serializer}; use serde_json::ser::{CompactFormatter, Formatter}; use std::sync::atomic::Ordering; +use crate::namespace::fence::outcome::FenceError; use crate::query_result_builder::{ Column, JsonFormatter, QueryBuilderConfig, QueryResultBuilder, QueryResultBuilderError, TOTAL_RESPONSE_SIZE, @@ -25,6 +26,9 @@ pub struct JsonHttpPayloadBuilder { step_row_count: usize, is_step_error: bool, is_step_empty: bool, + /// The first step error that was a fence denial. The legacy API has no per-step codes, so + /// a fenced batch is answered as a whole with the fence's status and code. + fence_denial: Option, } #[derive(Default)] @@ -112,8 +116,14 @@ impl JsonHttpPayloadBuilder { step_row_count: 0, is_step_error: false, is_step_empty: false, + fence_denial: None, } } + + /// The first fence denial reported as a step error, if any. + pub fn take_fence_denial(&mut self) -> Option { + self.fence_denial.take() + } } impl<'a> Serialize for HttpJsonValueSerializer<'a> { @@ -209,6 +219,9 @@ impl QueryResultBuilder for JsonHttpPayloadBuilder { } fn step_error(&mut self, error: crate::error::Error) -> Result<(), QueryResultBuilderError> { + if self.fence_denial.is_none() { + self.fence_denial = error.fence_error().cloned(); + } self.is_step_error = true; self.is_step_empty = false; self.buffer.truncate(self.checkpoint); diff --git a/libsql-server/src/namespace/fence/outcome.rs b/libsql-server/src/namespace/fence/outcome.rs index cf4df0c9af..debe1ba88f 100644 --- a/libsql-server/src/namespace/fence/outcome.rs +++ b/libsql-server/src/namespace/fence/outcome.rs @@ -275,6 +275,27 @@ impl FenceError { &self.message } + /// The status of this error on either HTTP API: the user API's mapping for a data-plane + /// denial (`423`), and the admin API's for any other outcome. + pub fn http_status(&self) -> StatusCode { + self.outcome + .user_http_status() + .unwrap_or_else(|| self.outcome.admin_http_status()) + } + + /// The JSON error body of the HTTP APIs for this error: the usual `error` message plus the + /// additive stable `code` and, when there is one, the bounded `detail`. + pub fn http_error_body(&self) -> serde_json::Value { + let mut body = serde_json::json!({ + "error": self.to_string(), + "code": self.outcome.as_str(), + }); + if let Some(detail) = self.detail { + body["detail"] = detail.as_str().into(); + } + body + } + /// A gRPC status for this error, if the outcome has a gRPC mapping. The code is in the /// [`GRPC_FENCE_CODE_METADATA`] entry and prefixes the message. pub fn to_grpc_status(&self) -> Option { diff --git a/libsql-server/src/namespace/fence/stream.rs b/libsql-server/src/namespace/fence/stream.rs index bd9ae56097..2711105a2a 100644 --- a/libsql-server/src/namespace/fence/stream.rs +++ b/libsql-server/src/namespace/fence/stream.rs @@ -326,6 +326,48 @@ mod tests { assert!(!String::from_utf8_lossy(&body).contains("COMMIT;")); } + /// The same cancellation seen through the HTTP response `/dump` returns: the body fails + /// instead of ending, which hyper sends as an aborted response (no final chunk), so a + /// client can never take the bytes it received for a complete dump. + #[tokio::test(flavor = "multi_thread")] + async fn dump_response_aborted_on_cancel() { + use axum::response::IntoResponse as _; + use hyper::body::HttpBody as _; + + let s = large_fenced_source().await; + let stream = dump(&s).await.unwrap(); + let response = axum::body::StreamBody::new(stream).into_response(); + assert_eq!(response.status(), hyper::StatusCode::OK); + let mut body = response.into_body(); + let mut head = Vec::new(); + while !String::from_utf8_lossy(&head).contains("INSERT INTO") { + head.extend_from_slice(&body.data().await.unwrap().unwrap()); + } + + let fenced = tokio::time::timeout(PROMPT, s.execute(read_fence(&s, 2, NOW))) + .await + .expect("the dump's lease was never released") + .unwrap(); + assert_eq!(fence_outcome(&fenced), FenceOutcome::Applied); + + let mut failed = false; + while let Some(chunk) = tokio::time::timeout(PROMPT, body.data()) + .await + .expect("the dump body stalled") + { + match chunk { + Ok(bytes) => head.extend_from_slice(&bytes), + Err(e) => { + assert!(e.to_string().contains("MIGRATION_READ_FENCED"), "{e}"); + failed = true; + break; + } + } + } + assert!(failed, "the response body ended cleanly"); + assert!(!String::from_utf8_lossy(&head).contains("COMMIT;")); + } + /// The read drain waits for a running dump rather than for time: the fence is acknowledged /// only once the dump has completed and released its lease. #[tokio::test(flavor = "multi_thread")] diff --git a/libsql-server/src/schema/error.rs b/libsql-server/src/schema/error.rs index 528251a2b3..b54cc7e87d 100644 --- a/libsql-server/src/schema/error.rs +++ b/libsql-server/src/schema/error.rs @@ -60,7 +60,7 @@ impl IntoResponse for &Error { self.format_err(StatusCode::BAD_REQUEST) } Error::MigrationExecuteError(e) => e.as_ref().into_response(), - Error::NamespaceFence(e) => self.format_err(e.outcome().admin_http_status()), + Error::NamespaceFence(e) => crate::error::fence_error_response(e), _ => self.format_err(StatusCode::INTERNAL_SERVER_ERROR), } } diff --git a/libsql-server/tests/fence/lifecycle.rs b/libsql-server/tests/fence/lifecycle.rs index 90ac6d9749..2ec5f77e54 100644 --- a/libsql-server/tests/fence/lifecycle.rs +++ b/libsql-server/tests/fence/lifecycle.rs @@ -145,6 +145,7 @@ fn lifecycle_rejected_while_fenced() { message.starts_with(code), "{what} on {ns}: expected {code}, got {body}" ); + assert_eq!(body["code"], code, "{what} on {ns}: {body}"); } // Nothing moved: same state and revision, and no copy was created. let (status, body) = admin.inspect(ns).await?; diff --git a/libsql-server/tests/fence/mod.rs b/libsql-server/tests/fence/mod.rs index c8be3cde94..32952a9ca9 100644 --- a/libsql-server/tests/fence/mod.rs +++ b/libsql-server/tests/fence/mod.rs @@ -4,11 +4,14 @@ mod admin; mod lifecycle; +mod protocol; use std::path::PathBuf; use std::time::Duration; use hyper::StatusCode; +use libsql_server::auth::user_auth_strategies::http_basic::HttpBasic; +use libsql_server::auth::Auth; use libsql_server::config::{AdminApiConfig, MetaStoreConfig, RpcServerConfig, UserApiConfig}; use s3s::header::AUTHORIZATION; use serde_json::{json, Value}; @@ -26,6 +29,8 @@ pub struct Primary { /// `None` starts the admin API without an auth key. pub admin_key: Option<&'static str>, pub fence_enabled: bool, + /// A basic-auth credential the user API requires; `None` leaves it unauthenticated. + pub user_credential: Option<&'static str>, } impl Default for Primary { @@ -33,6 +38,7 @@ impl Default for Primary { Self { admin_key: Some(ADMIN_KEY), fence_enabled: true, + user_credential: None, } } } @@ -49,13 +55,20 @@ pub fn make_primary(sim: &mut Sim, path: PathBuf, primary: Primary) { let Primary { admin_key, fence_enabled, + user_credential, } = primary; sim.host("primary", move || { let path = path.clone(); async move { let server = TestServer { path: path.into(), - user_api_config: UserApiConfig::default(), + user_api_config: UserApiConfig { + auth_strategy: match user_credential { + Some(credential) => Auth::new(HttpBasic::new(credential.into())), + None => UserApiConfig::::default().auth_strategy, + }, + ..Default::default() + }, admin_api_config: Some(AdminApiConfig { acceptor: TurmoilAcceptor::bind(([0, 0, 0, 0], 9090)).await?, connector: TurmoilConnector, diff --git a/libsql-server/tests/fence/protocol.rs b/libsql-server/tests/fence/protocol.rs new file mode 100644 index 0000000000..d7255fbd29 --- /dev/null +++ b/libsql-server/tests/fence/protocol.rs @@ -0,0 +1,853 @@ +//! Typed fence outcomes on the user-facing protocols: the legacy HTTP API, Hrana over HTTP +//! (`/v1`, `/v2`, `/v3`, cursors), Hrana over WebSocket and `/dump` +//! (`docs/NAMESPACE_FENCE.md` section 6; section 17 rows 3 and 18). + +use futures::SinkExt as _; +use hyper::{Body, Method, Request, StatusCode}; +use serde_json::{json, Value}; +use tempfile::tempdir; +use tokio_stream::StreamExt as _; +use tokio_tungstenite::tungstenite::{self, client::IntoClientRequest}; +use turmoil::net::TcpStream; +use uuid::Uuid; + +use super::{ + acquire_body, command_body, load_and_log_id, make_primary, sim, state_of, Admin, Primary, + ADMIN_KEY, +}; +use crate::common::net::TurmoilConnector; + +const WRITE_FENCED: &str = "MIGRATION_WRITE_FENCED"; +const READ_FENCED: &str = "MIGRATION_READ_FENCED"; +const QUARANTINED: &str = "MIGRATION_TARGET_QUARANTINED"; +const UNAVAILABLE: &str = "FENCE_STATE_UNAVAILABLE"; + +fn uuid(n: u128) -> Uuid { + Uuid::from_u128(n) +} + +/// The user API of `primary`, for any namespace, with an optional basic-auth credential. +struct User { + client: hyper::Client, + auth: Option, +} + +impl User { + fn new() -> Self { + Self::with_auth(None) + } + + fn with_auth(credential: Option<&str>) -> Self { + Self { + client: hyper::Client::builder().build(TurmoilConnector), + auth: credential.map(|c| format!("basic {c}")), + } + } + + async fn request( + &self, + method: Method, + ns: &str, + path: &str, + body: Option, + ) -> anyhow::Result<(StatusCode, String)> { + let mut request = Request::builder() + .method(method) + .uri(format!("http://{ns}.primary:8080{path}")); + if let Some(auth) = &self.auth { + request = request.header("authorization", auth.as_str()); + } + let request = match body { + Some(body) => request + .header("content-type", "application/json") + .body(Body::from(serde_json::to_vec(&body)?))?, + None => request.body(Body::empty())?, + }; + let response = self.client.request(request).await?; + let status = response.status(); + let body = hyper::body::to_bytes(response.into_body()).await?; + Ok((status, String::from_utf8_lossy(&body).into_owned())) + } + + async fn post(&self, ns: &str, path: &str, body: Value) -> anyhow::Result<(StatusCode, Value)> { + let (status, body) = self.request(Method::POST, ns, path, Some(body)).await?; + let value = serde_json::from_str(&body).unwrap_or(Value::String(body)); + Ok((status, value)) + } + + /// Legacy API: one batch of statements. + async fn legacy(&self, ns: &str, sql: &[&str]) -> anyhow::Result<(StatusCode, Value)> { + self.post(ns, "/", json!({ "statements": sql })).await + } + + /// Hrana 1 over HTTP: one statement. + async fn execute(&self, ns: &str, sql: &str) -> anyhow::Result<(StatusCode, Value)> { + self.post(ns, "/v1/execute", json!({ "stmt": { "sql": sql } })) + .await + } + + /// Hrana 1 over HTTP: one batch, each statement conditional on nothing. + async fn batch(&self, ns: &str, sql: &[&str]) -> anyhow::Result<(StatusCode, Value)> { + self.post(ns, "/v1/batch", json!({ "batch": batch(sql) })) + .await + } + + /// Hrana 2/3 over HTTP: one pipeline. + async fn pipeline( + &self, + ns: &str, + version: u8, + baton: Option<&str>, + requests: Value, + ) -> anyhow::Result<(StatusCode, Value)> { + self.post( + ns, + &format!("/v{version}/pipeline"), + json!({ "baton": baton, "requests": requests }), + ) + .await + } + + /// Hrana 3 over HTTP: a cursor over one batch, as the list of entries it returned. + async fn cursor(&self, ns: &str, sql: &[&str]) -> anyhow::Result<(StatusCode, Vec)> { + let (status, body) = self + .request( + Method::POST, + ns, + "/v3/cursor", + Some(json!({ "baton": null, "batch": batch(sql) })), + ) + .await?; + let entries = body + .lines() + .filter(|line| !line.trim().is_empty()) + .map(|line| serde_json::from_str(line).unwrap_or(Value::String(line.into()))) + .collect(); + Ok((status, entries)) + } + + async fn dump(&self, ns: &str) -> anyhow::Result<(StatusCode, String)> { + self.request(Method::GET, ns, "/dump", None).await + } +} + +/// A Hrana batch of unconditional steps. +fn batch(sql: &[&str]) -> Value { + json!({ "steps": sql.iter().map(|sql| json!({ "stmt": { "sql": sql } })).collect::>() }) +} + +fn execute_req(sql: &str) -> Value { + json!({ "type": "execute", "stmt": { "sql": sql } }) +} + +fn batch_req(sql: &[&str]) -> Value { + json!({ "type": "batch", "batch": batch(sql) }) +} + +/// A user-HTTP refusal by the fence: `423`, and the stable code in the additive `code` field. +#[track_caller] +fn assert_locked(what: &str, (status, body): &(StatusCode, Value), code: &str) { + assert_eq!(*status, StatusCode::LOCKED, "{what}: {body}"); + assert_eq!(body["code"], code, "{what}: {body}"); +} + +/// A Hrana error object carrying `code`. +#[track_caller] +fn assert_hrana_error(what: &str, error: &Value, code: &str) { + assert_eq!(error["code"], code, "{what}: {error}"); + assert!( + error["message"].as_str().unwrap_or_default().contains(code), + "{what}: {error}" + ); +} + +/// `ns` created, loaded with table `t` holding one row, and write-fenced by `op`. Returns the +/// fence's revision. +async fn write_fenced(admin: &Admin, ns: &str, op: Uuid) -> anyhow::Result { + admin.create_namespace(ns).await?; + let log_id = load_and_log_id(admin, ns).await?; + let (status, body) = admin + .command( + ns, + "source/acquire-write-fence", + acquire_body(op, uuid(op.as_u128() + 1), &log_id), + ) + .await?; + assert_eq!(status, StatusCode::OK, "{body}"); + assert_eq!(state_of(&body).0, "SOURCE_WRITE_FENCED", "{body}"); + Ok(state_of(&body).1) +} + +/// Runs the source command `route` from `state` at `revision` and returns the new revision. +async fn source_command( + admin: &Admin, + ns: &str, + op: Uuid, + command: u128, + route: &str, + (state, revision): (&str, u64), + expect: &str, +) -> anyhow::Result { + let (status, body) = admin + .command( + ns, + route, + command_body( + op, + uuid(command), + state, + revision, + json!({ "drain_policy": { "deadline_ms": 5000 } }), + ), + ) + .await?; + assert_eq!(status, StatusCode::OK, "{route}: {body}"); + assert_eq!(state_of(&body).0, expect, "{route}: {body}"); + Ok(state_of(&body).1) +} + +async fn read_fence(admin: &Admin, ns: &str, op: Uuid, revision: u64) -> anyhow::Result { + source_command( + admin, + ns, + op, + op.as_u128() + 2, + "source/set-read-fence", + ("SOURCE_WRITE_FENCED", revision), + "SOURCE_READ_FENCED", + ) + .await +} + +async fn clear_read_fence(admin: &Admin, ns: &str, op: Uuid, revision: u64) -> anyhow::Result { + let (status, body) = admin + .command( + ns, + "source/clear-read-fence", + command_body( + op, + uuid(op.as_u128() + 3), + "SOURCE_READ_FENCED", + revision, + json!({}), + ), + ) + .await?; + assert_eq!(status, StatusCode::OK, "{body}"); + assert_eq!(state_of(&body).0, "SOURCE_WRITE_FENCED", "{body}"); + Ok(state_of(&body).1) +} + +async fn release(admin: &Admin, ns: &str, op: Uuid, revision: u64) -> anyhow::Result<()> { + let (status, body) = admin + .command( + ns, + "source/release-write-fence", + command_body( + op, + uuid(op.as_u128() + 4), + "SOURCE_WRITE_FENCED", + revision, + json!({}), + ), + ) + .await?; + assert_eq!(status, StatusCode::OK, "{body}"); + assert_eq!(state_of(&body).0, "RELEASED", "{body}"); + Ok(()) +} + +async fn quarantined_target(admin: &Admin, ns: &str, op: Uuid) -> anyhow::Result<()> { + let (status, body) = admin + .command( + ns, + "target/create-quarantined", + command_body(op, uuid(op.as_u128() + 1), "ABSENT", 0, json!({})), + ) + .await?; + assert_eq!(status, StatusCode::OK, "{body}"); + assert_eq!(state_of(&body).0, "TARGET_QUARANTINED", "{body}"); + Ok(()) +} + +/// Every user HTTP entry point answers a fence denial with `423` and the stable code, whether +/// the fence refused one statement (a write) or the whole request (a read, a quarantined +/// target, a fence state the server cannot establish). +#[test] +fn http_codes() { + let mut sim = sim(); + let tmp = tempdir().unwrap(); + // A namespace directory whose fence marker cannot be decoded: its fence state cannot be + // established, so the server refuses it (section 13.3). + let broken = tmp.path().join("dbs").join("broken"); + std::fs::create_dir_all(&broken).unwrap(); + std::fs::write(broken.join(".fence"), b"garbage").unwrap(); + make_primary(&mut sim, tmp.path().to_path_buf(), Primary::default()); + sim.client("client", async { + let admin = Admin::new(Some(ADMIN_KEY)); + let user = User::new(); + let op = uuid(0x100); + let rev = write_fenced(&admin, "src", op).await?; + + // Write-fenced: writes are refused, reads are served. + assert_locked( + "legacy write", + &user.legacy("src", &["insert into t values (2)"]).await?, + WRITE_FENCED, + ); + assert_locked( + "legacy read then write", + &user + .legacy("src", &["select * from t", "insert into t values (2)"]) + .await?, + WRITE_FENCED, + ); + let (status, body) = user.legacy("src", &["select * from t"]).await?; + assert_eq!(status, StatusCode::OK, "{body}"); + assert_locked( + "v1 execute write", + &user.execute("src", "insert into t values (2)").await?, + WRITE_FENCED, + ); + let (status, body) = user.execute("src", "select * from t").await?; + assert_eq!(status, StatusCode::OK, "{body}"); + // A Hrana batch reports the refused step in its own error, as for any step error. + let (status, body) = user + .batch("src", &["select * from t", "insert into t values (2)"]) + .await?; + assert_eq!(status, StatusCode::OK, "{body}"); + assert!(body["result"]["step_results"][0].is_object(), "{body}"); + assert_hrana_error( + "v1 batch write step", + &body["result"]["step_errors"][1], + WRITE_FENCED, + ); + + // Read-fenced: reads are refused as a whole. + read_fence(&admin, "src", op, rev).await?; + assert_locked( + "legacy read", + &user.legacy("src", &["select * from t"]).await?, + READ_FENCED, + ); + assert_locked( + "v1 execute read", + &user.execute("src", "select * from t").await?, + READ_FENCED, + ); + assert_locked( + "v1 batch read", + &user.batch("src", &["select * from t"]).await?, + READ_FENCED, + ); + + // A quarantined target serves nothing. + quarantined_target(&admin, "tgt", uuid(0x200)).await?; + assert_locked( + "legacy on target", + &user.legacy("tgt", &["select 1"]).await?, + QUARANTINED, + ); + assert_locked( + "v1 execute on target", + &user.execute("tgt", "select 1").await?, + QUARANTINED, + ); + assert_locked( + "v1 batch on target", + &user.batch("tgt", &["select 1"]).await?, + QUARANTINED, + ); + + // A namespace whose fence state is unknown is refused before a connection exists. + for (what, response) in [ + ("legacy", user.legacy("broken", &["select 1"]).await?), + ("v1 execute", user.execute("broken", "select 1").await?), + ( + "v2 pipeline", + user.pipeline("broken", 2, None, json!([execute_req("select 1")])) + .await?, + ), + ] { + assert_locked(what, &response, UNAVAILABLE); + assert_eq!(response.1["detail"], "corrupt_record", "{what}"); + } + Ok(()) + }); + sim.run().unwrap(); +} + +/// Hrana over HTTP (`/v2`, `/v3`, `/v3/cursor`): a fence denial is a Hrana error with the +/// stable code, and the stream stays usable. +#[test] +fn hrana_http_codes() { + let mut sim = sim(); + let tmp = tempdir().unwrap(); + make_primary(&mut sim, tmp.path().to_path_buf(), Primary::default()); + sim.client("client", async { + let admin = Admin::new(Some(ADMIN_KEY)); + let user = User::new(); + let op = uuid(0x100); + let rev = write_fenced(&admin, "src", op).await?; + + for version in [2, 3] { + let what = format!("v{version}"); + let (status, body) = user + .pipeline( + "src", + version, + None, + json!([ + execute_req("insert into t values (2)"), + execute_req("select * from t"), + batch_req(&["select * from t", "insert into t values (2)"]), + ]), + ) + .await?; + assert_eq!(status, StatusCode::OK, "{what}: {body}"); + let results = &body["results"]; + assert_eq!(results[0]["type"], "error", "{what}: {body}"); + assert_hrana_error(&what, &results[0]["error"], WRITE_FENCED); + assert_eq!(results[1]["type"], "ok", "{what}: {body}"); + assert_eq!(results[2]["type"], "ok", "{what}: {body}"); + assert_hrana_error( + &what, + &results[2]["response"]["result"]["step_errors"][1], + WRITE_FENCED, + ); + // The stream survives the denial. + let baton = body["baton"].as_str().expect("the stream stays open"); + let (status, body) = user + .pipeline( + "src", + version, + Some(baton), + json!([execute_req("select * from t"), { "type": "close" }]), + ) + .await?; + assert_eq!(status, StatusCode::OK, "{what}: {body}"); + assert_eq!(body["results"][0]["type"], "ok", "{what}: {body}"); + } + let (status, entries) = user + .cursor("src", &["select * from t", "insert into t values (2)"]) + .await?; + assert_eq!(status, StatusCode::OK, "{entries:?}"); + let step_error = entries + .iter() + .find(|e| e["type"] == "step_error") + .unwrap_or_else(|| panic!("no step error: {entries:?}")); + assert_eq!(step_error["step"], 1, "{entries:?}"); + assert_hrana_error("cursor write step", &step_error["error"], WRITE_FENCED); + + read_fence(&admin, "src", op, rev).await?; + for version in [2, 3] { + let what = format!("v{version} read-fenced"); + let (status, body) = user + .pipeline( + "src", + version, + None, + json!([ + execute_req("select * from t"), + batch_req(&["select * from t"]), + { "type": "close" }, + ]), + ) + .await?; + assert_eq!(status, StatusCode::OK, "{what}: {body}"); + let results = &body["results"]; + assert_hrana_error(&what, &results[0]["error"], READ_FENCED); + assert_hrana_error(&what, &results[1]["error"], READ_FENCED); + assert_eq!(results[2]["type"], "ok", "{what}: {body}"); + } + let (status, entries) = user.cursor("src", &["select * from t"]).await?; + assert_eq!(status, StatusCode::OK, "{entries:?}"); + let error = entries + .iter() + .find(|e| e["type"] == "error") + .unwrap_or_else(|| panic!("no error entry: {entries:?}")); + assert_hrana_error("cursor read", &error["error"], READ_FENCED); + + quarantined_target(&admin, "tgt", uuid(0x200)).await?; + let (status, body) = user + .pipeline("tgt", 3, None, json!([execute_req("select 1")])) + .await?; + assert_eq!(status, StatusCode::OK, "{body}"); + assert_hrana_error("target", &body["results"][0]["error"], QUARANTINED); + Ok(()) + }); + sim.run().unwrap(); +} + +/// Acceptance test (section 17 row 3): in a batch, the steps before a write run, the write +/// step is refused with the fence code, and conditions see the refusal as a failed step. +#[test] +fn batch_denied_mid_batch() { + let mut sim = sim(); + let tmp = tempdir().unwrap(); + make_primary(&mut sim, tmp.path().to_path_buf(), Primary::default()); + sim.client("client", async { + let admin = Admin::new(Some(ADMIN_KEY)); + let user = User::new(); + write_fenced(&admin, "src", uuid(0x100)).await?; + + let batch = json!({ + "steps": [ + { "stmt": { "sql": "select count(*) from t" } }, + { "stmt": { "sql": "insert into t values (2)" } }, + { + "condition": { "type": "ok", "step": 1 }, + "stmt": { "sql": "insert into t values (3)" }, + }, + { + "condition": { "type": "not", "cond": { "type": "ok", "step": 1 } }, + "stmt": { "sql": "select count(*) from t" }, + }, + ], + }); + let (status, body) = user + .pipeline( + "src", + 3, + None, + json!([{ "type": "batch", "batch": batch }, { "type": "close" }]), + ) + .await?; + assert_eq!(status, StatusCode::OK, "{body}"); + let result = &body["results"][0]["response"]["result"]; + let count = |step: usize| result["step_results"][step]["rows"][0][0]["value"].clone(); + assert_eq!(count(0), "1", "{body}"); + assert!(result["step_errors"][0].is_null(), "{body}"); + assert!(result["step_results"][1].is_null(), "{body}"); + assert_hrana_error("write step", &result["step_errors"][1], WRITE_FENCED); + // The step conditional on the write did not run; the one conditional on its failure did. + assert!(result["step_results"][2].is_null(), "{body}"); + assert!(result["step_errors"][2].is_null(), "{body}"); + assert_eq!(count(3), "1", "{body}"); + Ok(()) + }); + sim.run().unwrap(); +} + +type Ws = tokio_tungstenite::WebSocketStream; + +/// A Hrana 3 WebSocket session to `ns`, after `hello`. +async fn ws_connect(ns: &str) -> anyhow::Result { + let mut request = format!("ws://{ns}.primary:8080").into_client_request()?; + request + .headers_mut() + .insert("sec-websocket-protocol", "hrana3".parse()?); + let conn = TcpStream::connect("primary:8080").await?; + let (mut ws, _) = tokio_tungstenite::client_async(request, conn).await?; + ws.send(tungstenite::Message::Text( + json!({ "type": "hello", "jwt": null }).to_string(), + )) + .await?; + let hello = ws_next(&mut ws).await?; + assert_eq!(hello["type"], "hello_ok", "{hello}"); + Ok(ws) +} + +async fn ws_next(ws: &mut Ws) -> anyhow::Result { + match ws.try_next().await? { + Some(tungstenite::Message::Text(text)) => Ok(serde_json::from_str(&text)?), + other => anyhow::bail!("unexpected WebSocket message {other:?}"), + } +} + +/// Sends one request and returns the server's answer to it. +async fn ws_request(ws: &mut Ws, request_id: u64, request: Value) -> anyhow::Result { + ws.send(tungstenite::Message::Text( + json!({ "type": "request", "request_id": request_id, "request": request }).to_string(), + )) + .await?; + let response = ws_next(ws).await?; + assert_eq!(response["request_id"], request_id, "{response}"); + Ok(response) +} + +async fn ws_execute(ws: &mut Ws, request_id: u64, sql: &str) -> anyhow::Result { + ws_request( + ws, + request_id, + json!({ "type": "execute", "stream_id": 1, "stmt": { "sql": sql } }), + ) + .await +} + +#[track_caller] +fn assert_ws_ok(what: &str, response: &Value) { + assert_eq!(response["type"], "response_ok", "{what}: {response}"); +} + +#[track_caller] +fn assert_ws_error(what: &str, response: &Value, code: &str) { + assert_eq!(response["type"], "response_error", "{what}: {response}"); + assert_hrana_error(what, &response["error"], code); +} + +/// Hrana over WebSocket: fence denials are request errors with the stable code; the connection +/// and its stream stay usable. +#[test] +fn hrana_ws_codes() { + let mut sim = sim(); + let tmp = tempdir().unwrap(); + make_primary(&mut sim, tmp.path().to_path_buf(), Primary::default()); + sim.client("client", async { + let admin = Admin::new(Some(ADMIN_KEY)); + let op = uuid(0x100); + let rev = write_fenced(&admin, "src", op).await?; + + let mut ws = ws_connect("src").await?; + let opened = + ws_request(&mut ws, 1, json!({ "type": "open_stream", "stream_id": 1 })).await?; + assert_ws_ok("open_stream", &opened); + let denied = ws_execute(&mut ws, 2, "insert into t values (2)").await?; + assert_ws_error("write", &denied, WRITE_FENCED); + assert_ws_ok("read", &ws_execute(&mut ws, 3, "select * from t").await?); + let batched = ws_request( + &mut ws, + 4, + json!({ + "type": "batch", + "stream_id": 1, + "batch": batch(&["select * from t", "insert into t values (2)"]), + }), + ) + .await?; + assert_ws_ok("batch", &batched); + assert_hrana_error( + "batch write step", + &batched["response"]["result"]["step_errors"][1], + WRITE_FENCED, + ); + + let rev = read_fence(&admin, "src", op, rev).await?; + let denied = ws_execute(&mut ws, 5, "select * from t").await?; + assert_ws_error("read-fenced read", &denied, READ_FENCED); + let denied = ws_request( + &mut ws, + 6, + json!({ "type": "batch", "stream_id": 1, "batch": batch(&["select * from t"]) }), + ) + .await?; + assert_ws_error("read-fenced batch", &denied, READ_FENCED); + + // Clearing the read fence makes the same stream readable again. + clear_read_fence(&admin, "src", op, rev).await?; + assert_ws_ok( + "read after clear", + &ws_execute(&mut ws, 7, "select * from t").await?, + ); + Ok(()) + }); + sim.run().unwrap(); +} + +/// Acceptance test (section 17 row 3): a WebSocket session whose transaction began before the +/// fence cannot write while the fence holds, nor after it is released; only a transaction +/// begun after the release writes. +#[test] +fn old_ws_session_cannot_write() { + let mut sim = sim(); + let tmp = tempdir().unwrap(); + make_primary(&mut sim, tmp.path().to_path_buf(), Primary::default()); + sim.client("client", async { + let admin = Admin::new(Some(ADMIN_KEY)); + admin.create_namespace("src").await?; + let log_id = load_and_log_id(&admin, "src").await?; + + let mut ws = ws_connect("src").await?; + let opened = + ws_request(&mut ws, 1, json!({ "type": "open_stream", "stream_id": 1 })).await?; + assert_ws_ok("open_stream", &opened); + assert_ws_ok("begin", &ws_execute(&mut ws, 2, "begin").await?); + assert_ws_ok("read", &ws_execute(&mut ws, 3, "select * from t").await?); + + let op = uuid(0x100); + let (status, body) = admin + .command( + "src", + "source/acquire-write-fence", + acquire_body(op, uuid(0x101), &log_id), + ) + .await?; + assert_eq!(status, StatusCode::OK, "{body}"); + let rev = state_of(&body).1; + + let denied = ws_execute(&mut ws, 4, "insert into t values (2)").await?; + assert_ws_error("write while fenced", &denied, WRITE_FENCED); + + release(&admin, "src", op, rev).await?; + // The transaction still began under the earlier write generation. + let denied = ws_execute(&mut ws, 5, "insert into t values (3)").await?; + assert_ws_error("write after release", &denied, WRITE_FENCED); + assert_ws_ok("rollback", &ws_execute(&mut ws, 6, "rollback").await?); + // A fresh transaction on the same session writes. + assert_ws_ok( + "fresh write", + &ws_execute(&mut ws, 7, "insert into t values (4)").await?, + ); + let rows = ws_execute(&mut ws, 8, "select x from t order by x").await?; + assert_ws_ok("rows", &rows); + let values: Vec = rows["response"]["result"]["rows"] + .as_array() + .unwrap() + .iter() + .map(|row| row[0]["value"].clone()) + .collect(); + assert_eq!(values, vec![json!("1"), json!("4")], "{rows}"); + Ok(()) + }); + sim.run().unwrap(); +} + +/// `/dump`: served in full under a write fence; refused with `423` and the code under a read +/// fence and on a quarantined target. (An export cancelled mid-way is an aborted response +/// body: `namespace::fence::stream::tests::dump_response_aborted_on_cancel`.) +#[test] +fn dump_codes() { + let mut sim = sim(); + let tmp = tempdir().unwrap(); + make_primary(&mut sim, tmp.path().to_path_buf(), Primary::default()); + sim.client("client", async { + let admin = Admin::new(Some(ADMIN_KEY)); + let user = User::new(); + let op = uuid(0x100); + let rev = write_fenced(&admin, "src", op).await?; + + let (status, dump) = user.dump("src").await?; + assert_eq!(status, StatusCode::OK, "{dump}"); + assert!(dump.contains("INSERT INTO t"), "{dump}"); + assert!(dump.trim_end().ends_with("COMMIT;"), "{dump}"); + + read_fence(&admin, "src", op, rev).await?; + let (status, body) = user.dump("src").await?; + assert_locked( + "read-fenced dump", + &(status, serde_json::from_str(&body)?), + READ_FENCED, + ); + + quarantined_target(&admin, "tgt", uuid(0x200)).await?; + let (status, body) = user.dump("tgt").await?; + assert_locked( + "target dump", + &(status, serde_json::from_str(&body)?), + QUARANTINED, + ); + Ok(()) + }); + sim.run().unwrap(); +} + +/// Section 17 row 18: authentication failures and missing namespaces keep their own statuses +/// and carry no fence code, so a client can tell them from a fence denial. +#[test] +fn auth_and_not_found_distinct() { + const CREDENTIAL: &str = "dXNlcjpwYXNz"; + let mut sim = sim(); + let tmp = tempdir().unwrap(); + make_primary( + &mut sim, + tmp.path().to_path_buf(), + Primary { + user_credential: Some(CREDENTIAL), + ..Default::default() + }, + ); + sim.client("client", async { + let admin = Admin::new(Some(ADMIN_KEY)); + let user = User::with_auth(Some(CREDENTIAL)); + admin.create_namespace("src").await?; + for sql in ["create table t (x)", "insert into t values (1)"] { + let (status, body) = user.execute("src", sql).await?; + assert_eq!(status, StatusCode::OK, "{body}"); + } + let (_, body) = admin.inspect("src").await?; + let log_id = body["fence"]["incarnation"]["current_log_id"] + .as_str() + .unwrap() + .to_string(); + let (status, body) = admin + .command( + "src", + "source/acquire-write-fence", + acquire_body(uuid(0x100), uuid(0x101), &log_id), + ) + .await?; + assert_eq!(status, StatusCode::OK, "{body}"); + + let write = "insert into t values (2)"; + let anonymous = User::new(); + let wrong = User::with_auth(Some("d3Jvbmc6d3Jvbmc=")); + for (what, response, expected) in [ + ( + "no credential", + anonymous.execute("src", write).await?, + StatusCode::UNAUTHORIZED, + ), + ( + "wrong credential", + wrong.execute("src", write).await?, + StatusCode::UNAUTHORIZED, + ), + ( + "no credential, legacy", + anonymous.legacy("src", &[write]).await?, + StatusCode::UNAUTHORIZED, + ), + ( + "no credential, v3", + anonymous + .pipeline("src", 3, None, json!([execute_req(write)])) + .await?, + StatusCode::UNAUTHORIZED, + ), + ( + "missing namespace", + user.execute("nope", write).await?, + StatusCode::NOT_FOUND, + ), + ( + "missing namespace, v3", + user.pipeline("nope", 3, None, json!([execute_req(write)])) + .await?, + StatusCode::NOT_FOUND, + ), + ] { + let (status, body) = &response; + assert_eq!(*status, expected, "{what}: {body}"); + assert!( + body.get("code").map_or(true, |code| !code + .as_str() + .unwrap_or_default() + .starts_with("MIGRATION_") + && code != UNAVAILABLE), + "{what}: {body}" + ); + } + // With the credential, on the existing namespace, it is the fence that answers. + assert_locked( + "fenced write", + &user.execute("src", write).await?, + WRITE_FENCED, + ); + assert_locked( + "fenced legacy write", + &user.legacy("src", &[write]).await?, + WRITE_FENCED, + ); + let (status, body) = user + .pipeline("src", 3, None, json!([execute_req(write)])) + .await?; + assert_eq!(status, StatusCode::OK, "{body}"); + assert_hrana_error( + "fenced v3 write", + &body["results"][0]["error"], + WRITE_FENCED, + ); + Ok(()) + }); + sim.run().unwrap(); +} From 5719ff364a20825c1590287554c1a3604c1e0f45 Mon Sep 17 00:00:00 2001 From: River Date: Wed, 30 Sep 2026 10:06:56 +0000 Subject: [PATCH 22/33] libsql-server: carry fence outcomes through RPC and the replica write proxy The primary's proxy service now fills the additive `stable_code` field (with `code = SQL_ERROR`) for fence denials, on step errors and program errors, streamed and unary. A fence denial before a program runs (the namespace or JWT-key lookup, connection creation, a unary program refused as a whole) is the typed `FAILED_PRECONDITION` status with the stable code in `x-libsql-fence-code`, never `UNAVAILABLE`, which the write proxy retries without bound. On a replica, the write proxy maps a proxied error or status carrying a fence code back to `Error::NamespaceFence`, so the replica answers its client exactly as the primary would: `423` with the `code` field on the HTTP APIs and the stable code as the Hrana error code. Errors from an older primary, which never sets the field, keep their old mapping. Capability discovery now reports `proxy_stable_code: true`. Co-authored-by: Tomasz Szymczyszyn --- docs/NAMESPACE_FENCE.md | 14 +- libsql-server/src/connection/write_proxy.rs | 8 +- libsql-server/src/error.rs | 26 ++ libsql-server/src/http/user/mod.rs | 2 +- libsql-server/src/namespace/fence/mod.rs | 2 +- libsql-server/src/namespace/fence/outcome.rs | 75 ++++ libsql-server/src/namespace/mod.rs | 3 + libsql-server/src/rpc/proxy.rs | 366 +++++++++++++++++-- libsql-server/src/rpc/streaming_exec.rs | 2 +- libsql-server/tests/fence/admin.rs | 2 +- libsql-server/tests/fence/mod.rs | 35 +- libsql-server/tests/fence/protocol.rs | 131 ++++++- 12 files changed, 621 insertions(+), 45 deletions(-) diff --git a/docs/NAMESPACE_FENCE.md b/docs/NAMESPACE_FENCE.md index 90cd8b6517..70e465025a 100644 --- a/docs/NAMESPACE_FENCE.md +++ b/docs/NAMESPACE_FENCE.md @@ -420,7 +420,13 @@ message Error { This is additive in proto3: older peers skip the unknown field; a newer replica treats an absent field as "no typed outcome". An error without a stable code encodes exactly as before. For fence denials the primary sets `code = SQL_ERROR` and `stable_code`. The replica threads `stable_code` through `Error::RpcQueryError` into the same user HTTP status and Hrana code the primary would have returned. A fence denial at proxy connection creation is returned as `FAILED_PRECONDITION` with the metadata above, never `UNAVAILABLE`, so the replica's write-proxy reconnect loop (which retries `UNAVAILABLE` without bound) does not spin on it. -Capability discovery reports `proxy_stable_code: true` only once a server both fills the field on fence denials and maps it on the replica side; a server that has the field in its protocol but does neither reports `false`. +Capability discovery reports `proxy_stable_code: true` only once a server both fills the field on fence denials and maps it on the replica side; a server that has the field in its protocol but does neither reports `false`. This server reports `true`. + +Implementation: + +- **Primary.** `From for proxy::Error` sets `code = SQL_ERROR` and `stable_code` for any error whose `Error::fence_error()` is a data-plane denial, so a write refused at the WAL (a step error) and a read refused at program start (a program error) both carry it, on the streaming (`stream_exec`) and the unary (`execute`) service alike. Everything else is encoded as before. A fence error before a program runs — the namespace lookup (including the lookup of its JWT key), connection creation, or a unary program refused as a whole — is the typed `FAILED_PRECONDITION` status (`FenceError::to_grpc_status`), where it used to be `INTERNAL`, `UNAVAILABLE` or `PERMISSION_DENIED`. +- **Replica.** `Error::from_proxy_error` turns a step or program error whose `stable_code` names a data-plane denial into `Error::NamespaceFence`, and `Error::from_proxy_status` does the same for the typed status at proxy connection creation. From there the replica answers exactly as the primary would (section 6.0: `423` + `code` on HTTP, the stable code on Hrana). The peer's message is kept; the bounded `detail` is not carried by the proxy protocol. An error without `stable_code` (an older primary), or with a code this server does not know, keeps its old mapping (`Error::RpcQueryError`). The write proxy's connection loop retries only `UNAVAILABLE`, so a denial is answered at once; each refused write is delegated to the primary exactly once. +- **Replica proxy for embedded replicas** (`rpc/replica_proxy.rs`) forwards requests and responses unchanged, so the primary's `stable_code` and typed statuses reach the embedded replica as they are. ### 6.2 Replicated configuration addition @@ -731,7 +737,7 @@ A namespace that startup cannot recover is registered `UNKNOWN_UNAVAILABLE` in m - `destroy_on_error`: the broken metastore is renamed to `metastore.broken-` rather than deleted (if the rename fails, startup fails instead), and directories with markers are registered `UNKNOWN_UNAVAILABLE` after the rebuild. Without fences it still deletes, as before. - A metastore without fence tables next to a directory that holds a marker (rebuilt, recovered, or restored from a backup taken before the tables existed) makes that namespace `UNKNOWN_UNAVAILABLE`, even when its config row is present. - Metastore restore from backup: the marker comparison (section 5.6) makes any namespace whose record went backwards or disappeared `UNKNOWN_UNAVAILABLE`. A restored record is never trusted over a newer marker. Surfacing the provenance (`restored_from_backup`, backup generation) in the capability endpoint, `InspectFence`, a metric and a startup line is part of the admin API (section 4). -- Replica-kind servers: lazy creation of a name refused by the primary with a fence code does not create a local default namespace. This needs the proxy's stable code (section 6.1) and lands with the protocol mappings. +- Replica-kind servers: lazy creation of a name refused by the primary with a fence code must not create a local default namespace. The primary's refusal arrives as the typed status of the replication `hello` (section 6.2), not through the write proxy; handling it on the replica is not implemented yet (planned with the replica's handling of the replicated fence). ### 13.4 Shared schema @@ -766,7 +772,7 @@ How each path that can reach namespace data or lifecycle is covered. File refere | `connection/connection_manager.rs` | Authoritative gate in `begin_write_txn` before `acquire()`, generation check, non-`BUSY` refusal; classed queue entries; queue wake on generation change; active-writer query, release notification, `abort_active()`. | | `connection/legacy.rs` | `FenceConnState` wired into every `LegacyConnection`; the controller is passed to `MakeLegacyConnection::new` before the first connection. `with_raw` users are covered by the WAL gate. | | HTTP `/`, `/v1/execute`, `/v1/batch`, Hrana `/v2`, `/v3`, cursors, WebSocket, dev route | Core checks and WAL gate; typed status and `code` field; Hrana codes, for step and whole-request denials alike, with the stream left usable (section 6.0). | -| `rpc/proxy.rs`, `rpc/streaming_exec.rs`, `rpc/replica_proxy.rs`, `connection/write_proxy.rs` | `stable_code` set on the primary and carried back on the replica; connection-creation denials are `FAILED_PRECONDITION`, not `UNAVAILABLE`. | +| `rpc/proxy.rs`, `rpc/streaming_exec.rs`, `rpc/replica_proxy.rs`, `connection/write_proxy.rs` | `stable_code` set on the primary for step and program denials (streamed and unary); namespace-lookup, JWT-key-lookup, connection-creation and unary whole-program denials are `FAILED_PRECONDITION` with the code, never `UNAVAILABLE`; the replica maps both back to `Error::NamespaceFence` and never retries them; the replica proxy forwards them unchanged (section 6.1). | | `namespace/meta_store.rs` `handle()`, `restore()`, `maybe_recover_from_fs`, `destroy_on_error`, `process`/`try_process`, `remove`, bottomless metastore restore | Non-creating lookups; fail-closed decoding; marker-aware recovery; rename-aside; publish only after commit; fence check inside the config and remove transactions; restore provenance surfaced. | | `namespace/store.rs` `with`, `load_namespace`, `make_namespace`, eviction | Registry check before setup; controller passed into setup; eviction keeps the registry entry. | | `store.rs` `create`, `destroy`, `reset`, `fork`, `checkpoint`, restore options | `check_lifecycle` (section 13.5): create refuses a name whose fence denies lifecycle work; `CreateTargetQuarantined` is the atomic quarantined create; destroy is refused in the metastore transaction; reset and fork as source are checked under the namespace's transition lock; fork's destination is checked before anything is stored and again under its entry lock; every restore (create with a dump, reset, fork to a point in time) is one of these paths and is refused there; checkpoint uses a non-creating lookup and skips vacuum. | @@ -831,7 +837,7 @@ Planned test names; the table is updated as tests land. | 15 | Lost `EnableTargetWrites` response resolved from receipt/state | landed: `namespace::fence::target::tests::enable_writes_response_loss_resolved` | | 16 | Read fence drains SQL, dump, `log_entries`, `snapshot`, including dead peers and forced termination | SQL landed: `namespace::fence::read::tests::{read_fence_waits_for_running_program, program_after_closing_gate_is_refused (parked after the read-closing gate, before the CAS), read_fence_cancels_at_deadline, unreleased_lease_answers_draining_and_replay_completes, idle_txn_fails_on_next_program, clear_read_fence_reopens_reads_not_writes, refused_read_fence_reopens_reads, attach_of_read_fenced_namespace_denied}`, `admin_shell::fence_tests::admin_shell_read_denied`; dump and replication landed: `namespace::fence::stream::tests::{dump_lease_released_on_cancel (a dump blocked mid-row on a peer that stopped reading is cancelled at the deadline, its lease released without the peer, the fence acknowledged, and the body ends with the fence error and no `COMMIT;`), read_fence_waits_for_dump, dump_refused_while_read_fenced, log_entries_stream_ends_typed, stream_lease_released_without_peer_read (a dead peer), snapshot_stream_ends_typed, replication_calls_denied_while_read_fenced, read_fence_forced_termination}`; the response `/dump` returns: `namespace::fence::stream::tests::dump_response_aborted_on_cancel` (a dump cancelled by the read drain fails its response body, with the fence error and no `COMMIT;`); over HTTP: `tests::fence::protocol::dump_codes` (complete under a write fence; `423` + code under a read fence and on a quarantined target) | | 17 | Delete, reset, fork, restore, config, schema mutation rejected | landed: `tests::fence::lifecycle::lifecycle_rejected_while_fenced` (over the admin API, for a write-fenced source and a quarantined target: delete, fork as source and as destination, create with a `dump_url` whose file does not exist, create over the record, linking to a shared schema at creation and config `POST` are all `423` with the fence code; state, revision and data are unchanged and no copy exists; a target cannot be created with a shared schema; after release the source takes writes and config, fork and delete work again); `namespace::store::fence_tests::{reset_refused_while_fenced (called directly and as the replicator's reset callback; after release reset works and wipes the data), lifecycle_refused_while_fenced (fork either side, create over, delete, config and shared-schema link in the metastore transaction)}`; `schema::scheduler::test::fence::{acquire_rejects_shared_schema, migration_not_registered_while_linked_namespace_fenced}` | -| 18 | Codes through HTTP, Hrana, RPC, dump, replication, replica write proxy; distinguishable from auth/timeout/not-found; old peers compatible; no retry loops | user protocols landed: `tests::fence::protocol::{http_codes (legacy `/`, `/v1/execute`, `/v1/batch` for a write-fenced write, a read-fenced read, a quarantined target and a namespace whose marker cannot be decoded: `423` with `code`, and `detail` where there is one), hrana_http_codes (`/v2`, `/v3` pipelines and `/v3/cursor`: step and whole-request errors carry the code and the baton stays usable), hrana_ws_codes (the same over a WebSocket, whose stream reads again after the read fence is cleared), dump_codes, auth_and_not_found_distinct (`401` without or with a wrong credential and `404` for a missing namespace, with no fence code, on the same fenced server)}`, `error::fence_tests::fence_errors_carry_code` (the body through every wrapper; `block_*`'s `Blocked` keeps its mapping); planned: `tests::fence::protocol::{rpc_codes, replication_codes, replica_proxy_preserves_code, denial_not_retried}`; landed: `libsql-replication` `rpc::test::{proxy_error_stable_code_is_additive, replicated_fence_is_additive}` (each new field is skipped by a peer that does not know it, absent from an older peer's message, and absent fields encode exactly as before), `namespace::fence::stream::tests::hello_carries_replicated_fence` (no fence before acquisition and after release; state and revision while write-fenced; the stored configuration never carries it) | +| 18 | Codes through HTTP, Hrana, RPC, dump, replication, replica write proxy; distinguishable from auth/timeout/not-found; old peers compatible; no retry loops | user protocols landed: `tests::fence::protocol::{http_codes (legacy `/`, `/v1/execute`, `/v1/batch` for a write-fenced write, a read-fenced read, a quarantined target and a namespace whose marker cannot be decoded: `423` with `code`, and `detail` where there is one), hrana_http_codes (`/v2`, `/v3` pipelines and `/v3/cursor`: step and whole-request errors carry the code and the baton stays usable), hrana_ws_codes (the same over a WebSocket, whose stream reads again after the read fence is cleared), dump_codes, auth_and_not_found_distinct (`401` without or with a wrong credential and `404` for a missing namespace, with no fence code, on the same fenced server)}`, `error::fence_tests::fence_errors_carry_code` (the body through every wrapper; `block_*`'s `Blocked` keeps its mapping); RPC and replica write proxy landed: `rpc::proxy::fence_tests::rpc_codes` (on the primary's proxy service: a write-fenced write is a step error with `SQL_ERROR` + `stable_code`, reads are served, a read-fenced read is a program error with the code when streamed and the typed `FAILED_PRECONDITION` status when unary, and a namespace whose fence state is unknown is refused with the typed status before any connection), `rpc::proxy::fence_tests::replica_maps_proxied_denials` (step, program and connection-status denials become the fence error on the replica; an older primary's error without `stable_code`, an unknown code and other errors keep their mapping), `namespace::fence::outcome::tests::peer_denials_round_trip`, `tests::fence::protocol::{replica_proxy_preserves_code (writes through a replica to a write-fenced primary: `423` + `MIGRATION_WRITE_FENCED` on legacy `/` and `/v1/execute`, the step error on `/v1/batch`, the Hrana error code on `/v2` and `/v3`; reads on the replica served; writes through the replica work after release), denial_not_retried (each refused write is delegated exactly once and answered well within the write proxy's first retry backoff)}`; capability: `tests::fence::admin::capabilities` asserts `proxy_stable_code: true`; planned: `tests::fence::protocol::replication_codes`; landed: `libsql-replication` `rpc::test::{proxy_error_stable_code_is_additive, replicated_fence_is_additive}` (each new field is skipped by a peer that does not know it, absent from an older peer's message, and absent fields encode exactly as before), `namespace::fence::stream::tests::hello_carries_replicated_fence` (no fence before acquisition and after release; state and revision while write-fenced; the stored configuration never carries it) | | 19 | Corrupt or unknown durable fence state fails closed | `fence::store::tests::corrupt_payload_fails_closed`, `unknown_format_version_fails_closed` | | 20 | Metrics and audit logs | `tests::fence::observability::metrics_and_labels`; `fence::audit::tests::audit_event_fields` | | 21 | Capability discovery and mixed-version protection | capability discovery landed: `tests::fence::admin::{capabilities, capabilities_when_disabled}`; planned: `fence::store::tests::legacy_mirror_and_fk_guard` (bounded, see section 18) | diff --git a/libsql-server/src/connection/write_proxy.rs b/libsql-server/src/connection/write_proxy.rs index 030920c6ec..cc86a2170d 100644 --- a/libsql-server/src/connection/write_proxy.rs +++ b/libsql-server/src/connection/write_proxy.rs @@ -266,13 +266,15 @@ impl RemoteConnection { let response_stream = match client.stream_exec(req).await { Ok(i) => i.into_inner(), Err(e) => { + // Only `UNAVAILABLE` is retried. A fence denial is `FAILED_PRECONDITION` + // with its stable code, answered to the client as the primary's denial. if e.code() == Code::Unavailable { tracing::error!("retrying proxy connection: {}", e); tokio::time::sleep(Duration::from_millis(500) * 2u32.pow(retries)).await; retries += 1; continue; } else { - return Err(e.into()); + return Err(Error::from_proxy_status(e)); } } }; @@ -387,7 +389,7 @@ where ) } exec_resp::Response::DescribeResp(_) => Err(Error::PrimaryStreamMisuse), - exec_resp::Response::Error(e) => Err(Error::RpcQueryError(e)), + exec_resp::Response::Error(e) => Err(Error::from_proxy_error(e)), } }; @@ -430,7 +432,7 @@ where Ok(false) } - exec_resp::Response::Error(e) => Err(Error::RpcQueryError(e)), + exec_resp::Response::Error(e) => Err(Error::from_proxy_error(e)), exec_resp::Response::ProgramResp(_) => Err(Error::PrimaryStreamMisuse), }; diff --git a/libsql-server/src/error.rs b/libsql-server/src/error.rs index 1408c67432..b6903c4a82 100644 --- a/libsql-server/src/error.rs +++ b/libsql-server/src/error.rs @@ -162,6 +162,32 @@ impl Error { _ => None, } } + + /// A step or program error the primary returned through the write proxy. A fence denial + /// from a primary that fills the additive `stable_code` field + /// (`docs/NAMESPACE_FENCE.md` section 6.1) becomes the same [`Error::NamespaceFence`] a + /// local denial is, so the replica answers its client exactly as the primary would. Any + /// other error, and every error from a primary that does not fill the field, stays + /// [`Error::RpcQueryError`]. + pub(crate) fn from_proxy_error(e: crate::rpc::proxy::rpc::Error) -> Self { + let fence = e.stable_code.as_deref().and_then(|code| { + crate::namespace::fence::outcome::FenceError::from_proxy_stable_code(code, &e.message) + }); + match fence { + Some(fence) => Error::NamespaceFence(fence), + None => Error::RpcQueryError(e), + } + } + + /// A gRPC status from the primary's proxy service: its typed fence denial + /// (`FAILED_PRECONDITION` with the stable code, section 6.1) as [`Error::NamespaceFence`], + /// anything else unchanged. + pub(crate) fn from_proxy_status(status: tonic::Status) -> Self { + match crate::namespace::fence::outcome::FenceError::from_grpc_status(&status) { + Some(fence) => Error::NamespaceFence(fence), + None => Error::RpcQueryExecutionError(status), + } + } } /// The HTTP response for a fence denial (`docs/NAMESPACE_FENCE.md` section 6): the fence status diff --git a/libsql-server/src/http/user/mod.rs b/libsql-server/src/http/user/mod.rs index 294c378e01..eed6e751b5 100644 --- a/libsql-server/src/http/user/mod.rs +++ b/libsql-server/src/http/user/mod.rs @@ -3,7 +3,7 @@ pub(crate) mod dump; mod extract; mod hrana_over_http_1; mod listen; -mod result_builder; +pub(crate) mod result_builder; mod trace; mod types; #[macro_use] diff --git a/libsql-server/src/namespace/fence/mod.rs b/libsql-server/src/namespace/fence/mod.rs index af5b76182e..43c384659a 100644 --- a/libsql-server/src/namespace/fence/mod.rs +++ b/libsql-server/src/namespace/fence/mod.rs @@ -49,7 +49,7 @@ pub const FENCE_PROTOCOL_VERSION: u32 = 1; /// Whether this server fills the proxy protocol's additive `Error.stable_code` field and maps /// it on the replica side (`docs/NAMESPACE_FENCE.md` section 6.1). Reported by capability /// discovery so that deployment tooling can check every server before fences are used. -pub const PROXY_STABLE_CODE: bool = false; +pub const PROXY_STABLE_CODE: bool = true; /// The identity of this server process: its build and an id generated once per process. It is /// written into records and receipts, and reported by the admin API. diff --git a/libsql-server/src/namespace/fence/outcome.rs b/libsql-server/src/namespace/fence/outcome.rs index debe1ba88f..723b7b5ab6 100644 --- a/libsql-server/src/namespace/fence/outcome.rs +++ b/libsql-server/src/namespace/fence/outcome.rs @@ -318,6 +318,38 @@ impl FenceError { .parse() .ok() } + + /// The fence denial a peer reported as a gRPC status produced by + /// [`FenceError::to_grpc_status`], with the peer's message. `None` for any other status, + /// including one whose code is not the fence mapping of the outcome it names. + pub fn from_grpc_status(status: &tonic::Status) -> Option { + let outcome = Self::outcome_from_grpc_status(status)?; + if outcome.grpc_code() != Some(status.code()) { + return None; + } + Some(Self::from_peer(outcome, status.message())) + } + + /// The fence denial a peer reported in the proxy protocol's `Error.stable_code` field + /// (section 6.1), with the peer's message. `None` for a code this server does not know or + /// that is not a data-plane denial, which the peer never sends: such an error keeps its + /// untyped mapping. + pub fn from_proxy_stable_code(stable_code: &str, message: &str) -> Option { + let outcome = stable_code.parse::().ok()?; + outcome.proxy_stable_code()?; + Some(Self::from_peer(outcome, message)) + } + + /// A denial reported by a peer. The peer's message is `": "` (this type's + /// `Display`); the prefix is dropped so that it is not repeated. The bounded `detail` is + /// not carried by the peer protocols. + fn from_peer(outcome: FenceOutcome, message: &str) -> FenceError { + let message = message + .strip_prefix(outcome.as_str()) + .and_then(|m| m.strip_prefix(": ")) + .unwrap_or(message); + Self::new(outcome, message) + } } #[cfg(test)] @@ -425,6 +457,49 @@ mod tests { assert!(control.to_grpc_status().is_none()); } + /// A denial crosses the proxy and RPC protocols as the same outcome and message, and + /// nothing else is mistaken for one. + #[test] + fn peer_denials_round_trip() { + let err = FenceError::new(FenceOutcome::MigrationWriteFenced, "writes are fenced"); + let status = err.to_grpc_status().unwrap(); + assert_eq!(FenceError::from_grpc_status(&status), Some(err.clone())); + // The metadata alone is not enough: the code must be the outcome's gRPC mapping. + let mut forged = tonic::Status::unavailable("MIGRATION_WRITE_FENCED: x"); + forged.metadata_mut().insert( + GRPC_FENCE_CODE_METADATA, + tonic::metadata::MetadataValue::from_static("MIGRATION_WRITE_FENCED"), + ); + assert_eq!(FenceError::from_grpc_status(&forged), None); + assert_eq!( + FenceError::from_grpc_status(&tonic::Status::failed_precondition("x")), + None + ); + + for outcome in FenceOutcome::ALL { + let decoded = FenceError::from_proxy_stable_code(outcome.as_str(), "m"); + match outcome.proxy_stable_code() { + Some(_) => assert_eq!(decoded, Some(FenceError::new(outcome, "m"))), + None => assert_eq!(decoded, None, "{outcome}"), + } + } + assert_eq!( + FenceError::from_proxy_stable_code("MIGRATION_READ_FENCED", &err_msg()), + Some(FenceError::new( + FenceOutcome::MigrationReadFenced, + "reads are fenced" + )) + ); + assert_eq!( + FenceError::from_proxy_stable_code("SOMETHING_NEW", "m"), + None + ); + + fn err_msg() -> String { + FenceError::new(FenceOutcome::MigrationReadFenced, "reads are fenced").to_string() + } + } + #[test] #[should_panic] fn success_is_not_an_error() { diff --git a/libsql-server/src/namespace/mod.rs b/libsql-server/src/namespace/mod.rs index f75dbd700f..69d93368f3 100644 --- a/libsql-server/src/namespace/mod.rs +++ b/libsql-server/src/namespace/mod.rs @@ -28,6 +28,9 @@ pub mod replication_wal; mod schema_lock; mod store; +#[cfg(test)] +pub(crate) use store::fence_tests::open_store as open_test_store; + pub type ResetCb = Box; /// Resolves a namespace that a program ATTACHes: its directory, and its fence controller, which /// admits the attachment as a read of that namespace (`docs/NAMESPACE_FENCE.md` section 9). diff --git a/libsql-server/src/rpc/proxy.rs b/libsql-server/src/rpc/proxy.rs index 32bc6aeb96..b189b1143e 100644 --- a/libsql-server/src/rpc/proxy.rs +++ b/libsql-server/src/rpc/proxy.rs @@ -40,7 +40,14 @@ pub mod rpc { impl From for Error { fn from(other: SqldError) -> Self { + // A fence denial is an ordinary SQL error to an older replica, and carries its + // stable code in the additive `stable_code` field for one that maps it + // (`docs/NAMESPACE_FENCE.md` section 6.1). + let stable_code = other + .fence_error() + .and_then(|e| e.outcome().proxy_stable_code()); let code = match other { + _ if stable_code.is_some() => ErrorCode::SqlError, SqldError::LibSqlInvalidQueryParams(_) => ErrorCode::SqlError, SqldError::LibSqlTxTimeout => ErrorCode::TxTimeout, SqldError::LibSqlTxBusy => ErrorCode::TxBusy, @@ -57,7 +64,7 @@ pub mod rpc { message: other.to_string(), code: code as i32, extended_code, - stable_code: None, + stable_code: stable_code.map(Into::into), } } } @@ -65,6 +72,7 @@ pub mod rpc { impl From for ErrorCode { fn from(other: SqldError) -> Self { match other { + _ if other.fence_error().is_some() => ErrorCode::SqlError, SqldError::LibSqlInvalidQueryParams(_) => ErrorCode::SqlError, SqldError::LibSqlTxTimeout => ErrorCode::TxTimeout, SqldError::LibSqlTxBusy => ErrorCode::TxBusy, @@ -316,6 +324,9 @@ impl ProxyService { Ok(Ok(None)) => self.user_auth_strategy.clone(), Err(e) => match e.as_ref() { crate::error::Error::NamespaceDoesntExist(_) => None, + // A namespace the fence refuses is refused with the typed status, never + // retried by the write proxy (`docs/NAMESPACE_FENCE.md` section 6.1). + e if fence_status(e).is_some() => Err(fence_status(e).unwrap())?, _ => Err(tonic::Status::internal(format!( "Error fetching jwt key for a namespace: {}", e @@ -568,6 +579,24 @@ pub async fn garbage_collect(clients: &mut HashMap> } } +/// The typed status of a fence denial on the proxy service (`docs/NAMESPACE_FENCE.md` section +/// 6.1): `FAILED_PRECONDITION` with the stable code, never `UNAVAILABLE`, which the write +/// proxy retries without bound. +fn fence_status(e: &crate::error::Error) -> Option { + e.fence_error()?.to_grpc_status() +} + +/// The status for an error looking up the namespace a proxy request names. +fn namespace_status(e: crate::error::Error) -> tonic::Status { + if let crate::error::Error::NamespaceDoesntExist(_) = e { + tonic::Status::failed_precondition(NAMESPACE_DOESNT_EXIST) + } else if let Some(status) = fence_status(&e) { + status + } else { + tonic::Status::internal(e.to_string()) + } +} + #[tonic::async_trait] impl Proxy for ProxyService { type StreamExecStream = Pin> + Send>>; @@ -586,18 +615,13 @@ impl Proxy for ProxyService { (connection_maker, notifier) }) .await - .map_err(|e| { - if let crate::error::Error::NamespaceDoesntExist(_) = e { - tonic::Status::failed_precondition(NAMESPACE_DOESNT_EXIST) - } else { - tonic::Status::internal(e.to_string()) - } - })?; + .map_err(namespace_status)?; - let conn = connection_maker - .create() - .await - .map_err(|e| tonic::Status::unavailable(format!("Unable to create DB: {:?}", e)))?; + let conn = connection_maker.create().await.map_err(|e| { + fence_status(&e).unwrap_or_else(|| { + tonic::Status::unavailable(format!("Unable to create DB: {:?}", e)) + }) + })?; let stream = make_proxy_stream(conn, ctx, req.into_inner()); @@ -618,13 +642,7 @@ impl Proxy for ProxyService { .namespaces .with(ctx.namespace().clone(), |ns| ns.db.connection_maker()) .await - .map_err(|e| { - if let crate::error::Error::NamespaceDoesntExist(_) = e { - tonic::Status::failed_precondition(NAMESPACE_DOESNT_EXIST) - } else { - tonic::Status::internal(e.to_string()) - } - })?; + .map_err(namespace_status)?; let conn = { let lock = self.clients.upgradable_read().await; @@ -646,7 +664,9 @@ impl Proxy for ProxyService { conn } Err(e) => { - return Err(tonic::Status::new(tonic::Code::Internal, e.to_string())) + return Err(fence_status(&e).unwrap_or_else(|| { + tonic::Status::new(tonic::Code::Internal, e.to_string()) + })) } } } @@ -660,7 +680,11 @@ impl Proxy for ProxyService { .execute_program(pgm, ctx, builder, None) .await // TODO: this is no necessarily a permission denied error! - .map_err(|e| tonic::Status::new(tonic::Code::PermissionDenied, e.to_string()))?; + .map_err(|e| { + fence_status(&e).unwrap_or_else(|| { + tonic::Status::new(tonic::Code::PermissionDenied, e.to_string()) + }) + })?; Ok(tonic::Response::new(builder.into_ret())) } @@ -692,13 +716,7 @@ impl Proxy for ProxyService { .namespaces .with(ctx.namespace().clone(), |ns| ns.db.connection_maker()) .await - .map_err(|e| { - if let crate::error::Error::NamespaceDoesntExist(_) = e { - tonic::Status::failed_precondition(NAMESPACE_DOESNT_EXIST) - } else { - tonic::Status::internal(e.to_string()) - } - })?; + .map_err(namespace_status)?; let DescribeRequest { client_id, stmt } = req.into_inner(); let client_id = Uuid::from_str(&client_id).unwrap(); @@ -720,7 +738,11 @@ impl Proxy for ProxyService { lock.insert(client_id, conn.clone()); conn } - Err(e) => return Err(tonic::Status::new(tonic::Code::Internal, e.to_string())), + Err(e) => { + return Err(fence_status(&e).unwrap_or_else(|| { + tonic::Status::new(tonic::Code::Internal, e.to_string()) + })) + } } } }; @@ -756,3 +778,289 @@ impl Proxy for ProxyService { })) } } + +/// Fence denials on the proxy protocol (`docs/NAMESPACE_FENCE.md` section 6.1): the primary +/// fills the additive `stable_code` of a step or program error and answers a namespace it +/// refuses with the typed `FAILED_PRECONDITION` status; the replica side turns both back into +/// the same fence denial, and an older primary's errors keep their untyped mapping. +#[cfg(test)] +mod fence_tests { + use libsql_replication::rpc::proxy::error::ErrorCode; + use libsql_replication::rpc::proxy::exec_resp; + use libsql_replication::rpc::proxy::{ + resp_step, ExecReq, ProgramResp, RespStep, StepError, StreamProgramReq, + }; + use libsql_replication::rpc::replication::NAMESPACE_METADATA_KEY; + use tempfile::tempdir; + use tokio_stream::wrappers::ReceiverStream; + use tokio_stream::StreamExt as _; + use tonic::metadata::BinaryMetadataValue; + + use super::*; + use crate::connection::program::Program; + use crate::error::Error; + use crate::namespace::fence::drain::tests::{fence_outcome, Source, LONG}; + use crate::namespace::fence::outcome::{FenceError, FenceOutcome, GRPC_FENCE_CODE_METADATA}; + use crate::namespace::fence::read::tests::{fenced_source, read_fence}; + use crate::namespace::fence::store as fence_store; + use crate::query_result_builder::QueryBuilderConfig; + + const WRITE: &str = "insert into t values (3)"; + const READ: &str = "select count(*) from t"; + + fn service(s: &NamespaceStore) -> ProxyService { + ProxyService::new(s.clone(), None, false) + } + + fn request(ns: &str, msg: T) -> tonic::Request { + let mut req = tonic::Request::new(msg); + req.metadata_mut().insert_bin( + NAMESPACE_METADATA_KEY, + BinaryMetadataValue::from_bytes(ns.as_bytes()), + ); + Authenticated::FullAccess.upgrade_grpc_request(&mut req); + req + } + + fn program_req(sql: &str) -> rpc::ProgramReq { + rpc::ProgramReq { + client_id: Uuid::new_v4().to_string(), + pgm: Some(Program::seq(&[sql]).into()), + } + } + + #[track_caller] + fn assert_fence_status(status: &tonic::Status, outcome: FenceOutcome) { + assert_eq!(status.code(), tonic::Code::FailedPrecondition, "{status:?}"); + assert_eq!( + status + .metadata() + .get(GRPC_FENCE_CODE_METADATA) + .and_then(|v| v.to_str().ok()), + Some(outcome.as_str()), + "{status:?}" + ); + assert_eq!( + FenceError::outcome_from_grpc_status(status), + Some(outcome), + "{status:?}" + ); + } + + #[track_caller] + fn assert_proxy_error(error: &rpc::Error, outcome: FenceOutcome) { + assert_eq!(error.code, ErrorCode::SqlError as i32, "{error:?}"); + assert_eq!(error.stable_code.as_deref(), Some(outcome.as_str())); + assert!(error.message.starts_with(outcome.as_str()), "{error:?}"); + } + + /// The step errors of a streamed program, run on `ns` through the primary's proxy stream. + async fn stream_program(s: &Source, sql: &str) -> Vec { + let conn = s + .store + .with("ns".into(), |ns| ns.db.connection_maker()) + .await + .unwrap() + .create() + .await + .unwrap(); + let ctx = RequestContext::new( + Authenticated::FullAccess, + "ns".into(), + s.store.meta_store().clone(), + ); + let (snd, rcv) = tokio::sync::mpsc::channel(1); + let stream = make_proxy_stream(conn, ctx, ReceiverStream::new(rcv)); + tokio::pin!(stream); + snd.send(Ok(ExecReq { + request_id: 0, + request: Some(libsql_replication::rpc::proxy::exec_req::Request::Execute( + StreamProgramReq { + pgm: Some(Program::seq(&[sql]).into()), + }, + )), + })) + .await + .unwrap(); + // The request stream stays open until the program is answered: a closed request stream + // ends the proxy stream. + let mut responses = Vec::new(); + while let Some(resp) = stream.next().await { + let resp = resp.unwrap().response.unwrap(); + let last = match &resp { + exec_resp::Response::ProgramResp(p) => p + .steps + .iter() + .any(|s| matches!(s.step, Some(resp_step::Step::Finish(_)))), + _ => true, + }; + responses.push(resp); + if last { + break; + } + } + drop(snd); + responses + } + + fn step_errors(responses: &[exec_resp::Response]) -> Vec<&rpc::Error> { + responses + .iter() + .filter_map(|r| match r { + exec_resp::Response::ProgramResp(p) => Some(p), + _ => None, + }) + .flat_map(|p| &p.steps) + .filter_map(|s| match &s.step { + Some(resp_step::Step::StepError(StepError { error: Some(e) })) => Some(e), + _ => None, + }) + .collect() + } + + /// Section 6.1 on the primary: a write refused at the WAL gate is a step error with the + /// stable code, a read refused by the read fence is a program error with it (streamed) or + /// the typed status (unary), and a namespace whose fence state is unknown is refused with + /// the typed status before any connection is made. + #[tokio::test] + async fn rpc_codes() { + let s = fenced_source().await; + + // Write-fenced: the write step carries MIGRATION_WRITE_FENCED, reads are served. + let streamed = stream_program(&s, WRITE).await; + let errors = step_errors(&streamed); + assert_eq!(errors.len(), 1, "{streamed:?}"); + assert_proxy_error(errors[0], FenceOutcome::MigrationWriteFenced); + assert!(step_errors(&stream_program(&s, READ).await).is_empty()); + + let unary = service(&s.store) + .execute(request("ns", program_req(WRITE))) + .await + .unwrap() + .into_inner(); + match &unary.results[..] { + [QueryResult { + row_result: Some(RowResult::Error(e)), + }] => assert_proxy_error(e, FenceOutcome::MigrationWriteFenced), + other => panic!("{other:?}"), + } + + // Read-fenced: the whole program is refused. + let fenced = s.execute(read_fence(&s, 2, LONG)).await.unwrap(); + assert_eq!(fence_outcome(&fenced), FenceOutcome::Applied); + match &stream_program(&s, READ).await[..] { + [exec_resp::Response::Error(e)] => { + assert_proxy_error(e, FenceOutcome::MigrationReadFenced) + } + other => panic!("{other:?}"), + } + let status = service(&s.store) + .execute(request("ns", program_req(READ))) + .await + .unwrap_err(); + assert_fence_status(&status, FenceOutcome::MigrationReadFenced); + + // A namespace whose fence state cannot be established. + let tmp = tempdir().unwrap(); + let broken = tmp.path().join("dbs").join("broken"); + std::fs::create_dir_all(&broken).unwrap(); + std::fs::write(broken.join(fence_store::MARKER_FILE_NAME), b"garbage").unwrap(); + let store = crate::namespace::open_test_store(tmp.path()).await; + let status = service(&store) + .execute(request("broken", program_req(READ))) + .await + .unwrap_err(); + assert_fence_status(&status, FenceOutcome::FenceStateUnavailable); + assert_eq!( + namespace_status(Error::NamespaceFence( + store + .with("broken".into(), |_| ()) + .await + .unwrap_err() + .fence_error() + .unwrap() + .clone() + )) + .code(), + tonic::Code::FailedPrecondition + ); + // A connection a fence refuses is never reported as `UNAVAILABLE`. + let e = Error::NamespaceFence(FenceError::new(FenceOutcome::MigrationWriteFenced, "x")); + assert_fence_status( + &fence_status(&e).unwrap(), + FenceOutcome::MigrationWriteFenced, + ); + assert!(fence_status(&Error::LibSqlTxBusy).is_none()); + } + + /// Section 6.1 on the replica: a proxied fence denial becomes the fence error the replica + /// answers its client with, whether it arrives as a step error, a program error or a + /// status; an older primary's errors (no `stable_code`) keep their untyped mapping. + #[tokio::test] + async fn replica_maps_proxied_denials() { + let denial = FenceError::new(FenceOutcome::MigrationWriteFenced, "writes are fenced"); + let proxied: rpc::Error = Error::NamespaceFence(denial.clone()).into(); + assert_eq!(proxied.code, ErrorCode::SqlError as i32); + match Error::from_proxy_error(proxied.clone()) { + Error::NamespaceFence(e) => assert_eq!(e, denial), + other => panic!("{other:?}"), + } + // An older primary: same error without the additive field. + let old = rpc::Error { + stable_code: None, + ..proxied.clone() + }; + assert!(matches!( + Error::from_proxy_error(old.clone()), + Error::RpcQueryError(e) if e == old + )); + // A code this server does not know stays untyped too. + let unknown = rpc::Error { + stable_code: Some("SOMETHING_NEW".into()), + ..proxied.clone() + }; + assert!(matches!( + Error::from_proxy_error(unknown), + Error::RpcQueryError(_) + )); + // Other errors are unchanged. + let busy: rpc::Error = Error::LibSqlTxBusy.into(); + assert_eq!(busy.stable_code, None); + assert_eq!(busy.code, ErrorCode::TxBusy as i32); + + // Status at connection time: typed denial, not retried; anything else unchanged. + match Error::from_proxy_status(denial.to_grpc_status().unwrap()) { + Error::NamespaceFence(e) => assert_eq!(e, denial), + other => panic!("{other:?}"), + } + assert!(matches!( + Error::from_proxy_status(tonic::Status::failed_precondition("x")), + Error::RpcQueryExecutionError(_) + )); + + // A step error applied to the replica's result builder. + let resp = ProgramResp { + steps: [ + resp_step::Step::Init(Default::default()), + resp_step::Step::BeginStep(Default::default()), + resp_step::Step::StepError(StepError { + error: Some(proxied), + }), + resp_step::Step::FinishStep(Default::default()), + resp_step::Step::Finish(Default::default()), + ] + .into_iter() + .map(|step| RespStep { step: Some(step) }) + .collect(), + }; + let mut builder = crate::http::user::result_builder::JsonHttpPayloadBuilder::new(); + crate::rpc::streaming_exec::apply_program_resp_to_builder( + &QueryBuilderConfig::default(), + &mut builder, + resp, + |_, _| (), + ) + .unwrap(); + assert_eq!(builder.take_fence_denial(), Some(denial)); + } +} diff --git a/libsql-server/src/rpc/streaming_exec.rs b/libsql-server/src/rpc/streaming_exec.rs index 482ce2074c..351fd7cb8d 100644 --- a/libsql-server/src/rpc/streaming_exec.rs +++ b/libsql-server/src/rpc/streaming_exec.rs @@ -225,7 +225,7 @@ pub fn apply_program_resp_to_builder( last_insert_rowid, }) => builder.finish_step(affected_row_count, last_insert_rowid)?, Step::StepError(StepError { error: Some(err) }) => { - builder.step_error(crate::error::Error::RpcQueryError(err))? + builder.step_error(crate::error::Error::from_proxy_error(err))? } Step::ColsDescription(ColsDescription { columns }) => { let cols = columns.iter().map(|c| Column { diff --git a/libsql-server/tests/fence/admin.rs b/libsql-server/tests/fence/admin.rs index 2c64935c46..5a5a2aec03 100644 --- a/libsql-server/tests/fence/admin.rs +++ b/libsql-server/tests/fence/admin.rs @@ -26,7 +26,7 @@ fn capabilities() { assert_eq!(body["fence_protocol_version"], 1); assert_eq!(body["enabled"], true); assert_eq!(body["active_fences"], 0); - assert_eq!(body["proxy_stable_code"], false); + assert_eq!(body["proxy_stable_code"], true); let commands: Vec<&str> = body["commands"] .as_array() .unwrap() diff --git a/libsql-server/tests/fence/mod.rs b/libsql-server/tests/fence/mod.rs index 32952a9ca9..2e9cbae183 100644 --- a/libsql-server/tests/fence/mod.rs +++ b/libsql-server/tests/fence/mod.rs @@ -12,7 +12,9 @@ use std::time::Duration; use hyper::StatusCode; use libsql_server::auth::user_auth_strategies::http_basic::HttpBasic; use libsql_server::auth::Auth; -use libsql_server::config::{AdminApiConfig, MetaStoreConfig, RpcServerConfig, UserApiConfig}; +use libsql_server::config::{ + AdminApiConfig, MetaStoreConfig, RpcClientConfig, RpcServerConfig, UserApiConfig, +}; use s3s::header::AUTHORIZATION; use serde_json::{json, Value}; use turmoil::{Builder, Sim}; @@ -93,6 +95,37 @@ pub fn make_primary(sim: &mut Sim, path: PathBuf, primary: Primary) { }); } +/// A replica of `primary` on host `replica0`: user API on 8080, admin API (no auth key) on +/// 9090. It creates a namespace lazily, on first use, by replicating it from the primary. +pub fn make_replica(sim: &mut Sim, path: PathBuf) { + init_tracing(); + sim.host("replica0", move || { + let path = path.clone(); + async move { + let server = TestServer { + path: path.into(), + user_api_config: UserApiConfig::default(), + admin_api_config: Some(AdminApiConfig { + acceptor: TurmoilAcceptor::bind(([0, 0, 0, 0], 9090)).await?, + connector: TurmoilConnector, + disable_metrics: true, + auth_key: None, + }), + rpc_client_config: Some(RpcClientConfig { + remote_url: "http://primary:4567".into(), + connector: TurmoilConnector, + tls_config: None, + }), + disable_namespaces: false, + disable_default_namespace: true, + ..Default::default() + }; + server.start_sim(8080).await?; + Ok(()) + } + }); +} + /// The admin API of `primary`, authenticating with `key` when there is one. pub struct Admin { client: Client, diff --git a/libsql-server/tests/fence/protocol.rs b/libsql-server/tests/fence/protocol.rs index d7255fbd29..7cf970fda3 100644 --- a/libsql-server/tests/fence/protocol.rs +++ b/libsql-server/tests/fence/protocol.rs @@ -12,8 +12,8 @@ use turmoil::net::TcpStream; use uuid::Uuid; use super::{ - acquire_body, command_body, load_and_log_id, make_primary, sim, state_of, Admin, Primary, - ADMIN_KEY, + acquire_body, command_body, load_and_log_id, make_primary, make_replica, sim, state_of, Admin, + Primary, ADMIN_KEY, }; use crate::common::net::TurmoilConnector; @@ -26,10 +26,12 @@ fn uuid(n: u128) -> Uuid { Uuid::from_u128(n) } -/// The user API of `primary`, for any namespace, with an optional basic-auth credential. +/// The user API of `primary` (or of another host), for any namespace, with an optional +/// basic-auth credential. struct User { client: hyper::Client, auth: Option, + host: &'static str, } impl User { @@ -41,6 +43,15 @@ impl User { Self { client: hyper::Client::builder().build(TurmoilConnector), auth: credential.map(|c| format!("basic {c}")), + host: "primary", + } + } + + /// The user API of `host` instead. + fn on(host: &'static str) -> Self { + Self { + host, + ..Self::new() } } @@ -53,7 +64,7 @@ impl User { ) -> anyhow::Result<(StatusCode, String)> { let mut request = Request::builder() .method(method) - .uri(format!("http://{ns}.primary:8080{path}")); + .uri(format!("http://{ns}.{}:8080{path}", self.host)); if let Some(auth) = &self.auth { request = request.header("authorization", auth.as_str()); } @@ -851,3 +862,115 @@ fn auth_and_not_found_distinct() { }); sim.run().unwrap(); } + +/// The replica's count of writes it delegated to the primary for `ns`. +async fn delegated_writes(ns: &str) -> anyhow::Result { + let resp = crate::common::http::Client::new() + .get(&format!("http://replica0:9090/v1/namespaces/{ns}/stats")) + .await?; + let body: Value = resp.json().await?; + body["write_requests_delegated"] + .as_u64() + .ok_or_else(|| anyhow::anyhow!("no write_requests_delegated: {body}")) +} + +/// A write sent to a replica is proxied to the primary; when the primary's fence refuses it, +/// the replica answers with the primary's denial: `423` and the stable code on HTTP, the +/// stable code as the Hrana error code (section 6.1). Reads stay local and are served, and +/// writes through the replica work again once the fence is released. +#[test] +fn replica_proxy_preserves_code() { + let mut sim = sim(); + let primary = tempdir().unwrap(); + let replica = tempdir().unwrap(); + make_primary(&mut sim, primary.path().to_path_buf(), Primary::default()); + make_replica(&mut sim, replica.path().to_path_buf()); + sim.client("client", async { + let admin = Admin::new(Some(ADMIN_KEY)); + let user = User::on("replica0"); + let op = uuid(0x100); + let rev = write_fenced(&admin, "src", op).await?; + let write = "insert into t values (2)"; + + assert_locked( + "legacy write", + &user.legacy("src", &[write]).await?, + WRITE_FENCED, + ); + assert_locked( + "v1 execute", + &user.execute("src", write).await?, + WRITE_FENCED, + ); + // A refused step of a `/v1` batch is a step error, as on the primary. + let (status, body) = user.batch("src", &["select 1", write]).await?; + assert_eq!(status, StatusCode::OK, "v1 batch: {body}"); + assert!( + body["result"]["step_errors"][0].is_null(), + "v1 batch: {body}" + ); + assert_hrana_error("v1 batch", &body["result"]["step_errors"][1], WRITE_FENCED); + for version in [2, 3] { + let what = format!("v{version}"); + let (status, body) = user + .pipeline( + "src", + version, + None, + json!([execute_req(write), execute_req("select * from t")]), + ) + .await?; + assert_eq!(status, StatusCode::OK, "{what}: {body}"); + let results = &body["results"]; + assert_eq!(results[0]["type"], "error", "{what}: {body}"); + assert_hrana_error(&what, &results[0]["error"], WRITE_FENCED); + assert_eq!(results[1]["type"], "ok", "{what}: {body}"); + } + let (status, body) = user.legacy("src", &["select count(*) from t"]).await?; + assert_eq!(status, StatusCode::OK, "{body}"); + + release(&admin, "src", op, rev).await?; + let (status, body) = user.execute("src", write).await?; + assert_eq!(status, StatusCode::OK, "after release: {body}"); + Ok(()) + }); + sim.run().unwrap(); +} + +/// A fence denial is final for the request: the replica delegates each refused write to the +/// primary exactly once and answers at once, with no reconnect or retry (the write proxy's +/// only retry, of `UNAVAILABLE`, backs off 500 ms first). +#[test] +fn denial_not_retried() { + let mut sim = sim(); + let primary = tempdir().unwrap(); + let replica = tempdir().unwrap(); + make_primary(&mut sim, primary.path().to_path_buf(), Primary::default()); + make_replica(&mut sim, replica.path().to_path_buf()); + sim.client("client", async { + let admin = Admin::new(Some(ADMIN_KEY)); + let user = User::on("replica0"); + write_fenced(&admin, "src", uuid(0x100)).await?; + // Load the namespace on the replica. + let (status, body) = user.legacy("src", &["select 1"]).await?; + assert_eq!(status, StatusCode::OK, "{body}"); + + let before = delegated_writes("src").await?; + for i in 0..3u64 { + let started = tokio::time::Instant::now(); + assert_locked( + "write", + &user.execute("src", "insert into t values (2)").await?, + WRITE_FENCED, + ); + assert!( + started.elapsed() < std::time::Duration::from_millis(500), + "{:?}", + started.elapsed() + ); + assert_eq!(delegated_writes("src").await?, before + i + 1); + } + Ok(()) + }); + sim.run().unwrap(); +} From efaf7d9de3fc60a1444c41cc25a530b207c1e522 Mon Sep 17 00:00:00 2001 From: River Date: Wed, 30 Sep 2026 10:24:22 +0000 Subject: [PATCH 23/33] libsql-server: honour the replicated fence on replica servers A replica server whose primary refuses replication of a namespace with a fence code (a refused `hello`, `log_entries` or `snapshot`, or a stream the primary ended with the typed status) now refuses local reads and streams of its copy with the same code, and asks the reads it had already admitted to stop. The denial is published on the namespace's fence controller on the replica, so every user protocol reports it as it does on the primary; writes keep going to the primary, which refuses them itself. A `hello` the primary answers lifts it. The refusal is no longer retried every second by the handshake loop: the replica's replication loop waits 1 s, doubling to 15 s, until the primary answers again, and counts each refused call in `libsql_server_replica_fence_refusals_total{code}`. Co-authored-by: Tomasz Szymczyszyn --- docs/NAMESPACE_FENCE.md | 18 +- .../src/namespace/configurator/replica.rs | 15 + .../src/namespace/fence/controller.rs | 44 +++ libsql-server/src/namespace/fence/mod.rs | 5 +- libsql-server/src/namespace/fence/replica.rs | 267 +++++++++++++ .../src/replication/replicator_client.rs | 138 +++++-- libsql-server/tests/fence/protocol.rs | 361 ++++++++++++++++++ 7 files changed, 817 insertions(+), 31 deletions(-) create mode 100644 libsql-server/src/namespace/fence/replica.rs diff --git a/docs/NAMESPACE_FENCE.md b/docs/NAMESPACE_FENCE.md index 70e465025a..b84f4fa332 100644 --- a/docs/NAMESPACE_FENCE.md +++ b/docs/NAMESPACE_FENCE.md @@ -438,6 +438,13 @@ Implementation: - **A fence change does not move the configuration version**, so it does not by itself invalidate a replica's session token or make it call `hello` again; the read fence reaches a replica by ending its streams. - **Older replica servers** skip the field. The read fence still stops their replication (refused `hello`, ended streams), but not their local reads of data they already hold (section 18). +Implementation on a replica server (`namespace/fence/replica.rs`, `replication/replicator_client.rs`, `namespace/configurator/replica.rs`): + +- **Local read denial.** The replicator publishes what it learns on the namespace's fence controller on the replica (`FenceController::observe_primary`), as `GateSnapshot::primary_denial`. While set, normal reads and streams of the local copy are refused with the primary's code (`MIGRATION_READ_FENCED`, `MIGRATION_TARGET_QUARANTINED` or `FENCE_STATE_UNAVAILABLE`), so every user protocol reports it exactly as section 6.0 describes; writes keep going to the primary, which refuses them itself (section 6.1). Publishing the denial asks every read lease held on the replica to stop, under the lease lock, so a read admitted concurrently is either refused or cancelled. It never moves the write generation and is never persisted. +- **What sets and clears it.** A `hello`, `log_entries` or `snapshot` call refused with the typed status, or a stream the primary ended with it, sets it (`PrimaryFenceRefusal`; any other status keeps its existing handling). A `hello` the primary answers clears it, unless its replicated fence names a state whose normal-read column denies (the primary does not send one; a state this server does not know is not treated as a denial, because the answer itself shows the primary admits replication). A replica that cannot reach its primary at all keeps what it last learned. +- **Paced reconnects.** A fence refusal is returned to the replica's replication loop instead of being retried by the handshake loop every second. The loop waits 1 s after the first refusal, doubling to at most 15 s between attempts, until the primary answers `hello` again, so a replica of a read-fenced namespace makes a handful of calls a minute, and resumes serving at most about 15 s after the read fence is cleared. Each refused call is counted (`libsql_server_replica_fence_refusals_total{code}`); the replica logs once when the denial is installed and once when it is lifted. +- **Asynchronous by nature.** The replica learns of the fence when its stream ends, which the primary's read drain waits for; it installs the denial as the terminal status arrives, not before the primary's drain completes. The drain proves that the primary serves nothing more; it does not prove that a replica has stopped serving its copy (section 18). + ## 7. FenceController (Design) ### 7.1 Registry @@ -561,11 +568,11 @@ This makes the WAL gate independent of statement classification: DDL, misclassif `ClearSourceReadFence` reopens reads (writes stay fenced) with a new revision. -Covered surfaces: HTTP (`/`, `/v1/execute`, `/v1/batch`), Hrana over HTTP (`/v2`, `/v3`, cursors), Hrana over WebSocket, the gRPC proxy (`execute`, `stream_exec`, `describe`), the admin shell (which runs raw SQL and is checked per query), `/beta/listen`, ATTACH from other namespaces, `/dump`, and both replication services (the internal one used by replica servers and the external one on the user port). A replica server that receives the terminal status installs a local read denial for that namespace, so it stops serving its local copy. +Covered surfaces: HTTP (`/`, `/v1/execute`, `/v1/batch`), Hrana over HTTP (`/v2`, `/v3`, cursors), Hrana over WebSocket, the gRPC proxy (`execute`, `stream_exec`, `describe`), the admin shell (which runs raw SQL and is checked per query), `/beta/listen`, ATTACH from other namespaces, `/dump`, and both replication services (the internal one used by replica servers and the external one on the user port). A replica server that receives the terminal status installs a local read denial for that namespace, so it stops serving its local copy, and paces its reconnects (section 6.2). Transport keepalive: when the fence is enabled (`--enable-namespace-fence`), the RPC server and the user-port HTTP server (which carries the user-facing gRPC services), including connections upgraded to `h2c`, send HTTP/2 keepalive pings (`--namespace-fence-keepalive-interval-s`, default 30 s; a ping unanswered for 20 s closes the connection) so dead peers are detected; lease release does not depend on it. -Replication calls denied by the fence are counted (`libsql_server_fence_denials_total{code, surface = "replication"}`) and logged at most once per namespace per minute, so repeated reconnects of replicas that do not understand the typed code are observable rather than noisy. Replica-side handling of the typed code (a local read denial, capped back-off) is part of the protocol work of section 6. +Replication calls denied by the fence are counted (`libsql_server_fence_denials_total{code, surface = "replication"}`) and logged at most once per namespace per minute, so repeated reconnects of replicas that do not understand the typed code are observable rather than noisy. Replica-side handling of the typed code (a local read denial, capped back-off) is described in section 6.2. Internal work that must keep running is classed `Maintenance` or `Observability` and holds no read lease: bottomless WAL upload, the storage monitor's read transaction, stats and metrics. @@ -780,6 +787,7 @@ How each path that can reach namespace data or lifecycle is covered. File refere | `http/admin/fence.rs` | Fence routes (section 4.5): capability discovery, `InspectFence`, one route per command through `NamespaceStore::execute_fence_command`, and `validation-query` through `open_validation_session`; admin auth key and primary required for every route that changes state or uses a capability. | | `http/user/dump.rs`, `connection/dump/exporter.rs` | Gate check (`Stream`) before connection creation (typed, no panic on create error); `Dump` lease held by the export; cancel checked before every row and by the pipe's pending write; the body stream ends with the fence error (aborted response) on cancellation. | | `rpc/replication/replication_log.rs` `hello`, `log_entries`, `batch_log_entries`, `snapshot` | Denied at request start (`FAILED_PRECONDITION` + `x-libsql-fence-code`, counted, rate-limited log); `FencedStream` replication leases for both streams and the batch; typed terminal status when the gate closes or at the deadline, lease released without the peer; `ReplicatedFence` in `hello`'s config while a fence is active (section 6.2; the gate is read without a lease, since `hello` streams no data). | +| `replication/replicator_client.rs`, `namespace/configurator/replica.rs` (replica servers) | A fence status from `hello`, `log_entries` or `snapshot`, or a stream ended with one, installs the local read denial (`FenceController::observe_primary`) and is returned as `PrimaryFenceRefusal`; the replication loop backs off 1 s → 15 s instead of reconnecting every second; an answered `hello` lifts the denial (section 6.2). | | `admin_shell.rs` | Writes denied at the WAL (no capability); reads checked against the gate, and a read lease held, per query (cancelled through the connection's interrupt handle). | | `schema/scheduler.rs`, `database/schema.rs` | Shared schema excluded from fencing; registering a migration job is refused while the schema or a linked namespace denies lifecycle work (section 13.4); migration writes are WAL-gated; the scheduler's `block_writes` flag is not treated as drain evidence. | | `namespace/configurator/helpers.rs` `load_dump`, `http/admin/mod.rs` `dump_stream_from_url` | Restore options and dump URLs are refused for fenced namespaces before the dump is fetched (section 13.5); outside a capability the loader's writes are refused at the WAL. Import goes through `ImportSession::load_dump`, which runs the same loader (`load_dump_sql`) on the capability connection. | @@ -837,7 +845,7 @@ Planned test names; the table is updated as tests land. | 15 | Lost `EnableTargetWrites` response resolved from receipt/state | landed: `namespace::fence::target::tests::enable_writes_response_loss_resolved` | | 16 | Read fence drains SQL, dump, `log_entries`, `snapshot`, including dead peers and forced termination | SQL landed: `namespace::fence::read::tests::{read_fence_waits_for_running_program, program_after_closing_gate_is_refused (parked after the read-closing gate, before the CAS), read_fence_cancels_at_deadline, unreleased_lease_answers_draining_and_replay_completes, idle_txn_fails_on_next_program, clear_read_fence_reopens_reads_not_writes, refused_read_fence_reopens_reads, attach_of_read_fenced_namespace_denied}`, `admin_shell::fence_tests::admin_shell_read_denied`; dump and replication landed: `namespace::fence::stream::tests::{dump_lease_released_on_cancel (a dump blocked mid-row on a peer that stopped reading is cancelled at the deadline, its lease released without the peer, the fence acknowledged, and the body ends with the fence error and no `COMMIT;`), read_fence_waits_for_dump, dump_refused_while_read_fenced, log_entries_stream_ends_typed, stream_lease_released_without_peer_read (a dead peer), snapshot_stream_ends_typed, replication_calls_denied_while_read_fenced, read_fence_forced_termination}`; the response `/dump` returns: `namespace::fence::stream::tests::dump_response_aborted_on_cancel` (a dump cancelled by the read drain fails its response body, with the fence error and no `COMMIT;`); over HTTP: `tests::fence::protocol::dump_codes` (complete under a write fence; `423` + code under a read fence and on a quarantined target) | | 17 | Delete, reset, fork, restore, config, schema mutation rejected | landed: `tests::fence::lifecycle::lifecycle_rejected_while_fenced` (over the admin API, for a write-fenced source and a quarantined target: delete, fork as source and as destination, create with a `dump_url` whose file does not exist, create over the record, linking to a shared schema at creation and config `POST` are all `423` with the fence code; state, revision and data are unchanged and no copy exists; a target cannot be created with a shared schema; after release the source takes writes and config, fork and delete work again); `namespace::store::fence_tests::{reset_refused_while_fenced (called directly and as the replicator's reset callback; after release reset works and wipes the data), lifecycle_refused_while_fenced (fork either side, create over, delete, config and shared-schema link in the metastore transaction)}`; `schema::scheduler::test::fence::{acquire_rejects_shared_schema, migration_not_registered_while_linked_namespace_fenced}` | -| 18 | Codes through HTTP, Hrana, RPC, dump, replication, replica write proxy; distinguishable from auth/timeout/not-found; old peers compatible; no retry loops | user protocols landed: `tests::fence::protocol::{http_codes (legacy `/`, `/v1/execute`, `/v1/batch` for a write-fenced write, a read-fenced read, a quarantined target and a namespace whose marker cannot be decoded: `423` with `code`, and `detail` where there is one), hrana_http_codes (`/v2`, `/v3` pipelines and `/v3/cursor`: step and whole-request errors carry the code and the baton stays usable), hrana_ws_codes (the same over a WebSocket, whose stream reads again after the read fence is cleared), dump_codes, auth_and_not_found_distinct (`401` without or with a wrong credential and `404` for a missing namespace, with no fence code, on the same fenced server)}`, `error::fence_tests::fence_errors_carry_code` (the body through every wrapper; `block_*`'s `Blocked` keeps its mapping); RPC and replica write proxy landed: `rpc::proxy::fence_tests::rpc_codes` (on the primary's proxy service: a write-fenced write is a step error with `SQL_ERROR` + `stable_code`, reads are served, a read-fenced read is a program error with the code when streamed and the typed `FAILED_PRECONDITION` status when unary, and a namespace whose fence state is unknown is refused with the typed status before any connection), `rpc::proxy::fence_tests::replica_maps_proxied_denials` (step, program and connection-status denials become the fence error on the replica; an older primary's error without `stable_code`, an unknown code and other errors keep their mapping), `namespace::fence::outcome::tests::peer_denials_round_trip`, `tests::fence::protocol::{replica_proxy_preserves_code (writes through a replica to a write-fenced primary: `423` + `MIGRATION_WRITE_FENCED` on legacy `/` and `/v1/execute`, the step error on `/v1/batch`, the Hrana error code on `/v2` and `/v3`; reads on the replica served; writes through the replica work after release), denial_not_retried (each refused write is delegated exactly once and answered well within the write proxy's first retry backoff)}`; capability: `tests::fence::admin::capabilities` asserts `proxy_stable_code: true`; planned: `tests::fence::protocol::replication_codes`; landed: `libsql-replication` `rpc::test::{proxy_error_stable_code_is_additive, replicated_fence_is_additive}` (each new field is skipped by a peer that does not know it, absent from an older peer's message, and absent fields encode exactly as before), `namespace::fence::stream::tests::hello_carries_replicated_fence` (no fence before acquisition and after release; state and revision while write-fenced; the stored configuration never carries it) | +| 18 | Codes through HTTP, Hrana, RPC, dump, replication, replica write proxy; distinguishable from auth/timeout/not-found; old peers compatible; no retry loops | user protocols landed: `tests::fence::protocol::{http_codes (legacy `/`, `/v1/execute`, `/v1/batch` for a write-fenced write, a read-fenced read, a quarantined target and a namespace whose marker cannot be decoded: `423` with `code`, and `detail` where there is one), hrana_http_codes (`/v2`, `/v3` pipelines and `/v3/cursor`: step and whole-request errors carry the code and the baton stays usable), hrana_ws_codes (the same over a WebSocket, whose stream reads again after the read fence is cleared), dump_codes, auth_and_not_found_distinct (`401` without or with a wrong credential and `404` for a missing namespace, with no fence code, on the same fenced server)}`, `error::fence_tests::fence_errors_carry_code` (the body through every wrapper; `block_*`'s `Blocked` keeps its mapping); RPC and replica write proxy landed: `rpc::proxy::fence_tests::rpc_codes` (on the primary's proxy service: a write-fenced write is a step error with `SQL_ERROR` + `stable_code`, reads are served, a read-fenced read is a program error with the code when streamed and the typed `FAILED_PRECONDITION` status when unary, and a namespace whose fence state is unknown is refused with the typed status before any connection), `rpc::proxy::fence_tests::replica_maps_proxied_denials` (step, program and connection-status denials become the fence error on the replica; an older primary's error without `stable_code`, an unknown code and other errors keep their mapping), `namespace::fence::outcome::tests::peer_denials_round_trip`, `tests::fence::protocol::{replica_proxy_preserves_code (writes through a replica to a write-fenced primary: `423` + `MIGRATION_WRITE_FENCED` on legacy `/` and `/v1/execute`, the step error on `/v1/batch`, the Hrana error code on `/v2` and `/v3`; reads on the replica served; writes through the replica work after release), denial_not_retried (each refused write is delegated exactly once and answered well within the write proxy's first retry backoff)}`; capability: `tests::fence::admin::capabilities` asserts `proxy_stable_code: true`; replication and replica servers landed: `tests::fence::protocol::{replication_codes (a raw peer of the primary's internal replication service: `hello` carries the write fence's state and revision, an open `log_entries` stream ends with `FAILED_PRECONDITION` + `x-libsql-fence-code` `MIGRATION_READ_FENCED` under the read fence, `hello`, `log_entries` and `snapshot` are then refused with it, `hello` is answered again after the clear and carries no fence after release, and a quarantined target refuses `hello` with `MIGRATION_TARGET_QUARANTINED`), replica_reads_denied_while_source_read_fenced (the replica refuses local reads with `423` + `MIGRATION_READ_FENCED` on legacy `/` and `/v1/execute` and the Hrana code on `/v2` within 500 ms of simulated time after the read fence is acknowledged, and still 30 s later), replica_backs_off_on_fence_code (4 to 9 refused, counted attempts over 60 s of simulated time; a fixed 1 s retry makes 57), replica_resumes_after_clear_read_fence (reads served again within 16 s of the clear, and a write after release is replicated)}`, `namespace::fence::replica::tests::{refusal_from_typed_status_only, hello_fence_denies_only_read_denying_states, backoff_doubles_to_its_cap, observed_denial_refuses_local_reads_and_cancels_leases}`; landed: `libsql-replication` `rpc::test::{proxy_error_stable_code_is_additive, replicated_fence_is_additive}` (each new field is skipped by a peer that does not know it, absent from an older peer's message, and absent fields encode exactly as before), `namespace::fence::stream::tests::hello_carries_replicated_fence` (no fence before acquisition and after release; state and revision while write-fenced; the stored configuration never carries it) | | 19 | Corrupt or unknown durable fence state fails closed | `fence::store::tests::corrupt_payload_fails_closed`, `unknown_format_version_fails_closed` | | 20 | Metrics and audit logs | `tests::fence::observability::metrics_and_labels`; `fence::audit::tests::audit_event_fields` | | 21 | Capability discovery and mixed-version protection | capability discovery landed: `tests::fence::admin::{capabilities, capabilities_when_disabled}`; planned: `fence::store::tests::legacy_mirror_and_fk_guard` (bounded, see section 18) | @@ -855,6 +863,7 @@ What this design and its tests do not prove: - **Two-person adoption** is a separate secret plus a recorded two-approver request, not verified identities. - **Delivered data** cannot be recalled: bytes already sent, frames held by embedded replicas, and a replica server partitioned from the primary are outside the server's reach. - **Older replica servers** stop replicating under a read fence but keep serving local reads of what they already hold; only a replica server that understands the replicated fence (section 6.2) denies them. +- **A replica server's local read denial is not part of the positive read drain.** The primary proves that its own reads and streams have ended; a newer replica installs its denial when its ended stream reaches it, a moment after the drain completes, and one partitioned from the primary keeps serving what it holds. The operation proves replica convergence separately. - **Metastore rollback detection** relies on the namespace directory's marker. If both the metastore and the namespace directory are lost or restored from backup together, the server cannot detect that a newer fence existed; the caller's durable intent is authoritative then. - **Release and pinning** of a server build that contains the fence are outside this change. @@ -876,6 +885,7 @@ libsql-server/src/namespace/fence/ drain.rs FenceController::execute, source write drain read.rs source read fence and its drain stream.rs stream leases: FencedStream for replication, dump cancel + replica.rs a replica server's view of the primary's fence, reconnect back-off target.rs quarantined target creation, ValidationSession capability.rs MigrationCapability, CapabilityPurpose, ImportWriter import.rs ImportSession, SealTargetImport drain @@ -918,7 +928,7 @@ Planned commits, each leaving the crate building with its tests passing: | Open-default `handle()`, including ATTACH authorisation and replica lazy creation | Non-creating lookups on read paths; creating paths refuse names with a record or marker. | | `process()` publishing unpersisted config | Fixed for all config writes. | | The scheduler's second metastore connection | Fence CAS and config writes use `BEGIN IMMEDIATE`; shared schema is excluded. | -| Replica servers inherit config | Legacy `block_*` mirror plus additive `ReplicatedFence`; typed terminal status installs a local read denial. | +| Replica servers inherit config | Additive `ReplicatedFence` in `hello`; the typed terminal status installs a local read denial on a newer replica server, with a capped reconnect back-off (section 6.2). The legacy `block_*` mirror exists only in the stored config row (section 13.2), so it does not reach replica servers. | | Proxy errors lose their type | Additive `stable_code = 4`. | | Write proxy retries `UNAVAILABLE` forever | Fence denials are never `UNAVAILABLE`. | | Admin API has no principal and may have no key | Fence mutators require a configured key; adoption requires a second key and two recorded approvers. | diff --git a/libsql-server/src/namespace/configurator/replica.rs b/libsql-server/src/namespace/configurator/replica.rs index adea0fd406..986d5880f2 100644 --- a/libsql-server/src/namespace/configurator/replica.rs +++ b/libsql-server/src/namespace/configurator/replica.rs @@ -19,6 +19,7 @@ use crate::database::{Database, ReplicaDatabase}; use crate::namespace::broadcasters::BroadcasterHandle; use crate::namespace::configurator::helpers::{make_stats, run_storage_monitor}; use crate::namespace::fence::controller::FenceController; +use crate::namespace::fence::replica::{refusal_backoff, PrimaryFenceRefusal}; use crate::namespace::meta_store::MetaStoreHandle; use crate::namespace::{Namespace, NamespaceBottomlessDbIdInit, RestoreOption}; use crate::namespace::{NamespaceName, NamespaceStore, ResetCb, ResetOp, ResolveNamespacePathFn}; @@ -77,6 +78,7 @@ impl ConfigureNamespace for ReplicaConfigurator { meta_store_handle.clone(), store.clone(), WalImpl::new_sqlite(&db_path, new_frame_sender).await?, + fence.clone(), ) .await?; let mut replicator = libsql_replication::replicator::Replicator::new_sqlite( @@ -126,6 +128,19 @@ impl ConfigureNamespace for ReplicaConfigurator { loop { match replicator.run().await { err @ Error::Fatal(_) => Err(err)?, + e @ Error::Internal(_) if PrimaryFenceRefusal::of(&e).is_some() => { + // The primary's fence refuses replication of this namespace + // (`docs/NAMESPACE_FENCE.md` section 6.2). The client has published + // the local read denial; retry at a capped, growing interval rather + // than at once, until the primary answers `hello` again. + let refusals = replicator.client_mut().fence_refusals(); + let delay = refusal_backoff(refusals); + tracing::debug!( + "{e}; retrying replication of {namespace} in {delay:?} \ + ({refusals} refusals in a row)" + ); + tokio::time::sleep(delay).await; + } _err @ Error::NamespaceDoesntExist => { // TODO(lucio): there is a bug where a primary will report that a valid // namespace doesn't exist when it does and causes the replicate to diff --git a/libsql-server/src/namespace/fence/controller.rs b/libsql-server/src/namespace/fence/controller.rs index e61421f341..bc358818a4 100644 --- a/libsql-server/src/namespace/fence/controller.rs +++ b/libsql-server/src/namespace/fence/controller.rs @@ -66,6 +66,11 @@ pub struct GateSnapshot { /// observability is refused with `MIGRATION_TARGET_QUARANTINED`, and the namespace is not /// set up. Never persisted; replaced by the record the command's commit publishes. pub creating_target: Option, + /// On a replica server only: the primary's fence denies reads of this namespace (section + /// 6.2), as the replicator last learned it from a refused replication call or from the + /// fence `hello` replicated. Normal reads and streams of the local copy are refused with + /// it. Never persisted and never set on a primary. + pub primary_denial: Option, } impl GateSnapshot { @@ -77,6 +82,7 @@ impl GateSnapshot { installing: None, closing_reads: None, creating_target: None, + primary_denial: None, } } @@ -156,6 +162,11 @@ impl GateSnapshot { )); } } + if let Some(denial) = &self.primary_denial { + if matches!(class, OperationClass::NormalRead | OperationClass::Stream) { + return Err(denial.clone()); + } + } Ok(()) } @@ -502,6 +513,39 @@ impl FenceController { asked } + /// On a replica server: publish what the replicator learned of the primary's fence + /// (section 6.2). `Some` denies normal reads and streams of the local copy with that error + /// and asks every read lease held now to stop, so that work admitted before the replica + /// learned of the fence does not outlive it; `None` admits them again. The denial is + /// published under the lease lock, so a read admitted concurrently is either refused or + /// counted and cancelled. A denial with the code already published leaves the gate as it + /// is. Returns whether the gate changed. + pub fn observe_primary(&self, denial: Option) -> bool { + let leases = self.read_leases.lock(); + let deny = denial.is_some(); + let changed = self.gate.send_if_modified(|gate| { + // A denial with the same code is the same denial, whichever call reported it. + let same = match (&gate.primary_denial, &denial) { + (Some(old), Some(new)) => old.outcome() == new.outcome(), + (None, None) => true, + _ => false, + }; + if same { + return false; + } + gate.primary_denial = denial; + true + }); + if changed && deny { + for entry in leases.live.values() { + if !entry.cancelled.swap(true, Ordering::AcqRel) { + (entry.cancel)(); + } + } + } + changed + } + /// Notified on every read-lease release. Enable the notification before checking /// [`read_lease_counts`](Self::read_lease_counts), so a release in between is not missed. pub(crate) fn read_released(&self) -> &Notify { diff --git a/libsql-server/src/namespace/fence/mod.rs b/libsql-server/src/namespace/fence/mod.rs index 43c384659a..8c89d79f5e 100644 --- a/libsql-server/src/namespace/fence/mod.rs +++ b/libsql-server/src/namespace/fence/mod.rs @@ -11,8 +11,8 @@ //! on them: the per-namespace [`controller`] with its gate and read leases, the positive write //! [`drain`], the source [`read`] fence and its //! [`stream`] leases for dump and replication, quarantined migration [`target`]s with their -//! [`capability`]-scoped [`import`] sessions and seal drain, the [`registry`] that holds the controllers outside the namespace cache, and the test [`hooks`] -//! on their paths. +//! [`capability`]-scoped [`import`] sessions and seal drain, the [`registry`] that holds the controllers outside the namespace cache, the +//! [`replica`]-server view of a primary's fence, and the test [`hooks`] on their paths. // The persistence, controller and protocol layers that consume these types land in the // following commits of this series; until then most of the module is unused by the rest of @@ -29,6 +29,7 @@ pub mod outcome; pub mod read; pub mod record; pub mod registry; +pub mod replica; pub mod state; pub mod store; pub mod stream; diff --git a/libsql-server/src/namespace/fence/replica.rs b/libsql-server/src/namespace/fence/replica.rs new file mode 100644 index 0000000000..bf198f5bce --- /dev/null +++ b/libsql-server/src/namespace/fence/replica.rs @@ -0,0 +1,267 @@ +//! The primary's fence as a replica server sees it (`docs/NAMESPACE_FENCE.md` section 6.2). +//! +//! A replica server holds a copy of a primary's namespace and serves reads of it locally. When +//! the primary's fence denies reads (a source read fence, a quarantined or aborted target, a +//! fence state the primary cannot establish), the primary refuses the replica's replication +//! calls and ends its streams with a typed `FAILED_PRECONDITION` status, and a `hello` it does +//! answer carries the fence it has. This module turns both into the local read denial the +//! replica publishes on its own fence controller +//! ([`FenceController::observe_primary`](super::controller::FenceController::observe_primary)), +//! and paces the replicator's reconnects while the primary keeps refusing. + +use std::time::Duration; + +use libsql_replication::replicator::Error as ReplicatorError; +use libsql_replication::rpc::metadata::ReplicatedFence; + +use super::outcome::{FenceError, OutcomeKind}; +use super::state::{FenceState, OperationClass}; + +/// The first pause after the primary refuses replication with a fence code. +pub const REFUSAL_BACKOFF_INITIAL: Duration = Duration::from_secs(1); +/// The longest pause between two replication attempts the primary's fence refused. It bounds +/// how long a replica keeps denying reads after the primary admits them again. +pub const REFUSAL_BACKOFF_MAX: Duration = Duration::from_secs(15); + +/// A replication call the primary refused, or a stream it ended, because its fence denies +/// replication of the namespace. Carried in [`ReplicatorError::Internal`], which the +/// replicator's handshake loop does not retry by itself, so that the replica's own loop can +/// pace the next attempt. +#[derive(Debug, Clone, PartialEq, Eq, thiserror::Error)] +#[error("the primary refused replication: {0}")] +pub struct PrimaryFenceRefusal(pub FenceError); + +impl PrimaryFenceRefusal { + /// The refusal a replication status reports: a data-plane fence denial in the typed form of + /// section 6 (`FAILED_PRECONDITION` with the stable code in `x-libsql-fence-code`). `None` + /// for every other status, which keeps its existing handling. + pub fn from_status(status: &tonic::Status) -> Option { + let denial = FenceError::from_grpc_status(status)?; + (denial.outcome().kind() == OutcomeKind::DataPlane).then_some(Self(denial)) + } + + /// The refusal `error` carries, if it is one. + pub fn of(error: &ReplicatorError) -> Option<&Self> { + match error { + ReplicatorError::Internal(e) => e.downcast_ref::(), + _ => None, + } + } + + /// The local read denial it implies. Replication is `Stream` work, whose column of the + /// permission matrix equals the normal-read column, so every data-plane refusal of it + /// means the primary denies reads. + pub fn local_denial(&self) -> FenceError { + FenceError::new( + self.0.outcome(), + format!( + "the primary denies reads of this namespace: {}", + self.0.message() + ), + ) + } +} + +/// `status` as a replicator error: a [`PrimaryFenceRefusal`] for a fence denial, the +/// replicator's own mapping otherwise. +pub fn replicator_error(status: tonic::Status) -> ReplicatorError { + match PrimaryFenceRefusal::from_status(&status) { + Some(refusal) => ReplicatorError::Internal(Box::new(refusal)), + None => status.into(), + } +} + +/// The local read denial the fence a primary's `hello` replicated implies: `Some` only for a +/// known state whose normal-read column denies. The primary answers `hello` only where it +/// admits streams, so this is `None` in practice; a state this server does not know is not +/// a denial for the same reason. +pub fn denial_from_hello(fence: Option<&ReplicatedFence>) -> Option { + let fence = fence?; + let state = fence.state.parse::().ok()?; + let outcome = state.permits(OperationClass::NormalRead).err()?; + Some(FenceError::new( + outcome, + format!( + "the primary's fence is {} at revision {}", + fence.state, fence.revision + ), + )) +} + +/// The pause before the next replication attempt after `consecutive` refusals in a row +/// (counting the one just received): doubling from [`REFUSAL_BACKOFF_INITIAL`] up to +/// [`REFUSAL_BACKOFF_MAX`]. +pub fn refusal_backoff(consecutive: u32) -> Duration { + let doublings = consecutive.saturating_sub(1).min(16); + REFUSAL_BACKOFF_INITIAL + .saturating_mul(1 << doublings) + .min(REFUSAL_BACKOFF_MAX) +} + +#[cfg(test)] +mod tests { + use tonic::Code; + + use super::super::outcome::FenceOutcome; + use super::*; + + fn status(outcome: FenceOutcome) -> tonic::Status { + FenceError::new(outcome, "no") + .to_grpc_status() + .expect("data-plane outcomes have a gRPC mapping") + } + + #[test] + fn refusal_from_typed_status_only() { + for outcome in [ + FenceOutcome::MigrationReadFenced, + FenceOutcome::MigrationTargetQuarantined, + FenceOutcome::FenceStateUnavailable, + ] { + let refusal = PrimaryFenceRefusal::from_status(&status(outcome)).unwrap(); + assert_eq!(refusal.0.outcome(), outcome); + let denial = refusal.local_denial(); + assert_eq!(denial.outcome(), outcome); + assert!( + denial.message().contains("the primary denies reads"), + "{denial}" + ); + + let error = replicator_error(status(outcome)); + assert_eq!(PrimaryFenceRefusal::of(&error), Some(&refusal)); + } + + // Untyped statuses keep the replicator's own mapping. + for status in [ + tonic::Status::new(Code::FailedPrecondition, "MIGRATION_READ_FENCED: no"), + tonic::Status::new(Code::Unavailable, "down"), + tonic::Status::new( + Code::FailedPrecondition, + libsql_replication::rpc::replication::NAMESPACE_DOESNT_EXIST, + ), + ] { + assert!(PrimaryFenceRefusal::from_status(&status).is_none()); + assert!(PrimaryFenceRefusal::of(&replicator_error(status)).is_none()); + } + assert!(matches!( + replicator_error(tonic::Status::new( + Code::FailedPrecondition, + libsql_replication::rpc::replication::NEED_SNAPSHOT_ERROR_MSG + )), + ReplicatorError::NeedSnapshot + )); + + // A control outcome is never a replication refusal. + let mut control = tonic::Status::new(Code::FailedPrecondition, "x"); + control.metadata_mut().insert( + super::super::outcome::GRPC_FENCE_CODE_METADATA, + tonic::metadata::MetadataValue::from_static("FENCE_PRECONDITION_FAILED"), + ); + assert!(PrimaryFenceRefusal::from_status(&control).is_none()); + } + + #[test] + fn hello_fence_denies_only_read_denying_states() { + let fence = |state: &str| ReplicatedFence { + state: state.into(), + revision: 7, + }; + assert_eq!(denial_from_hello(None), None); + for state in [ + "SOURCE_DRAINING", + "SOURCE_WRITE_FENCED", + "TARGET_WRITE_FENCED", + "SOMETHING_NEWER", + ] { + assert_eq!(denial_from_hello(Some(&fence(state))), None, "{state}"); + } + for (state, outcome) in [ + ("SOURCE_READ_DRAINING", FenceOutcome::MigrationReadFenced), + ("SOURCE_READ_FENCED", FenceOutcome::MigrationReadFenced), + ( + "TARGET_QUARANTINED", + FenceOutcome::MigrationTargetQuarantined, + ), + ("UNKNOWN_UNAVAILABLE", FenceOutcome::FenceStateUnavailable), + ] { + let denial = denial_from_hello(Some(&fence(state))).unwrap(); + assert_eq!(denial.outcome(), outcome, "{state}"); + assert!(denial.message().contains("revision 7"), "{denial}"); + } + } + + #[test] + fn backoff_doubles_to_its_cap() { + let delays: Vec<_> = (1..=7).map(refusal_backoff).collect(); + assert_eq!( + delays, + [1, 2, 4, 8, 15, 15, 15].map(Duration::from_secs).to_vec() + ); + assert_eq!(refusal_backoff(0), REFUSAL_BACKOFF_INITIAL); + assert_eq!(refusal_backoff(u32::MAX), REFUSAL_BACKOFF_MAX); + } + + #[test] + fn observed_denial_refuses_local_reads_and_cancels_leases() { + use std::sync::atomic::{AtomicUsize, Ordering}; + use std::sync::Arc; + + use super::super::controller::{FenceController, LeaseKind}; + use crate::namespace::NamespaceName; + + let fence = FenceController::unfenced(NamespaceName::from_string("ns".into()).unwrap()); + let cancels = Arc::new(AtomicUsize::new(0)); + let lease = fence + .acquire_read_lease(OperationClass::NormalRead, LeaseKind::Sql, { + let cancels = cancels.clone(); + move || { + cancels.fetch_add(1, Ordering::SeqCst); + } + }) + .unwrap(); + let generation = fence.write_generation(); + let refusal = + PrimaryFenceRefusal::from_status(&status(FenceOutcome::MigrationReadFenced)).unwrap(); + + assert!(fence.observe_primary(Some(refusal.local_denial()))); + assert_eq!(cancels.load(Ordering::SeqCst), 1); + assert!(lease.cancelled_by_fence()); + for class in [OperationClass::NormalRead, OperationClass::Stream] { + let err = fence.permits(class).unwrap_err(); + assert_eq!( + err.outcome(), + FenceOutcome::MigrationReadFenced, + "{class:?}" + ); + } + // Writes are the primary's to refuse; maintenance goes on. + for class in [OperationClass::NormalWrite, OperationClass::Maintenance] { + assert!(fence.permits(class).is_ok(), "{class:?}"); + } + assert!(fence + .acquire_read_lease(OperationClass::Stream, LeaseKind::Dump, || ()) + .is_err()); + // The same code again, from another call, is the same denial. + assert!(!fence.observe_primary(Some(refusal.local_denial()))); + assert_eq!(cancels.load(Ordering::SeqCst), 1); + // A different code replaces it. + let quarantined = + PrimaryFenceRefusal::from_status(&status(FenceOutcome::MigrationTargetQuarantined)) + .unwrap(); + assert!(fence.observe_primary(Some(quarantined.local_denial()))); + assert_eq!( + fence + .permits(OperationClass::NormalRead) + .unwrap_err() + .outcome(), + FenceOutcome::MigrationTargetQuarantined + ); + + assert!(fence.observe_primary(None)); + assert!(fence.permits(OperationClass::NormalRead).is_ok()); + assert!(!fence.observe_primary(None)); + // Only reads were affected: the write generation never moved. + assert_eq!(fence.write_generation(), generation); + drop(lease); + } +} diff --git a/libsql-server/src/replication/replicator_client.rs b/libsql-server/src/replication/replicator_client.rs index fb8154824d..9d9a6df78f 100644 --- a/libsql-server/src/replication/replicator_client.rs +++ b/libsql-server/src/replication/replicator_client.rs @@ -1,5 +1,6 @@ use std::path::Path; use std::pin::Pin; +use std::sync::Arc; use bytes::Bytes; use chrono::{DateTime, Utc}; @@ -23,6 +24,9 @@ use crate::connection::config::DatabaseConfig; use crate::metrics::{ REPLICATION_LATENCY, REPLICATION_LATENCY_CACHE_MISS, REPLICATION_LATENCY_OUT_OF_SYNC, }; +use crate::namespace::fence::controller::FenceController; +use crate::namespace::fence::outcome::FenceError; +use crate::namespace::fence::replica::{self, PrimaryFenceRefusal}; use crate::namespace::meta_store::MetaStoreHandle; use crate::namespace::{NamespaceName, NamespaceStore}; use crate::replication::FrameNo; @@ -107,6 +111,11 @@ pub struct Client { store: NamespaceStore, wal_impl: WalImpl, first_sync_since_handshake: bool, + /// The namespace's fence controller on this replica server, on which the primary's fence + /// is published as a local read denial (`docs/NAMESPACE_FENCE.md` section 6.2). + fence: Arc, + /// Replication calls the primary's fence refused since the last `hello` it answered. + fence_refusals: u32, } impl Client { @@ -116,6 +125,7 @@ impl Client { meta_store_handle: MetaStoreHandle, store: NamespaceStore, wal_flavor: WalImpl, + fence: Arc, ) -> crate::Result { Ok(Self { namespace, @@ -126,6 +136,78 @@ impl Client { store, wal_impl: wal_flavor, first_sync_since_handshake: true, + fence, + fence_refusals: 0, + }) + } + + /// Replication calls the primary's fence refused in a row, since the last `hello` it + /// answered. The replica's replication loop paces its reconnects by it. + pub(crate) fn fence_refusals(&self) -> u32 { + self.fence_refusals + } + + /// Publish what the primary said of its fence as this replica's local read denial, logging + /// when that changes. + fn observe_primary_fence(&self, denial: Option) { + let denies = denial.as_ref().map(|d| d.outcome()); + if self.fence.observe_primary(denial) { + match denies { + Some(code) => tracing::warn!( + namespace = %self.namespace, + "the primary's namespace fence denies reads ({code}): local reads of this \ + replica are refused until the primary admits replication again" + ), + None => tracing::info!( + namespace = %self.namespace, + "the primary admits replication again: local reads are served" + ), + } + } + } + + /// Map a status of a replication call: a fence refusal is published as the local read + /// denial, counted, and returned as a [`PrimaryFenceRefusal`]. + fn status_error(&mut self, status: Status) -> Error { + let error = replica::replicator_error(status); + if let Some(refusal) = PrimaryFenceRefusal::of(&error) { + self.fence_refusals = self.fence_refusals.saturating_add(1); + metrics::increment_counter!( + "libsql_server_replica_fence_refusals_total", + "code" => refusal.0.outcome().as_str(), + ); + tracing::debug!(namespace = %self.namespace, "{refusal}"); + self.observe_primary_fence(Some(refusal.local_denial())); + } + error + } + + /// A stream of the primary's that ends with a fence refusal publishes it as the local read + /// denial. + fn fenced_frames( + &self, + stream: tonic::Streaming, + ) -> impl Stream> + Send + 'static { + let fence = self.fence.clone(); + let namespace = self.namespace.clone(); + stream.map_err(move |status| { + let error = replica::replicator_error(status); + if let Some(refusal) = PrimaryFenceRefusal::of(&error) { + metrics::increment_counter!( + "libsql_server_replica_fence_refusals_total", + "code" => refusal.0.outcome().as_str(), + ); + if fence.observe_primary(Some(refusal.local_denial())) { + tracing::warn!( + namespace = %namespace, + "the primary ended replication because its namespace fence denies \ + reads ({}): local reads of this replica are refused until the primary \ + admits replication again", + refusal.0.outcome() + ); + } + } + error }) } @@ -164,9 +246,17 @@ impl ReplicatorClient for Client { self.first_sync_since_handshake = true; tracing::debug!("Attempting to perform handshake with primary."); let req = self.make_request(HelloRequest::new()); - let resp = self.client.hello(req).await?; + let resp = match self.client.hello(req).await { + Ok(resp) => resp, + Err(status) => return Err(self.status_error(status)), + }; let hello = resp.into_inner(); verify_session_token(&hello.session_token).map_err(Error::Client)?; + // The primary answers `hello` only where its fence admits replication. + self.fence_refusals = 0; + self.observe_primary_fence(replica::denial_from_hello( + hello.config.as_ref().and_then(|c| c.fence.as_ref()), + )); self.primary_replication_index = hello.current_replication_index; self.session_token.replace(hello.session_token.clone()); @@ -207,32 +297,30 @@ impl ReplicatorClient for Client { }; let req = self.make_request(offset); - let stream = self - .client - .log_entries(req) - .await? - .into_inner() - .inspect_ok(|f| { - match f.timestamp { - Some(ts_millis) => { - if let Some(commited_at) = DateTime::from_timestamp_millis(ts_millis) { - let lat = Utc::now() - commited_at; - match lat.to_std() { - Ok(lat) => { - // we can record negative values if the clocks are out-of-sync. There is not - // point in recording those values. - REPLICATION_LATENCY.record(lat); - } - Err(_) => { - REPLICATION_LATENCY_OUT_OF_SYNC.increment(1); - } + let stream = match self.client.log_entries(req).await { + Ok(resp) => resp.into_inner(), + Err(status) => return Err(self.status_error(status)), + }; + let stream = self.fenced_frames(stream).inspect_ok(|f| { + match f.timestamp { + Some(ts_millis) => { + if let Some(commited_at) = DateTime::from_timestamp_millis(ts_millis) { + let lat = Utc::now() - commited_at; + match lat.to_std() { + Ok(lat) => { + // we can record negative values if the clocks are out-of-sync. There is not + // point in recording those values. + REPLICATION_LATENCY.record(lat); + } + Err(_) => { + REPLICATION_LATENCY_OUT_OF_SYNC.increment(1); } } } - None => REPLICATION_LATENCY_CACHE_MISS.increment(1), } - }) - .map_err(Into::into); + None => REPLICATION_LATENCY_CACHE_MISS.increment(1), + } + }); Ok(Box::pin(stream)) } @@ -245,11 +333,11 @@ impl ReplicatorClient for Client { let req = self.make_request(offset); match self.client.snapshot(req).await { Ok(resp) => { - let stream = resp.into_inner().map_err(Into::into); + let stream = self.fenced_frames(resp.into_inner()); Ok(Box::pin(stream)) } Err(e) if e.code() == Code::Unavailable => Err(Error::SnapshotPending), - Err(e) => return Err(e.into()), + Err(e) => Err(self.status_error(e)), } } diff --git a/libsql-server/tests/fence/protocol.rs b/libsql-server/tests/fence/protocol.rs index 7cf970fda3..a7b7b85d0f 100644 --- a/libsql-server/tests/fence/protocol.rs +++ b/libsql-server/tests/fence/protocol.rs @@ -974,3 +974,364 @@ fn denial_not_retried() { }); sim.run().unwrap(); } + +/// The primary's internal replication service (the one replica servers use), as a raw client +/// that sees statuses and their metadata. +struct Replication { + client: libsql_replication::rpc::replication::replication_log_client::ReplicationLogClient< + tonic::transport::Channel, + >, + ns: String, + token: Option, +} + +impl Replication { + fn new(ns: &str) -> anyhow::Result { + use tower::ServiceExt as _; + let uri = tonic::transport::Uri::from_static("http://primary:4567"); + let channel = tonic::transport::Channel::builder(uri.clone()).connect_with_connector_lazy( + TurmoilConnector.map_err(|e| -> Box { e.into() }), + ); + Ok(Self { + client: libsql_replication::rpc::replication::replication_log_client::ReplicationLogClient::with_origin(channel, uri), + ns: ns.into(), + token: None, + }) + } + + fn request(&self, msg: T) -> tonic::Request { + use libsql_replication::rpc::replication::{NAMESPACE_METADATA_KEY, SESSION_TOKEN_KEY}; + let mut req = tonic::Request::new(msg); + req.metadata_mut().insert_bin( + NAMESPACE_METADATA_KEY, + tonic::metadata::BinaryMetadataValue::from_bytes(self.ns.as_bytes()), + ); + if let Some(token) = &self.token { + req.metadata_mut().insert( + SESSION_TOKEN_KEY, + tonic::metadata::AsciiMetadataValue::try_from(token.as_ref()).unwrap(), + ); + } + req + } + + /// `hello`; on success keeps the session token and returns the replicated fence. + async fn hello( + &mut self, + ) -> Result, tonic::Status> { + let req = self.request(libsql_replication::rpc::replication::HelloRequest::new()); + let hello = self.client.hello(req).await?.into_inner(); + self.token = Some(hello.session_token.clone()); + Ok(hello.config.and_then(|c| c.fence)) + } + + fn offset(&self) -> tonic::Request { + self.request(libsql_replication::rpc::replication::LogOffset { + next_offset: 0, + wal_flavor: None, + }) + } + + async fn log_entries( + &mut self, + ) -> Result, tonic::Status> { + let req = self.offset(); + Ok(self.client.log_entries(req).await?.into_inner()) + } + + async fn snapshot( + &mut self, + ) -> Result, tonic::Status> { + let req = self.offset(); + Ok(self.client.snapshot(req).await?.into_inner()) + } +} + +/// A status in the typed form of section 6: `FAILED_PRECONDITION`, the stable code in +/// `x-libsql-fence-code`, and the code prefixing the message. +#[track_caller] +fn assert_fence_status(what: &str, status: &tonic::Status, code: &str) { + assert_eq!( + status.code(), + tonic::Code::FailedPrecondition, + "{what}: {status:?}" + ); + assert_eq!( + status + .metadata() + .get("x-libsql-fence-code") + .and_then(|v| v.to_str().ok()), + Some(code), + "{what}: {status:?}" + ); + assert!(status.message().starts_with(code), "{what}: {status:?}"); +} + +/// Replication as a raw peer of the primary sees it (sections 6.2 and 9): `hello` carries the +/// replicated fence while the primary admits replication; under a read fence an open stream +/// ends with the typed status and every call is refused with it; a quarantined target refuses +/// with its own code; after the read fence is cleared `hello` is answered again. +#[test] +fn replication_codes() { + let mut sim = sim(); + let tmp = tempdir().unwrap(); + make_primary(&mut sim, tmp.path().to_path_buf(), Primary::default()); + sim.client("client", async { + let admin = Admin::new(Some(ADMIN_KEY)); + let op = uuid(0x100); + admin.create_namespace("plain").await?; + load_and_log_id(&admin, "plain").await?; + assert_eq!(Replication::new("plain")?.hello().await?, None, "unfenced"); + + let mut repl = Replication::new("src")?; + let rev = write_fenced(&admin, "src", op).await?; + let fence = repl.hello().await?.expect("hello carries the write fence"); + assert_eq!( + (fence.state.as_str(), fence.revision), + ("SOURCE_WRITE_FENCED", rev) + ); + let mut tail = repl.log_entries().await?; + let frame = tail + .next() + .await + .expect("a frame") + .expect("frames are served"); + assert!(!frame.data.is_empty()); + + let rev = read_fence(&admin, "src", op, rev).await?; + // The open stream ends with the terminal status (frames already buffered first). + let ended = loop { + match tail.next().await { + Some(Ok(_)) => continue, + Some(Err(status)) => break status, + None => panic!("the stream ended without a status"), + } + }; + assert_fence_status("open log_entries", &ended, READ_FENCED); + assert!(tail.next().await.is_none()); + assert_fence_status("hello", &repl.hello().await.unwrap_err(), READ_FENCED); + assert_fence_status( + "log_entries", + &repl.log_entries().await.unwrap_err(), + READ_FENCED, + ); + assert_fence_status("snapshot", &repl.snapshot().await.unwrap_err(), READ_FENCED); + + let rev = clear_read_fence(&admin, "src", op, rev).await?; + let fence = repl.hello().await?.expect("hello carries the write fence"); + assert_eq!( + (fence.state.as_str(), fence.revision), + ("SOURCE_WRITE_FENCED", rev) + ); + release(&admin, "src", op, rev).await?; + assert_eq!(repl.hello().await?, None, "released"); + + quarantined_target(&admin, "dst", uuid(0x200)).await?; + let mut target = Replication::new("dst")?; + assert_fence_status( + "target hello", + &target.hello().await.unwrap_err(), + QUARANTINED, + ); + Ok(()) + }); + sim.run().unwrap(); +} + +/// The first `/v2` result of `sql` on `user`'s host: `Ok(())` or the Hrana error code. +async fn v2_read(user: &User, ns: &str, sql: &str) -> anyhow::Result> { + let (status, body) = user + .pipeline(ns, 2, None, json!([execute_req(sql)])) + .await?; + assert_eq!(status, StatusCode::OK, "{body}"); + let result = &body["results"][0]; + Ok(match result["type"].as_str() { + Some("ok") => Ok(()), + _ => Err(result["error"]["code"] + .as_str() + .unwrap_or_else(|| panic!("no error code: {body}")) + .to_string()), + }) +} + +/// Poll the legacy API on `user`'s host with `sql` until it answers `status`, for at most +/// `within` of simulated time, and return that answer and how long it took. +async fn poll_until( + user: &User, + ns: &str, + sql: &str, + status: StatusCode, + within: std::time::Duration, +) -> anyhow::Result<(Value, std::time::Duration)> { + let started = tokio::time::Instant::now(); + loop { + let (got, body) = user.legacy(ns, &[sql]).await?; + if got == status { + return Ok((body, started.elapsed())); + } + assert!( + started.elapsed() < within, + "still {got} after {:?}: {body}", + started.elapsed() + ); + tokio::time::sleep(std::time::Duration::from_millis(50)).await; + } +} + +/// `src` write-fenced on the primary and loaded on the replica, then read-fenced; returns once +/// the replica denies local reads, with the fence's revision. +async fn replica_read_fenced(admin: &Admin, replica: &User, op: Uuid) -> anyhow::Result { + let rev = write_fenced(admin, "src", op).await?; + let (status, body) = replica.legacy("src", &["select count(*) from t"]).await?; + assert_eq!( + status, + StatusCode::OK, + "replica before the read fence: {body}" + ); + let rev = read_fence(admin, "src", op, rev).await?; + // The replica learns of the fence when its replication stream ends, which the primary's + // read drain waits for; the denial is published as the terminal status arrives. + let (body, took) = poll_until( + replica, + "src", + "select count(*) from t", + StatusCode::LOCKED, + std::time::Duration::from_secs(2), + ) + .await?; + assert_eq!(body["code"], READ_FENCED, "{body}"); + assert!( + took < std::time::Duration::from_millis(500), + "took {took:?}" + ); + Ok(rev) +} + +/// A replica server denies local reads of a namespace whose primary is read-fenced, on every +/// read surface, with the primary's code (section 6.2). +#[test] +fn replica_reads_denied_while_source_read_fenced() { + let mut sim = sim(); + let primary = tempdir().unwrap(); + let replica = tempdir().unwrap(); + make_primary(&mut sim, primary.path().to_path_buf(), Primary::default()); + make_replica(&mut sim, replica.path().to_path_buf()); + sim.client("client", async { + let admin = Admin::new(Some(ADMIN_KEY)); + let user = User::on("replica0"); + replica_read_fenced(&admin, &user, uuid(0x100)).await?; + + assert_locked( + "legacy read", + &user.legacy("src", &["select * from t"]).await?, + READ_FENCED, + ); + assert_locked( + "v1 execute", + &user.execute("src", "select * from t").await?, + READ_FENCED, + ); + assert_eq!( + v2_read(&user, "src", "select * from t").await?, + Err(READ_FENCED.to_string()) + ); + // `/dump` is not served by a replica server at all ("database is not a primary"). + // Still denied later: the replica does not forget while the primary keeps refusing. + tokio::time::sleep(std::time::Duration::from_secs(30)).await; + assert_locked( + "legacy read later", + &user.legacy("src", &["select * from t"]).await?, + READ_FENCED, + ); + Ok(()) + }); + sim.run().unwrap(); +} + +/// While the primary's fence refuses replication, the replica retries at a growing interval +/// capped at 15 s, not every second or in a tight loop: over 60 s of simulated time it makes +/// a handful of attempts (1 + 2 + 4 + 8 + 15 + 15 + 15 s of pauses), each counted. +#[test] +fn replica_backs_off_on_fence_code() { + let mut sim = sim(); + let primary = tempdir().unwrap(); + let replica = tempdir().unwrap(); + make_primary(&mut sim, primary.path().to_path_buf(), Primary::default()); + make_replica(&mut sim, replica.path().to_path_buf()); + sim.client("client", async { + let admin = Admin::new(Some(ADMIN_KEY)); + let user = User::on("replica0"); + replica_read_fenced(&admin, &user, uuid(0x100)).await?; + + let refusals = || { + crate::common::snapshot_metrics() + .get_counter_label( + "libsql_server_replica_fence_refusals_total", + ("code", READ_FENCED), + ) + .unwrap_or(0) + }; + let before = refusals(); + assert!(before >= 1, "the ended stream is counted"); + tokio::time::sleep(std::time::Duration::from_secs(60)).await; + let attempts = refusals() - before; + assert!( + (4..=9).contains(&attempts), + "{attempts} refused attempts in 60 s" + ); + Ok(()) + }); + sim.run().unwrap(); +} + +/// Once the primary clears the read fence the replica answers `hello` again within the +/// back-off cap, serves local reads, and replicates new writes once the source is released. +#[test] +fn replica_resumes_after_clear_read_fence() { + let mut sim = sim(); + let primary = tempdir().unwrap(); + let replica = tempdir().unwrap(); + make_primary(&mut sim, primary.path().to_path_buf(), Primary::default()); + make_replica(&mut sim, replica.path().to_path_buf()); + sim.client("client", async { + let admin = Admin::new(Some(ADMIN_KEY)); + let user = User::on("replica0"); + let op = uuid(0x100); + let rev = replica_read_fenced(&admin, &user, op).await?; + // Let the back-off grow to its cap before clearing. + tokio::time::sleep(std::time::Duration::from_secs(40)).await; + + let rev = clear_read_fence(&admin, "src", op, rev).await?; + let (_, took) = poll_until( + &user, + "src", + "select count(*) from t", + StatusCode::OK, + std::time::Duration::from_secs(20), + ) + .await?; + assert!(took <= std::time::Duration::from_secs(16), "took {took:?}"); + assert_eq!(v2_read(&user, "src", "select * from t").await?, Ok(())); + + release(&admin, "src", op, rev).await?; + let (status, body) = User::new() + .legacy("src", &["insert into t values (2)"]) + .await?; + assert_eq!(status, StatusCode::OK, "{body}"); + let started = tokio::time::Instant::now(); + loop { + let (status, body) = user.legacy("src", &["select count(*) from t"]).await?; + assert_eq!(status, StatusCode::OK, "{body}"); + if body[0]["results"]["rows"][0][0] == 2 { + break; + } + assert!( + started.elapsed() < std::time::Duration::from_secs(10), + "not replicated: {body}" + ); + tokio::time::sleep(std::time::Duration::from_millis(100)).await; + } + Ok(()) + }); + sim.run().unwrap(); +} From 7dd414dafee7c3ec83ab6c2e1d7d61581d8714c0 Mon Sep 17 00:00:00 2001 From: River Date: Wed, 30 Sep 2026 10:36:32 +0000 Subject: [PATCH 24/33] libsql-server: do not create a replica namespace the primary's fence refuses A replica server creating a namespace lazily for a name the primary's fence refuses (a quarantined or aborted target, a read-fenced source, a fence state the primary cannot establish) now fails the request at once with the primary's code (Error::NamespaceFence: 423 and the stable code) instead of retrying the handshake, and leaves no local namespace behind: the namespace directory the setup created is removed (a directory that already existed is kept), and NamespaceStore::with forgets the metastore entry handle() added and the controller the attempt created, when neither holds anything durable or is in use by another attempt. A later request, once the primary admits the name, creates it normally. Co-authored-by: Tomasz Szymczyszyn --- docs/NAMESPACE_FENCE.md | 8 +- .../src/namespace/configurator/replica.rs | 26 +++- libsql-server/src/namespace/fence/registry.rs | 60 +++++++++ libsql-server/src/namespace/meta_store.rs | 69 +++++++++- libsql-server/src/namespace/store.rs | 41 +++++- libsql-server/tests/fence/protocol.rs | 125 ++++++++++++++++++ 6 files changed, 317 insertions(+), 12 deletions(-) diff --git a/docs/NAMESPACE_FENCE.md b/docs/NAMESPACE_FENCE.md index b84f4fa332..195e5b0745 100644 --- a/docs/NAMESPACE_FENCE.md +++ b/docs/NAMESPACE_FENCE.md @@ -744,7 +744,7 @@ A namespace that startup cannot recover is registered `UNKNOWN_UNAVAILABLE` in m - `destroy_on_error`: the broken metastore is renamed to `metastore.broken-` rather than deleted (if the rename fails, startup fails instead), and directories with markers are registered `UNKNOWN_UNAVAILABLE` after the rebuild. Without fences it still deletes, as before. - A metastore without fence tables next to a directory that holds a marker (rebuilt, recovered, or restored from a backup taken before the tables existed) makes that namespace `UNKNOWN_UNAVAILABLE`, even when its config row is present. - Metastore restore from backup: the marker comparison (section 5.6) makes any namespace whose record went backwards or disappeared `UNKNOWN_UNAVAILABLE`. A restored record is never trusted over a newer marker. Surfacing the provenance (`restored_from_backup`, backup generation) in the capability endpoint, `InspectFence`, a metric and a startup line is part of the admin API (section 4). -- Replica-kind servers: lazy creation of a name refused by the primary with a fence code must not create a local default namespace. The primary's refusal arrives as the typed status of the replication `hello` (section 6.2), not through the write proxy; handling it on the replica is not implemented yet (planned with the replica's handling of the replicated fence). +- Replica-kind servers: lazy creation of a name the primary refuses with a fence code (a quarantined or aborted target, a read-fenced source, a fence state the primary cannot establish) does not create a local namespace. The refusal arrives as the typed status of the replication `hello` (section 6.2), not through the write proxy. `ReplicaConfigurator::setup` fails at once with that code (`Error::NamespaceFence`, so the client gets `423` and the code of section 6.0) instead of retrying the handshake, and removes the namespace directory if the setup created it (a directory that was already there is left as it is). `NamespaceStore::with` then forgets what the attempt left in memory: the metastore entry `handle()` added (`MetaStore::forget_unstored`, only when no config row is stored for the name and no other handle is subscribed to it, so a concurrent creation or a persisted config keeps it) and the name's controller (`FenceRegistry::forget_idle`, only when it holds no fence record and no in-memory gate and nothing else refers to it). `exists()` and `lookup()` therefore do not report the name, and a later request, once the primary admits the name, creates it normally. ### 13.4 Shared schema @@ -787,7 +787,7 @@ How each path that can reach namespace data or lifecycle is covered. File refere | `http/admin/fence.rs` | Fence routes (section 4.5): capability discovery, `InspectFence`, one route per command through `NamespaceStore::execute_fence_command`, and `validation-query` through `open_validation_session`; admin auth key and primary required for every route that changes state or uses a capability. | | `http/user/dump.rs`, `connection/dump/exporter.rs` | Gate check (`Stream`) before connection creation (typed, no panic on create error); `Dump` lease held by the export; cancel checked before every row and by the pipe's pending write; the body stream ends with the fence error (aborted response) on cancellation. | | `rpc/replication/replication_log.rs` `hello`, `log_entries`, `batch_log_entries`, `snapshot` | Denied at request start (`FAILED_PRECONDITION` + `x-libsql-fence-code`, counted, rate-limited log); `FencedStream` replication leases for both streams and the batch; typed terminal status when the gate closes or at the deadline, lease released without the peer; `ReplicatedFence` in `hello`'s config while a fence is active (section 6.2; the gate is read without a lease, since `hello` streams no data). | -| `replication/replicator_client.rs`, `namespace/configurator/replica.rs` (replica servers) | A fence status from `hello`, `log_entries` or `snapshot`, or a stream ended with one, installs the local read denial (`FenceController::observe_primary`) and is returned as `PrimaryFenceRefusal`; the replication loop backs off 1 s → 15 s instead of reconnecting every second; an answered `hello` lifts the denial (section 6.2). | +| `replication/replicator_client.rs`, `namespace/configurator/replica.rs` (replica servers) | A fence status from `hello`, `log_entries` or `snapshot`, or a stream ended with one, installs the local read denial (`FenceController::observe_primary`) and is returned as `PrimaryFenceRefusal`; the replication loop backs off 1 s → 15 s instead of reconnecting every second; an answered `hello` lifts the denial (section 6.2). A lazy creation whose first `hello` the primary refuses fails at once with the code and leaves no directory, metastore entry or controller behind (section 13.3). | | `admin_shell.rs` | Writes denied at the WAL (no capability); reads checked against the gate, and a read lease held, per query (cancelled through the connection's interrupt handle). | | `schema/scheduler.rs`, `database/schema.rs` | Shared schema excluded from fencing; registering a migration job is refused while the schema or a linked namespace denies lifecycle work (section 13.4); migration writes are WAL-gated; the scheduler's `block_writes` flag is not treated as drain evidence. | | `namespace/configurator/helpers.rs` `load_dump`, `http/admin/mod.rs` `dump_stream_from_url` | Restore options and dump URLs are refused for fenced namespaces before the dump is fetched (section 13.5); outside a capability the loader's writes are refused at the WAL. Import goes through `ImportSession::load_dump`, which runs the same loader (`load_dump_sql`) on the capability connection. | @@ -845,7 +845,7 @@ Planned test names; the table is updated as tests land. | 15 | Lost `EnableTargetWrites` response resolved from receipt/state | landed: `namespace::fence::target::tests::enable_writes_response_loss_resolved` | | 16 | Read fence drains SQL, dump, `log_entries`, `snapshot`, including dead peers and forced termination | SQL landed: `namespace::fence::read::tests::{read_fence_waits_for_running_program, program_after_closing_gate_is_refused (parked after the read-closing gate, before the CAS), read_fence_cancels_at_deadline, unreleased_lease_answers_draining_and_replay_completes, idle_txn_fails_on_next_program, clear_read_fence_reopens_reads_not_writes, refused_read_fence_reopens_reads, attach_of_read_fenced_namespace_denied}`, `admin_shell::fence_tests::admin_shell_read_denied`; dump and replication landed: `namespace::fence::stream::tests::{dump_lease_released_on_cancel (a dump blocked mid-row on a peer that stopped reading is cancelled at the deadline, its lease released without the peer, the fence acknowledged, and the body ends with the fence error and no `COMMIT;`), read_fence_waits_for_dump, dump_refused_while_read_fenced, log_entries_stream_ends_typed, stream_lease_released_without_peer_read (a dead peer), snapshot_stream_ends_typed, replication_calls_denied_while_read_fenced, read_fence_forced_termination}`; the response `/dump` returns: `namespace::fence::stream::tests::dump_response_aborted_on_cancel` (a dump cancelled by the read drain fails its response body, with the fence error and no `COMMIT;`); over HTTP: `tests::fence::protocol::dump_codes` (complete under a write fence; `423` + code under a read fence and on a quarantined target) | | 17 | Delete, reset, fork, restore, config, schema mutation rejected | landed: `tests::fence::lifecycle::lifecycle_rejected_while_fenced` (over the admin API, for a write-fenced source and a quarantined target: delete, fork as source and as destination, create with a `dump_url` whose file does not exist, create over the record, linking to a shared schema at creation and config `POST` are all `423` with the fence code; state, revision and data are unchanged and no copy exists; a target cannot be created with a shared schema; after release the source takes writes and config, fork and delete work again); `namespace::store::fence_tests::{reset_refused_while_fenced (called directly and as the replicator's reset callback; after release reset works and wipes the data), lifecycle_refused_while_fenced (fork either side, create over, delete, config and shared-schema link in the metastore transaction)}`; `schema::scheduler::test::fence::{acquire_rejects_shared_schema, migration_not_registered_while_linked_namespace_fenced}` | -| 18 | Codes through HTTP, Hrana, RPC, dump, replication, replica write proxy; distinguishable from auth/timeout/not-found; old peers compatible; no retry loops | user protocols landed: `tests::fence::protocol::{http_codes (legacy `/`, `/v1/execute`, `/v1/batch` for a write-fenced write, a read-fenced read, a quarantined target and a namespace whose marker cannot be decoded: `423` with `code`, and `detail` where there is one), hrana_http_codes (`/v2`, `/v3` pipelines and `/v3/cursor`: step and whole-request errors carry the code and the baton stays usable), hrana_ws_codes (the same over a WebSocket, whose stream reads again after the read fence is cleared), dump_codes, auth_and_not_found_distinct (`401` without or with a wrong credential and `404` for a missing namespace, with no fence code, on the same fenced server)}`, `error::fence_tests::fence_errors_carry_code` (the body through every wrapper; `block_*`'s `Blocked` keeps its mapping); RPC and replica write proxy landed: `rpc::proxy::fence_tests::rpc_codes` (on the primary's proxy service: a write-fenced write is a step error with `SQL_ERROR` + `stable_code`, reads are served, a read-fenced read is a program error with the code when streamed and the typed `FAILED_PRECONDITION` status when unary, and a namespace whose fence state is unknown is refused with the typed status before any connection), `rpc::proxy::fence_tests::replica_maps_proxied_denials` (step, program and connection-status denials become the fence error on the replica; an older primary's error without `stable_code`, an unknown code and other errors keep their mapping), `namespace::fence::outcome::tests::peer_denials_round_trip`, `tests::fence::protocol::{replica_proxy_preserves_code (writes through a replica to a write-fenced primary: `423` + `MIGRATION_WRITE_FENCED` on legacy `/` and `/v1/execute`, the step error on `/v1/batch`, the Hrana error code on `/v2` and `/v3`; reads on the replica served; writes through the replica work after release), denial_not_retried (each refused write is delegated exactly once and answered well within the write proxy's first retry backoff)}`; capability: `tests::fence::admin::capabilities` asserts `proxy_stable_code: true`; replication and replica servers landed: `tests::fence::protocol::{replication_codes (a raw peer of the primary's internal replication service: `hello` carries the write fence's state and revision, an open `log_entries` stream ends with `FAILED_PRECONDITION` + `x-libsql-fence-code` `MIGRATION_READ_FENCED` under the read fence, `hello`, `log_entries` and `snapshot` are then refused with it, `hello` is answered again after the clear and carries no fence after release, and a quarantined target refuses `hello` with `MIGRATION_TARGET_QUARANTINED`), replica_reads_denied_while_source_read_fenced (the replica refuses local reads with `423` + `MIGRATION_READ_FENCED` on legacy `/` and `/v1/execute` and the Hrana code on `/v2` within 500 ms of simulated time after the read fence is acknowledged, and still 30 s later), replica_backs_off_on_fence_code (4 to 9 refused, counted attempts over 60 s of simulated time; a fixed 1 s retry makes 57), replica_resumes_after_clear_read_fence (reads served again within 16 s of the clear, and a write after release is replicated)}`, `namespace::fence::replica::tests::{refusal_from_typed_status_only, hello_fence_denies_only_read_denying_states, backoff_doubles_to_its_cap, observed_denial_refuses_local_reads_and_cancels_leases}`; landed: `libsql-replication` `rpc::test::{proxy_error_stable_code_is_additive, replicated_fence_is_additive}` (each new field is skipped by a peer that does not know it, absent from an older peer's message, and absent fields encode exactly as before), `namespace::fence::stream::tests::hello_carries_replicated_fence` (no fence before acquisition and after release; state and revision while write-fenced; the stored configuration never carries it) | +| 18 | Codes through HTTP, Hrana, RPC, dump, replication, replica write proxy; distinguishable from auth/timeout/not-found; old peers compatible; no retry loops | user protocols landed: `tests::fence::protocol::{http_codes (legacy `/`, `/v1/execute`, `/v1/batch` for a write-fenced write, a read-fenced read, a quarantined target and a namespace whose marker cannot be decoded: `423` with `code`, and `detail` where there is one), hrana_http_codes (`/v2`, `/v3` pipelines and `/v3/cursor`: step and whole-request errors carry the code and the baton stays usable), hrana_ws_codes (the same over a WebSocket, whose stream reads again after the read fence is cleared), dump_codes, auth_and_not_found_distinct (`401` without or with a wrong credential and `404` for a missing namespace, with no fence code, on the same fenced server)}`, `error::fence_tests::fence_errors_carry_code` (the body through every wrapper; `block_*`'s `Blocked` keeps its mapping); RPC and replica write proxy landed: `rpc::proxy::fence_tests::rpc_codes` (on the primary's proxy service: a write-fenced write is a step error with `SQL_ERROR` + `stable_code`, reads are served, a read-fenced read is a program error with the code when streamed and the typed `FAILED_PRECONDITION` status when unary, and a namespace whose fence state is unknown is refused with the typed status before any connection), `rpc::proxy::fence_tests::replica_maps_proxied_denials` (step, program and connection-status denials become the fence error on the replica; an older primary's error without `stable_code`, an unknown code and other errors keep their mapping), `namespace::fence::outcome::tests::peer_denials_round_trip`, `tests::fence::protocol::{replica_proxy_preserves_code (writes through a replica to a write-fenced primary: `423` + `MIGRATION_WRITE_FENCED` on legacy `/` and `/v1/execute`, the step error on `/v1/batch`, the Hrana error code on `/v2` and `/v3`; reads on the replica served; writes through the replica work after release), denial_not_retried (each refused write is delegated exactly once and answered well within the write proxy's first retry backoff)}`; capability: `tests::fence::admin::capabilities` asserts `proxy_stable_code: true`; replication and replica servers landed: `tests::fence::protocol::{replication_codes (a raw peer of the primary's internal replication service: `hello` carries the write fence's state and revision, an open `log_entries` stream ends with `FAILED_PRECONDITION` + `x-libsql-fence-code` `MIGRATION_READ_FENCED` under the read fence, `hello`, `log_entries` and `snapshot` are then refused with it, `hello` is answered again after the clear and carries no fence after release, and a quarantined target refuses `hello` with `MIGRATION_TARGET_QUARANTINED`), replica_reads_denied_while_source_read_fenced (the replica refuses local reads with `423` + `MIGRATION_READ_FENCED` on legacy `/` and `/v1/execute` and the Hrana code on `/v2` within 500 ms of simulated time after the read fence is acknowledged, and still 30 s later), replica_backs_off_on_fence_code (4 to 9 refused, counted attempts over 60 s of simulated time; a fixed 1 s retry makes 57), replica_resumes_after_clear_read_fence (reads served again within 16 s of the clear, and a write after release is replicated), replica_lazy_creation_refused_by_fence (a read of a quarantined target through a replica that has never loaded it answers `423` + `MIGRATION_TARGET_QUARANTINED` within 500 ms of simulated time, twice, leaving no `dbs/` directory; a directory that was already there is kept; after publication the replica creates and serves the name, and after enable-writes a write through it succeeds)}`, `namespace::meta_store::fence_tests::forget_unstored_only_unused_unstored_entries`, `namespace::fence::registry::tests::forget_idle_only_unreferenced_plain_controllers`, `namespace::fence::replica::tests::{refusal_from_typed_status_only, hello_fence_denies_only_read_denying_states, backoff_doubles_to_its_cap, observed_denial_refuses_local_reads_and_cancels_leases}`; landed: `libsql-replication` `rpc::test::{proxy_error_stable_code_is_additive, replicated_fence_is_additive}` (each new field is skipped by a peer that does not know it, absent from an older peer's message, and absent fields encode exactly as before), `namespace::fence::stream::tests::hello_carries_replicated_fence` (no fence before acquisition and after release; state and revision while write-fenced; the stored configuration never carries it) | | 19 | Corrupt or unknown durable fence state fails closed | `fence::store::tests::corrupt_payload_fails_closed`, `unknown_format_version_fails_closed` | | 20 | Metrics and audit logs | `tests::fence::observability::metrics_and_labels`; `fence::audit::tests::audit_event_fields` | | 21 | Capability discovery and mixed-version protection | capability discovery landed: `tests::fence::admin::{capabilities, capabilities_when_disabled}`; planned: `fence::store::tests::legacy_mirror_and_fk_guard` (bounded, see section 18) | @@ -925,7 +925,7 @@ Planned commits, each leaving the crate building with its tests passing: | Frozen boundary | Captured with no writer holding the manager slot, after `insert_frames` published it. | | `VACUUM` | Not maintenance; skipped while writes are fenced. | | ATTACH of a fenced namespace | Checked as a `NormalRead` of the attached namespace through a non-creating lookup. | -| Open-default `handle()`, including ATTACH authorisation and replica lazy creation | Non-creating lookups on read paths; creating paths refuse names with a record or marker. | +| Open-default `handle()`, including ATTACH authorisation and replica lazy creation | Non-creating lookups on read paths; creating paths refuse names with a record or marker; a replica's lazy creation that the primary's fence refuses fails with the code and is undone (section 13.3). | | `process()` publishing unpersisted config | Fixed for all config writes. | | The scheduler's second metastore connection | Fence CAS and config writes use `BEGIN IMMEDIATE`; shared schema is excluded. | | Replica servers inherit config | Additive `ReplicatedFence` in `hello`; the typed terminal status installs a local read denial on a newer replica server, with a capped reconnect back-off (section 6.2). The legacy `block_*` mirror exists only in the stored config row (section 13.2), so it does not reach replica servers. | diff --git a/libsql-server/src/namespace/configurator/replica.rs b/libsql-server/src/namespace/configurator/replica.rs index 986d5880f2..9846e652f5 100644 --- a/libsql-server/src/namespace/configurator/replica.rs +++ b/libsql-server/src/namespace/configurator/replica.rs @@ -67,6 +67,9 @@ impl ConfigureNamespace for ReplicaConfigurator { Box::pin(async move { tracing::debug!("creating replica namespace"); let db_path = self.base.base_path.join("dbs").join(name.as_str()); + // Whether this server already had a copy of the namespace: a directory this setup + // creates is removed again if the primary's fence refuses the namespace. + let had_copy = db_path.try_exists()?; let channel = self.channel.clone(); let uri = self.uri.clone(); @@ -112,7 +115,28 @@ impl ConfigureNamespace for ReplicaConfigurator { ) .await; } - Err(e) => Err(e)?, + Err(e) => { + if let Some(refusal) = PrimaryFenceRefusal::of(&e) { + // The primary's fence refuses the namespace (a quarantined target, a + // read-fenced source, a fence state it cannot establish; section 13.3): + // fail at once with its code rather than retrying the handshake, and + // leave no local copy behind that this setup created. + let denial = refusal.0.clone(); + drop(replicator); + if !had_copy { + if let Err(e) = tokio::fs::remove_dir_all(&db_path).await { + if e.kind() != std::io::ErrorKind::NotFound { + tracing::warn!( + "failed to remove {} after the primary refused {name}: {e}", + db_path.display() + ); + } + } + } + return Err(crate::Error::NamespaceFence(denial)); + } + Err(e)? + } Ok(_) => (), } diff --git a/libsql-server/src/namespace/fence/registry.rs b/libsql-server/src/namespace/fence/registry.rs index 24dd0ea23e..7dd52425ba 100644 --- a/libsql-server/src/namespace/fence/registry.rs +++ b/libsql-server/src/namespace/fence/registry.rs @@ -66,6 +66,29 @@ impl FenceRegistry { self.controllers.lock().remove(namespace) } + /// Forget `namespace`'s controller if it holds nothing worth keeping: no fence record and + /// no in-memory gate (only what a replica learned of its primary's fence, which the next + /// answered `hello` would replace), and nothing but the registry refers to it. For a name + /// whose setup failed before it was ever served, such as a replica's lazy creation that + /// the primary's fence refused. Returns whether it was forgotten. + pub fn forget_idle(&self, namespace: &NamespaceName) -> bool { + let mut controllers = self.controllers.lock(); + let idle = controllers.get(namespace).is_some_and(|controller| { + let gate = controller.gate(); + // Under the registry lock nobody can take another reference to it. + Arc::strong_count(controller) == 1 + && matches!(gate.fence, StoredFence::None { .. }) + && gate.indeterminate.is_none() + && gate.installing.is_none() + && gate.closing_reads.is_none() + && gate.creating_target.is_none() + }); + if idle { + controllers.remove(namespace); + } + idle + } + /// Refuse a namespace whose fence state is `UNKNOWN_UNAVAILABLE`, or that is being created /// as a quarantined target, before any work is done to serve it. pub fn check_available(&self, namespace: &NamespaceName) -> Result<(), FenceError> { @@ -245,4 +268,41 @@ mod tests { assert!(registry.remove(&"ns".into()).is_some()); assert!(!Arc::ptr_eq(®istry.controller(&"ns".into()), &a)); } + + /// A replica's lazy creation that the primary refused leaves nothing behind in the registry, + /// unless the controller holds fence state or somebody else still refers to it. + #[test] + fn forget_idle_only_unreferenced_plain_controllers() { + let registry = FenceRegistry::default(); + assert!(!registry.forget_idle(&"missing".into())); + + // What a refused replication taught it does not keep it. + let refused = registry.controller(&"refused".into()); + refused.observe_primary(Some(FenceError::new( + FenceOutcome::MigrationTargetQuarantined, + "quarantined on the primary", + ))); + // Still referenced: kept. + assert!(!registry.forget_idle(&"refused".into())); + drop(refused); + assert!(registry.forget_idle(&"refused".into())); + assert!(registry.get(&"refused".into()).is_none()); + // The next use starts from a fresh UNFENCED controller. + assert!(registry + .controller(&"refused".into()) + .permits(OperationClass::NormalRead) + .is_ok()); + + // A controller with fence state is never forgotten. + let registry = FenceRegistry::seeded([( + "lost".into(), + StoredFence::Unavailable { + detail: FenceDetail::CorruptRecord, + reason: "test".into(), + marker: None, + }, + )]); + assert!(!registry.forget_idle(&"lost".into())); + assert!(registry.get(&"lost".into()).is_some()); + } } diff --git a/libsql-server/src/namespace/meta_store.rs b/libsql-server/src/namespace/meta_store.rs index 17d23db89c..49fd1d7404 100644 --- a/libsql-server/src/namespace/meta_store.rs +++ b/libsql-server/src/namespace/meta_store.rs @@ -15,7 +15,7 @@ use libsql_sys::wal::{ }; use parking_lot::Mutex; use prost::Message; -use rusqlite::TransactionBehavior; +use rusqlite::{OptionalExtension, TransactionBehavior}; use tokio::sync::oneshot; use tokio::sync::{ mpsc, @@ -1297,6 +1297,41 @@ impl MetaStore { r } + /// Take out the in-memory entry that [`handle`](Self::handle) put in the map for a + /// namespace whose creation then failed, so that [`exists`](Self::exists) and + /// [`lookup`](Self::lookup) do not report a namespace that was never created + /// (`docs/NAMESPACE_FENCE.md` section 13.3, replica lazy creation). Only an entry that no + /// handle is subscribed to any more and that has no stored config row is removed: a config + /// that was persisted, or a creation of the same name still in progress, keeps its entry. + /// Returns whether the entry was removed. + pub async fn forget_unstored(&self, namespace: NamespaceName) -> Result { + let inner = self.inner.clone(); + tokio::task::spawn_blocking(move || -> std::result::Result { + // The connection lock first, as everywhere else that takes both: a config being + // persisted concurrently is either already in its row here, or finds no entry when + // it publishes and inserts its own. + let conn = inner.conn.blocking_lock(); + let stored = conn + .query_row( + "SELECT 1 FROM namespace_configs WHERE namespace = ?1", + [namespace.as_str()], + |_| Ok(()), + ) + .optional()? + .is_some(); + let mut configs = inner.configs.blocking_lock(); + match configs.get(&namespace) { + Some(sender) if !stored && sender.receiver_count() == 0 => { + configs.remove(&namespace); + Ok(true) + } + _ => Ok(false), + } + }) + .await? + .map_err(fence_store_error) + } + // TODO: we need to either make sure that the metastore is restored // before we start accepting connections or we need to contact bottomless // here to check if a namespace exists. Preferably the former. @@ -1732,6 +1767,38 @@ mod fence_tests { } } + /// The in-memory entry a failed creation left is forgotten, so `exists()` and `lookup()` do + /// not report the name; a stored config, or a handle still held, keeps the entry. + #[tokio::test] + async fn forget_unstored_only_unused_unstored_entries() { + let tmp = tempdir().unwrap(); + let meta = open(tmp.path(), true).await; + let ns = NamespaceName::from("lazy"); + + assert!(!meta.forget_unstored(ns.clone()).await.unwrap()); + + let handle = meta.handle(ns.clone()).await.unwrap(); + assert!(meta.exists(&ns).await); + // A handle is still held (a creation in progress): kept. + assert!(!meta.forget_unstored(ns.clone()).await.unwrap()); + assert!(meta.exists(&ns).await); + drop(handle); + assert!(meta.forget_unstored(ns.clone()).await.unwrap()); + assert!(!meta.exists(&ns).await); + assert!(meta.lookup(&ns).await.unwrap().is_none()); + + // A stored config is never forgotten. + let stored = NamespaceName::from("stored"); + meta.handle(stored.clone()) + .await + .unwrap() + .store(DatabaseConfig::default()) + .await + .unwrap(); + assert!(!meta.forget_unstored(stored.clone()).await.unwrap()); + assert!(meta.lookup(&stored).await.unwrap().is_some()); + } + fn request( ns: &'static str, op: Uuid, diff --git a/libsql-server/src/namespace/store.rs b/libsql-server/src/namespace/store.rs index 9a83252b53..aa22de9e26 100644 --- a/libsql-server/src/namespace/store.rs +++ b/libsql-server/src/namespace/store.rs @@ -400,17 +400,46 @@ impl NamespaceStore { // A lookup that cannot create: only the default namespace and lazy creation create a // namespace here, and those refuse a name whose fence state is not established. - let handle = match self.inner.metadata.lookup(&namespace).await? { - Some(handle) => handle, + let (handle, created) = match self.inner.metadata.lookup(&namespace).await? { + Some(handle) => (handle, false), None if namespace == NamespaceName::default() || self.inner.allow_lazy_creation => { - self.inner.metadata.handle(namespace.clone()).await? + (self.inner.metadata.handle(namespace.clone()).await?, true) } None => return Err(Error::NamespaceDoesntExist(namespace.to_string())), }; - f(self + let entry = match self .load_namespace(&namespace, handle, RestoreOption::Latest) - .await?) - .await + .await + { + Ok(entry) => entry, + Err(e) => { + if created && e.fence_error().is_some() { + self.forget_refused_creation(&namespace).await; + } + return Err(e); + } + }; + f(entry).await + } + + /// Undo what a lazy creation that a fence refused left in memory (on a replica server, the + /// primary refused to replicate the name; `docs/NAMESPACE_FENCE.md` section 13.3): the + /// metastore entry [`MetaStore::handle`] added, which would otherwise make the name look + /// like an existing namespace, and the controller the attempt created. Both are kept when + /// they hold anything durable or are in use by another attempt. + async fn forget_refused_creation(&self, namespace: &NamespaceName) { + match self.inner.metadata.forget_unstored(namespace.clone()).await { + Ok(forgotten) => { + let controller = self.inner.fences.forget_idle(namespace); + tracing::debug!( + "refused creation of {namespace}: metastore entry forgotten: {forgotten}, \ + controller forgotten: {controller}" + ); + } + Err(e) => { + tracing::warn!("failed to forget the refused creation of {namespace}: {e}") + } + } } fn resolve_attach_fn(&self) -> ResolveNamespacePathFn { diff --git a/libsql-server/tests/fence/protocol.rs b/libsql-server/tests/fence/protocol.rs index a7b7b85d0f..50ae54e158 100644 --- a/libsql-server/tests/fence/protocol.rs +++ b/libsql-server/tests/fence/protocol.rs @@ -1335,3 +1335,128 @@ fn replica_resumes_after_clear_read_fence() { }); sim.run().unwrap(); } + +/// Walks the quarantined target `ns` of `op` to `TARGET_WRITE_FENCED` (readable): seal the +/// (empty) import, record a successful validation, publish. Returns the revision. +async fn publish_target(admin: &Admin, ns: &str, op: Uuid) -> anyhow::Result { + let (_, body) = admin.inspect(ns).await?; + let (state, mut rev) = state_of(&body); + assert_eq!(state, "TARGET_QUARANTINED", "{body}"); + for (n, route, from, extra, to) in [ + ( + 2, + "target/seal-import", + "TARGET_QUARANTINED", + json!({}), + "TARGET_VALIDATING", + ), + ( + 3, + "target/validation-receipt", + "TARGET_VALIDATING", + json!({ "result": "ok", "summary": "empty" }), + "TARGET_VALIDATING", + ), + ( + 4, + "target/publish-readable", + "TARGET_VALIDATING", + json!({}), + "TARGET_WRITE_FENCED", + ), + ] { + let (status, body) = admin + .command( + ns, + route, + command_body(op, uuid(op.as_u128() + n), from, rev, extra), + ) + .await?; + assert_eq!(status, StatusCode::OK, "{route}: {body}"); + assert_eq!(state_of(&body).0, to, "{route}: {body}"); + rev = state_of(&body).1; + } + Ok(rev) +} + +/// A replica server creating a namespace lazily, for a name the primary's fence refuses to +/// replicate (here a quarantined target), fails the request at once with the primary's code +/// instead of retrying the handshake, and leaves no local copy: no namespace directory it +/// created (a directory that was already there stays) and no metastore entry. Once the target +/// is published the same name is created and served normally (section 13.3). +#[test] +fn replica_lazy_creation_refused_by_fence() { + let mut sim = sim(); + let primary = tempdir().unwrap(); + let replica = tempdir().unwrap(); + // A directory that was on the replica before: a refused creation must not delete it. + let kept = replica.path().join("dbs").join("kept"); + std::fs::create_dir_all(&kept).unwrap(); + std::fs::write(kept.join("sentinel"), b"x").unwrap(); + let dbs = replica.path().join("dbs"); + make_primary(&mut sim, primary.path().to_path_buf(), Primary::default()); + make_replica(&mut sim, replica.path().to_path_buf()); + sim.client("client", async move { + let admin = Admin::new(Some(ADMIN_KEY)); + let user = User::on("replica0"); + let op = uuid(0x100); + quarantined_target(&admin, "tgt", op).await?; + quarantined_target(&admin, "kept", uuid(0x200)).await?; + + for attempt in 0..2 { + let started = tokio::time::Instant::now(); + assert_locked( + &format!("replica read of a quarantined target, attempt {attempt}"), + &user.legacy("tgt", &["select 1"]).await?, + QUARANTINED, + ); + // Refused at the first handshake, not after a second of retries per attempt. + let took = started.elapsed(); + assert!( + took < std::time::Duration::from_millis(500), + "took {took:?}" + ); + assert!( + !dbs.join("tgt").exists(), + "the refused creation left dbs/tgt behind" + ); + } + assert_locked( + "replica read of a quarantined target with a directory", + &user.legacy("kept", &["select 1"]).await?, + QUARANTINED, + ); + assert!( + kept.join("sentinel").exists(), + "a pre-existing directory was removed" + ); + + let rev = publish_target(&admin, "tgt", op).await?; + let (status, body) = user.legacy("tgt", &["select 1"]).await?; + assert_eq!(status, StatusCode::OK, "after publication: {body}"); + assert!( + dbs.join("tgt").exists(), + "published target not created on the replica" + ); + + // Writes through the replica reach the primary once writes are enabled. + let (status, body) = admin + .command( + "tgt", + "target/enable-writes", + command_body( + op, + uuid(op.as_u128() + 5), + "TARGET_WRITE_FENCED", + rev, + json!({}), + ), + ) + .await?; + assert_eq!(status, StatusCode::OK, "{body}"); + let (status, body) = user.legacy("tgt", &["create table t (x)"]).await?; + assert_eq!(status, StatusCode::OK, "write through the replica: {body}"); + Ok(()) + }); + sim.run().unwrap(); +} From d1ae89881013b9ae285d09f49fa675067bd72d04 Mon Sep 17 00:00:00 2001 From: River Date: Wed, 30 Sep 2026 10:52:46 +0000 Subject: [PATCH 25/33] libsql-server: surface metastore restore provenance and live log id Keep what the metastore's bottomless restore reports at startup (whether it recovered the database and the generation it restored from) instead of discarding it, record it on the MetaStore, and report it in the fence capability endpoint (`metastore`), in every fence view (`provenance`), in the `libsql_server_metastore_restored_from_backup` gauge and in a startup warning. When destroy_on_error rebuilds the metastore, the restore of the rebuilt metastore is what is reported. Tests cover a real bottomless restore against a local S3 endpoint, the admin rendering, and that `incarnation.current_log_id` names the rebuilt replication log after a dirty restart while the stored record keeps the log the source was acquired on. Co-authored-by: Tomasz Szymczyszyn --- docs/NAMESPACE_FENCE.md | 22 +- libsql-server/src/http/admin/fence.rs | 101 +++++++++- libsql-server/src/lib.rs | 12 +- libsql-server/src/metrics.rs | 8 + libsql-server/src/namespace/fence/tests.rs | 40 ++++ libsql-server/src/namespace/meta_store.rs | 222 ++++++++++++++++++++- libsql-server/tests/fence/admin.rs | 33 ++- 7 files changed, 411 insertions(+), 27 deletions(-) diff --git a/docs/NAMESPACE_FENCE.md b/docs/NAMESPACE_FENCE.md index 195e5b0745..be1e6bea13 100644 --- a/docs/NAMESPACE_FENCE.md +++ b/docs/NAMESPACE_FENCE.md @@ -169,7 +169,11 @@ Success (`APPLIED`, `ALREADY_APPLIED`) is `200`; `DRAINING` is `202`. Every resp "created_at": "2026-01-01T00:00:00Z", "last_transition_at": "2026-01-01T00:00:05Z", "server": { "build": " ()", "instance_id": "…uuid…" }, - "provenance": { "metastore_restored_from_backup": false, "marker": "consistent" } + "provenance": { + "metastore_restored_from_backup": false, + "metastore_restored_generation": null, + "marker": "consistent" + } }, "receipt": { "operation_id": "5f0c8a1e-...", @@ -187,6 +191,8 @@ Success (`APPLIED`, `ALREADY_APPLIED`) is `200`; `DRAINING` is `202`. Every resp `frozen_boundary.frame_no` is the last frame committed to the source's replication log, or `null` when the log has no frames. +`provenance.metastore_restored_from_backup` is `true` when this server restored its metastore from the metastore's bottomless backup at startup, with the backup generation in `metastore_restored_generation`; a record reported then may be older than one the server acknowledged before the restore, unless the marker comparison (section 13.3) caught it. `provenance.marker` is `consistent` when the record and its marker agree, `null` when there is no record, and the detail of section 13.3 for `UNKNOWN_UNAVAILABLE`. + Errors use the same shape with `"outcome": ""`, plus `"error": ""` and, where useful, `"detail"` (a bounded reason string such as `role_mismatch` or `namespace_identity_mismatch`). ### 4.4 Routes @@ -264,10 +270,10 @@ The routes are in `libsql-server/src/http/admin/fence.rs`, behind the admin list - **Availability.** `GET /v1/fence/capabilities` is always served. Command routes and `validation-query` answer `404` unless `--enable-namespace-fence` is on. `InspectFence` is also served while the metastore holds fence tables with the flag off (fences are enforced either way, section 13.1), and answers `404` on a server that never used fences. - **Order of checks.** Admin authentication (middleware, `401`), then the `404` above, then `not_primary` on a replica-kind server, then `admin_auth_required` for command routes and `validation-query` when no admin auth key is configured (`InspectFence` is read-only and does not need one), then the request body. Every command runs through `NamespaceStore::execute_fence_command` (which routes `CreateTargetQuarantined` to the atomic creation and adds the server's validation snapshot to `RecordTargetValidation`), so replay handling precedes every other check as section 5.3 requires. - **Request bodies** are strict: an unparsable body, a malformed id, an unknown `expected_state`, a missing required field or an unknown field is `FENCE_PRECONDITION_FAILED` with detail `invalid_argument`. `drain_policy.on_deadline` defaults to `fail`. `create-quarantined` refuses `dump_url`, `restore`, `restore_option`, `timestamp` and `from_backup` with `restore_not_allowed`, and `shared_schema` / `shared_schema_name` with `shared_schema_unsupported`; `max_db_size` takes the same byte-size values as `/v1/namespaces/:namespace/create`, and a `jwt_key` is parsed before anything is written. -- **Fence view.** Beyond the fields of section 4.3 it reports `incarnation.current_log_id` (the replication log id of the namespace as loaded now, `null` when it is not loaded; a caller can take `expected_namespace_identity.log_id` from it), `admission.indeterminate`, `drain_started_at`, the owning operation's latest `validation` (with the server snapshot), `last_command_id`, `written_by`, `adoptions`, and for `UNKNOWN_UNAVAILABLE` the `detail`, `reason` and the record the marker holds (`marker_record`). A namespace being created as a target reports `TARGET_QUARANTINED` with the durable revision it has so far. `provenance.metastore_restored_from_backup` is `false` until restore provenance is tracked. Error responses carry the live view from the namespace's controller when it has one, otherwise what the metastore holds, or `null` when the namespace name itself is invalid. +- **Fence view.** Beyond the fields of section 4.3 it reports `incarnation.current_log_id` (the replication log id of the namespace as loaded now, `null` when it is not loaded; a caller can take `expected_namespace_identity.log_id` from it), `admission.indeterminate`, `drain_started_at`, the owning operation's latest `validation` (with the server snapshot), `last_command_id`, `written_by`, `adoptions`, and for `UNKNOWN_UNAVAILABLE` the `detail`, `reason` and the record the marker holds (`marker_record`). A namespace being created as a target reports `TARGET_QUARANTINED` with the durable revision it has so far. `provenance` reports the metastore's restore provenance as recorded at startup (`MetaStore::restore_provenance`, section 13.3), the same for every namespace. Error responses carry the live view from the namespace's controller when it has one, otherwise what the metastore holds, or `null` when the namespace name itself is invalid. - **`InspectFence`** reads the metastore and the live controller; it neither loads the namespace nor creates a controller, so `drain` counters are zero for a namespace that is not loaded. Without `?receipts=all` only the owning operation's receipts are listed; `?receipts` with any other value is `invalid_argument`. - **`validation-query`** opens a `ValidationSession` (section 10.3) for the request and closes it afterwards. `stmts` follow the `/v1/execute` statement shape (`sql`, positional `args` or `named_args`, `want_rows`; `sql_id` is not supported). The response is `{"results": [...], "fence": ..., "drain": ...}`, one result per statement with `cols` and `rows` in the Hrana value encoding. A request whose statements return more than 10 000 rows in total is refused with `invalid_argument` rather than truncated, so a validation never looks at part of a result. A statement that tries to write is `OPERATION_CAPABILITY_REQUIRED` (`403`); other SQLite errors are `invalid_argument`. `expected_state`, when given, must match the live state (`FENCE_REVISION_MISMATCH` otherwise). -- **Capability discovery** lists in `commands` the commands this server serves (`AdoptFence` is added with its route, section 12), every state in `states`, and counts `active_fences` from the fence registry, which is seeded from the metastore at startup: records in any state but `RELEASED` or `TARGET_WRITABLE`, unavailable names, targets being created and indeterminate commits. `proxy_stable_code` reports whether the proxy's `stable_code` (section 6.1) is supported. `metastore` reports `restored_from_backup: false` until restore provenance is tracked. +- **Capability discovery** lists in `commands` the commands this server serves (`AdoptFence` is added with its route, section 12), every state in `states`, and counts `active_fences` from the fence registry, which is seeded from the metastore at startup: records in any state but `RELEASED` or `TARGET_WRITABLE`, unavailable names, targets being created and indeterminate commits. `proxy_stable_code` reports whether the proxy's `stable_code` (section 6.1) is supported. `metastore` reports whether the metastore was restored from its backup at startup and the generation it was restored from (`restored_generation`, `null` when it was not restored). ## 5. Durable state (Design, with contract points marked) @@ -548,7 +554,7 @@ This makes the WAL gate independent of statement classification: DDL, misclassif - **A crash rebuilds the source's replication log.** A namespace that was not shut down cleanly is recovered by rebuilding its replication log from the database file under a new `log_id` (existing behaviour, not specific to fences). For the fence this means: - Crash before `SOURCE_DRAINING` committed: nothing was written or acknowledged. A replay of the same acquisition is refused with `FENCE_PRECONDITION_FAILED`/`namespace_identity_mismatch`, before anything is written, because the log id the caller observed is gone; the caller reads the new identity and acquires with a new command. - Crash in `SOURCE_DRAINING`: the replay completes the drain at once, and the frozen boundary names the rebuilt log (its `log_id` and last frame). The record's `identity.log_id` keeps the log the caller acquired against; the two differ exactly when the log was rebuilt during the drain. Write admission was durably closed from the `SOURCE_DRAINING` commit on and the lifecycle paths that could replace the database are denied, so the data at the boundary is what was committed before the cutoff. The server logs a warning when it records such a boundary. - - Crash after `SOURCE_WRITE_FENCED` committed: the stored boundary names the log that was live when the drain was proven. After the restart the live log has a new id but the same data (writes stayed closed). A caller that compares the boundary's `log_id` with the source's current log id sees the rebuild and copies from a snapshot of the unchanged database rather than from the old log's frames. + - Crash after `SOURCE_WRITE_FENCED` committed: the stored boundary names the log that was live when the drain was proven. After the restart the live log has a new id but the same data (writes stayed closed). A caller that compares the boundary's `log_id` with the source's current log id sees the rebuild and copies from a snapshot of the unchanged database rather than from the old log's frames. The current log id is `incarnation.current_log_id` in `InspectFence` and in every admin response (section 4.5), so no separate probe is needed. ## 9. Source read fence (Design) @@ -743,7 +749,7 @@ A namespace that startup cannot recover is registered `UNKNOWN_UNAVAILABLE` in m - `maybe_recover_from_fs`: a directory with a marker is not recovered from `config.json` (or a default); it is registered `UNKNOWN_UNAVAILABLE` (`metastore_behind_marker`). Directories without markers keep today's behaviour: they are legacy, unfenced namespaces. - `destroy_on_error`: the broken metastore is renamed to `metastore.broken-` rather than deleted (if the rename fails, startup fails instead), and directories with markers are registered `UNKNOWN_UNAVAILABLE` after the rebuild. Without fences it still deletes, as before. - A metastore without fence tables next to a directory that holds a marker (rebuilt, recovered, or restored from a backup taken before the tables existed) makes that namespace `UNKNOWN_UNAVAILABLE`, even when its config row is present. -- Metastore restore from backup: the marker comparison (section 5.6) makes any namespace whose record went backwards or disappeared `UNKNOWN_UNAVAILABLE`. A restored record is never trusted over a newer marker. Surfacing the provenance (`restored_from_backup`, backup generation) in the capability endpoint, `InspectFence`, a metric and a startup line is part of the admin API (section 4). +- Metastore restore from backup: the marker comparison (section 5.6) makes any namespace whose record went backwards or disappeared `UNKNOWN_UNAVAILABLE`. A restored record is never trusted over a newer marker. The restore itself is reported: `metastore_connection_maker_with_provenance` keeps what the bottomless restore returns (whether it recovered the database, and the generation it restored from, which is the replicator's generation before a new one is started), and the server records it on the metastore right after opening it (`MetaStore::record_restore_provenance`). It is surfaced in the capability endpoint (`metastore`), in every fence view (`provenance`), in the `libsql_server_metastore_restored_from_backup` gauge (0 or 1, set at every start) and, after a restore, in a startup warning naming the generation and the number of namespaces startup registered `UNKNOWN_UNAVAILABLE`. When `destroy_on_error` rebuilds the metastore, the rebuilt metastore is restored from the backup again, and that second restore is what is reported. - Replica-kind servers: lazy creation of a name the primary refuses with a fence code (a quarantined or aborted target, a read-fenced source, a fence state the primary cannot establish) does not create a local namespace. The refusal arrives as the typed status of the replication `hello` (section 6.2), not through the write proxy. `ReplicaConfigurator::setup` fails at once with that code (`Error::NamespaceFence`, so the client gets `423` and the code of section 6.0) instead of retrying the handshake, and removes the namespace directory if the setup created it (a directory that was already there is left as it is). `NamespaceStore::with` then forgets what the attempt left in memory: the metastore entry `handle()` added (`MetaStore::forget_unstored`, only when no config row is stored for the name and no other handle is subscribed to it, so a concurrent creation or a persisted config keeps it) and the name's controller (`FenceRegistry::forget_idle`, only when it holds no fence record and no in-memory gate and nothing else refers to it). `exists()` and `lookup()` therefore do not report the name, and a later request, once the primary admits the name, creates it normally. ### 13.4 Shared schema @@ -780,7 +786,7 @@ How each path that can reach namespace data or lifecycle is covered. File refere | `connection/legacy.rs` | `FenceConnState` wired into every `LegacyConnection`; the controller is passed to `MakeLegacyConnection::new` before the first connection. `with_raw` users are covered by the WAL gate. | | HTTP `/`, `/v1/execute`, `/v1/batch`, Hrana `/v2`, `/v3`, cursors, WebSocket, dev route | Core checks and WAL gate; typed status and `code` field; Hrana codes, for step and whole-request denials alike, with the stream left usable (section 6.0). | | `rpc/proxy.rs`, `rpc/streaming_exec.rs`, `rpc/replica_proxy.rs`, `connection/write_proxy.rs` | `stable_code` set on the primary for step and program denials (streamed and unary); namespace-lookup, JWT-key-lookup, connection-creation and unary whole-program denials are `FAILED_PRECONDITION` with the code, never `UNAVAILABLE`; the replica maps both back to `Error::NamespaceFence` and never retries them; the replica proxy forwards them unchanged (section 6.1). | -| `namespace/meta_store.rs` `handle()`, `restore()`, `maybe_recover_from_fs`, `destroy_on_error`, `process`/`try_process`, `remove`, bottomless metastore restore | Non-creating lookups; fail-closed decoding; marker-aware recovery; rename-aside; publish only after commit; fence check inside the config and remove transactions; restore provenance surfaced. | +| `namespace/meta_store.rs` `handle()`, `restore()`, `maybe_recover_from_fs`, `destroy_on_error`, `process`/`try_process`, `remove`, bottomless metastore restore | Non-creating lookups; fail-closed decoding; marker-aware recovery; rename-aside; publish only after commit; fence check inside the config and remove transactions; the bottomless restore's outcome is kept (`metastore_connection_maker_with_provenance`) and reported in the admin API, a gauge and a startup warning (section 13.3). | | `namespace/store.rs` `with`, `load_namespace`, `make_namespace`, eviction | Registry check before setup; controller passed into setup; eviction keeps the registry entry. | | `store.rs` `create`, `destroy`, `reset`, `fork`, `checkpoint`, restore options | `check_lifecycle` (section 13.5): create refuses a name whose fence denies lifecycle work; `CreateTargetQuarantined` is the atomic quarantined create; destroy is refused in the metastore transaction; reset and fork as source are checked under the namespace's transition lock; fork's destination is checked before anything is stored and again under its entry lock; every restore (create with a dump, reset, fork to a point in time) is one of these paths and is refused there; checkpoint uses a non-creating lookup and skips vacuum. | | `http/admin/mod.rs` config, create, fork, delete, checkpoint, stats | Config POST and create (before a `dump_url` is fetched) are checked before the namespace is loaded; fork and delete are refused by the store; all of them return the fence error with `423`. Config GET, stats and checkpoint are allowed. Fence routes live in `http/admin/fence.rs`. | @@ -834,9 +840,9 @@ Planned test names; the table is updated as tests land. | 4 | Program that captured config before the fence is rejected at the WAL | landed: `connection::connection_manager::fence_tests::wal_gate_rejects_program_admitted_before_fence` (a SQL function parks the program between admission and its write while the fence is acquired and released) | | 5 | Pre-fence transactions cannot write after release or publication | landed: `connection::connection_manager::fence_tests::stale_generation_cannot_write_after_release`, `namespace::fence::target::tests::stale_generation_cannot_write_after_enable_writes`; unfenced behaviour unchanged: `unfenced_namespace_is_unchanged` | | 6 | Acquisition timeout returns `DRAINING`, admission stays closed | landed: `namespace::fence::drain::tests::{deadline_returns_draining_and_stays_closed, replay_of_draining_resumes_and_completes}` | -| 7 | Restart at every persistence boundary; indeterminate persistence keeps the gate closed until same-command reconciliation | landed: `namespace::fence::tests::restart_at_each_boundary` (a crash at each of 17 boundaries of `AcquireSourceWriteFence` and `ReleaseSourceWriteFence`: after the `INSTALLING` gate, before, at and after each metastore commit, with and without a lagging marker, before publication, before boundary capture and before the response; the restart recovers the prior or the committed state, admits writes only if an opening transition committed, repairs the marker, and a replay finishes the command), `restart_in_draining_waits_for_the_same_command` (a writer active at the crash; nothing advances until the same command is replayed, which completes at once), `indeterminate_commit_keeps_gate_closed` (through the drain path, both when the commit happened and when it did not), `acquire_response_loss_resolved_by_replay_and_inspect`; `fence::controller::tests::{indeterminate_commit_keeps_writes_closed_until_replayed, indeterminate_commit_that_did_not_apply_is_retried_by_replay, failed_before_commit_leaves_gate_unchanged, publication_happens_before_the_response, committed_command_is_published_when_the_caller_goes_away}`, `namespace::store::fence_tests::restart_installs_the_durable_gate_before_serving` | +| 7 | Restart at every persistence boundary; indeterminate persistence keeps the gate closed until same-command reconciliation | landed: `namespace::fence::tests::restart_at_each_boundary` (a crash at each of 17 boundaries of `AcquireSourceWriteFence` and `ReleaseSourceWriteFence`: after the `INSTALLING` gate, before, at and after each metastore commit, with and without a lagging marker, before publication, before boundary capture and before the response; the restart recovers the prior or the committed state, admits writes only if an opening transition committed, repairs the marker, and a replay finishes the command), `restart_in_draining_waits_for_the_same_command` (a writer active at the crash; nothing advances until the same command is replayed, which completes at once), `indeterminate_commit_keeps_gate_closed` (through the drain path, both when the commit happened and when it did not), `acquire_response_loss_resolved_by_replay_and_inspect`; `fence::controller::tests::{indeterminate_commit_keeps_writes_closed_until_replayed, indeterminate_commit_that_did_not_apply_is_retried_by_replay, failed_before_commit_leaves_gate_unchanged, publication_happens_before_the_response, committed_command_is_published_when_the_caller_goes_away}`, `namespace::store::fence_tests::restart_installs_the_durable_gate_before_serving`; the rebuilt log after a restart: `namespace::fence::tests::current_log_id_is_the_rebuilt_log_after_restart` (the stored record keeps the acquisition log and its boundary; the controller's `current_log_id`, which the admin view reports, names the rebuilt log) | | 8 | Evict and lazily reload a fenced namespace; identical admission | `tests::fence::lifecycle::evicted_namespace_reloads_same_gate`; landed at unit level: `namespace::store::fence_tests::evicted_namespace_reloads_with_the_same_controller`, `fence::registry::tests::seeded_from_load_fences_including_recovered_names` | -| 9 | Filesystem recovery, `destroy_on_error`, undecodable records, missing target quarantine, metastore backup rollback fail closed with provenance | `meta_store::fence_tests::recovery::{fs_recovery_with_marker_unavailable, destroy_on_error_keeps_fenced_unavailable, undecodable_row_unavailable, incomplete_target_unavailable, metastore_rollback_detected_by_marker, lookup_never_creates, undecodable_name_with_fence_fails_startup, marker_in_invalid_directory_fails_startup}`; legacy behaviour kept: `destroy_on_error_without_fences_is_unchanged`, `undecodable_row_without_fences_is_skipped_as_before`; `meta_store::fence_tests::corrupt_fence_row_fails_closed` | +| 9 | Filesystem recovery, `destroy_on_error`, undecodable records, missing target quarantine, metastore backup rollback fail closed with provenance | `meta_store::fence_tests::recovery::{fs_recovery_with_marker_unavailable, destroy_on_error_keeps_fenced_unavailable, undecodable_row_unavailable, incomplete_target_unavailable, metastore_rollback_detected_by_marker, lookup_never_creates, undecodable_name_with_fence_fails_startup, marker_in_invalid_directory_fails_startup}`; legacy behaviour kept: `destroy_on_error_without_fences_is_unchanged`, `undecodable_row_without_fences_is_skipped_as_before`; `meta_store::fence_tests::corrupt_fence_row_fails_closed`; restore provenance: `namespace::meta_store::fence_tests::provenance::{bottomless_restore_is_reported (a real bottomless restore against a local S3 endpoint: a metastore opened on an empty directory from a backup reports the restore and its generation and holds the backed-up namespace; one opened with nothing to restore does not), generation_only_after_a_recovery, not_restored_until_recorded_and_first_record_wins}`, `http::admin::fence::tests::restore_provenance_is_reported` (fence view and capability `metastore` object), `tests::fence::admin::capabilities` (not restored: capability endpoint, fence views and the gauge report it) | | 10 | Wrong owner, stale revision, invalid role/state, replay, command-id reuse; replay before revision check | `fence::transition::tests::*` (exhaustive over states × commands) | | 11 | Target creation raced with SQL, dump, replication, lifecycle never observable as writable or readable | landed: `namespace::fence::target::tests::create_race_never_observable` (parked after the rows commit and before the config is published and the namespace loaded: SQL connections, stats, replication `hello` (never `UNAVAILABLE`), create, delete and fork of the name are denied or find nothing; afterwards the target is loaded behind the quarantine gate, SQL reads and WAL writes are refused, and lifecycle and replication are refused with `MIGRATION_TARGET_QUARANTINED`), `creating_gate_refuses_before_commit` (parked before the metastore transaction: the same attempts are refused and no database file is created), `create_replay_completes_interrupted_creation` (marker only, after a restart), `create_completes_when_the_caller_goes_away`, `indeterminate_create_is_completed_by_replay`, `create_rejects_existing_name` (a loaded or cold existing name, whose gate never moves, and another operation's target), `abort_keeps_traffic_denied` (also across a restart) | | 12 | Only the matching import capability writes a quarantined target; admin credentials and admin shell cannot | landed: `namespace::fence::import::tests::import_requires_matching_capability` (plain connections with and without raw access, as the admin shell uses; issuing to another operation or at another revision; at the WAL, capabilities the server never issued, of another operation, at another revision, for validation, and revoked); `namespace::fence::target::tests::{create_race_never_observable, abort_keeps_traffic_denied}` (raw DDL refused); planned: `tests::fence::admin::admin_shell_cannot_write_quarantined` | diff --git a/libsql-server/src/http/admin/fence.rs b/libsql-server/src/http/admin/fence.rs index ca0955706a..b47b906390 100644 --- a/libsql-server/src/http/admin/fence.rs +++ b/libsql-server/src/http/admin/fence.rs @@ -32,7 +32,7 @@ use crate::namespace::fence::record::{CommandReceipt, NamespaceFenceRecord, Serv use crate::namespace::fence::state::FenceState; use crate::namespace::fence::store::{StoredFence, StoredReceipt}; use crate::namespace::fence::{server_identity, FENCE_PROTOCOL_VERSION, PROXY_STABLE_CODE}; -use crate::namespace::meta_store::FenceCommit; +use crate::namespace::meta_store::{FenceCommit, MetastoreProvenance}; use crate::namespace::NamespaceName; use crate::net::Connector; @@ -132,8 +132,7 @@ async fn handle_capabilities(State(state): State>>) -> Json( let body = json!({ "outcome": FenceOutcome::Applied.as_str(), "replayed": false, - "fence": fence_json(&namespace, &inspection.fence, controller.as_deref()), + "fence": fence_json( + &namespace, + &inspection.fence, + controller.as_deref(), + &meta.restore_provenance(), + ), "receipts": receipts, "drain": drain_json(controller.as_deref()), }); @@ -307,9 +311,10 @@ async fn handle_validation_query( drop(session); let controller = state.namespaces.existing_fence_controller(&namespace); + let provenance = state.namespaces.meta_store().restore_provenance(); let fence = controller .as_ref() - .map(|c| fence_json(&namespace, &c.gate().fence, Some(c))) + .map(|c| fence_json(&namespace, &c.gate().fence, Some(c), &provenance)) .unwrap_or(Value::Null); let body = json!({ "results": results, @@ -396,14 +401,20 @@ async fn error_reply( error: FenceError, ) -> Response { let mut reply = ErrorReply::new(error); + let provenance = state.namespaces.meta_store().restore_provenance(); match state.namespaces.existing_fence_controller(namespace) { Some(controller) => { - reply.fence = fence_json(namespace, &controller.gate().fence, Some(&controller)); + reply.fence = fence_json( + namespace, + &controller.gate().fence, + Some(&controller), + &provenance, + ); reply.drain = drain_json(Some(&controller)); } None => { if let Ok((inspection, _)) = state.namespaces.inspect_fence(namespace).await { - reply.fence = fence_json(namespace, &inspection.fence, None); + reply.fence = fence_json(namespace, &inspection.fence, None, &provenance); } } } @@ -424,13 +435,15 @@ fn success_reply( commit: FenceCommit, ) -> Response { let controller = state.namespaces.existing_fence_controller(namespace); + let provenance = state.namespaces.meta_store().restore_provenance(); let fence = match (&commit.record, &controller) { (Some(record), _) => fence_json( namespace, &StoredFence::Record(record.clone()), controller.as_deref(), + &provenance, ), - (None, Some(c)) => fence_json(namespace, &c.gate().fence, Some(c)), + (None, Some(c)) => fence_json(namespace, &c.gate().fence, Some(c), &provenance), (None, None) => Value::Null, }; let outcome = commit.receipt.outcome; @@ -916,11 +929,13 @@ fn record_fields(record: &NamespaceFenceRecord, out: &mut Map) { } /// The fence view of section 4.3. `fence` is the durable state being reported; `controller`, -/// when the namespace has one, supplies the live admission and the live log id. +/// when the namespace has one, supplies the live admission and the live log id; `provenance` +/// says whether the metastore that holds `fence` was restored from its backup at startup. fn fence_json( namespace: &NamespaceName, fence: &StoredFence, controller: Option<&FenceController>, + provenance: &MetastoreProvenance, ) -> Value { let gate = controller.map(|c| c.gate()); let mut out = Map::new(); @@ -990,11 +1005,23 @@ fn fence_json( out.insert("server".into(), server_json(&server_identity())); out.insert( "provenance".into(), - json!({ "metastore_restored_from_backup": false, "marker": marker }), + json!({ + "metastore_restored_from_backup": provenance.restored_from_backup, + "metastore_restored_generation": provenance.restored_generation.map(|g| g.to_string()), + "marker": marker, + }), ); Value::Object(out) } +/// The `metastore` object of the capability endpoint (section 4.4). +fn metastore_provenance_json(provenance: &MetastoreProvenance) -> Value { + json!({ + "restored_from_backup": provenance.restored_from_backup, + "restored_generation": provenance.restored_generation.map(|g| g.to_string()), + }) +} + fn receipt_json(receipt: &CommandReceipt) -> Value { json!({ "operation_id": receipt.operation_id.to_string(), @@ -1022,3 +1049,57 @@ fn stored_receipt_json(stored: &StoredReceipt) -> Value { }), } } + +#[cfg(test)] +mod tests { + use super::*; + + /// A metastore restored from its backup is reported in the fence view (section 4.3) and + /// in the capability endpoint's `metastore` object (section 4.4), with the generation. + #[test] + fn restore_provenance_is_reported() { + let generation = Uuid::from_u128(0x77); + let restored = MetastoreProvenance { + restored_from_backup: true, + restored_generation: Some(generation), + }; + let namespace = NamespaceName::from_string("db".into()).unwrap(); + let unavailable = StoredFence::Unavailable { + detail: FenceDetail::MetastoreBehindMarker, + reason: "the metastore is behind the marker".into(), + marker: None, + }; + + let view = fence_json(&namespace, &unavailable, None, &restored); + assert_eq!( + view["provenance"], + json!({ + "metastore_restored_from_backup": true, + "metastore_restored_generation": generation.to_string(), + "marker": "metastore_behind_marker", + }) + ); + assert_eq!(view["state"], "UNKNOWN_UNAVAILABLE"); + assert_eq!( + metastore_provenance_json(&restored), + json!({ "restored_from_backup": true, "restored_generation": generation.to_string() }) + ); + + let plain = StoredFence::None { + namespace_exists: true, + }; + let view = fence_json(&namespace, &plain, None, &MetastoreProvenance::default()); + assert_eq!( + view["provenance"], + json!({ + "metastore_restored_from_backup": false, + "metastore_restored_generation": null, + "marker": null, + }) + ); + assert_eq!( + metastore_provenance_json(&MetastoreProvenance::default()), + json!({ "restored_from_backup": false, "restored_generation": null }) + ); + } +} diff --git a/libsql-server/src/lib.rs b/libsql-server/src/lib.rs index 66fbbcf762..a3480eae32 100644 --- a/libsql-server/src/lib.rs +++ b/libsql-server/src/lib.rs @@ -11,7 +11,7 @@ use crate::connection::{Connection, MakeConnection}; use crate::database::DatabaseKind; use crate::error::Error; use crate::migration::maybe_migrate; -use crate::namespace::meta_store::{metastore_connection_maker, MetaStore}; +use crate::namespace::meta_store::{metastore_connection_maker_with_provenance, MetaStore}; use crate::net::Accept; use crate::pager::{make_pager, PAGER_CACHE_SIZE}; use crate::rpc::proxy::rpc::proxy_server::Proxy; @@ -611,9 +611,12 @@ where connection_creation_timeout: self.db_config.connection_creation_timeout, }; - let (metastore_conn_maker, meta_store_wal_manager) = - metastore_connection_maker(self.meta_store_config.bottomless.clone(), &self.path) - .await?; + let (metastore_conn_maker, meta_store_wal_manager, metastore_provenance) = + metastore_connection_maker_with_provenance( + self.meta_store_config.bottomless.clone(), + &self.path, + ) + .await?; let meta_conn = metastore_conn_maker()?; let meta_store = MetaStore::new( self.meta_store_config.clone(), @@ -623,6 +626,7 @@ where db_kind, ) .await?; + meta_store.record_restore_provenance(metastore_provenance); let (configurators, make_replication_svc) = self .make_configurators_and_replication_svc( diff --git a/libsql-server/src/metrics.rs b/libsql-server/src/metrics.rs index e5de216b79..69cd49bb32 100644 --- a/libsql-server/src/metrics.rs +++ b/libsql-server/src/metrics.rs @@ -32,6 +32,14 @@ pub static TOTAL_RESPONSE_SIZE_HIST: Lazy = Lazy::new(|| { describe_histogram!(NAME, "total response size value before connection lock"); register_histogram!(NAME) }); +pub static METASTORE_RESTORED_FROM_BACKUP: Lazy = Lazy::new(|| { + const NAME: &str = "libsql_server_metastore_restored_from_backup"; + describe_gauge!( + NAME, + "1 when the metastore was restored from its backup at startup, 0 otherwise" + ); + register_gauge!(NAME) +}); pub static STREAM_HANDLES_COUNT: Lazy = Lazy::new(|| { const NAME: &str = "libsql_server_stream_handles"; describe_gauge!(NAME, "amount of in-memory stream handles"); diff --git a/libsql-server/src/namespace/fence/tests.rs b/libsql-server/src/namespace/fence/tests.rs index bb41fbeecd..2ea08f8ec3 100644 --- a/libsql-server/src/namespace/fence/tests.rs +++ b/libsql-server/src/namespace/fence/tests.rs @@ -631,6 +631,46 @@ fn restart_at(case: &Boundary) { server.crash(); } +/// A dirty restart rebuilds the replication log under a new id. The stored record keeps the +/// log the source was acquired on and the boundary frozen in it; the controller's +/// `current_log_id`, which `InspectFence` and every admin response report as +/// `incarnation.current_log_id` (section 4.5), names the rebuilt log, so a caller can see the +/// rebuild without a separate probe. +#[test] +fn current_log_id_is_the_rebuilt_log_after_restart() { + let dir = tempdir().unwrap(); + + let server = Server::boot(dir.path()); + server.create_source(); + let (acquired_on, _) = server.run(server.log()); + let commit = server.fence_source(); + assert_eq!( + commit.record.as_ref().unwrap().state, + FenceState::SourceWriteFenced + ); + server.run(async { + assert_eq!(server.fence().await.current_log_id(), Some(acquired_on)); + }); + server.crash(); + + let server = Server::boot(dir.path()); + server.run(async { + let fence = server.fence().await; + let (live, _) = server.log().await; + assert_ne!(live, acquired_on, "a dirty restart rebuilds the log"); + assert_eq!(fence.current_log_id(), Some(live)); + + let inspected = server.inspect().await; + let record = inspected.fence.record().unwrap(); + assert_eq!(record.state, FenceState::SourceWriteFenced); + assert_eq!(record.identity.log_id, Some(acquired_on)); + assert_eq!(boundary(&commit).log_id, acquired_on); + // The data is the same; only the log was rebuilt. + assert_eq!(server.count().await, ROWS); + }); + server.crash(); +} + /// After a restart in `SOURCE_DRAINING` with a writer that was active at the crash, nothing /// advances by itself: the namespace stays closed, other commands cannot move it on, and only /// the replay of the same acquisition completes the drain, at once. diff --git a/libsql-server/src/namespace/meta_store.rs b/libsql-server/src/namespace/meta_store.rs index 49fd1d7404..a5de15a4bf 100644 --- a/libsql-server/src/namespace/meta_store.rs +++ b/libsql-server/src/namespace/meta_store.rs @@ -97,6 +97,33 @@ struct MetaStoreInner { /// record for. They are refused by lookups and by every config or lifecycle change, and /// never default-created. A fence command that commits for the name takes it out. recovered: Mutex>, + /// Where this metastore's contents came from at startup, recorded once the server has + /// opened it (section 13.3). + restore_provenance: std::sync::OnceLock, +} + +/// Whether the metastore was restored from its bottomless backup when the server started +/// (`docs/NAMESPACE_FENCE.md` sections 4.3, 4.4 and 13.3). A restored metastore can hold fence +/// records older than the namespace markers; the marker comparison makes those namespaces +/// `UNKNOWN_UNAVAILABLE`, and this is what the admin API, the startup log and the +/// `libsql_server_metastore_restored_from_backup` gauge report about the restore itself. +#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)] +pub struct MetastoreProvenance { + /// The metastore database was restored from its backup at startup. + pub restored_from_backup: bool, + /// The backup generation it was restored from, when known. + pub restored_generation: Option, +} + +impl MetastoreProvenance { + /// The provenance of a bottomless restore that reported `did_recover`, where `generation` + /// is the replicator's generation right after the restore: the generation restored from. + pub fn from_restore(did_recover: bool, generation: Option) -> Self { + Self { + restored_from_backup: did_recover, + restored_generation: generation.filter(|_| did_recover), + } + } } /// How this metastore treats namespace fences (`docs/NAMESPACE_FENCE.md` section 13.1). @@ -140,15 +167,33 @@ fn setup_connection(conn: &rusqlite::Connection) -> Result<()> { Ok(()) } +/// [`metastore_connection_maker_with_provenance`] without the provenance, for tests. +#[cfg(test)] pub async fn metastore_connection_maker( config: Option, base_path: &Path, ) -> crate::Result<( impl Fn() -> crate::Result, MetaStoreWalManager, +)> { + let (maker, wal_manager, _) = + metastore_connection_maker_with_provenance(config, base_path).await?; + Ok((maker, wal_manager)) +} + +/// [`metastore_connection_maker`], also returning whether the bottomless restore recovered the +/// metastore from its backup, for [`MetaStore::record_restore_provenance`]. +pub async fn metastore_connection_maker_with_provenance( + config: Option, + base_path: &Path, +) -> crate::Result<( + impl Fn() -> crate::Result, + MetaStoreWalManager, + MetastoreProvenance, )> { let db_path = base_path.join("metastore"); tokio::fs::create_dir_all(&db_path).await?; + let mut provenance = MetastoreProvenance::default(); let replicator = match config { Some(config) => { let options = bottomless::replicator::Options { @@ -175,7 +220,11 @@ pub async fn metastore_connection_maker( options, ) .await?; - let (action, _did_recover) = replicator.restore(None, None).await?; + let (action, did_recover) = replicator.restore(None, None).await?; + // A restore that recovered the database leaves the replicator on the generation it + // restored from; a new generation, if any, is only started below. + provenance = + MetastoreProvenance::from_restore(did_recover, replicator.generation().ok()); // TODO: this logic should probably be moved to bottomless. match action { bottomless::replicator::RestoreAction::SnapshotMainDbFile => { @@ -213,7 +262,7 @@ pub async fn metastore_connection_maker( } }; - Ok((maker, wal_manager)) + Ok((maker, wal_manager, provenance)) } impl MetaStoreInner { @@ -254,6 +303,7 @@ impl MetaStoreInner { dbs_path, fence, recovered: Default::default(), + restore_provenance: Default::default(), }; if config.allow_recover_from_fs { @@ -1137,8 +1187,13 @@ impl MetaStore { ); } - let (maker, wal) = - metastore_connection_maker(config.bottomless.clone(), base_path).await?; + // The rebuilt metastore is restored from the backup again, so what the + // server reports is this restore, not the one of the broken metastore. + let (maker, wal, provenance) = metastore_connection_maker_with_provenance( + config.bottomless.clone(), + base_path, + ) + .await?; let conn = maker()?; @@ -1150,6 +1205,7 @@ impl MetaStore { }) .await .unwrap()?; + let _ = inner.restore_provenance.set(provenance); tracing::info!("metastore destroy on error successful"); @@ -1344,6 +1400,41 @@ impl MetaStore { self.inner.fence.enabled } + /// Records where this metastore's contents came from at startup, as reported by + /// [`metastore_connection_maker_with_provenance`], then sets the + /// `libsql_server_metastore_restored_from_backup` gauge and, after a restore from backup, + /// logs a startup warning. What was recorded first wins: when `destroy_on_error` rebuilt + /// the metastore while opening it, the restore of the rebuilt metastore is already recorded + /// and is what is reported. The server calls this once, right after opening the metastore. + pub fn record_restore_provenance(&self, provenance: MetastoreProvenance) { + let provenance = *self.inner.restore_provenance.get_or_init(|| provenance); + crate::metrics::METASTORE_RESTORED_FROM_BACKUP.set(if provenance.restored_from_backup { + 1.0 + } else { + 0.0 + }); + if provenance.restored_from_backup { + tracing::warn!( + restored_generation = ?provenance.restored_generation, + fence_tables = self.inner.fence.tables, + unavailable_namespaces = self.inner.recovered.lock().len(), + "the metastore was restored from its backup at startup; fence records the backup \ + does not hold are detected through the namespace markers and reported as \ + UNKNOWN_UNAVAILABLE" + ); + } + } + + /// Where this metastore's contents came from at startup; the default (not restored) until + /// [`record_restore_provenance`](Self::record_restore_provenance) is called. + pub fn restore_provenance(&self) -> MetastoreProvenance { + self.inner + .restore_provenance + .get() + .copied() + .unwrap_or_default() + } + /// The drain policy of an `AcquireSourceWriteFence` that names none: the configured /// deadline, then `DRAINING`. pub fn fence_default_write_drain(&self) -> DrainPolicy { @@ -2951,4 +3042,127 @@ mod fence_tests { assert_eq!(e.detail(), Some(FenceDetail::MetastoreBehindMarker)); } } + + /// Metastore restore provenance (`docs/NAMESPACE_FENCE.md` sections 4.3, 4.4 and 13.3). + mod provenance { + use super::*; + + #[test] + fn generation_only_after_a_recovery() { + let generation = Uuid::from_u128(0x77); + assert_eq!( + MetastoreProvenance::from_restore(true, Some(generation)), + MetastoreProvenance { + restored_from_backup: true, + restored_generation: Some(generation), + } + ); + // A restore that found the local database up to date (or nothing to restore) did + // not recover anything, whatever generation the replicator is on. + assert_eq!( + MetastoreProvenance::from_restore(false, Some(generation)), + MetastoreProvenance::default() + ); + assert_eq!( + MetastoreProvenance::from_restore(true, None), + MetastoreProvenance { + restored_from_backup: true, + restored_generation: None, + } + ); + } + + #[tokio::test] + async fn not_restored_until_recorded_and_first_record_wins() { + let dir = tempdir().unwrap(); + let store = open(dir.path(), true).await; + assert_eq!(store.restore_provenance(), MetastoreProvenance::default()); + + let restored = MetastoreProvenance { + restored_from_backup: true, + restored_generation: Some(Uuid::from_u128(0x77)), + }; + store.record_restore_provenance(restored); + assert_eq!(store.restore_provenance(), restored); + // Every clone of the store reports the same provenance. + assert_eq!(store.clone().restore_provenance(), restored); + + store.record_restore_provenance(MetastoreProvenance::default()); + assert_eq!(store.restore_provenance(), restored); + } + + /// An S3 endpoint backed by a temporary directory, on a port of its own. + async fn mock_s3() -> (tempfile::TempDir, String) { + use s3s::auth::SimpleAuth; + use s3s::service::S3ServiceBuilder; + + let root = tempdir().unwrap(); + let mut s3 = S3ServiceBuilder::new(s3s_fs::FileSystem::new(root.path()).unwrap()); + s3.set_auth(SimpleAuth::from_single("key", "secret")); + let service = s3.build().into_shared().into_make_service(); + let server = hyper::Server::bind(&([127, 0, 0, 1], 0).into()).serve(service); + let endpoint = format!("http://{}", server.local_addr()); + tokio::spawn(server); + (root, endpoint) + } + + fn bottomless(endpoint: String) -> BottomlessConfig { + BottomlessConfig { + access_key_id: "key".into(), + secret_access_key: "secret".into(), + session_token: None, + region: "us-east-1".into(), + backup_id: "metastore-provenance".into(), + bucket_name: "provenance".into(), + backup_interval: Duration::from_millis(100), + bucket_endpoint: endpoint, + } + } + + async fn open_bottomless( + dir: &Path, + config: &BottomlessConfig, + ) -> (MetaStore, MetastoreProvenance) { + let (maker, manager, provenance) = + metastore_connection_maker_with_provenance(Some(config.clone()), dir) + .await + .unwrap(); + let store = MetaStore::new( + MetaStoreConfig::default(), + dir, + maker().unwrap(), + manager, + DatabaseKind::Primary, + ) + .await + .unwrap(); + store.record_restore_provenance(provenance); + (store, provenance) + } + + /// A metastore opened on an empty directory from a backup that holds one reports the + /// restore and the generation it came from; one opened with nothing to restore does + /// not. + #[tokio::test(flavor = "multi_thread", worker_threads = 2)] + async fn bottomless_restore_is_reported() { + let (_s3, endpoint) = mock_s3().await; + let config = bottomless(endpoint); + + let first = tempdir().unwrap(); + let (store, provenance) = open_bottomless(first.path(), &config).await; + assert_eq!(provenance, MetastoreProvenance::default()); + assert_eq!(store.restore_provenance(), MetastoreProvenance::default()); + let _handle = create_namespace(&store, "db").await; + // Uploads everything the backup does not hold yet. + store.shutdown().await.unwrap(); + + let second = tempdir().unwrap(); + let (store, provenance) = open_bottomless(second.path(), &config).await; + assert!(provenance.restored_from_backup, "{provenance:?}"); + assert!(provenance.restored_generation.is_some(), "{provenance:?}"); + assert_eq!(store.restore_provenance(), provenance); + // The restored metastore holds what the first one backed up. + assert!(store.exists(&"db".into()).await); + } + } } diff --git a/libsql-server/tests/fence/admin.rs b/libsql-server/tests/fence/admin.rs index 5a5a2aec03..49669fa168 100644 --- a/libsql-server/tests/fence/admin.rs +++ b/libsql-server/tests/fence/admin.rs @@ -56,11 +56,34 @@ fn capabilities() { .unwrap() .starts_with("sqld ")); Uuid::parse_str(body["server"]["instance_id"].as_str().unwrap())?; - assert_eq!(body["metastore"]["restored_from_backup"], false); + // Metastore restore provenance: this server's metastore was not restored from a + // backup, which the capability endpoint, every fence view and the gauge all report. + assert_eq!(body["metastore"]["restored_from_backup"], false, "{body}"); + assert_eq!( + body["metastore"]["restored_generation"], + json!(null), + "{body}" + ); + assert_eq!( + crate::common::snapshot_metrics() + .get_gauge("libsql_server_metastore_restored_from_backup"), + Some(0.0) + ); // An active fence is counted. admin.create_namespace("src").await?; let log_id = load_and_log_id(&admin, "src").await?; + let (status, body) = admin.inspect("src").await?; + assert_eq!(status, StatusCode::OK, "{body}"); + assert_eq!( + body["fence"]["provenance"], + json!({ + "metastore_restored_from_backup": false, + "metastore_restored_generation": null, + "marker": null, + }), + "{body}" + ); let (status, body) = admin .command( "src", @@ -69,6 +92,14 @@ fn capabilities() { ) .await?; assert_eq!(status, StatusCode::OK, "{body}"); + assert_eq!( + body["fence"]["provenance"]["metastore_restored_from_backup"], false, + "{body}" + ); + assert_eq!( + body["fence"]["provenance"]["marker"], "consistent", + "{body}" + ); let (_, body) = admin.get("/v1/fence/capabilities").await?; assert_eq!(body["active_fences"], 1, "{body}"); From 1ce9f2f9bce138199b27e2ab94c735fc09839af4 Mon Sep 17 00:00:00 2001 From: River Date: Wed, 30 Sep 2026 11:14:22 +0000 Subject: [PATCH 26/33] libsql-server: namespace fence adoption Serve AdoptFence on the admin API (POST /v1/namespaces/:ns/fence/adopt). It needs the admin credential and the separate adoption key configured with --namespace-fence-adoption-key (SQLD_NAMESPACE_FENCE_ADOPTION_KEY), presented in the x-libsql-fence-adoption-key header. The key is kept as a SHA-256 digest and compared without an early exit; without a key, adoption is refused with adoption_not_authorised. The request names the current owner, two distinct approvers, an incident reference and a reason, which are stored in the record and receipt and written to one audit event (target libsql_server::fence::audit). Adoption moves the owner and the revision and nothing else: every admission stays as it was, capabilities of the old owner are revoked, and finished operations cannot be adopted. After a metastore rollback it re-establishes the record the marker holds under the new owner, settles the unavailable name and restores the namespace's own block_* values in memory. When the metastore holds no config row for the name, adoption is refused with namespace_config_missing instead of inventing a configuration. Co-authored-by: Tomasz Szymczyszyn --- docs/NAMESPACE_FENCE.md | 20 +- libsql-server/src/config.rs | 44 ++ libsql-server/src/http/admin/fence.rs | 99 ++- libsql-server/src/main.rs | 22 +- libsql-server/src/namespace/fence/audit.rs | 156 +++++ libsql-server/src/namespace/fence/mod.rs | 4 +- libsql-server/src/namespace/fence/outcome.rs | 4 + libsql-server/src/namespace/fence/target.rs | 8 +- libsql-server/src/namespace/fence/tests.rs | 609 +++++++++++++++++++ libsql-server/src/namespace/meta_store.rs | 47 +- libsql-server/src/namespace/store.rs | 31 +- libsql-server/tests/fence/admin.rs | 183 ++++++ libsql-server/tests/fence/mod.rs | 24 +- 13 files changed, 1203 insertions(+), 48 deletions(-) create mode 100644 libsql-server/src/namespace/fence/audit.rs diff --git a/docs/NAMESPACE_FENCE.md b/docs/NAMESPACE_FENCE.md index be1e6bea13..2844770d29 100644 --- a/docs/NAMESPACE_FENCE.md +++ b/docs/NAMESPACE_FENCE.md @@ -273,7 +273,7 @@ The routes are in `libsql-server/src/http/admin/fence.rs`, behind the admin list - **Fence view.** Beyond the fields of section 4.3 it reports `incarnation.current_log_id` (the replication log id of the namespace as loaded now, `null` when it is not loaded; a caller can take `expected_namespace_identity.log_id` from it), `admission.indeterminate`, `drain_started_at`, the owning operation's latest `validation` (with the server snapshot), `last_command_id`, `written_by`, `adoptions`, and for `UNKNOWN_UNAVAILABLE` the `detail`, `reason` and the record the marker holds (`marker_record`). A namespace being created as a target reports `TARGET_QUARANTINED` with the durable revision it has so far. `provenance` reports the metastore's restore provenance as recorded at startup (`MetaStore::restore_provenance`, section 13.3), the same for every namespace. Error responses carry the live view from the namespace's controller when it has one, otherwise what the metastore holds, or `null` when the namespace name itself is invalid. - **`InspectFence`** reads the metastore and the live controller; it neither loads the namespace nor creates a controller, so `drain` counters are zero for a namespace that is not loaded. Without `?receipts=all` only the owning operation's receipts are listed; `?receipts` with any other value is `invalid_argument`. - **`validation-query`** opens a `ValidationSession` (section 10.3) for the request and closes it afterwards. `stmts` follow the `/v1/execute` statement shape (`sql`, positional `args` or `named_args`, `want_rows`; `sql_id` is not supported). The response is `{"results": [...], "fence": ..., "drain": ...}`, one result per statement with `cols` and `rows` in the Hrana value encoding. A request whose statements return more than 10 000 rows in total is refused with `invalid_argument` rather than truncated, so a validation never looks at part of a result. A statement that tries to write is `OPERATION_CAPABILITY_REQUIRED` (`403`); other SQLite errors are `invalid_argument`. `expected_state`, when given, must match the live state (`FENCE_REVISION_MISMATCH` otherwise). -- **Capability discovery** lists in `commands` the commands this server serves (`AdoptFence` is added with its route, section 12), every state in `states`, and counts `active_fences` from the fence registry, which is seeded from the metastore at startup: records in any state but `RELEASED` or `TARGET_WRITABLE`, unavailable names, targets being created and indeterminate commits. `proxy_stable_code` reports whether the proxy's `stable_code` (section 6.1) is supported. `metastore` reports whether the metastore was restored from its backup at startup and the generation it was restored from (`restored_generation`, `null` when it was not restored). +- **Capability discovery** lists in `commands` the commands this server serves (including `AdoptFence`, which is served, and refused, even where no adoption key is configured; section 12), every state in `states`, and counts `active_fences` from the fence registry, which is seeded from the metastore at startup: records in any state but `RELEASED` or `TARGET_WRITABLE`, unavailable names, targets being created and indeterminate commits. `proxy_stable_code` reports whether the proxy's `stable_code` (section 6.1) is supported. `metastore` reports whether the metastore was restored from its backup at startup and the generation it was restored from (`restored_generation`, `null` when it was not restored). ## 5. Durable state (Design, with contract points marked) @@ -387,7 +387,7 @@ Notes: - `FENCE_COMMAND_CONFLICT`: a `command_id` reused with a different request fingerprint. - `FENCE_COMMIT_INDETERMINATE`: the server could not establish whether its own commit landed (for example an I/O error on `COMMIT`). Admission stays closed; only a replay of the same `command_id` reconciles it; every other command on the namespace receives this code until then. -- `FENCE_PRECONDITION_FAILED` carries `detail`, one of: `admin_auth_required`, `fence_disabled`, `not_primary`, `shared_schema_unsupported`, `namespace_identity_mismatch`, `namespace_exists`, `validation_receipt_required`, `restore_not_allowed`, `adoption_not_authorised`, `invalid_argument`. `INVALID_FENCE_TRANSITION` may carry `role_mismatch` or `operation_finished`. `FENCE_STATE_UNAVAILABLE` carries the reason the state cannot be established: `corrupt_record`, `unsupported_format_version`, `incomplete_target_creation`, `metastore_behind_marker` or `indeterminate_commit`. `MIGRATION_WRITE_FENCED` carries `stale_transaction` when the gate itself admits writes but the transaction (or the program) attempting one began under an earlier write generation (section 8.1): the client must roll back and begin a new transaction. +- `FENCE_PRECONDITION_FAILED` carries `detail`, one of: `admin_auth_required`, `fence_disabled`, `not_primary`, `shared_schema_unsupported`, `namespace_identity_mismatch`, `namespace_exists`, `validation_receipt_required`, `restore_not_allowed`, `adoption_not_authorised`, `namespace_config_missing`, `invalid_argument`. `INVALID_FENCE_TRANSITION` may carry `role_mismatch` or `operation_finished`. `FENCE_STATE_UNAVAILABLE` carries the reason the state cannot be established: `corrupt_record`, `unsupported_format_version`, `incomplete_target_creation`, `metastore_behind_marker` or `indeterminate_commit`. `MIGRATION_WRITE_FENCED` carries `stale_transaction` when the gate itself admits writes but the transaction (or the program) attempting one began under an earlier write generation (section 8.1): the client must roll back and begin a new transaction. - **Data-plane denials are never `500`, `503`, `429` or gRPC `UNAVAILABLE`.** `423 Locked` is chosen because common HTTP clients do not retry it. The JSON error body of the user HTTP API gains an additive `"code"` field (`{"error": "...", "code": "MIGRATION_WRITE_FENCED"}`); the existing `Blocked` error (from `block_reads`/`block_writes`) keeps its current mapping. - gRPC statuses carry the code in the `x-libsql-fence-code` metadata entry and as the message prefix `": "`. - Authentication (`401`), missing namespace (`404`), timeouts and transport errors are distinct from all of the above. @@ -715,6 +715,16 @@ impl ValidationSession { The server cannot verify who the approvers are: the admin API has one shared key and no principal. "Two-person" is enforced as a separate secret plus a recorded two-approver request; real two-person control belongs to whatever holds those secrets. +Request body: the common fields of section 4.2 (`operation_id` is the adopting operation), plus `current_operation_id`, `approvers` (an array of exactly two strings, distinct after trimming and neither empty), `incident_ref` and `reason` (neither empty after trimming). Unknown fields are refused like on every other route. + +Implementation (`http/admin/fence.rs`, `NamespaceStore::execute_fence_command_authorised`, `namespace/fence/audit.rs`): + +- **The key.** `--namespace-fence-adoption-key` (env `SQLD_NAMESPACE_FENCE_ADOPTION_KEY`, hidden from `--help`; an empty value is a startup error) is kept only as its SHA-256 digest, and the header's value is hashed and compared digest to digest without an early exit. Whether it matched is handed to the transition function, so the order of section 5.3 holds: a replay of a committed adoption returns its receipt even without the key (it changes nothing), and the key, the approvers, the incident reference and the reason are checked after the owner (`current_operation_id` must be the record's owner, `FENCE_OWNED_BY_ANOTHER_OPERATION` otherwise), after the finished-operation check and after `expected_state` / `expected_revision`, all of which the admin credential already lets a caller read with `InspectFence`. An operation cannot adopt its own fence. +- **What changes.** The owner, the revision, `last_command_id`, `written_by` and the appended adoption entry; the receipt carries the same entry. The state, and so every admission, is unchanged, as are the identity, the frozen boundary, the validation and the saved legacy values; the legacy mirror in the config row names the new owner. Because the owner changed, the write generation moves (section 7.2: nothing can write in a state adoption can act on, so this refuses nothing that was admitted) and every capability issued to the old owner is revoked: the old owner's import or validation session stops working and the new owner opens its own. `RELEASED`, `TARGET_WRITABLE` and `TARGET_ABORTED` are refused with `INVALID_FENCE_TRANSITION` / `operation_finished`. +- **After a metastore rollback.** When the marker is ahead of the metastore (`metastore_behind_marker`: the fence row is missing or older), the adoption names the marker's state, revision and owner (the `marker_record` that `InspectFence` reports), and commits the marker's record with the new owner at the marker's revision + 1, as an update of the older row or an insert when the row is gone. The name leaves the unavailable set, its gate is the re-established record's, and its in-memory config gets the namespace's own `block_*` values back (the saved values in the record), as startup does for every established record. No other unavailable reason can be adopted: a corrupt record, an unsupported format version, an interrupted target creation (which only its own replay completes) and an indeterminate commit (which only its own replay reconciles) keep refusing it. +- **A name the metastore holds no configuration for.** A metastore restored from a backup older than the namespace itself has neither the fence row nor the config row. Adoption is then refused with `FENCE_PRECONDITION_FAILED` / `namespace_config_missing`, after the authorisation checks, and writes nothing; the name stays `UNKNOWN_UNAVAILABLE`. The marker holds the fence record, not the namespace's configuration (its JWT key, size limit, durability mode, backup id, attach and shared-schema settings), and re-creating the config row from defaults would silently change who can read the namespace and how it is stored once it is released. Recovering such a namespace is an operator decision outside the fence: put its config row back (for example from a newer metastore backup), after which the adoption applies; or discard the namespace directory, marker included. +- **Audit.** Every committed adoption (not a replay, not a refusal) emits one `info` event with target `libsql_server::fence::audit` and `event = "namespace_fence_adopted"`, carrying the namespace, command, outcome, state, previous and new operation id, command id, approvers, incident reference, reason, revisions before and after and the server instance. Section 15's events for the other transitions use the same target. + ## 13. Deployment and compatibility (Contract) ### 13.1 Deployment flag @@ -742,7 +752,7 @@ This is best effort. An older binary applies `block_*` at statement level only, Recovery fails closed when the flag is on, when the fence tables exist, or when any namespace directory holds a marker. Otherwise (a server that never used fences) recovery behaves as before. Startup scans `dbs/` for markers first; a marker in a directory whose name is not a valid namespace name stops startup with an operator error. -A namespace that startup cannot recover is registered `UNKNOWN_UNAVAILABLE` in memory. It is never given a default config, never default-created, and every lookup, config write and delete of it is refused with `FENCE_STATE_UNAVAILABLE` and a detail. `InspectFence` and the fence list report it. Nothing is written for it: the registration is recomputed from the metastore and the markers at every start, and a fence command that commits for the name (the replay that completes an interrupted target creation, later an adoption) settles it. +A namespace that startup cannot recover is registered `UNKNOWN_UNAVAILABLE` in memory. It is never given a default config, never default-created, and every lookup, config write and delete of it is refused with `FENCE_STATE_UNAVAILABLE` and a detail. `InspectFence` and the fence list report it. Nothing is written for it: the registration is recomputed from the metastore and the markers at every start, and a fence command that commits for the name (the replay that completes an interrupted target creation, or an adoption after a metastore rollback, section 12) settles it. - `MetaStore::handle()` no longer default-creates entries on read paths. `NamespaceStore::with` (and so every SQL, Hrana, dump and replication entry point), fork's source and ATTACH authorisation (`check_program_auth`) use the non-creating `MetaStore::lookup`, which returns the existing handle, nothing, or the fence error. Only create, fork destination, reset, the default namespace and lazy creation call `handle()`, which refuses a registered name, and a name without a config whose directory holds a marker (a target being created, or a namespace the metastore lost). `NamespaceStore::checkpoint` does not touch the metastore and needs no lookup. - `restore()`: an undecodable config row marks its namespace `UNKNOWN_UNAVAILABLE` (`corrupt_record`) and the row is left as it is. An undecodable namespace name cannot be addressed by any request, so a legacy row is still skipped; if the metastore holds a fence row for such a name, startup fails with an operator error. An undecodable, unknown-version or inconsistent fence row, or an unreadable marker, marks the namespace `UNKNOWN_UNAVAILABLE` (section 5.6). @@ -790,7 +800,7 @@ How each path that can reach namespace data or lifecycle is covered. File refere | `namespace/store.rs` `with`, `load_namespace`, `make_namespace`, eviction | Registry check before setup; controller passed into setup; eviction keeps the registry entry. | | `store.rs` `create`, `destroy`, `reset`, `fork`, `checkpoint`, restore options | `check_lifecycle` (section 13.5): create refuses a name whose fence denies lifecycle work; `CreateTargetQuarantined` is the atomic quarantined create; destroy is refused in the metastore transaction; reset and fork as source are checked under the namespace's transition lock; fork's destination is checked before anything is stored and again under its entry lock; every restore (create with a dump, reset, fork to a point in time) is one of these paths and is refused there; checkpoint uses a non-creating lookup and skips vacuum. | | `http/admin/mod.rs` config, create, fork, delete, checkpoint, stats | Config POST and create (before a `dump_url` is fetched) are checked before the namespace is loaded; fork and delete are refused by the store; all of them return the fence error with `423`. Config GET, stats and checkpoint are allowed. Fence routes live in `http/admin/fence.rs`. | -| `http/admin/fence.rs` | Fence routes (section 4.5): capability discovery, `InspectFence`, one route per command through `NamespaceStore::execute_fence_command`, and `validation-query` through `open_validation_session`; admin auth key and primary required for every route that changes state or uses a capability. | +| `http/admin/fence.rs` | Fence routes (section 4.5): capability discovery, `InspectFence`, one route per command through `NamespaceStore::execute_fence_command` (`AdoptFence` through `execute_fence_command_authorised` with the result of the adoption key check, section 12), and `validation-query` through `open_validation_session`; admin auth key and primary required for every route that changes state or uses a capability. | | `http/user/dump.rs`, `connection/dump/exporter.rs` | Gate check (`Stream`) before connection creation (typed, no panic on create error); `Dump` lease held by the export; cancel checked before every row and by the pipe's pending write; the body stream ends with the fence error (aborted response) on cancellation. | | `rpc/replication/replication_log.rs` `hello`, `log_entries`, `batch_log_entries`, `snapshot` | Denied at request start (`FAILED_PRECONDITION` + `x-libsql-fence-code`, counted, rate-limited log); `FencedStream` replication leases for both streams and the batch; typed terminal status when the gate closes or at the deadline, lease released without the peer; `ReplicatedFence` in `hello`'s config while a fence is active (section 6.2; the gate is read without a lease, since `hello` streams no data). | | `replication/replicator_client.rs`, `namespace/configurator/replica.rs` (replica servers) | A fence status from `hello`, `log_entries` or `snapshot`, or a stream ended with one, installs the local read denial (`FenceController::observe_primary`) and is returned as `PrimaryFenceRefusal`; the replication loop backs off 1 s → 15 s instead of reconnecting every second; an answered `hello` lifts the denial (section 6.2). A lazy creation whose first `hello` the primary refuses fails at once with the code and leaves no directory, metastore entry or controller behind (section 13.3). | @@ -855,7 +865,7 @@ Planned test names; the table is updated as tests land. | 19 | Corrupt or unknown durable fence state fails closed | `fence::store::tests::corrupt_payload_fails_closed`, `unknown_format_version_fails_closed` | | 20 | Metrics and audit logs | `tests::fence::observability::metrics_and_labels`; `fence::audit::tests::audit_event_fields` | | 21 | Capability discovery and mixed-version protection | capability discovery landed: `tests::fence::admin::{capabilities, capabilities_when_disabled}`; planned: `fence::store::tests::legacy_mirror_and_fk_guard` (bounded, see section 18) | -| 22 | Adoption is two-person/audited, keeps admission closed, cannot reverse publication | `fence::tests::adopt_requires_key_and_two_approvers`, `adopt_keeps_gates_closed`, `adopt_cannot_touch_writable` | +| 22 | Adoption is two-person/audited, keeps admission closed, cannot reverse publication | landed: `namespace::fence::tests::adoption::{adopt_requires_key_and_two_approvers (no key, one approver, duplicate or blank approvers, three approvers, blank incident or reason: `adoption_not_authorised`, nothing changes), adopt_keeps_gates_closed (write-fenced source: owner and revision move, state, admissions, boundary and saved values do not, writes still refused, replay with or without the key returns the receipt, old owner `FENCE_OWNED_BY_ANOTHER_OPERATION`, new owner releases), adopt_quarantined_target (the import capability moves to the new owner; SQL still `MIGRATION_TARGET_QUARANTINED`), adopt_cannot_touch_writable (`TARGET_WRITABLE`, `TARGET_ABORTED`), adopt_cannot_touch_released, adopt_recovers_metastore_rollback (fence row gone, and fence row at an older revision: re-established from the marker, served again behind the same gate, own `block_*` values back in memory, new owner finishes), adopt_recovered_name_without_config_row (`namespace_config_missing`, nothing written, still unavailable), adoption_key_matching}`, `namespace::fence::audit::tests::adoption_event_fields`; pure transition: `namespace::fence::transition::tests::{adopt_requires_key_and_two_approvers, adopt_keeps_gates_closed}`; over HTTP: `tests::fence::admin::{adopt_over_http (no key, wrong key, no admin credential, bad approvers, unknown field, success, replay, old and new owner, finished operation), adopt_disabled_without_key}` | | — | Import API usable by bulk import | landed: `namespace::fence::import::tests::import_session_loads_dump_into_quarantined_target` (a dump exported by the server from a source with tables, keys, a foreign key, an index, an autoincrement table, a trigger, a view and an FTS5 table loads through `ImportSession::load_dump`; after the seal the target's schema, rows, view and full-text results equal the source's) | ## 18. Limits diff --git a/libsql-server/src/config.rs b/libsql-server/src/config.rs index 48dee38dfa..ceaae19b3a 100644 --- a/libsql-server/src/config.rs +++ b/libsql-server/src/config.rs @@ -199,6 +199,50 @@ pub struct MetaStoreConfig { /// How long `SetSourceReadFence` waits for running reads and streams before it cancels /// them, when the request names no drain policy. `None` is the default of 30 seconds. pub namespace_fence_default_read_drain: Option, + /// The separate secret that authorises namespace fence adoption (`AdoptFence`), presented + /// in the `x-libsql-fence-adoption-key` header beside the admin credential. `None` + /// disables adoption. + pub namespace_fence_adoption_key: Option, +} + +/// The secret that authorises namespace fence adoption (`docs/NAMESPACE_FENCE.md` section 12). +/// +/// Only its SHA-256 digest is kept, and a presented key is compared digest to digest without +/// an early exit, so the comparison takes the same time whatever the presented key is. `Debug` +/// never prints it. +#[derive(Clone)] +pub struct FenceAdoptionKey(Arc<[u8; 32]>); + +impl FenceAdoptionKey { + /// `None` for an empty key, which would authorise nothing. + pub fn new(key: &str) -> Option { + if key.is_empty() { + return None; + } + Some(Self(Arc::new(Self::digest(key.as_bytes())))) + } + + /// Whether `presented` is the configured key. + pub fn matches(&self, presented: &[u8]) -> bool { + let presented = Self::digest(presented); + let difference = self + .0 + .iter() + .zip(presented.iter()) + .fold(0u8, |acc, (a, b)| acc | (a ^ b)); + difference == 0 + } + + fn digest(bytes: &[u8]) -> [u8; 32] { + use sha2::Digest as _; + sha2::Sha256::digest(bytes).into() + } +} + +impl std::fmt::Debug for FenceAdoptionKey { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.write_str("FenceAdoptionKey()") + } } #[derive(Debug, Clone)] diff --git a/libsql-server/src/http/admin/fence.rs b/libsql-server/src/http/admin/fence.rs index b47b906390..a4b94b86bb 100644 --- a/libsql-server/src/http/admin/fence.rs +++ b/libsql-server/src/http/admin/fence.rs @@ -4,7 +4,8 @@ //! server was started with `--enable-namespace-fence`, except `InspectFence`, which is also //! served while fence state exists in the metastore with the flag off (fences are enforced //! either way, section 13.1). Every mutating route runs through -//! [`NamespaceStore::execute_fence_command`], which owns replay, the drains and target creation. +//! [`NamespaceStore::execute_fence_command`], which owns replay, the drains and target creation +//! (`AdoptFence` through `execute_fence_command_authorised`, with the adoption key check). use std::sync::Arc; @@ -13,7 +14,7 @@ use axum::response::{IntoResponse, Response}; use axum::routing::{get, post}; use axum::Json; use bytes::Bytes; -use hyper::StatusCode; +use hyper::{HeaderMap, StatusCode}; use serde::de::DeserializeOwned; use serde::Deserialize; use serde_json::{json, Map, Value}; @@ -23,12 +24,14 @@ use crate::auth::parse_jwt_keys; use crate::error::Error; use crate::hrana::proto; use crate::namespace::fence::command::{ - CommandKind, DrainPolicy, FenceCommand, FenceRequest, OnDeadline, TargetConfig, + AdoptArgs, CommandKind, DrainPolicy, FenceCommand, FenceRequest, OnDeadline, TargetConfig, ValidationResult, }; use crate::namespace::fence::controller::{DrainCounters, FenceController}; use crate::namespace::fence::outcome::{FenceDetail, FenceError, FenceOutcome}; -use crate::namespace::fence::record::{CommandReceipt, NamespaceFenceRecord, ServerIdentity}; +use crate::namespace::fence::record::{ + Adoption, CommandReceipt, NamespaceFenceRecord, ServerIdentity, +}; use crate::namespace::fence::state::FenceState; use crate::namespace::fence::store::{StoredFence, StoredReceipt}; use crate::namespace::fence::{server_identity, FENCE_PROTOCOL_VERSION, PROXY_STABLE_CODE}; @@ -43,9 +46,11 @@ use super::AppState; /// looks at part of a result. pub const MAX_VALIDATION_QUERY_ROWS: usize = 10_000; +/// The header that carries the adoption key (section 12). +pub const ADOPTION_KEY_HEADER: &str = "x-libsql-fence-adoption-key"; + /// The commands this server serves over the admin API, reported by capability discovery. -/// Adoption is served once its route exists. -const SERVED_COMMANDS: [&str; 11] = [ +const SERVED_COMMANDS: [&str; 12] = [ "InspectFence", CommandKind::AcquireSourceWriteFence.as_str(), CommandKind::SetSourceReadFence.as_str(), @@ -57,6 +62,7 @@ const SERVED_COMMANDS: [&str; 11] = [ CommandKind::PublishTargetReadableWriteFenced.as_str(), CommandKind::EnableTargetWrites.as_str(), CommandKind::AbortQuarantinedTarget.as_str(), + CommandKind::AdoptFence.as_str(), ]; /// The fence routes, added to the admin router. @@ -66,7 +72,7 @@ pub(super) fn routes() -> axum::Router>> { move |State(state): State>>, Path(namespace): Path, body: Bytes| async move { - handle_command(state, namespace, kind, body).await + handle_command(state, namespace, kind, body, false).await }, ) }; @@ -117,6 +123,7 @@ pub(super) fn routes() -> axum::Router>> { "/v1/namespaces/:namespace/fence/target/validation-query", post(handle_validation_query), ) + .route("/v1/namespaces/:namespace/fence/adopt", post(handle_adopt)) } // --------------------------------------------------------------------------------------------- @@ -215,11 +222,30 @@ async fn handle_inspect( (StatusCode::OK, Json(body)).into_response() } +/// `AdoptFence` (section 12): the admin credential (checked by the middleware) and the separate +/// adoption key in [`ADOPTION_KEY_HEADER`]. Whether the key matched is handed to the transition, +/// which refuses an unauthorised adoption with `adoption_not_authorised` after replay handling, +/// like every other check. +async fn handle_adopt( + State(state): State>>, + Path(namespace): Path, + headers: HeaderMap, + body: Bytes, +) -> Response { + let presented = headers.get(ADOPTION_KEY_HEADER).map(|v| v.as_bytes()); + let authorised = state + .namespaces + .meta_store() + .fence_adoption_authorised(presented); + handle_command(state, namespace, CommandKind::AdoptFence, body, authorised).await +} + async fn handle_command( state: Arc>, namespace: String, kind: CommandKind, body: Bytes, + adoption_authorised: bool, ) -> Response { if !state.namespaces.meta_store().fence_enabled() { return StatusCode::NOT_FOUND.into_response(); @@ -235,11 +261,18 @@ async fn handle_command( Ok(request) => request, Err(e) => return error_reply(&state, &namespace, e).await, }; - match state - .namespaces - .execute_fence_command(request, server_identity()) - .await - { + let result = if kind == CommandKind::AdoptFence { + state + .namespaces + .execute_fence_command_authorised(request, server_identity(), adoption_authorised) + .await + } else { + state + .namespaces + .execute_fence_command(request, server_identity()) + .await + }; + match result { Ok(commit) => success_reply(&state, &namespace, commit), Err(e) => fence_or_error(&state, &namespace, e).await, } @@ -650,9 +683,12 @@ fn parse_command( } CommandKind::EnableTargetWrites => FenceCommand::EnableTargetWrites, CommandKind::AbortQuarantinedTarget => FenceCommand::AbortQuarantinedTarget, - CommandKind::AdoptFence => { - return Err(invalid("adoption is not served by this route")); - } + CommandKind::AdoptFence => FenceCommand::AdoptFence(AdoptArgs { + current_operation_id: body.uuid("current_operation_id")?, + approvers: body.req("approvers")?, + incident_ref: body.req("incident_ref")?, + reason: body.req("reason")?, + }), }; body.finish()?; Ok(FenceRequest { @@ -907,24 +943,7 @@ fn record_fields(record: &NamespaceFenceRecord, out: &mut Map) { out.insert("written_by".into(), server_json(&record.written_by)); out.insert( "adoptions".into(), - Value::Array( - record - .adoptions - .iter() - .map(|a| { - json!({ - "previous_operation_id": a.previous_operation_id.to_string(), - "new_operation_id": a.new_operation_id.to_string(), - "command_id": a.command_id.to_string(), - "approvers": a.approvers, - "incident_ref": a.incident_ref, - "reason": a.reason, - "at": timestamp(a.at_ms), - "revision": a.revision, - }) - }) - .collect(), - ), + Value::Array(record.adoptions.iter().map(adoption_json).collect()), ); } @@ -1022,6 +1041,19 @@ fn metastore_provenance_json(provenance: &MetastoreProvenance) -> Value { }) } +fn adoption_json(a: &Adoption) -> Value { + json!({ + "previous_operation_id": a.previous_operation_id.to_string(), + "new_operation_id": a.new_operation_id.to_string(), + "command_id": a.command_id.to_string(), + "approvers": a.approvers, + "incident_ref": a.incident_ref, + "reason": a.reason, + "at": timestamp(a.at_ms), + "revision": a.revision, + }) +} + fn receipt_json(receipt: &CommandReceipt) -> Value { json!({ "operation_id": receipt.operation_id.to_string(), @@ -1034,6 +1066,7 @@ fn receipt_json(receipt: &CommandReceipt) -> Value { "state_after": receipt.state_after.as_str(), "applied_at": timestamp(receipt.applied_at_ms), "instance_id": receipt.instance_id.to_string(), + "adoption": receipt.adoption.as_ref().map(adoption_json), }) } diff --git a/libsql-server/src/main.rs b/libsql-server/src/main.rs index 02a8011c0f..7358b3482f 100644 --- a/libsql-server/src/main.rs +++ b/libsql-server/src/main.rs @@ -16,8 +16,8 @@ use tracing_subscriber::Layer; use tracing_subscriber::{prelude::*, EnvFilter}; use libsql_server::config::{ - AdminApiConfig, BottomlessConfig, DbConfig, HeartbeatConfig, MetaStoreConfig, RpcClientConfig, - RpcServerConfig, TlsConfig, UserApiConfig, + AdminApiConfig, BottomlessConfig, DbConfig, FenceAdoptionKey, HeartbeatConfig, MetaStoreConfig, + RpcClientConfig, RpcServerConfig, TlsConfig, UserApiConfig, }; use libsql_server::net::AddrIncoming; use libsql_server::version::Version; @@ -287,6 +287,17 @@ struct Cli { #[clap(long, env = "SQLD_NAMESPACE_FENCE_KEEPALIVE_INTERVAL_S")] namespace_fence_keepalive_interval_s: Option, + /// The separate secret that authorises namespace fence adoption (incident recovery of an + /// unfinished operation whose owner was lost), presented in the + /// `x-libsql-fence-adoption-key` header beside the admin credential. Adoption is disabled + /// when this is not set. + #[clap( + long, + env = "SQLD_NAMESPACE_FENCE_ADOPTION_KEY", + hide_env_values = true + )] + namespace_fence_adoption_key: Option, + /// Shutdown timeout duration in seconds, defaults to 30 seconds. #[clap(long, env = "SQLD_SHUTDOWN_TIMEOUT")] shutdown_timeout: Option, @@ -689,6 +700,13 @@ fn make_meta_store_config(config: &Cli) -> anyhow::Result { namespace_fence_default_read_drain: config .namespace_fence_default_read_drain_ms .map(Duration::from_millis), + namespace_fence_adoption_key: match config.namespace_fence_adoption_key.as_deref() { + None => None, + Some(key) => Some( + FenceAdoptionKey::new(key) + .context("--namespace-fence-adoption-key must not be empty")?, + ), + }, }) } diff --git a/libsql-server/src/namespace/fence/audit.rs b/libsql-server/src/namespace/fence/audit.rs new file mode 100644 index 0000000000..e727a0fea7 --- /dev/null +++ b/libsql-server/src/namespace/fence/audit.rs @@ -0,0 +1,156 @@ +//! The fence audit log (`docs/NAMESPACE_FENCE.md` sections 12 and 15): structured events under +//! the tracing target [`AUDIT_TARGET`], so that a log pipeline can route them apart from the +//! server's operational logs. + +use crate::namespace::meta_store::FenceCommit; +use crate::namespace::NamespaceName; + +/// The tracing target of every fence audit event. +pub const AUDIT_TARGET: &str = "libsql_server::fence::audit"; + +/// One audit event for a committed `AdoptFence`: who adopted what from whom, the two recorded +/// approvers, the incident reference and the reason, and the revisions. The server cannot +/// verify the approvers (section 12); the event is the record of what the request claimed. +pub fn adoption(namespace: &NamespaceName, commit: &FenceCommit) { + let receipt = &commit.receipt; + let Some(adoption) = &receipt.adoption else { + return; + }; + tracing::info!( + target: AUDIT_TARGET, + event = "namespace_fence_adopted", + namespace = %namespace, + command = receipt.command.as_str(), + outcome = receipt.outcome.as_str(), + state = receipt.state_after.as_str(), + previous_operation_id = %adoption.previous_operation_id, + operation_id = %adoption.new_operation_id, + command_id = %adoption.command_id, + approvers = ?adoption.approvers, + incident_ref = %adoption.incident_ref, + reason = %adoption.reason, + revision_before = receipt.revision_before, + revision_after = receipt.revision_after, + server_instance = %receipt.instance_id, + "namespace fence adopted" + ); +} + +#[cfg(test)] +mod tests { + use std::io::Write; + use std::sync::{Arc, Mutex}; + + use uuid::Uuid; + + use super::*; + use crate::namespace::fence::command::{AdoptArgs, FenceCommand, FenceRequest}; + use crate::namespace::fence::outcome::FenceOutcome; + use crate::namespace::fence::record::{Adoption, CommandReceipt}; + use crate::namespace::fence::state::FenceState; + use crate::namespace::meta_store::FenceCommitKind; + + #[derive(Clone, Default)] + struct Captured(Arc>>); + + impl Write for Captured { + fn write(&mut self, buf: &[u8]) -> std::io::Result { + self.0.lock().unwrap().extend_from_slice(buf); + Ok(buf.len()) + } + + fn flush(&mut self) -> std::io::Result<()> { + Ok(()) + } + } + + fn capture(f: impl FnOnce()) -> String { + let captured = Captured::default(); + let writer = captured.clone(); + let subscriber = tracing_subscriber::fmt() + .with_ansi(false) + .with_max_level(tracing::Level::INFO) + .with_writer(move || writer.clone()) + .finish(); + tracing::subscriber::with_default(subscriber, f); + let bytes = captured.0.lock().unwrap().clone(); + String::from_utf8(bytes).unwrap() + } + + fn commit(adoption: Option) -> FenceCommit { + let args = AdoptArgs { + current_operation_id: Uuid::from_u128(0xa), + approvers: vec!["alice".into(), "bob".into()], + incident_ref: "INC-1".into(), + reason: "control record lost".into(), + }; + let request = FenceRequest { + namespace: "ns".into(), + operation_id: Uuid::from_u128(0xb), + command_id: Uuid::from_u128(3), + expected_state: FenceState::SourceWriteFenced, + expected_revision: 2, + command: FenceCommand::AdoptFence(args), + }; + FenceCommit { + kind: FenceCommitKind::Committed, + receipt: CommandReceipt { + namespace: "ns".into(), + operation_id: request.operation_id, + command_id: request.command_id, + command: request.command.kind(), + fingerprint: request.fingerprint(), + outcome: FenceOutcome::Applied, + revision_before: 2, + revision_after: 3, + state_after: FenceState::SourceWriteFenced, + applied_at_ms: 1_000, + instance_id: Uuid::from_u128(0x99), + adoption, + }, + record: None, + created_config: None, + } + } + + /// A committed adoption is one event under the audit target with every field section 12 + /// asks for; a receipt without an adoption emits nothing. + #[test] + fn adoption_event_fields() { + let entry = Adoption { + previous_operation_id: Uuid::from_u128(0xa), + new_operation_id: Uuid::from_u128(0xb), + command_id: Uuid::from_u128(3), + approvers: vec!["alice".into(), "bob".into()], + incident_ref: "INC-1".into(), + reason: "control record lost".into(), + at_ms: 1_000, + revision: 3, + }; + let out = capture(|| super::adoption(&"ns".into(), &commit(Some(entry.clone())))); + assert_eq!(out.lines().count(), 1, "{out}"); + for expected in [ + AUDIT_TARGET, + "namespace fence adopted", + "event=\"namespace_fence_adopted\"", + "namespace=ns", + "command=\"AdoptFence\"", + "outcome=\"APPLIED\"", + "state=\"SOURCE_WRITE_FENCED\"", + &format!("previous_operation_id={}", Uuid::from_u128(0xa)), + &format!("operation_id={}", Uuid::from_u128(0xb)), + &format!("command_id={}", Uuid::from_u128(3)), + "approvers=[\"alice\", \"bob\"]", + "incident_ref=INC-1", + "reason=control record lost", + "revision_before=2", + "revision_after=3", + &format!("server_instance={}", Uuid::from_u128(0x99)), + ] { + assert!(out.contains(expected), "`{expected}` missing from {out}"); + } + + let out = capture(|| super::adoption(&"ns".into(), &commit(None))); + assert!(out.is_empty(), "{out}"); + } +} diff --git a/libsql-server/src/namespace/fence/mod.rs b/libsql-server/src/namespace/fence/mod.rs index 8c89d79f5e..46ed08eb5e 100644 --- a/libsql-server/src/namespace/fence/mod.rs +++ b/libsql-server/src/namespace/fence/mod.rs @@ -12,13 +12,15 @@ //! [`drain`], the source [`read`] fence and its //! [`stream`] leases for dump and replication, quarantined migration [`target`]s with their //! [`capability`]-scoped [`import`] sessions and seal drain, the [`registry`] that holds the controllers outside the namespace cache, the -//! [`replica`]-server view of a primary's fence, and the test [`hooks`] on their paths. +//! [`replica`]-server view of a primary's fence, the [`audit`] log, and the test [`hooks`] on +//! their paths. // The persistence, controller and protocol layers that consume these types land in the // following commits of this series; until then most of the module is unused by the rest of // the crate. This attribute is removed once they are wired. #![allow(dead_code)] +pub mod audit; pub mod capability; pub mod command; pub mod controller; diff --git a/libsql-server/src/namespace/fence/outcome.rs b/libsql-server/src/namespace/fence/outcome.rs index 723b7b5ab6..5b4a18a7fd 100644 --- a/libsql-server/src/namespace/fence/outcome.rs +++ b/libsql-server/src/namespace/fence/outcome.rs @@ -191,6 +191,9 @@ pub enum FenceDetail { ValidationReceiptRequired, RestoreNotAllowed, AdoptionNotAuthorised, + /// Adoption would have to re-establish a record for a namespace the metastore holds no + /// configuration for (section 12). + NamespaceConfigMissing, InvalidArgument, // INVALID_FENCE_TRANSITION RoleMismatch, @@ -217,6 +220,7 @@ impl FenceDetail { FenceDetail::ValidationReceiptRequired => "validation_receipt_required", FenceDetail::RestoreNotAllowed => "restore_not_allowed", FenceDetail::AdoptionNotAuthorised => "adoption_not_authorised", + FenceDetail::NamespaceConfigMissing => "namespace_config_missing", FenceDetail::InvalidArgument => "invalid_argument", FenceDetail::RoleMismatch => "role_mismatch", FenceDetail::OperationFinished => "operation_finished", diff --git a/libsql-server/src/namespace/fence/target.rs b/libsql-server/src/namespace/fence/target.rs index 38729dd191..05901d8e25 100644 --- a/libsql-server/src/namespace/fence/target.rs +++ b/libsql-server/src/namespace/fence/target.rs @@ -219,7 +219,7 @@ pub(crate) mod tests { tokio::spawn(async move { store.create_target_quarantined(req, server()).await }) } - fn target_command( + pub(crate) fn target_command( command_id: u128, expected_state: FenceState, expected_revision: u64, @@ -235,7 +235,7 @@ pub(crate) mod tests { } } - fn execute( + pub(crate) fn execute( store: &NamespaceStore, request: FenceRequest, ) -> tokio::task::JoinHandle> { @@ -303,7 +303,7 @@ pub(crate) mod tests { } /// A target with successful validation, published readable and write-fenced at revision 5. - async fn write_fenced_target() -> (TempDir, NamespaceStore, Arc) { + pub(crate) async fn write_fenced_target() -> (TempDir, NamespaceStore, Arc) { let (dir, store, fence) = validating_target().await; record_validation(&store, 20, 3, ValidationResult::Ok).await; let publish = target_command( @@ -320,7 +320,7 @@ pub(crate) mod tests { (dir, store, fence) } - fn enable_request(command_id: u128) -> FenceRequest { + pub(crate) fn enable_request(command_id: u128) -> FenceRequest { target_command( command_id, FenceState::TargetWriteFenced, diff --git a/libsql-server/src/namespace/fence/tests.rs b/libsql-server/src/namespace/fence/tests.rs index 2ea08f8ec3..0d5aeb7ae7 100644 --- a/libsql-server/src/namespace/fence/tests.rs +++ b/libsql-server/src/namespace/fence/tests.rs @@ -936,3 +936,612 @@ fn acquire_response_loss_resolved_by_replay_and_inspect() { server.crash(); } } + +// --------------------------------------------------------------------------------------------- +// Incident adoption (section 12; section 17 row 22) + +mod adoption { + use super::*; + use crate::config::FenceAdoptionKey; + use crate::namespace::fence::command::AdoptArgs; + use crate::namespace::fence::target::tests::{ + create, create_request, enable_request, execute as execute_target, target_command, + write_fenced_target, + }; + use crate::namespace::meta_store::metastore_connection_maker; + + fn adopt_args(approvers: &[&str]) -> AdoptArgs { + AdoptArgs { + current_operation_id: OP, + approvers: approvers.iter().map(|a| a.to_string()).collect(), + incident_ref: "INC-1".into(), + reason: "the operation's control record was lost".into(), + } + } + + fn adopt( + ns: &'static str, + command_id: u128, + expected_state: FenceState, + expected_revision: u64, + args: AdoptArgs, + ) -> FenceRequest { + FenceRequest { + namespace: ns.into(), + operation_id: OTHER_OP, + command_id: Uuid::from_u128(command_id), + expected_state, + expected_revision, + command: FenceCommand::AdoptFence(args), + } + } + + async fn run_adopt( + store: &NamespaceStore, + request: FenceRequest, + authorised: bool, + ) -> crate::Result { + let store = store.clone(); + tokio::spawn(async move { + store + .execute_fence_command_authorised(request, server_identity(), authorised) + .await + }) + .await + .unwrap() + } + + fn release_by(operation_id: Uuid, command_id: u128, revision: u64) -> FenceRequest { + FenceRequest { + operation_id, + ..release(command_id, revision) + } + } + + /// The adoption key is kept as a digest and compared whole; an empty key authorises + /// nothing; `Debug` does not print it; a metastore without a key refuses every presented + /// value. + #[test] + fn adoption_key_matching() { + let key = FenceAdoptionKey::new("s3cret").unwrap(); + assert!(key.matches(b"s3cret")); + for wrong in [&b""[..], b"s3cre", b"s3cret ", b"S3CRET"] { + assert!(!key.matches(wrong)); + } + assert!(FenceAdoptionKey::new("").is_none()); + assert!(!format!("{key:?}").contains("s3cret")); + + let dir = tempdir().unwrap(); + let server = Server::boot(dir.path()); + let meta = server.store.meta_store(); + assert!(!meta.fence_adoption_authorised(None)); + assert!(!meta.fence_adoption_authorised(Some(b"s3cret"))); + server.crash(); + } + + /// Adoption without the key, with one approver, with the same approver twice (also after + /// trimming) or without an incident reference or reason is refused with + /// `adoption_not_authorised`, and changes nothing. + #[test] + fn adopt_requires_key_and_two_approvers() { + let dir = tempdir().unwrap(); + let server = Server::boot(dir.path()); + server.create_source(); + let fenced = server.fence_source(); + let revision = fenced.record.as_ref().unwrap().revision; + server.run(async { + let state = FenceState::SourceWriteFenced; + let refusals = [ + (adopt_args(&["alice", "bob"]), false), + (adopt_args(&["alice"]), true), + (adopt_args(&["alice", "alice"]), true), + (adopt_args(&["alice", " alice "]), true), + (adopt_args(&["alice", ""]), true), + (adopt_args(&["alice", "bob", "carol"]), true), + ( + AdoptArgs { + incident_ref: " ".into(), + ..adopt_args(&["alice", "bob"]) + }, + true, + ), + ( + AdoptArgs { + reason: String::new(), + ..adopt_args(&["alice", "bob"]) + }, + true, + ), + ]; + for (i, (args, authorised)) in refusals.into_iter().enumerate() { + let request = adopt("ns", 10 + i as u128, state, revision, args.clone()); + let result = run_adopt(&server.store, request, authorised).await; + let e = fence_error(&result); + assert_eq!( + e.outcome(), + FenceOutcome::FencePreconditionFailed, + "{args:?}" + ); + assert_eq!( + e.detail(), + Some(FenceDetail::AdoptionNotAuthorised), + "{args:?}" + ); + } + // Nothing changed: owner, revision, receipts. + let inspection = server.inspect().await; + let record = inspection.fence.record().unwrap(); + assert_eq!((record.operation_id, record.revision), (OP, revision)); + assert!(record.adoptions.is_empty()); + assert!(inspection + .receipts + .iter() + .all(|r| r.operation_id == OP.to_string())); + }); + server.crash(); + } + + /// Adopting a write-fenced source changes the owner and the revision and nothing else: the + /// state and every admission stay as they were (the write generation moves, as on every + /// change of owner, section 7.2), SQL writes are still refused, the old owner is refused + /// and the new owner finishes the operation. A replay returns the stored receipt. + #[test] + fn adopt_keeps_gates_closed() { + let dir = tempdir().unwrap(); + let server = Server::boot(dir.path()); + server.create_source(); + let fenced = server.fence_source(); + let before = fenced.record.clone().unwrap(); + server.run(async { + let fence = server.fence().await; + let generation = fence.write_generation(); + let request = adopt( + "ns", + 10, + FenceState::SourceWriteFenced, + before.revision, + adopt_args(&["alice", "bob"]), + ); + let commit = run_adopt(&server.store, request.clone(), true) + .await + .unwrap(); + assert_eq!(commit.kind, FenceCommitKind::Committed); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + let after = commit.record.clone().unwrap(); + assert_eq!(after.operation_id, OTHER_OP); + assert_eq!(after.revision, before.revision + 1); + assert_eq!(after.state, before.state); + assert_eq!(after.frozen_boundary, before.frozen_boundary); + assert_eq!(after.identity, before.identity); + assert_eq!(after.legacy_blocks, before.legacy_blocks); + assert_eq!(after.adoptions.len(), 1); + let adoption = commit.receipt.adoption.as_ref().unwrap(); + assert_eq!(adoption.previous_operation_id, OP); + assert_eq!(adoption.approvers, vec!["alice", "bob"]); + assert_eq!(&after.adoptions[0], adoption); + + // The gate: same state and admissions, the new owner and revision. + let gate = fence.gate(); + assert_eq!(gate.state(), FenceState::SourceWriteFenced); + assert_eq!(gate.operation_id(), Some(OTHER_OP)); + assert_eq!(gate.revision(), before.revision + 1); + assert!(!gate.write().is_open()); + assert!(gate.read().is_open()); + assert_eq!(fence.write_generation(), generation + 1); + assert!(!server.writes_admitted().await); + assert_eq!(server.count().await, ROWS); + // The marker follows the record. + let marker = fence_store::read_marker(&dir.path().join("dbs"), &"ns".into()) + .unwrap() + .unwrap(); + assert_eq!(marker.unwrap().record, after); + + // Replay, with or without the key: the stored receipt, nothing moves. + for authorised in [true, false] { + let replay = run_adopt(&server.store, request.clone(), authorised) + .await + .unwrap(); + assert_eq!(replay.kind, FenceCommitKind::Replayed); + assert_eq!(replay.receipt, commit.receipt); + } + assert_eq!(fence.write_generation(), generation + 1); + + // The old owner is refused. + let e = server + .execute(release(20, after.revision)) + .await + .unwrap() + .unwrap_err(); + assert_eq!( + fence_error(&Err(e)).outcome(), + FenceOutcome::FenceOwnedByAnotherOperation + ); + assert!(!server.writes_admitted().await); + // The new owner finishes the operation. + let released = server + .execute(release_by(OTHER_OP, 21, after.revision)) + .await + .unwrap() + .unwrap(); + assert_eq!(released.receipt.outcome, FenceOutcome::Applied); + assert_eq!(fence.gate().state(), FenceState::Released); + assert!(server.writes_admitted().await); + }); + server.crash(); + } + + /// Adopting a quarantined target moves its import capability to the new owner: the old + /// owner can no longer open an import session, the new owner can and imports, and normal + /// SQL is still refused with `MIGRATION_TARGET_QUARANTINED`. + #[tokio::test(flavor = "multi_thread")] + async fn adopt_quarantined_target() { + let dir = tempdir().unwrap(); + let store = open_store(dir.path()).await; + create(&store, create_request("tgt", 1)) + .await + .unwrap() + .unwrap(); + let fence = store.fence_controller(&"tgt".into()); + let request = adopt( + "tgt", + 10, + FenceState::TargetQuarantined, + 1, + adopt_args(&["alice", "bob"]), + ); + let commit = run_adopt(&store, request, true).await.unwrap(); + let after = commit.record.unwrap(); + assert_eq!( + (after.state, after.revision, after.operation_id), + (FenceState::TargetQuarantined, 2, OTHER_OP) + ); + assert_eq!(fence.gate().state(), FenceState::TargetQuarantined); + for class in [OperationClass::NormalRead, OperationClass::NormalWrite] { + assert_eq!( + fence.permits(class).unwrap_err().outcome(), + FenceOutcome::MigrationTargetQuarantined + ); + } + + let e = store + .open_import_session("tgt".into(), OP, 2) + .await + .err() + .expect("the old owner has no import capability"); + assert_eq!( + fence_error(&Err(e)).outcome(), + FenceOutcome::FenceOwnedByAnotherOperation + ); + let mut session = store + .open_import_session("tgt".into(), OTHER_OP, 2) + .await + .unwrap(); + session + .with_raw(|conn| conn.execute_batch("create table t (x); insert into t values (1)")) + .await + .unwrap() + .unwrap(); + } + + /// Finished operations cannot be adopted: `TARGET_WRITABLE`, `TARGET_ABORTED` and a + /// `RELEASED` source are refused with `operation_finished`, and nothing changes. + #[tokio::test(flavor = "multi_thread")] + async fn adopt_cannot_touch_writable() { + // TARGET_WRITABLE. + let (_dir, store, fence) = write_fenced_target().await; + execute_target(&store, enable_request(30)) + .await + .unwrap() + .unwrap(); + assert_eq!(fence.gate().state(), FenceState::TargetWritable); + let revision = fence.gate().revision(); + let request = adopt( + "tgt", + 40, + FenceState::TargetWritable, + revision, + adopt_args(&["alice", "bob"]), + ); + let e = fence_error(&run_adopt(&store, request, true).await).clone(); + assert_eq!(e.outcome(), FenceOutcome::InvalidFenceTransition); + assert_eq!(e.detail(), Some(FenceDetail::OperationFinished)); + assert_eq!(fence.gate().operation_id(), Some(OP)); + assert_eq!(fence.gate().revision(), revision); + + // TARGET_ABORTED. + let dir = tempdir().unwrap(); + let store = open_store(dir.path()).await; + create(&store, create_request("tgt", 1)) + .await + .unwrap() + .unwrap(); + let abort = target_command( + 2, + FenceState::TargetQuarantined, + 1, + FenceCommand::AbortQuarantinedTarget, + ); + execute_target(&store, abort).await.unwrap().unwrap(); + let fence = store.fence_controller(&"tgt".into()); + assert_eq!(fence.gate().state(), FenceState::TargetAborted); + let request = adopt( + "tgt", + 3, + FenceState::TargetAborted, + 2, + adopt_args(&["alice", "bob"]), + ); + let e = fence_error(&run_adopt(&store, request, true).await).clone(); + assert_eq!(e.outcome(), FenceOutcome::InvalidFenceTransition); + assert_eq!(e.detail(), Some(FenceDetail::OperationFinished)); + assert_eq!(fence.gate().revision(), 2); + } + + /// A released source cannot be adopted either. + #[test] + fn adopt_cannot_touch_released() { + let dir = tempdir().unwrap(); + let server = Server::boot(dir.path()); + server.create_source(); + let revision = server.fence_source().record.unwrap().revision; + server.run(async { + server.execute(release(2, revision)).await.unwrap().unwrap(); + let request = adopt( + "ns", + 3, + FenceState::Released, + revision + 1, + adopt_args(&["alice", "bob"]), + ); + let e = fence_error(&run_adopt(&server.store, request, true).await).clone(); + assert_eq!(e.detail(), Some(FenceDetail::OperationFinished)); + }); + server.crash(); + } + + async fn raw_meta(dir: &Path) -> crate::namespace::meta_store::MetaStoreConnection { + let (maker, _) = metastore_connection_maker(None, dir).await.unwrap(); + maker().unwrap() + } + + fn unavailable_detail(result: crate::Result<()>) -> FenceDetail { + match result { + Err(Error::NamespaceFence(e)) => { + assert_eq!(e.outcome(), FenceOutcome::FenceStateUnavailable, "{e}"); + e.detail().unwrap() + } + other => panic!("expected FENCE_STATE_UNAVAILABLE, got {other:?}"), + } + } + + /// After a metastore rollback (a restore from a backup older than the fence), the + /// namespace is `UNKNOWN_UNAVAILABLE` with `metastore_behind_marker`. Adoption re-establishes + /// the record the marker holds, under the adopting operation, at the marker's revision + 1, + /// both when the fence row is gone (the backup predates the fence) and when it is at an + /// older revision (the backup predates the last transition). The namespace is then served + /// again behind the same gate, with its own `block_*` values in memory, and the new owner + /// finishes the operation. + #[test] + fn adopt_recovers_metastore_rollback() { + for row_behind in [false, true] { + let dir = tempdir().unwrap(); + let server = Server::boot(dir.path()); + server.create_source(); + let fenced = server.fence_source().record.unwrap(); + let saved_row: (i64, i64, Vec) = server.run(async { + raw_meta(dir.path()) + .await + .query_row( + "SELECT format_version, revision, record FROM namespace_fences", + (), + |r| Ok((r.get(0)?, r.get(1)?, r.get(2)?)), + ) + .unwrap() + }); + // With the row kept behind, one more transition moves the marker on. + let marker_record = if row_behind { + let read = FenceRequest { + namespace: "ns".into(), + operation_id: OP, + command_id: Uuid::from_u128(2), + expected_state: FenceState::SourceWriteFenced, + expected_revision: fenced.revision, + command: FenceCommand::SetSourceReadFence { + drain_policy: Some(LONG), + }, + }; + let commit = server.run(async { server.execute(read).await.unwrap().unwrap() }); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + commit.record.unwrap() + } else { + fenced.clone() + }; + server.crash(); + + // What restoring the metastore from an older backup leaves: no receipts, and the + // fence row gone or at the older revision. The config row keeps the legacy + // mirror the acquisition wrote into it (block_writes on). + server_run_blocking(|| async { + let conn = raw_meta(dir.path()).await; + conn.execute("DELETE FROM namespace_fence_receipts", ()) + .unwrap(); + if row_behind { + conn.execute( + "UPDATE namespace_fences SET format_version = ?1, revision = ?2, \ + record = ?3", + (saved_row.0, saved_row.1, &saved_row.2), + ) + .unwrap(); + } else { + conn.execute("DELETE FROM namespace_fences", ()).unwrap(); + } + }); + + let server = Server::boot(dir.path()); + server.run(async { + let store = &server.store; + assert_eq!( + unavailable_detail(store.with("ns".into(), |_| ()).await), + FenceDetail::MetastoreBehindMarker + ); + assert!(store.meta_store().lookup(&"ns".into()).await.is_err()); + + // The adoption has to name the marker's state, revision and owner. + let wrong_revision = adopt( + "ns", + 10, + marker_record.state, + saved_row.1 as u64, + adopt_args(&["alice", "bob"]), + ); + if row_behind { + let e = fence_error(&run_adopt(store, wrong_revision, true).await).clone(); + assert_eq!(e.outcome(), FenceOutcome::FenceRevisionMismatch); + } + let request = adopt( + "ns", + 11, + marker_record.state, + marker_record.revision, + adopt_args(&["alice", "bob"]), + ); + let e = fence_error(&run_adopt(store, request.clone(), false).await).clone(); + assert_eq!(e.detail(), Some(FenceDetail::AdoptionNotAuthorised)); + + let commit = run_adopt(store, request, true).await.unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + let adopted = commit.record.unwrap(); + assert_eq!(adopted.operation_id, OTHER_OP); + assert_eq!(adopted.revision, marker_record.revision + 1); + assert_eq!(adopted.state, marker_record.state); + assert_eq!(adopted.frozen_boundary, marker_record.frozen_boundary); + assert_eq!(adopted.adoptions.len(), 1); + + // Established again: durable, served, and behind the same gate. + let inspection = server.inspect().await; + assert_eq!(inspection.fence.record(), Some(&adopted)); + let fence = server.fence().await; + assert_eq!(fence.gate().state(), marker_record.state); + assert_eq!(fence.gate().operation_id(), Some(OTHER_OP)); + assert!(!server.writes_admitted().await); + let config = store.meta_store().lookup(&"ns".into()).await.unwrap(); + let config = config.unwrap().get(); + assert!(!config.block_writes, "the namespace's own values are back"); + assert!(!config.block_reads); + let marker = fence_store::read_marker(&dir.path().join("dbs"), &"ns".into()) + .unwrap() + .unwrap(); + assert_eq!(marker.unwrap().record, adopted); + + // The new owner finishes the operation. + let mut revision = adopted.revision; + if row_behind { + let clear = FenceRequest { + namespace: "ns".into(), + operation_id: OTHER_OP, + command_id: Uuid::from_u128(20), + expected_state: FenceState::SourceReadFenced, + expected_revision: revision, + command: FenceCommand::ClearSourceReadFence, + }; + revision = server + .execute(clear) + .await + .unwrap() + .unwrap() + .record + .unwrap() + .revision; + } + server + .execute(release_by(OTHER_OP, 21, revision)) + .await + .unwrap() + .unwrap(); + assert!(server.writes_admitted().await); + assert_eq!(server.count().await, ROWS); + }); + server.crash(); + } + } + + /// Run `f` on a throwaway runtime (between two server lifetimes). + fn server_run_blocking(f: F) + where + F: FnOnce() -> Fut, + Fut: Future, + { + tokio::runtime::Builder::new_current_thread() + .enable_all() + .build() + .unwrap() + .block_on(f()); + } + + /// A metastore restored from a backup that predates the namespace itself holds neither a + /// fence row nor a config row for it. Adoption is refused with + /// `FENCE_PRECONDITION_FAILED` / `namespace_config_missing` (after the authorisation + /// checks): the marker holds the fence record, not the namespace's configuration, and + /// adoption does not invent one. Nothing is written and the name stays unavailable. + #[test] + fn adopt_recovered_name_without_config_row() { + let dir = tempdir().unwrap(); + let server = Server::boot(dir.path()); + server.create_source(); + let fenced = server.fence_source().record.unwrap(); + server.crash(); + let marker_before = read_marker_bytes(&dir.path().join("dbs")); + server_run_blocking(|| async { + let conn = raw_meta(dir.path()).await; + for sql in [ + "DELETE FROM namespace_fence_receipts", + "DELETE FROM namespace_fences", + "DELETE FROM namespace_configs", + ] { + conn.execute(sql, ()).unwrap(); + } + }); + + let server = Server::boot(dir.path()); + server.run(async { + let store = &server.store; + assert_eq!( + unavailable_detail(store.with("ns".into(), |_| ()).await), + FenceDetail::MetastoreBehindMarker + ); + let request = adopt( + "ns", + 10, + fenced.state, + fenced.revision, + adopt_args(&["alice", "bob"]), + ); + let e = fence_error(&run_adopt(store, request.clone(), false).await).clone(); + assert_eq!(e.detail(), Some(FenceDetail::AdoptionNotAuthorised)); + let e = fence_error(&run_adopt(store, request, true).await).clone(); + assert_eq!(e.outcome(), FenceOutcome::FencePreconditionFailed); + assert_eq!(e.detail(), Some(FenceDetail::NamespaceConfigMissing)); + + // Nothing was written, and the name is still unavailable. + let conn = raw_meta(dir.path()).await; + for table in [ + "namespace_fences", + "namespace_fence_receipts", + "namespace_configs", + ] { + let n: i64 = conn + .query_row(&format!("SELECT count(*) FROM {table}"), (), |r| r.get(0)) + .unwrap(); + assert_eq!(n, 0, "{table}"); + } + assert_eq!(read_marker_bytes(&dir.path().join("dbs")), marker_before); + assert_eq!( + unavailable_detail(store.with("ns".into(), |_| ()).await), + FenceDetail::MetastoreBehindMarker + ); + assert!(store.meta_store().handle("ns".into()).await.is_err()); + assert!(store.fence_controller(&"ns".into()).gate().is_unavailable()); + }); + server.crash(); + } +} diff --git a/libsql-server/src/namespace/meta_store.rs b/libsql-server/src/namespace/meta_store.rs index a5de15a4bf..fe162f64a4 100644 --- a/libsql-server/src/namespace/meta_store.rs +++ b/libsql-server/src/namespace/meta_store.rs @@ -23,7 +23,7 @@ use tokio::sync::{ }; use uuid::Uuid; -use crate::config::BottomlessConfig; +use crate::config::{BottomlessConfig, FenceAdoptionKey}; use crate::connection::config::DatabaseConfig; use crate::database::DatabaseKind; use crate::schema::{MigrationDetails, MigrationSummary}; @@ -100,6 +100,8 @@ struct MetaStoreInner { /// Where this metastore's contents came from at startup, recorded once the server has /// opened it (section 13.3). restore_provenance: std::sync::OnceLock, + /// The secret that authorises `AdoptFence` (section 12); `None` disables adoption. + fence_adoption_key: Option, } /// Whether the metastore was restored from its bottomless backup when the server started @@ -304,6 +306,7 @@ impl MetaStoreInner { fence, recovered: Default::default(), restore_provenance: Default::default(), + fence_adoption_key: config.namespace_fence_adoption_key.clone(), }; if config.allow_recover_from_fs { @@ -967,6 +970,19 @@ fn apply_fence_command( created_config = Some(Arc::new(logical)); } else { let Some(config) = &config else { + if let FenceCommand::AdoptFence(_) = &request.command { + // Section 12: adoption changes the owner and nothing else. The marker holds + // the fence record, not the namespace's configuration (its JWT key, size + // limit, durability, backup id), so re-establishing the record would mean + // inventing a configuration. The namespace stays unavailable. + return Err(FenceError::new( + FenceOutcome::FencePreconditionFailed, + "the metastore holds no configuration for this namespace; adoption \ + re-establishes a fence, not a namespace configuration", + ) + .with_detail(FenceDetail::NamespaceConfigMissing) + .into()); + } return Err(FenceError::new( FenceOutcome::FenceStateUnavailable, "the fenced namespace has no config row", @@ -992,6 +1008,13 @@ fn apply_fence_command( // The command established the fence from the durable state; whatever startup could not // recover about this name is settled. inner.recovered.lock().remove(ns); + if let (StoredFence::Unavailable { .. }, Some(next)) = (&stored, &record) { + // An adoption re-established the record the marker held (section 12). While the name + // was unavailable its in-memory config kept whatever the stale config row held; from + // now on it carries the namespace's own values, as `restore_fences` gives every + // established record at startup. + restore_own_blocks(inner, ns, next); + } let current = record.or_else(|| stored.record().cloned()); after_fence_commit( @@ -1009,6 +1032,18 @@ fn apply_fence_command( }) } +/// Put the namespace's own `block_*` values (the record's saved values) into its in-memory +/// config, if it has one. The caller holds the connection lock, which is taken before the +/// config map's everywhere. +fn restore_own_blocks(inner: &MetaStoreInner, ns: &NamespaceName, record: &NamespaceFenceRecord) { + let configs = inner.configs.blocking_lock(); + if let Some(sender) = configs.get(ns) { + let config = sender.borrow().config.clone(); + let config = fence_store::with_legacy_blocks(&config, &record.legacy_blocks); + sender.send_modify(|c| c.config = Arc::new(config)); + } +} + /// Whether the marker has to be written after a commit: the record changed, or it had fallen /// behind. fn record_changed( @@ -1400,6 +1435,16 @@ impl MetaStore { self.inner.fence.enabled } + /// Whether `presented`, the value of a request's `x-libsql-fence-adoption-key` header, + /// authorises `AdoptFence` (section 12): an adoption key is configured and `presented` is + /// that key. Compared in constant time. + pub fn fence_adoption_authorised(&self, presented: Option<&[u8]>) -> bool { + match (&self.inner.fence_adoption_key, presented) { + (Some(key), Some(presented)) => key.matches(presented), + _ => false, + } + } + /// Records where this metastore's contents came from at startup, as reported by /// [`metastore_connection_maker_with_provenance`], then sets the /// `libsql_server_metastore_restored_from_backup` gauge and, after a restore from backup, diff --git a/libsql-server/src/namespace/store.rs b/libsql-server/src/namespace/store.rs index aa22de9e26..0ac3c42465 100644 --- a/libsql-server/src/namespace/store.rs +++ b/libsql-server/src/namespace/store.rs @@ -32,7 +32,9 @@ use super::fence::registry::FenceRegistry; use super::fence::state::{FenceState, Role}; use super::fence::store::StoredFence; use super::fence::target::{self, CreateTargetRequest, ValidationSession}; -use super::meta_store::{FenceCommit, FenceContext, FenceInspection, MetaStore, MetaStoreHandle}; +use super::meta_store::{ + FenceCommit, FenceCommitKind, FenceContext, FenceInspection, MetaStore, MetaStoreHandle, +}; use super::schema_lock::SchemaLocksRegistry; use super::{Namespace, ResetCb, ResetOp, ResolveNamespacePathFn, RestoreOption}; @@ -633,6 +635,33 @@ impl NamespaceStore { request: FenceRequest, server: ServerIdentity, ) -> crate::Result { + self.execute_fence_command_authorised(request, server, false) + .await + } + + /// [`execute_fence_command`](Self::execute_fence_command) for a request that may carry + /// the adoption key: `adoption_authorised` says whether it did (section 12). Only + /// `AdoptFence` looks at it. A committed adoption is written to the audit log (target + /// `libsql_server::fence::audit`) with its approvers, incident reference and reason. + pub(crate) async fn execute_fence_command_authorised( + &self, + request: FenceRequest, + server: ServerIdentity, + adoption_authorised: bool, + ) -> crate::Result { + if let FenceCommand::AdoptFence(_) = &request.command { + let namespace = request.namespace.clone(); + let controller = self.inner.fences.controller(&namespace); + let mut ctx = FenceContext::now(server, None); + ctx.adoption_authorised = adoption_authorised; + let commit = controller + .execute(&self.inner.metadata, request, ctx) + .await?; + if commit.kind == FenceCommitKind::Committed { + super::fence::audit::adoption(&namespace, &commit); + } + return Ok(commit); + } let controller = match request.command { FenceCommand::CreateTargetQuarantined { .. } => { return self.run_create_target(request, server).await diff --git a/libsql-server/tests/fence/admin.rs b/libsql-server/tests/fence/admin.rs index 49669fa168..365ecb1fae 100644 --- a/libsql-server/tests/fence/admin.rs +++ b/libsql-server/tests/fence/admin.rs @@ -45,6 +45,7 @@ fn capabilities() { "PublishTargetReadableWriteFenced", "EnableTargetWrites", "AbortQuarantinedTarget", + "AdoptFence", ] { assert!(commands.contains(&command), "{command} missing: {body}"); } @@ -735,3 +736,185 @@ fn inspect_reports_drain_counters() { }); sim.run().unwrap(); } + +const ADOPTION_KEY: &str = "fence-adoption-key"; + +fn adopt_body( + new_op: Uuid, + cmd: Uuid, + state: &str, + rev: u64, + current_op: Uuid, +) -> serde_json::Value { + command_body( + new_op, + cmd, + state, + rev, + json!({ + "current_operation_id": current_op.to_string(), + "approvers": ["alice", "bob"], + "incident_ref": "INC-1", + "reason": "the operation's control record was lost", + }), + ) +} + +/// `AdoptFence` over HTTP (section 12): the admin credential and the adoption key header are +/// both required, two distinct approvers are required, and an adoption moves the owner and +/// nothing else. The old owner is refused afterwards and the new one finishes the operation. +#[test] +fn adopt_over_http() { + let mut sim = sim(); + let tmp = tempdir().unwrap(); + make_primary( + &mut sim, + tmp.path().to_path_buf(), + Primary { + adoption_key: Some(ADOPTION_KEY), + ..Default::default() + }, + ); + sim.client("client", async { + let admin = Admin::new(Some(ADMIN_KEY)); + admin.create_namespace("src").await?; + let log_id = load_and_log_id(&admin, "src").await?; + let (old, new) = (uuid(0xa), uuid(0xb)); + let conn = connect("src")?; + let (status, acquired) = admin + .command( + "src", + "source/acquire-write-fence", + acquire_body(old, uuid(1), &log_id), + ) + .await?; + assert_eq!(status, StatusCode::OK, "{acquired}"); + assert_eq!(state_of(&acquired), ("SOURCE_WRITE_FENCED", 2)); + + let body = adopt_body(new, uuid(2), "SOURCE_WRITE_FENCED", 2, old); + // No key, a wrong key, and the admin credential missing. + for key in [None, Some("not-the-key")] { + let (status, refused) = admin.adopt("src", body.clone(), key).await?; + assert_eq!(status, StatusCode::PRECONDITION_FAILED, "{refused}"); + assert_eq!(refused["outcome"], "FENCE_PRECONDITION_FAILED"); + assert_eq!(refused["detail"], "adoption_not_authorised", "{refused}"); + assert_eq!(state_of(&refused), ("SOURCE_WRITE_FENCED", 2), "{refused}"); + } + let (status, _) = Admin::new(None) + .adopt("src", body.clone(), Some(ADOPTION_KEY)) + .await?; + assert_eq!(status, StatusCode::UNAUTHORIZED); + // One approver, the same approver twice, and an unknown field. + for approvers in [json!(["alice"]), json!(["alice", "alice"])] { + let mut bad = adopt_body(new, uuid(3), "SOURCE_WRITE_FENCED", 2, old); + bad["approvers"] = approvers; + let (status, refused) = admin.adopt("src", bad, Some(ADOPTION_KEY)).await?; + assert_eq!(status, StatusCode::PRECONDITION_FAILED, "{refused}"); + assert_eq!(refused["detail"], "adoption_not_authorised", "{refused}"); + } + let mut bad = adopt_body(new, uuid(3), "SOURCE_WRITE_FENCED", 2, old); + bad["gate"] = json!("open"); + let (status, refused) = admin.adopt("src", bad, Some(ADOPTION_KEY)).await?; + assert_eq!(status, StatusCode::PRECONDITION_FAILED, "{refused}"); + assert_eq!(refused["detail"], "invalid_argument", "{refused}"); + + let (status, adopted) = admin.adopt("src", body.clone(), Some(ADOPTION_KEY)).await?; + assert_eq!(status, StatusCode::OK, "{adopted}"); + assert_eq!(adopted["outcome"], "APPLIED"); + assert_eq!(adopted["replayed"], false); + assert_eq!(state_of(&adopted), ("SOURCE_WRITE_FENCED", 3)); + assert_eq!(adopted["fence"]["operation_id"], new.to_string()); + assert_eq!(adopted["fence"]["admission"]["write"], "closed"); + assert_eq!(adopted["fence"]["admission"]["read"], "open"); + assert_eq!( + adopted["fence"]["frozen_boundary"], + acquired["fence"]["frozen_boundary"] + ); + assert_eq!(adopted["receipt"]["command"], "AdoptFence"); + let adoption = &adopted["receipt"]["adoption"]; + assert_eq!(adoption["previous_operation_id"], old.to_string()); + assert_eq!(adoption["new_operation_id"], new.to_string()); + assert_eq!(adoption["approvers"], json!(["alice", "bob"])); + assert_eq!(adoption["incident_ref"], "INC-1"); + assert_eq!(adopted["fence"]["adoptions"][0], *adoption); + // Writes stay closed; reads stay open. + assert!(conn.execute("insert into t values (2)", ()).await.is_err()); + conn.query("select * from t", ()).await?; + + // A replay returns the stored receipt, even without the key. + let (status, replay) = admin.adopt("src", body.clone(), None).await?; + assert_eq!(status, StatusCode::OK, "{replay}"); + assert_eq!(replay["replayed"], true); + assert_eq!(replay["receipt"], adopted["receipt"]); + + // The old owner is refused; the new owner finishes the operation. + let (status, refused) = admin + .command( + "src", + "source/release-write-fence", + command_body(old, uuid(4), "SOURCE_WRITE_FENCED", 3, json!({})), + ) + .await?; + assert_eq!(status, StatusCode::CONFLICT, "{refused}"); + assert_eq!(refused["outcome"], "FENCE_OWNED_BY_ANOTHER_OPERATION"); + let (status, released) = admin + .command( + "src", + "source/release-write-fence", + command_body(new, uuid(5), "SOURCE_WRITE_FENCED", 3, json!({})), + ) + .await?; + assert_eq!(status, StatusCode::OK, "{released}"); + assert_eq!(state_of(&released), ("RELEASED", 4)); + conn.execute("insert into t values (3)", ()).await?; + + // A finished operation cannot be adopted. + let (status, refused) = admin + .adopt( + "src", + adopt_body(uuid(0xc), uuid(6), "RELEASED", 4, new), + Some(ADOPTION_KEY), + ) + .await?; + assert_eq!(status, StatusCode::CONFLICT, "{refused}"); + assert_eq!(refused["outcome"], "INVALID_FENCE_TRANSITION"); + assert_eq!(refused["detail"], "operation_finished", "{refused}"); + Ok(()) + }); + sim.run().unwrap(); +} + +/// Without `--namespace-fence-adoption-key`, adoption is disabled whatever the request presents. +#[test] +fn adopt_disabled_without_key() { + let mut sim = sim(); + let tmp = tempdir().unwrap(); + make_primary(&mut sim, tmp.path().to_path_buf(), Primary::default()); + sim.client("client", async { + let admin = Admin::new(Some(ADMIN_KEY)); + admin.create_namespace("src").await?; + let log_id = load_and_log_id(&admin, "src").await?; + let (status, acquired) = admin + .command( + "src", + "source/acquire-write-fence", + acquire_body(uuid(0xa), uuid(1), &log_id), + ) + .await?; + assert_eq!(status, StatusCode::OK, "{acquired}"); + for key in [None, Some(""), Some(ADOPTION_KEY)] { + let (status, refused) = admin + .adopt( + "src", + adopt_body(uuid(0xb), uuid(2), "SOURCE_WRITE_FENCED", 2, uuid(0xa)), + key, + ) + .await?; + assert_eq!(status, StatusCode::PRECONDITION_FAILED, "{refused}"); + assert_eq!(refused["detail"], "adoption_not_authorised", "{refused}"); + assert_eq!(refused["fence"]["operation_id"], uuid(0xa).to_string()); + } + Ok(()) + }); + sim.run().unwrap(); +} diff --git a/libsql-server/tests/fence/mod.rs b/libsql-server/tests/fence/mod.rs index 2e9cbae183..5e97ed7ec1 100644 --- a/libsql-server/tests/fence/mod.rs +++ b/libsql-server/tests/fence/mod.rs @@ -13,7 +13,8 @@ use hyper::StatusCode; use libsql_server::auth::user_auth_strategies::http_basic::HttpBasic; use libsql_server::auth::Auth; use libsql_server::config::{ - AdminApiConfig, MetaStoreConfig, RpcClientConfig, RpcServerConfig, UserApiConfig, + AdminApiConfig, FenceAdoptionKey, MetaStoreConfig, RpcClientConfig, RpcServerConfig, + UserApiConfig, }; use s3s::header::AUTHORIZATION; use serde_json::{json, Value}; @@ -33,6 +34,8 @@ pub struct Primary { pub fence_enabled: bool, /// A basic-auth credential the user API requires; `None` leaves it unauthenticated. pub user_credential: Option<&'static str>, + /// The fence adoption key; `None` leaves adoption disabled. + pub adoption_key: Option<&'static str>, } impl Default for Primary { @@ -41,6 +44,7 @@ impl Default for Primary { admin_key: Some(ADMIN_KEY), fence_enabled: true, user_credential: None, + adoption_key: None, } } } @@ -58,6 +62,7 @@ pub fn make_primary(sim: &mut Sim, path: PathBuf, primary: Primary) { admin_key, fence_enabled, user_credential, + adoption_key, } = primary; sim.host("primary", move || { let path = path.clone(); @@ -83,6 +88,7 @@ pub fn make_primary(sim: &mut Sim, path: PathBuf, primary: Primary) { }), meta_store_config: MetaStoreConfig { namespace_fence: fence_enabled, + namespace_fence_adoption_key: adoption_key.and_then(FenceAdoptionKey::new), ..Default::default() }, disable_namespaces: false, @@ -195,6 +201,22 @@ impl Admin { self.get(&format!("/v1/namespaces/{ns}/fence")).await } + /// `AdoptFence`, presenting `adoption_key` (if any) in the adoption key header. + pub async fn adopt( + &self, + ns: &str, + body: Value, + adoption_key: Option<&str>, + ) -> anyhow::Result<(StatusCode, Value)> { + let url = format!("http://primary:9090/v1/namespaces/{ns}/fence/adopt"); + let mut headers = self.headers(); + let name = hyper::header::HeaderName::from_static("x-libsql-fence-adoption-key"); + if let Some(key) = adoption_key { + headers.push((name, key)); + } + Self::json(self.client.post_with_headers(&url, &headers, body).await?).await + } + /// A fence command: `route` is the part after `/fence/`. pub async fn command( &self, From 7835acccaa7e912ff045d042b79acd4b486344e7 Mon Sep 17 00:00:00 2001 From: River Date: Wed, 30 Sep 2026 11:36:57 +0000 Subject: [PATCH 27/33] libsql-server: test legacy fence mirror and read/target crash boundaries Walk a source and two targets through every stored fence state and check the protection an older binary gets (docs/NAMESPACE_FENCE.md section 13.2): the config row's block_* fields hold the fence's mirror and nothing else in the row changes, a config write through the metastore is refused and changes nothing, and an older binary's delete of the config row fails on the fence row's foreign key. Release and write enable put the namespace's own values back. The test found that a restart overwrote the in-memory config of a released source or a writable target with the block_* values saved when the fence was acquired, although config writes after the operation had stored the namespace's own values in the row since: a namespace blocked after its release could come back unblocked. Loading the metastore, the target config publication and adoption now take the namespace's own config through fence_store::own_config, which uses the row as it is once the record no longer mirrors the fence. Crash the server at each persistence boundary of SetSourceReadFence, SealTargetImport and EnableTargetWrites, and while a read or seal drain waits for a reader or an import call, then restart it on the same directory: the prior or the committed state is recovered with no in-memory gate, reader or import call left, reads and import stay closed once their draining state committed, target writes open only if TARGET_WRITABLE committed, the marker is repaired, and a replay of the same command completes an interrupted drain at once. Co-authored-by: Tomasz Szymczyszyn --- docs/NAMESPACE_FENCE.md | 12 +- libsql-server/src/namespace/fence/record.rs | 14 +- libsql-server/src/namespace/fence/store.rs | 12 + libsql-server/src/namespace/fence/tests.rs | 1122 ++++++++++++++++++- libsql-server/src/namespace/meta_store.rs | 20 +- 5 files changed, 1158 insertions(+), 22 deletions(-) diff --git a/docs/NAMESPACE_FENCE.md b/docs/NAMESPACE_FENCE.md index 2844770d29..eda47c556c 100644 --- a/docs/NAMESPACE_FENCE.md +++ b/docs/NAMESPACE_FENCE.md @@ -344,7 +344,7 @@ All namespace-config writes already pass through the metastore's single `MetaSto Ordinary config writes (`try_process`) and `remove` read the fence row inside their own transaction and refuse when the state denies lifecycle operations. Because both take `BEGIN IMMEDIATE` on the same database, a config write cannot interleave with a fence transition. -While a record is in force, the stored config row carries the legacy mirror (section 13.2) and the in-memory config is the namespace's own configuration: a transition overlays the mirror on the config row *as read inside its transaction*, so a config change committed by another metastore connection is preserved underneath it, and loading the metastore puts the saved `block_*` values back into the in-memory config. A namespace whose fence cannot be established keeps its stored row, mirror included, in memory. +While a record is in force, the stored config row carries the legacy mirror (section 13.2) and the in-memory config is the namespace's own configuration: a transition overlays the mirror on the config row *as read inside its transaction*, so a config change committed by another metastore connection is preserved underneath it, and loading the metastore puts the saved `block_*` values back into the in-memory config. Once the operation has finished (`RELEASED`, `TARGET_WRITABLE`) the row holds the namespace's own values again, including any config written since, so loading the metastore uses the row as it is (`fence_store::own_config`) rather than the values saved when the fence was acquired. A namespace whose fence cannot be established keeps its stored row, mirror included, in memory. A fence command that returns an error writes nothing. The store answers a replay or a resumed drain without writing; otherwise it commits the record (when it changed), the receipt and the mirror together, writes the marker, and only then returns. `CreateTargetQuarantined` returns the created namespace config without publishing it; the caller publishes it after installing the target's gate (section 10.1). @@ -549,7 +549,7 @@ This makes the WAL gate independent of statement classification: DDL, misclassif - At startup the registry is built from the metastore and markers before `NamespaceStore` serves anything. `make_namespace` takes the controller from the registry and passes it into the configurator's `setup()` (primary, schema and replica), down to `MakeLegacyConnection::new`, which binds a `FenceConnState` to it for every `LegacyConnection` it opens, **starting with** the maker's held `_db` connection. The `Namespace` keeps the same controller (`Namespace::fence()`), which is how dump, replication and lifecycle code, all of which reach a namespace through `NamespaceStore::with`, read its gate. - `NamespaceStore::with` and `make_namespace` check the registry before `lookup()`, `handle()` or any setup: `UNKNOWN_UNAVAILABLE` is refused before any setup work. - Idle or capacity eviction shuts the namespace down but leaves the controller in the registry; a lazy reload reinstalls the identical gate, revision and generation. A drain waiter that holds the evicted manager sees its connections close and is notified. -- Namespaces in `SOURCE_DRAINING` or `TARGET_IMPORT_DRAINING` after a restart stay closed until the same command is replayed. Nothing advances in the background. +- Namespaces in `SOURCE_DRAINING`, `SOURCE_READ_DRAINING` or `TARGET_IMPORT_DRAINING` after a restart stay closed until the same command is replayed. Nothing advances in the background. No reader, stream or import call survives a restart, so the replay completes the drain at once. - A restart at any persistence boundary recovers either the state before the command or the state it committed, never anything in between, and never an open namespace unless an opening transition had committed: the `INSTALLING` gate and an indeterminate flag are in memory only, so a crash before the commit recovers the prior state (which was never acknowledged as closed), and a crash after it recovers the committed one. A marker that fell behind (a crash between the metastore commit and the marker write) is repaired when the fence is loaded. - **A crash rebuilds the source's replication log.** A namespace that was not shut down cleanly is recovered by rebuilding its replication log from the database file under a new `log_id` (existing behaviour, not specific to fences). For the fence this means: - Crash before `SOURCE_DRAINING` committed: nothing was written or acknowledged. A replay of the same acquisition is refused with `FENCE_PRECONDITION_FAILED`/`namespace_identity_mismatch`, before anything is written, because the log id the caller observed is gone; the caller reads the new identity and acquires with a new command. @@ -743,8 +743,8 @@ Upgrade order: deploy a binary with capability discovery and proxy `stable_code` An older binary does not know the fence tables. While a record is active: -- the config row's `block_reads`, `block_writes` and `block_reason` are mirrored from the fence state in the same transaction (`block_writes` in every state that denies writes, `block_reads` in every state that denies reads, `block_reason = "namespace fence: (operation )"`), and restored when the operation finishes; -- the foreign key from `namespace_fences` to `namespace_configs` makes an older binary's namespace delete fail. +- the config row's `block_reads`, `block_writes` and `block_reason` are mirrored from the fence state in the same transaction (`block_writes` in every state that denies writes, `block_reads` in every state that denies reads, `block_reason = "namespace fence: (operation )"`), and restored when the operation finishes (`ReleaseSourceWriteFence`, `EnableTargetWrites`). Nothing else in the row changes, and while the fence denies lifecycle work a config write through the metastore is refused in its transaction, so it cannot overwrite the mirror. After the operation has finished, config writes follow the existing policy again and the row holds whatever they store; +- the foreign key from `namespace_fences` to `namespace_configs` (`ON DELETE RESTRICT`) makes an older binary's namespace delete fail with `FOREIGN KEY constraint failed` (the older binary enables foreign keys on its metastore connection). The fence row, and so the guard, stays until this binary deletes the namespace, which removes the fence row and its receipts in the same transaction: an older binary cannot delete a namespace whose operation has finished either. This is best effort. An older binary applies `block_*` at statement level only, lets its admin shell and `/dump` bypass them, and would let a config update overwrite them. It is a mitigation for an accidental rollback, not a guarantee; the guarantee is the deployment order above. @@ -850,7 +850,7 @@ Planned test names; the table is updated as tests land. | 4 | Program that captured config before the fence is rejected at the WAL | landed: `connection::connection_manager::fence_tests::wal_gate_rejects_program_admitted_before_fence` (a SQL function parks the program between admission and its write while the fence is acquired and released) | | 5 | Pre-fence transactions cannot write after release or publication | landed: `connection::connection_manager::fence_tests::stale_generation_cannot_write_after_release`, `namespace::fence::target::tests::stale_generation_cannot_write_after_enable_writes`; unfenced behaviour unchanged: `unfenced_namespace_is_unchanged` | | 6 | Acquisition timeout returns `DRAINING`, admission stays closed | landed: `namespace::fence::drain::tests::{deadline_returns_draining_and_stays_closed, replay_of_draining_resumes_and_completes}` | -| 7 | Restart at every persistence boundary; indeterminate persistence keeps the gate closed until same-command reconciliation | landed: `namespace::fence::tests::restart_at_each_boundary` (a crash at each of 17 boundaries of `AcquireSourceWriteFence` and `ReleaseSourceWriteFence`: after the `INSTALLING` gate, before, at and after each metastore commit, with and without a lagging marker, before publication, before boundary capture and before the response; the restart recovers the prior or the committed state, admits writes only if an opening transition committed, repairs the marker, and a replay finishes the command), `restart_in_draining_waits_for_the_same_command` (a writer active at the crash; nothing advances until the same command is replayed, which completes at once), `indeterminate_commit_keeps_gate_closed` (through the drain path, both when the commit happened and when it did not), `acquire_response_loss_resolved_by_replay_and_inspect`; `fence::controller::tests::{indeterminate_commit_keeps_writes_closed_until_replayed, indeterminate_commit_that_did_not_apply_is_retried_by_replay, failed_before_commit_leaves_gate_unchanged, publication_happens_before_the_response, committed_command_is_published_when_the_caller_goes_away}`, `namespace::store::fence_tests::restart_installs_the_durable_gate_before_serving`; the rebuilt log after a restart: `namespace::fence::tests::current_log_id_is_the_rebuilt_log_after_restart` (the stored record keeps the acquisition log and its boundary; the controller's `current_log_id`, which the admin view reports, names the rebuilt log) | +| 7 | Restart at every persistence boundary; indeterminate persistence keeps the gate closed until same-command reconciliation | landed: `namespace::fence::tests::restart_at_each_boundary` (a crash at each of 17 boundaries of `AcquireSourceWriteFence` and `ReleaseSourceWriteFence`: after the `INSTALLING` gate, before, at and after each metastore commit, with and without a lagging marker, before publication, before boundary capture and before the response; the restart recovers the prior or the committed state, admits writes only if an opening transition committed, repairs the marker, and a replay finishes the command), `restart_in_draining_waits_for_the_same_command` (a writer active at the crash; nothing advances until the same command is replayed, which completes at once), `indeterminate_commit_keeps_gate_closed` (through the drain path, both when the commit happened and when it did not), `acquire_response_loss_resolved_by_replay_and_inspect`; `fence::controller::tests::{indeterminate_commit_keeps_writes_closed_until_replayed, indeterminate_commit_that_did_not_apply_is_retried_by_replay, failed_before_commit_leaves_gate_unchanged, publication_happens_before_the_response, committed_command_is_published_when_the_caller_goes_away}`, `namespace::store::fence_tests::restart_installs_the_durable_gate_before_serving`; the read fence, the seal and write enable: `namespace::fence::tests::read_and_target_boundaries::restart_at_each_read_and_target_boundary` (a crash at each of 29 boundaries of `SetSourceReadFence`, `SealTargetImport` and `EnableTargetWrites`: after the read-closing or `INSTALLING` gate, before, at and after each metastore commit, with and without a lagging marker, before publication, before the response, and while the drain waits for a reader or an import call that was running when it started; the restart recovers the prior or the committed state with no in-memory gate, reader or import call surviving, keeps reads closed once `SOURCE_READ_DRAINING` committed and import closed once `TARGET_IMPORT_DRAINING` committed, admits target writes only if `TARGET_WRITABLE` committed, repairs the marker, and a replay completes an interrupted drain at once or returns the stored result); the rebuilt log after a restart: `namespace::fence::tests::current_log_id_is_the_rebuilt_log_after_restart` (the stored record keeps the acquisition log and its boundary; the controller's `current_log_id`, which the admin view reports, names the rebuilt log) | | 8 | Evict and lazily reload a fenced namespace; identical admission | `tests::fence::lifecycle::evicted_namespace_reloads_same_gate`; landed at unit level: `namespace::store::fence_tests::evicted_namespace_reloads_with_the_same_controller`, `fence::registry::tests::seeded_from_load_fences_including_recovered_names` | | 9 | Filesystem recovery, `destroy_on_error`, undecodable records, missing target quarantine, metastore backup rollback fail closed with provenance | `meta_store::fence_tests::recovery::{fs_recovery_with_marker_unavailable, destroy_on_error_keeps_fenced_unavailable, undecodable_row_unavailable, incomplete_target_unavailable, metastore_rollback_detected_by_marker, lookup_never_creates, undecodable_name_with_fence_fails_startup, marker_in_invalid_directory_fails_startup}`; legacy behaviour kept: `destroy_on_error_without_fences_is_unchanged`, `undecodable_row_without_fences_is_skipped_as_before`; `meta_store::fence_tests::corrupt_fence_row_fails_closed`; restore provenance: `namespace::meta_store::fence_tests::provenance::{bottomless_restore_is_reported (a real bottomless restore against a local S3 endpoint: a metastore opened on an empty directory from a backup reports the restore and its generation and holds the backed-up namespace; one opened with nothing to restore does not), generation_only_after_a_recovery, not_restored_until_recorded_and_first_record_wins}`, `http::admin::fence::tests::restore_provenance_is_reported` (fence view and capability `metastore` object), `tests::fence::admin::capabilities` (not restored: capability endpoint, fence views and the gauge report it) | | 10 | Wrong owner, stale revision, invalid role/state, replay, command-id reuse; replay before revision check | `fence::transition::tests::*` (exhaustive over states × commands) | @@ -864,7 +864,7 @@ Planned test names; the table is updated as tests land. | 18 | Codes through HTTP, Hrana, RPC, dump, replication, replica write proxy; distinguishable from auth/timeout/not-found; old peers compatible; no retry loops | user protocols landed: `tests::fence::protocol::{http_codes (legacy `/`, `/v1/execute`, `/v1/batch` for a write-fenced write, a read-fenced read, a quarantined target and a namespace whose marker cannot be decoded: `423` with `code`, and `detail` where there is one), hrana_http_codes (`/v2`, `/v3` pipelines and `/v3/cursor`: step and whole-request errors carry the code and the baton stays usable), hrana_ws_codes (the same over a WebSocket, whose stream reads again after the read fence is cleared), dump_codes, auth_and_not_found_distinct (`401` without or with a wrong credential and `404` for a missing namespace, with no fence code, on the same fenced server)}`, `error::fence_tests::fence_errors_carry_code` (the body through every wrapper; `block_*`'s `Blocked` keeps its mapping); RPC and replica write proxy landed: `rpc::proxy::fence_tests::rpc_codes` (on the primary's proxy service: a write-fenced write is a step error with `SQL_ERROR` + `stable_code`, reads are served, a read-fenced read is a program error with the code when streamed and the typed `FAILED_PRECONDITION` status when unary, and a namespace whose fence state is unknown is refused with the typed status before any connection), `rpc::proxy::fence_tests::replica_maps_proxied_denials` (step, program and connection-status denials become the fence error on the replica; an older primary's error without `stable_code`, an unknown code and other errors keep their mapping), `namespace::fence::outcome::tests::peer_denials_round_trip`, `tests::fence::protocol::{replica_proxy_preserves_code (writes through a replica to a write-fenced primary: `423` + `MIGRATION_WRITE_FENCED` on legacy `/` and `/v1/execute`, the step error on `/v1/batch`, the Hrana error code on `/v2` and `/v3`; reads on the replica served; writes through the replica work after release), denial_not_retried (each refused write is delegated exactly once and answered well within the write proxy's first retry backoff)}`; capability: `tests::fence::admin::capabilities` asserts `proxy_stable_code: true`; replication and replica servers landed: `tests::fence::protocol::{replication_codes (a raw peer of the primary's internal replication service: `hello` carries the write fence's state and revision, an open `log_entries` stream ends with `FAILED_PRECONDITION` + `x-libsql-fence-code` `MIGRATION_READ_FENCED` under the read fence, `hello`, `log_entries` and `snapshot` are then refused with it, `hello` is answered again after the clear and carries no fence after release, and a quarantined target refuses `hello` with `MIGRATION_TARGET_QUARANTINED`), replica_reads_denied_while_source_read_fenced (the replica refuses local reads with `423` + `MIGRATION_READ_FENCED` on legacy `/` and `/v1/execute` and the Hrana code on `/v2` within 500 ms of simulated time after the read fence is acknowledged, and still 30 s later), replica_backs_off_on_fence_code (4 to 9 refused, counted attempts over 60 s of simulated time; a fixed 1 s retry makes 57), replica_resumes_after_clear_read_fence (reads served again within 16 s of the clear, and a write after release is replicated), replica_lazy_creation_refused_by_fence (a read of a quarantined target through a replica that has never loaded it answers `423` + `MIGRATION_TARGET_QUARANTINED` within 500 ms of simulated time, twice, leaving no `dbs/` directory; a directory that was already there is kept; after publication the replica creates and serves the name, and after enable-writes a write through it succeeds)}`, `namespace::meta_store::fence_tests::forget_unstored_only_unused_unstored_entries`, `namespace::fence::registry::tests::forget_idle_only_unreferenced_plain_controllers`, `namespace::fence::replica::tests::{refusal_from_typed_status_only, hello_fence_denies_only_read_denying_states, backoff_doubles_to_its_cap, observed_denial_refuses_local_reads_and_cancels_leases}`; landed: `libsql-replication` `rpc::test::{proxy_error_stable_code_is_additive, replicated_fence_is_additive}` (each new field is skipped by a peer that does not know it, absent from an older peer's message, and absent fields encode exactly as before), `namespace::fence::stream::tests::hello_carries_replicated_fence` (no fence before acquisition and after release; state and revision while write-fenced; the stored configuration never carries it) | | 19 | Corrupt or unknown durable fence state fails closed | `fence::store::tests::corrupt_payload_fails_closed`, `unknown_format_version_fails_closed` | | 20 | Metrics and audit logs | `tests::fence::observability::metrics_and_labels`; `fence::audit::tests::audit_event_fields` | -| 21 | Capability discovery and mixed-version protection | capability discovery landed: `tests::fence::admin::{capabilities, capabilities_when_disabled}`; planned: `fence::store::tests::legacy_mirror_and_fk_guard` (bounded, see section 18) | +| 21 | Capability discovery and mixed-version protection | capability discovery landed: `tests::fence::admin::{capabilities, capabilities_when_disabled}`; legacy mirror and foreign-key guard landed: `namespace::fence::tests::legacy_mirror::legacy_mirror_and_fk_guard` (a source and two targets walked through every stored state: the config row's `block_*` fields hold the mirror of section 13.2 and nothing else in the row changes, a config write is refused and changes nothing, an older binary's delete fails on the foreign key; release and write enable restore the namespace's own values, later config writes are stored as written, and after a restart the rows are unchanged and the in-memory config holds the namespace's own values, including a config written after the release); an older binary is not run (bounded, see section 18) | | 22 | Adoption is two-person/audited, keeps admission closed, cannot reverse publication | landed: `namespace::fence::tests::adoption::{adopt_requires_key_and_two_approvers (no key, one approver, duplicate or blank approvers, three approvers, blank incident or reason: `adoption_not_authorised`, nothing changes), adopt_keeps_gates_closed (write-fenced source: owner and revision move, state, admissions, boundary and saved values do not, writes still refused, replay with or without the key returns the receipt, old owner `FENCE_OWNED_BY_ANOTHER_OPERATION`, new owner releases), adopt_quarantined_target (the import capability moves to the new owner; SQL still `MIGRATION_TARGET_QUARANTINED`), adopt_cannot_touch_writable (`TARGET_WRITABLE`, `TARGET_ABORTED`), adopt_cannot_touch_released, adopt_recovers_metastore_rollback (fence row gone, and fence row at an older revision: re-established from the marker, served again behind the same gate, own `block_*` values back in memory, new owner finishes), adopt_recovered_name_without_config_row (`namespace_config_missing`, nothing written, still unavailable), adoption_key_matching}`, `namespace::fence::audit::tests::adoption_event_fields`; pure transition: `namespace::fence::transition::tests::{adopt_requires_key_and_two_approvers, adopt_keeps_gates_closed}`; over HTTP: `tests::fence::admin::{adopt_over_http (no key, wrong key, no admin credential, bad approvers, unknown field, success, replay, old and new owner, finished operation), adopt_disabled_without_key}` | | — | Import API usable by bulk import | landed: `namespace::fence::import::tests::import_session_loads_dump_into_quarantined_target` (a dump exported by the server from a source with tables, keys, a foreign key, an index, an autoincrement table, a trigger, a view and an FTS5 table loads through `ImportSession::load_dump`; after the seal the target's schema, rows, view and full-text results equal the source's) | diff --git a/libsql-server/src/namespace/fence/record.rs b/libsql-server/src/namespace/fence/record.rs index 247b481ad2..a8baa25e60 100644 --- a/libsql-server/src/namespace/fence/record.rs +++ b/libsql-server/src/namespace/fence/record.rs @@ -118,12 +118,24 @@ impl NamespaceFenceRecord { self.state.read_admission() } + /// Whether the stored config's `block_*` fields hold the fence's mirror (section 13.2) + /// rather than the namespace's own values. They stop holding it when the operation releases + /// the namespace or enables target writes: that transition puts the saved values back, and + /// from then on config writes store the namespace's own values in the row again, so the row + /// is authoritative and `legacy_blocks` may be out of date. + pub fn mirrors_legacy_blocks(&self) -> bool { + !matches!( + self.state, + FenceState::Released | FenceState::TargetWritable + ) + } + /// Values of the legacy `block_*` configuration fields while this record is in force: the /// fence state mirrored for an older binary, or the pre-fence values once the operation /// has released the namespace. pub fn legacy_mirror(&self) -> LegacyBlocks { match self.state { - FenceState::Released | FenceState::TargetWritable => self.legacy_blocks.clone(), + _ if !self.mirrors_legacy_blocks() => self.legacy_blocks.clone(), state => LegacyBlocks { block_reads: !state.read_admission().is_open(), block_writes: !state.write_admission().is_open(), diff --git a/libsql-server/src/namespace/fence/store.rs b/libsql-server/src/namespace/fence/store.rs index bb7a92a20f..fc7e60e062 100644 --- a/libsql-server/src/namespace/fence/store.rs +++ b/libsql-server/src/namespace/fence/store.rs @@ -675,6 +675,18 @@ pub fn with_legacy_blocks(config: &DatabaseConfig, blocks: &LegacyBlocks) -> Dat } } +/// The namespace's own config, given its stored config row `stored` and its fence record: the +/// row with the record's saved `block_*` values in place of the mirror while the record +/// mirrors them, and the row itself once the operation has finished (section 13.2), since +/// config writes after a release or a write enable store the namespace's own values. +pub fn own_config(stored: &DatabaseConfig, record: &NamespaceFenceRecord) -> DatabaseConfig { + if record.mirrors_legacy_blocks() { + with_legacy_blocks(stored, &record.legacy_blocks) + } else { + stored.clone() + } +} + /// The `block_*` fields of `config`. pub fn legacy_blocks_of(config: &DatabaseConfig) -> LegacyBlocks { LegacyBlocks { diff --git a/libsql-server/src/namespace/fence/tests.rs b/libsql-server/src/namespace/fence/tests.rs index 0d5aeb7ae7..8791ee2e54 100644 --- a/libsql-server/src/namespace/fence/tests.rs +++ b/libsql-server/src/namespace/fence/tests.rs @@ -114,16 +114,24 @@ impl Server { /// The namespace's controller, loading the namespace if it is not loaded. async fn fence(&self) -> Arc { + self.fence_of("ns").await + } + + async fn fence_of(&self, ns: &'static str) -> Arc { self.store - .with("ns".into(), |ns| ns.fence().clone()) + .with(ns.into(), |ns| ns.fence().clone()) .await .unwrap() } async fn conn(&self) -> Arc { + self.conn_to("ns").await + } + + async fn conn_to(&self, ns: &'static str) -> Arc { let maker = self .store - .with("ns".into(), |ns| ns.db.connection_maker()) + .with(ns.into(), |ns| ns.db.connection_maker()) .await .unwrap(); Arc::new(maker.create().await.unwrap()) @@ -157,15 +165,23 @@ impl Server { } async fn inspect(&self) -> FenceInspection { + self.inspect_of("ns").await + } + + async fn inspect_of(&self, ns: &'static str) -> FenceInspection { self.store .meta_store() - .inspect_fence("ns".into()) + .inspect_fence(ns.into()) .await .unwrap() } async fn count(&self) -> i64 { - let conn = self.conn().await; + self.count_in("ns").await + } + + async fn count_in(&self, ns: &'static str) -> i64 { + let conn = self.conn_to(ns).await; tokio::task::spawn_blocking(move || { conn.with_raw(|c| c.query_row("select count(*) from t", (), |r| r.get(0))) }) @@ -176,7 +192,11 @@ impl Server { /// Whether a new connection may begin a write transaction. Writes nothing. async fn writes_admitted(&self) -> bool { - let conn = self.conn().await; + self.writes_admitted_in("ns").await + } + + async fn writes_admitted_in(&self, ns: &'static str) -> bool { + let conn = self.conn_to(ns).await; match raw(&conn, "begin immediate; rollback;").await { Ok(()) => true, Err(rusqlite::Error::SqliteFailure(e, _)) @@ -254,7 +274,11 @@ fn boundary(commit: &FenceCommit) -> FrozenBoundary { /// The marker file's bytes, if there is one. fn read_marker_bytes(dbs: &Path) -> Option> { - match std::fs::read(fence_store::marker_path(dbs, &"ns".into())) { + read_marker_bytes_of(dbs, "ns") +} + +fn read_marker_bytes_of(dbs: &Path, ns: &'static str) -> Option> { + match std::fs::read(fence_store::marker_path(dbs, &ns.into())) { Ok(bytes) => Some(bytes), Err(e) if e.kind() == std::io::ErrorKind::NotFound => None, Err(e) => panic!("{e}"), @@ -263,7 +287,11 @@ fn read_marker_bytes(dbs: &Path) -> Option> { /// Put the marker file back to `bytes` (`None`: no marker). fn restore_marker_bytes(dbs: &Path, bytes: Option<&[u8]>) { - let path = fence_store::marker_path(dbs, &"ns".into()); + restore_marker_bytes_of(dbs, "ns", bytes) +} + +fn restore_marker_bytes_of(dbs: &Path, ns: &'static str, bytes: Option<&[u8]>) { + let path = fence_store::marker_path(dbs, &ns.into()); match bytes { Some(bytes) => std::fs::write(path, bytes).unwrap(), None => std::fs::remove_file(path).unwrap(), @@ -937,6 +965,1086 @@ fn acquire_response_loss_resolved_by_replay_and_inspect() { } } +// --------------------------------------------------------------------------------------------- +// Read fence, seal and write enable: restart at each persistence boundary (section 8.5) + +mod read_and_target_boundaries { + use super::*; + use crate::namespace::fence::command::ValidationResult; + use crate::namespace::fence::controller::LeaseKind; + use crate::namespace::fence::target::tests::{create_request, enable_request, target_command}; + + #[derive(Debug, Clone, Copy, PartialEq, Eq)] + enum Later { + /// `SetSourceReadFence` on the write-fenced source `ns`. + ReadFence, + /// `SealTargetImport` on the quarantined target `tgt`. + Seal, + /// `EnableTargetWrites` on the write-fenced target `tgt`. + Enable, + } + + impl Later { + fn namespace(self) -> &'static str { + match self { + Later::ReadFence => "ns", + Later::Seal | Later::Enable => "tgt", + } + } + + /// State and revision before the command. + fn before(self) -> (FenceState, u64) { + match self { + Later::ReadFence => (FenceState::SourceWriteFenced, 2), + Later::Seal => (FenceState::TargetQuarantined, 1), + Later::Enable => (FenceState::TargetWriteFenced, 5), + } + } + + /// State and revision while the command drains. + fn draining(self) -> (FenceState, u64) { + match self { + Later::ReadFence => (FenceState::SourceReadDraining, 3), + Later::Seal => (FenceState::TargetImportDraining, 2), + Later::Enable => unreachable!("EnableTargetWrites does not drain"), + } + } + + /// State and revision once the command has applied. + fn applied(self) -> (FenceState, u64) { + match self { + Later::ReadFence => (FenceState::SourceReadFenced, 4), + Later::Seal => (FenceState::TargetValidating, 3), + Later::Enable => (FenceState::TargetWritable, 6), + } + } + + fn state_after(self, durable: Durable) -> (FenceState, u64) { + match durable { + Durable::Nothing => self.before(), + Durable::Draining => self.draining(), + Durable::Final => self.applied(), + } + } + + fn request(self) -> FenceRequest { + match self { + Later::ReadFence => FenceRequest { + namespace: "ns".into(), + operation_id: OP, + command_id: Uuid::from_u128(2), + expected_state: FenceState::SourceWriteFenced, + expected_revision: 2, + command: FenceCommand::SetSourceReadFence { + drain_policy: Some(LONG), + }, + }, + Later::Seal => target_command( + 10, + FenceState::TargetQuarantined, + 1, + FenceCommand::SealTargetImport { + drain_policy: Some(LONG), + }, + ), + Later::Enable => enable_request(30), + } + } + } + + /// Where the command is when the process dies. + #[derive(Debug, Clone, Copy)] + enum Park { + /// At a point of the command's first (or only) commit. + First(HookPoint), + /// At a point of the drain's completion, the command's second commit. + Second(HookPoint), + /// In the drain, waiting for a reader (read fence) or an import call (seal) that was + /// already running when the command started. + WaitingForHolder, + } + + #[derive(Debug, Clone, Copy)] + struct LaterBoundary { + name: &'static str, + command: Later, + park: Park, + /// The marker file is put back to what it held before the commit, as a crash between + /// the metastore commit and the marker write leaves it. + marker_lags: bool, + durable: Durable, + } + + const fn case( + name: &'static str, + command: Later, + park: Park, + marker_lags: bool, + durable: Durable, + ) -> LaterBoundary { + LaterBoundary { + name, + command, + park, + marker_lags, + durable, + } + } + + use Durable::{Draining, Final, Nothing}; + use HookPoint::{ + AfterClosingReads, AfterInstallingGate, AfterMetastoreCommit, BeforeGatePublish, + BeforeMetastoreCommit, BeforeResponse, + }; + use Later::{Enable, ReadFence, Seal}; + use Park::{First, Second, WaitingForHolder}; + + const LATER_BOUNDARIES: &[LaterBoundary] = &[ + case( + "read/after-closing-reads", + ReadFence, + First(AfterClosingReads), + false, + Nothing, + ), + case( + "read/before-draining-commit", + ReadFence, + First(BeforeMetastoreCommit), + false, + Nothing, + ), + case( + "read/after-draining-commit", + ReadFence, + First(AfterMetastoreCommit), + false, + Draining, + ), + case( + "read/after-draining-commit/marker-lags", + ReadFence, + First(AfterMetastoreCommit), + true, + Draining, + ), + case( + "read/before-draining-publish", + ReadFence, + First(BeforeGatePublish), + false, + Draining, + ), + case( + "read/draining-published", + ReadFence, + First(BeforeResponse), + false, + Draining, + ), + case( + "read/waiting-for-reader", + ReadFence, + WaitingForHolder, + false, + Draining, + ), + case( + "read/before-fenced-commit", + ReadFence, + Second(BeforeMetastoreCommit), + false, + Draining, + ), + case( + "read/after-fenced-commit", + ReadFence, + Second(AfterMetastoreCommit), + false, + Final, + ), + case( + "read/after-fenced-commit/marker-lags", + ReadFence, + Second(AfterMetastoreCommit), + true, + Final, + ), + case( + "read/before-fenced-publish", + ReadFence, + Second(BeforeGatePublish), + false, + Final, + ), + case( + "read/before-fenced-response", + ReadFence, + Second(BeforeResponse), + false, + Final, + ), + case( + "seal/after-installing-gate", + Seal, + First(AfterInstallingGate), + false, + Nothing, + ), + case( + "seal/before-draining-commit", + Seal, + First(BeforeMetastoreCommit), + false, + Nothing, + ), + case( + "seal/after-draining-commit", + Seal, + First(AfterMetastoreCommit), + false, + Draining, + ), + case( + "seal/after-draining-commit/marker-lags", + Seal, + First(AfterMetastoreCommit), + true, + Draining, + ), + case( + "seal/before-draining-publish", + Seal, + First(BeforeGatePublish), + false, + Draining, + ), + case( + "seal/draining-published", + Seal, + First(BeforeResponse), + false, + Draining, + ), + case( + "seal/waiting-for-import-call", + Seal, + WaitingForHolder, + false, + Draining, + ), + case( + "seal/before-validating-commit", + Seal, + Second(BeforeMetastoreCommit), + false, + Draining, + ), + case( + "seal/after-validating-commit", + Seal, + Second(AfterMetastoreCommit), + false, + Final, + ), + case( + "seal/after-validating-commit/marker-lags", + Seal, + Second(AfterMetastoreCommit), + true, + Final, + ), + case( + "seal/before-validating-publish", + Seal, + Second(BeforeGatePublish), + false, + Final, + ), + case( + "seal/before-validating-response", + Seal, + Second(BeforeResponse), + false, + Final, + ), + case( + "enable/before-commit", + Enable, + First(BeforeMetastoreCommit), + false, + Nothing, + ), + case( + "enable/after-commit", + Enable, + First(AfterMetastoreCommit), + false, + Final, + ), + case( + "enable/after-commit/marker-lags", + Enable, + First(AfterMetastoreCommit), + true, + Final, + ), + case( + "enable/before-publish", + Enable, + First(BeforeGatePublish), + false, + Final, + ), + case( + "enable/before-response", + Enable, + First(BeforeResponse), + false, + Final, + ), + ]; + + /// Create the quarantined target `tgt` (revision 1) and import a table `t` of [`ROWS`] + /// rows into it. + fn create_target(server: &Server) { + server.run(async { + let commit = server + .store + .create_target_quarantined(create_request("tgt", 1), server_identity()) + .await + .unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + let mut session = server + .store + .open_import_session("tgt".into(), OP, 1) + .await + .unwrap(); + session + .with_raw(|c| { + c.execute_batch("create table t (x)")?; + for _ in 0..ROWS { + c.execute("insert into t values (1)", ())?; + } + Ok::<_, rusqlite::Error>(()) + }) + .await + .unwrap() + .unwrap(); + }) + } + + /// Seal, validate and publish `tgt`: `TARGET_WRITE_FENCED` at revision 5. + fn write_fence_target(server: &Server) { + server.run(async { + let steps = [ + target_command( + 10, + FenceState::TargetQuarantined, + 1, + FenceCommand::SealTargetImport { + drain_policy: Some(LONG), + }, + ), + target_command( + 20, + FenceState::TargetValidating, + 3, + FenceCommand::RecordTargetValidation { + result: ValidationResult::Ok, + summary: "rows match".into(), + }, + ), + target_command( + 21, + FenceState::TargetValidating, + 4, + FenceCommand::PublishTargetReadableWriteFenced, + ), + ]; + for step in steps { + let commit = server.execute(step).await.unwrap().unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + } + }) + } + + fn prepare(server: &Server, command: Later) { + match command { + Later::ReadFence => { + server.create_source(); + server.fence_source(); + } + Later::Seal => create_target(server), + Later::Enable => { + create_target(server); + write_fence_target(server); + } + } + } + + /// What the drain of [`Park::WaitingForHolder`] waits for: a read lease, as a running SQL + /// program holds one, or an admitted import call, as a running `ImportSession::with_raw` + /// holds one. Neither holds a SQLite lock, so it can outlive the crash without disturbing + /// the next lifetime's recovery. Only dropped. + type Holder = Box; + + async fn hold(server: &Server, fence: &Arc, command: Later) -> Holder { + match command { + Later::ReadFence => Box::new( + fence + .acquire_read_lease(OperationClass::NormalRead, LeaseKind::Sql, || {}) + .unwrap(), + ), + Later::Seal => { + let session = server + .store + .open_import_session("tgt".into(), OP, 1) + .await + .unwrap(); + let call = fence.begin_import_write(session.capability()).unwrap(); + drop(session); + assert_eq!(fence.import_writers(), 1); + Box::new(call) + } + Later::Enable => unreachable!("EnableTargetWrites does not drain"), + } + } + + /// Admission of new work in the gate's current state: writes only on a writable target, + /// normal reads wherever the state admits them. + async fn assert_admission( + server: &Server, + fence: &Arc, + ns: &'static str, + name: &str, + ) { + let state = fence.gate().state(); + let writes = state == FenceState::TargetWritable; + let reads = matches!( + state, + FenceState::SourceWriteFenced + | FenceState::TargetWriteFenced + | FenceState::TargetWritable + ); + assert_eq!( + server.writes_admitted_in(ns).await, + writes, + "{name}: write admission in {state}" + ); + let lease = fence.acquire_read_lease(OperationClass::NormalRead, LeaseKind::Sql, || {}); + assert_eq!(lease.is_ok(), reads, "{name}: read admission in {state}"); + } + + /// Kill the process at every point where `SetSourceReadFence`, `SealTargetImport` and + /// `EnableTargetWrites` persist, publish, answer or wait, and restart it on the same + /// directory. As for the source write fence (`restart_at_each_boundary`), the restarted + /// server recovers exactly the state before the command or the state it committed and + /// installs that gate before serving the namespace: reads stay closed once + /// `SOURCE_READ_DRAINING` committed, import stays closed once `TARGET_IMPORT_DRAINING` + /// committed, and target writes open only if `TARGET_WRITABLE` committed. No reader or + /// import call survives a restart, so a replay of the same command completes an + /// interrupted drain at once; a replay of a finished one returns its stored result. + #[test] + fn restart_at_each_read_and_target_boundary() { + for case in LATER_BOUNDARIES { + restart_later_at(case); + } + } + + fn restart_later_at(case: &LaterBoundary) { + let name = case.name; + let ns = case.command.namespace(); + let dir = tempdir().unwrap(); + let dbs = dir.path().join("dbs"); + let request = case.command.request(); + let before = case.command.before(); + + // First lifetime: run the command until it reaches the boundary, then crash. + let server = Server::boot(dir.path()); + prepare(&server, case.command); + let holder = server.run(async { + let fence = server.fence_of(ns).await; + assert_eq!( + (fence.gate().state(), fence.gate().revision()), + before, + "{name}" + ); + let hooks = fence.hooks(); + let mut marker_before = read_marker_bytes_of(&dbs, ns); + let mut holder = None; + let paused = match case.park { + Park::First(point) => { + let paused = hooks.pause_at(point); + let task = server.execute(request.clone()); + reached(&paused, name, point).await; + drop(task); + Some(paused) + } + Park::Second(point) => { + // The first commit's response point comes after its publication and + // before the drain, which has nothing to wait for. + let first = hooks.pause_at(HookPoint::BeforeResponse); + let task = server.execute(request.clone()); + reached(&first, name, HookPoint::BeforeResponse).await; + marker_before = read_marker_bytes_of(&dbs, ns); + let paused = hooks.pause_at(point); + first.resume(); + reached(&paused, name, point).await; + drop(task); + Some(paused) + } + Park::WaitingForHolder => { + holder = Some(hold(&server, &fence, case.command).await); + let (draining, _) = case.command.draining(); + let mut gate = fence.subscribe(); + let task = server.execute(request.clone()); + tokio::time::timeout(PROMPT, gate.wait_for(|g| g.state() == draining)) + .await + .unwrap_or_else(|_| panic!("{name}: the command never started draining")) + .unwrap(); + assert!(!task.is_finished(), "{name}: the drain did not wait"); + drop(task); + None + } + }; + if case.marker_lags { + restore_marker_bytes_of(&dbs, ns, marker_before.as_deref()); + } + // What the metastore holds at the moment of the crash. + let inspected = server.inspect_of(ns).await; + assert_eq!( + (inspected.fence.state(), inspected.fence.revision()), + case.command.state_after(case.durable), + "{name}: durable at the crash" + ); + drop(paused); + holder + }); + server.crash(); + // The reader or import call belonged to the dead process. + drop(holder); + + // Second lifetime. + let server = Server::boot(dir.path()); + server.run(async { + let recovered = case.command.state_after(case.durable); + let fence = server.fence_of(ns).await; + let gate = fence.gate(); + assert_eq!( + (gate.state(), gate.revision()), + recovered, + "{name}: recovered state" + ); + assert!( + gate.indeterminate.is_none() + && !gate.is_installing() + && gate.closing_reads.is_none(), + "{name}: an in-memory gate survived the restart" + ); + assert_eq!(fence.read_lease_counts().total(), 0, "{name}"); + assert_eq!(fence.import_writers(), 0, "{name}"); + assert_admission(&server, &fence, ns, name).await; + // Committed data survived, nothing else was written. + assert_eq!(server.count_in(ns).await, ROWS, "{name}"); + // The marker was repaired if it had fallen behind. + let marker = fence_store::read_marker(&dbs, &ns.into()).unwrap(); + assert_eq!( + marker.and_then(|m| m.ok()).map(|m| m.record.revision), + gate.fence.record().map(|r| r.revision), + "{name}: marker" + ); + if case.command == Later::Seal { + // Import resumes only if the seal never committed. + let session = server.store.open_import_session(ns.into(), OP, 1).await; + match case.durable { + Durable::Nothing => drop(session.unwrap()), + Durable::Draining | Durable::Final => { + assert!( + matches!(session, Err(Error::NamespaceFence(_))), + "{name}: import reopened after the seal committed" + ); + } + } + } + + let replay = server.execute(request.clone()).await.unwrap().unwrap(); + let kind = if case.durable == Durable::Final { + FenceCommitKind::Replayed + } else { + FenceCommitKind::Committed + }; + assert_eq!(replay.kind, kind, "{name}"); + assert_eq!(replay.receipt.outcome, FenceOutcome::Applied, "{name}"); + assert_eq!(replay.receipt.command_id, request.command_id, "{name}"); + assert_eq!(replay.receipt.revision_before, before.1, "{name}"); + assert_eq!( + (replay.receipt.state_after, replay.receipt.revision_after), + case.command.applied(), + "{name}" + ); + + // Settled: the gate is the durable state, and a further replay answers the same. + let gate = fence.gate(); + let durable = server.inspect_of(ns).await; + assert_eq!(gate.fence, durable.fence, "{name}"); + assert_eq!( + (gate.state(), gate.revision()), + case.command.applied(), + "{name}" + ); + assert_admission(&server, &fence, ns, name).await; + assert_eq!(server.count_in(ns).await, ROWS, "{name}"); + let again = server.execute(request.clone()).await.unwrap().unwrap(); + assert_eq!(again.kind, FenceCommitKind::Replayed, "{name}"); + assert_eq!(again.receipt, replay.receipt, "{name}"); + }); + server.crash(); + } +} + +// --------------------------------------------------------------------------------------------- +// Protection against an older binary (section 13.2) + +mod legacy_mirror { + use super::*; + use crate::namespace::fence::command::ValidationResult; + use crate::namespace::fence::target::tests::{create_request, enable_request, target_command}; + use crate::namespace::meta_store::{metastore_connection_maker, MetaStoreConnection}; + use libsql_replication::rpc::metadata; + + /// A metastore connection set up the way an older binary sets up its own: foreign keys on, + /// and no knowledge of the fence tables. + async fn older_binary(dir: &Path) -> MetaStoreConnection { + let (maker, _) = metastore_connection_maker(None, dir).await.unwrap(); + let conn = maker().unwrap(); + conn.execute("PRAGMA foreign_keys=ON", ()).unwrap(); + conn + } + + fn encoded(config: &DatabaseConfig) -> metadata::DatabaseConfig { + metadata::DatabaseConfig::from(config) + } + + fn stored(meta: &rusqlite::Connection, ns: &'static str) -> DatabaseConfig { + fence_store::read_config_row(meta, &ns.into()) + .unwrap() + .expect("the namespace has a config row") + } + + /// The `block_*` values section 13.2 says a fenced namespace's stored config holds in + /// `state`, derived from the permission matrix (section 3.3) rather than from the record. + fn mirror(state: FenceState) -> (bool, bool, Option) { + let (block_reads, block_writes) = match state { + FenceState::SourceDraining + | FenceState::SourceWriteFenced + | FenceState::TargetWriteFenced => (false, true), + FenceState::SourceReadDraining + | FenceState::SourceReadFenced + | FenceState::TargetQuarantined + | FenceState::TargetImportDraining + | FenceState::TargetValidating + | FenceState::TargetAborted => (true, true), + other => unreachable!("{other} does not mirror the fence"), + }; + let reason = format!("namespace fence: {state} (operation {OP})"); + (block_reads, block_writes, Some(reason)) + } + + /// In `state`, the stored config of `ns` is the namespace's own config `own` with the fence + /// mirrored into its `block_*` fields (or with its own values once the operation has + /// finished); the in-memory config is `own`; a config write through the metastore is + /// refused and changes nothing while the fence denies lifecycle work; and an older + /// binary's delete of the config row fails on the fence row's foreign key. + async fn check( + server: &Server, + meta: &rusqlite::Connection, + ns: &'static str, + own: &DatabaseConfig, + state: FenceState, + ) { + let fence = server.fence_of(ns).await; + assert_eq!(fence.gate().state(), state, "{ns}"); + let row = stored(meta, ns); + let blocks = (row.block_reads, row.block_writes, row.block_reason.clone()); + let finished = matches!( + state, + FenceState::Unfenced | FenceState::Released | FenceState::TargetWritable + ); + if finished { + assert_eq!( + blocks, + (own.block_reads, own.block_writes, own.block_reason.clone()), + "{ns} in {state}: the namespace's own block_* values" + ); + } else { + assert_eq!(blocks, mirror(state), "{ns} in {state}: the legacy mirror"); + } + // Only the block_* fields carry the mirror. + assert_eq!( + encoded(&fence_store::with_legacy_blocks( + &row, + &fence_store::legacy_blocks_of(own) + )), + encoded(own), + "{ns} in {state}" + ); + let handle = server + .store + .meta_store() + .lookup(&ns.into()) + .await + .unwrap() + .expect("the namespace has a config"); + assert_eq!( + encoded(&handle.get()), + encoded(own), + "{ns} in {state}: in memory" + ); + + if !finished { + let overwrite = DatabaseConfig { + block_reads: false, + block_writes: false, + block_reason: None, + max_db_pages: own.max_db_pages + 1, + ..own.clone() + }; + match handle.store(overwrite).await { + Err(Error::NamespaceFence(_)) => (), + other => panic!("{ns} in {state}: config write not refused: {other:?}"), + } + assert_eq!(encoded(&stored(meta, ns)), encoded(&row), "{ns} in {state}"); + assert_eq!(encoded(&handle.get()), encoded(own), "{ns} in {state}"); + } + + if state != FenceState::Unfenced { + // SQLite enforces `ON DELETE RESTRICT` with an action trigger, so the refusal is + // `SQLITE_CONSTRAINT_TRIGGER` carrying the foreign key message. + match meta.execute("DELETE FROM namespace_configs WHERE namespace = ?1", [ns]) { + Err(rusqlite::Error::SqliteFailure(e, message)) => { + assert_eq!( + e.code, + ErrorCode::ConstraintViolation, + "{ns} in {state}: {e}" + ); + assert_eq!( + message.as_deref(), + Some("FOREIGN KEY constraint failed"), + "{ns} in {state}" + ); + } + other => { + panic!("{ns} in {state}: an older binary's delete was not refused: {other:?}") + } + } + assert_eq!(encoded(&stored(meta, ns)), encoded(&row), "{ns} in {state}"); + } + } + + fn applied(result: Result, tokio::task::JoinError>) -> FenceCommit { + let commit = result.unwrap().unwrap(); + assert_eq!(commit.receipt.outcome, FenceOutcome::Applied); + commit + } + + fn source_command( + command_id: u128, + expected_state: FenceState, + expected_revision: u64, + command: FenceCommand, + ) -> FenceRequest { + FenceRequest { + namespace: "ns".into(), + operation_id: OP, + command_id: Uuid::from_u128(command_id), + expected_state, + expected_revision, + command, + } + } + + /// Store `config` through the metastore, as `POST /v1/namespaces/:ns/config` does. + async fn store_config(server: &Server, ns: &'static str, config: &DatabaseConfig) { + server + .store + .meta_store() + .lookup(&ns.into()) + .await + .unwrap() + .unwrap() + .store(config.clone()) + .await + .unwrap(); + } + + /// Walk a source and two targets through every stored state: in each, the config row + /// carries the legacy mirror of section 13.2 (reads blocked where the state denies reads, + /// writes blocked where it denies writes, and a reason naming the state and the + /// operation), nothing else in the row changes, a config write cannot overwrite it, and the + /// foreign key refuses an older binary's delete. Release and write enable put the + /// namespace's own values back, after which config writes follow the existing policy; the + /// foreign key stays with the fence row. A restart keeps the rows and gives the in-memory + /// config the namespace's own values. + #[test] + fn legacy_mirror_and_fk_guard() { + let dir = tempdir().unwrap(); + let server = Server::boot(dir.path()); + server.create_source(); + let owns = server.run(async { + let meta = older_binary(dir.path()).await; + let mut own = DatabaseConfig { + block_reason: Some("pre-fence note".into()), + max_db_pages: 1234, + ..(*server + .store + .meta_store() + .lookup(&"ns".into()) + .await + .unwrap() + .unwrap() + .get()) + .clone() + }; + store_config(&server, "ns", &own).await; + check(&server, &meta, "ns", &own, FenceState::Unfenced).await; + + // SOURCE_DRAINING, parked after its commit and before the boundary is captured. + let fence = server.fence().await; + let (log_id, _) = server.log().await; + let paused = fence.hooks().pause_at(HookPoint::BeforeBoundaryCapture); + let task = server.execute(acquire(log_id, 1, LONG)); + reached(&paused, "acquire", HookPoint::BeforeBoundaryCapture).await; + check(&server, &meta, "ns", &own, FenceState::SourceDraining).await; + paused.resume(); + applied(task.await); + check(&server, &meta, "ns", &own, FenceState::SourceWriteFenced).await; + + // SOURCE_READ_DRAINING, parked after its publication and before the drain. + let paused = fence.hooks().pause_at(HookPoint::BeforeResponse); + let task = server.execute(source_command( + 2, + FenceState::SourceWriteFenced, + 2, + FenceCommand::SetSourceReadFence { + drain_policy: Some(LONG), + }, + )); + reached(&paused, "read fence", HookPoint::BeforeResponse).await; + check(&server, &meta, "ns", &own, FenceState::SourceReadDraining).await; + paused.resume(); + applied(task.await); + check(&server, &meta, "ns", &own, FenceState::SourceReadFenced).await; + + applied( + server + .execute(source_command( + 3, + FenceState::SourceReadFenced, + 4, + FenceCommand::ClearSourceReadFence, + )) + .await, + ); + check(&server, &meta, "ns", &own, FenceState::SourceWriteFenced).await; + applied(server.execute(release(4, 5)).await); + check(&server, &meta, "ns", &own, FenceState::Released).await; + // Released: config writes follow the existing policy and are stored as written. + own = DatabaseConfig { + block_writes: true, + block_reason: Some("after the operation".into()), + max_db_pages: 4321, + ..own + }; + store_config(&server, "ns", &own).await; + check(&server, &meta, "ns", &own, FenceState::Released).await; + + // A target, created with the default block_* values. + applied(Ok(server + .store + .create_target_quarantined(create_request("tgt", 1), server_identity()) + .await)); + let mut own_target = (*server + .store + .meta_store() + .lookup(&"tgt".into()) + .await + .unwrap() + .unwrap() + .get()) + .clone(); + assert!( + !own_target.block_reads + && !own_target.block_writes + && own_target.block_reason.is_none() + ); + check( + &server, + &meta, + "tgt", + &own_target, + FenceState::TargetQuarantined, + ) + .await; + + // TARGET_IMPORT_DRAINING, parked after its publication and before the drain. + let target = server.fence_of("tgt").await; + let paused = target.hooks().pause_at(HookPoint::BeforeResponse); + let task = server.execute(target_command( + 10, + FenceState::TargetQuarantined, + 1, + FenceCommand::SealTargetImport { + drain_policy: Some(LONG), + }, + )); + reached(&paused, "seal", HookPoint::BeforeResponse).await; + check( + &server, + &meta, + "tgt", + &own_target, + FenceState::TargetImportDraining, + ) + .await; + paused.resume(); + applied(task.await); + check( + &server, + &meta, + "tgt", + &own_target, + FenceState::TargetValidating, + ) + .await; + + applied( + server + .execute(target_command( + 20, + FenceState::TargetValidating, + 3, + FenceCommand::RecordTargetValidation { + result: ValidationResult::Ok, + summary: "rows match".into(), + }, + )) + .await, + ); + check( + &server, + &meta, + "tgt", + &own_target, + FenceState::TargetValidating, + ) + .await; + applied( + server + .execute(target_command( + 21, + FenceState::TargetValidating, + 4, + FenceCommand::PublishTargetReadableWriteFenced, + )) + .await, + ); + check( + &server, + &meta, + "tgt", + &own_target, + FenceState::TargetWriteFenced, + ) + .await; + applied(server.execute(enable_request(30)).await); + check( + &server, + &meta, + "tgt", + &own_target, + FenceState::TargetWritable, + ) + .await; + own_target = DatabaseConfig { + max_db_pages: 777, + ..own_target + }; + store_config(&server, "tgt", &own_target).await; + check( + &server, + &meta, + "tgt", + &own_target, + FenceState::TargetWritable, + ) + .await; + + // An aborted target keeps everything blocked. + applied(Ok(server + .store + .create_target_quarantined(create_request("tgt2", 40), server_identity()) + .await)); + let own_aborted = (*server + .store + .meta_store() + .lookup(&"tgt2".into()) + .await + .unwrap() + .unwrap() + .get()) + .clone(); + applied( + server + .execute(FenceRequest { + namespace: "tgt2".into(), + operation_id: OP, + command_id: Uuid::from_u128(41), + expected_state: FenceState::TargetQuarantined, + expected_revision: 1, + command: FenceCommand::AbortQuarantinedTarget, + }) + .await, + ); + check( + &server, + &meta, + "tgt2", + &own_aborted, + FenceState::TargetAborted, + ) + .await; + (own, own_target, own_aborted) + }); + server.crash(); + + // The rows are kept across a restart, and the in-memory config is the namespace's own. + let (own, own_target, own_aborted) = owns; + let server = Server::boot(dir.path()); + server.run(async { + let meta = older_binary(dir.path()).await; + check(&server, &meta, "ns", &own, FenceState::Released).await; + check( + &server, + &meta, + "tgt", + &own_target, + FenceState::TargetWritable, + ) + .await; + check( + &server, + &meta, + "tgt2", + &own_aborted, + FenceState::TargetAborted, + ) + .await; + }); + server.crash(); + } +} + // --------------------------------------------------------------------------------------------- // Incident adoption (section 12; section 17 row 22) diff --git a/libsql-server/src/namespace/meta_store.rs b/libsql-server/src/namespace/meta_store.rs index fe162f64a4..ff8c9f3ec1 100644 --- a/libsql-server/src/namespace/meta_store.rs +++ b/libsql-server/src/namespace/meta_store.rs @@ -519,8 +519,11 @@ impl MetaStoreInner { /// Load every namespace's fence after the configs (section 5.6). The stored config row of /// a fenced namespace carries the legacy mirror of the fence in its `block_*` fields - /// (section 13.2); the in-memory config is the namespace's own configuration, so those - /// fields are put back to the values the record saved. A marker that fell behind its + /// (section 13.2) while the record is in force; the in-memory config is the namespace's own + /// configuration, so those fields are put back to the values the record saved + /// ([`fence_store::own_config`]). Once the operation has released the namespace or enabled + /// target writes, the row holds the namespace's own values (including any config written + /// since) and is used as it is. A marker that fell behind its /// record is rewritten. A namespace whose fence cannot be established is logged and keeps /// its stored config, mirror included. fn restore_fences(&mut self) -> Result<()> { @@ -554,7 +557,7 @@ impl MetaStoreInner { } let sender = self.configs.get_mut().get_mut(&ns).expect("listed above"); let config = sender.borrow().config.clone(); - let config = fence_store::with_legacy_blocks(&config, &record.legacy_blocks); + let config = fence_store::own_config(&config, record); sender.send_modify(|c| c.config = Arc::new(config)); } StoredFence::Unavailable { @@ -1032,14 +1035,14 @@ fn apply_fence_command( }) } -/// Put the namespace's own `block_*` values (the record's saved values) into its in-memory +/// Put the namespace's own `block_*` values ([`fence_store::own_config`]) into its in-memory /// config, if it has one. The caller holds the connection lock, which is taken before the /// config map's everywhere. fn restore_own_blocks(inner: &MetaStoreInner, ns: &NamespaceName, record: &NamespaceFenceRecord) { let configs = inner.configs.blocking_lock(); if let Some(sender) = configs.get(ns) { let config = sender.borrow().config.clone(); - let config = fence_store::with_legacy_blocks(&config, &record.legacy_blocks); + let config = fence_store::own_config(&config, record); sender.send_modify(|c| c.config = Arc::new(config)); } } @@ -1552,8 +1555,9 @@ impl MetaStore { /// Make a migration target that the metastore holds visible in the in-memory config map, /// which is what makes `exists()` and `lookup()` find it (section 10.1, step 5). The config - /// published is the stored row with the record's own `block_*` values in place of the - /// legacy mirror, as `restore_fences` does at startup. The caller has already installed + /// published is the namespace's own config ([`fence_store::own_config`]): the stored row + /// with the record's saved `block_*` values in place of the legacy mirror, or the row + /// itself once target writes are enabled, as `restore_fences` does at startup. The caller has already installed /// the target's gate. Returns whether the map changed; `false` also when the namespace is /// not a target with a stored config. pub async fn publish_target_config(&self, namespace: NamespaceName) -> Result { @@ -1571,7 +1575,7 @@ impl MetaStore { return Ok(false); }; drop(tx); - let config = Arc::new(fence_store::with_legacy_blocks(&row, &record.legacy_blocks)); + let config = Arc::new(fence_store::own_config(&row, &record)); let mut configs = inner.configs.blocking_lock(); match configs.get_mut(&namespace) { Some(sender) From 64580572a127081565931fb4bf950abefdde4b23 Mon Sep 17 00:00:00 2001 From: River Date: Wed, 30 Sep 2026 11:54:21 +0000 Subject: [PATCH 28/33] libsql-server: fence restart and eviction integration tests Co-authored-by: Tomasz Szymczyszyn --- docs/NAMESPACE_FENCE.md | 8 +- libsql-server/tests/fence/lifecycle.rs | 386 ++++++++++++++++++++++++- libsql-server/tests/fence/mod.rs | 113 ++++++-- 3 files changed, 470 insertions(+), 37 deletions(-) diff --git a/docs/NAMESPACE_FENCE.md b/docs/NAMESPACE_FENCE.md index eda47c556c..a757c61e46 100644 --- a/docs/NAMESPACE_FENCE.md +++ b/docs/NAMESPACE_FENCE.md @@ -548,7 +548,7 @@ This makes the WAL gate independent of statement classification: DDL, misclassif - At startup the registry is built from the metastore and markers before `NamespaceStore` serves anything. `make_namespace` takes the controller from the registry and passes it into the configurator's `setup()` (primary, schema and replica), down to `MakeLegacyConnection::new`, which binds a `FenceConnState` to it for every `LegacyConnection` it opens, **starting with** the maker's held `_db` connection. The `Namespace` keeps the same controller (`Namespace::fence()`), which is how dump, replication and lifecycle code, all of which reach a namespace through `NamespaceStore::with`, read its gate. - `NamespaceStore::with` and `make_namespace` check the registry before `lookup()`, `handle()` or any setup: `UNKNOWN_UNAVAILABLE` is refused before any setup work. -- Idle or capacity eviction shuts the namespace down but leaves the controller in the registry; a lazy reload reinstalls the identical gate, revision and generation. A drain waiter that holds the evicted manager sees its connections close and is notified. +- Idle or capacity eviction shuts the namespace down but leaves the controller in the registry; a lazy reload reinstalls the identical gate, revision and generation. A drain waiter that holds the evicted manager sees its connections close and is notified. `tests::fence::lifecycle::evicted_namespace_reloads_same_gate` exercises this over the user/admin protocols with a one-entry cache and capacity pressure from four other namespaces. - Namespaces in `SOURCE_DRAINING`, `SOURCE_READ_DRAINING` or `TARGET_IMPORT_DRAINING` after a restart stay closed until the same command is replayed. Nothing advances in the background. No reader, stream or import call survives a restart, so the replay completes the drain at once. - A restart at any persistence boundary recovers either the state before the command or the state it committed, never anything in between, and never an open namespace unless an opening transition had committed: the `INSTALLING` gate and an indeterminate flag are in memory only, so a crash before the commit recovers the prior state (which was never acknowledged as closed), and a crash after it recovers the committed one. A marker that fell behind (a crash between the metastore commit and the marker write) is repaired when the fence is loaded. - **A crash rebuilds the source's replication log.** A namespace that was not shut down cleanly is recovered by rebuilding its replication log from the database file under a new `log_id` (existing behaviour, not specific to fences). For the fence this means: @@ -850,8 +850,8 @@ Planned test names; the table is updated as tests land. | 4 | Program that captured config before the fence is rejected at the WAL | landed: `connection::connection_manager::fence_tests::wal_gate_rejects_program_admitted_before_fence` (a SQL function parks the program between admission and its write while the fence is acquired and released) | | 5 | Pre-fence transactions cannot write after release or publication | landed: `connection::connection_manager::fence_tests::stale_generation_cannot_write_after_release`, `namespace::fence::target::tests::stale_generation_cannot_write_after_enable_writes`; unfenced behaviour unchanged: `unfenced_namespace_is_unchanged` | | 6 | Acquisition timeout returns `DRAINING`, admission stays closed | landed: `namespace::fence::drain::tests::{deadline_returns_draining_and_stays_closed, replay_of_draining_resumes_and_completes}` | -| 7 | Restart at every persistence boundary; indeterminate persistence keeps the gate closed until same-command reconciliation | landed: `namespace::fence::tests::restart_at_each_boundary` (a crash at each of 17 boundaries of `AcquireSourceWriteFence` and `ReleaseSourceWriteFence`: after the `INSTALLING` gate, before, at and after each metastore commit, with and without a lagging marker, before publication, before boundary capture and before the response; the restart recovers the prior or the committed state, admits writes only if an opening transition committed, repairs the marker, and a replay finishes the command), `restart_in_draining_waits_for_the_same_command` (a writer active at the crash; nothing advances until the same command is replayed, which completes at once), `indeterminate_commit_keeps_gate_closed` (through the drain path, both when the commit happened and when it did not), `acquire_response_loss_resolved_by_replay_and_inspect`; `fence::controller::tests::{indeterminate_commit_keeps_writes_closed_until_replayed, indeterminate_commit_that_did_not_apply_is_retried_by_replay, failed_before_commit_leaves_gate_unchanged, publication_happens_before_the_response, committed_command_is_published_when_the_caller_goes_away}`, `namespace::store::fence_tests::restart_installs_the_durable_gate_before_serving`; the read fence, the seal and write enable: `namespace::fence::tests::read_and_target_boundaries::restart_at_each_read_and_target_boundary` (a crash at each of 29 boundaries of `SetSourceReadFence`, `SealTargetImport` and `EnableTargetWrites`: after the read-closing or `INSTALLING` gate, before, at and after each metastore commit, with and without a lagging marker, before publication, before the response, and while the drain waits for a reader or an import call that was running when it started; the restart recovers the prior or the committed state with no in-memory gate, reader or import call surviving, keeps reads closed once `SOURCE_READ_DRAINING` committed and import closed once `TARGET_IMPORT_DRAINING` committed, admits target writes only if `TARGET_WRITABLE` committed, repairs the marker, and a replay completes an interrupted drain at once or returns the stored result); the rebuilt log after a restart: `namespace::fence::tests::current_log_id_is_the_rebuilt_log_after_restart` (the stored record keeps the acquisition log and its boundary; the controller's `current_log_id`, which the admin view reports, names the rebuilt log) | -| 8 | Evict and lazily reload a fenced namespace; identical admission | `tests::fence::lifecycle::evicted_namespace_reloads_same_gate`; landed at unit level: `namespace::store::fence_tests::evicted_namespace_reloads_with_the_same_controller`, `fence::registry::tests::seeded_from_load_fences_including_recovered_names` | +| 7 | Restart at every persistence boundary; indeterminate persistence keeps the gate closed until same-command reconciliation | landed: `namespace::fence::tests::restart_at_each_boundary` (a crash at each of 17 boundaries of `AcquireSourceWriteFence` and `ReleaseSourceWriteFence`: after the `INSTALLING` gate, before, at and after each metastore commit, with and without a lagging marker, before publication, before boundary capture and before the response; the restart recovers the prior or the committed state, admits writes only if an opening transition committed, repairs the marker, and a replay finishes the command), `restart_in_draining_waits_for_the_same_command` (a writer active at the crash; nothing advances until the same command is replayed, which completes at once), `indeterminate_commit_keeps_gate_closed` (through the drain path, both when the commit happened and when it did not), `acquire_response_loss_resolved_by_replay_and_inspect`; `fence::controller::tests::{indeterminate_commit_keeps_writes_closed_until_replayed, indeterminate_commit_that_did_not_apply_is_retried_by_replay, failed_before_commit_leaves_gate_unchanged, publication_happens_before_the_response, committed_command_is_published_when_the_caller_goes_away}`, `namespace::store::fence_tests::restart_installs_the_durable_gate_before_serving`; the read fence, the seal and write enable: `namespace::fence::tests::read_and_target_boundaries::restart_at_each_read_and_target_boundary` (a crash at each of 29 boundaries of `SetSourceReadFence`, `SealTargetImport` and `EnableTargetWrites`: after the read-closing or `INSTALLING` gate, before, at and after each metastore commit, with and without a lagging marker, before publication, before the response, and while the drain waits for a reader or an import call that was running when it started; the restart recovers the prior or the committed state with no in-memory gate, reader or import call surviving, keeps reads closed once `SOURCE_READ_DRAINING` committed and import closed once `TARGET_IMPORT_DRAINING` committed, admits target writes only if `TARGET_WRITABLE` committed, repairs the marker, and a replay completes an interrupted drain at once or returns the stored result); the rebuilt log after a restart: `namespace::fence::tests::current_log_id_is_the_rebuilt_log_after_restart` (the stored record keeps the acquisition log and its boundary; the controller's `current_log_id`, which the admin view reports, names the rebuilt log); integration restarts: `tests::fence::lifecycle::{restart_keeps_fence (a graceful same-path TestServer restart reconstructs SOURCE_WRITE_FENCED, SOURCE_READ_FENCED, TARGET_QUARANTINED and TARGET_WRITE_FENCED admission before traffic, exact command replays return their stored receipts, and the user protocol observes the same denials), restart_after_enable_writes_stays_writable (TARGET_WRITABLE stays readable/writable after restart and exact EnableTargetWrites replay returns the stored receipt)}` | +| 8 | Evict and lazily reload a fenced namespace; identical admission | landed: `tests::fence::lifecycle::evicted_namespace_reloads_same_gate` (a one-entry namespace cache is put under capacity pressure by four other loaded namespaces; a user-protocol reload of the fenced namespace keeps the exact revision and generation, serves reads and returns `423 MIGRATION_WRITE_FENCED` for writes); unit-level identity: `namespace::store::fence_tests::evicted_namespace_reloads_with_the_same_controller`, `fence::registry::tests::seeded_from_load_fences_including_recovered_names` | | 9 | Filesystem recovery, `destroy_on_error`, undecodable records, missing target quarantine, metastore backup rollback fail closed with provenance | `meta_store::fence_tests::recovery::{fs_recovery_with_marker_unavailable, destroy_on_error_keeps_fenced_unavailable, undecodable_row_unavailable, incomplete_target_unavailable, metastore_rollback_detected_by_marker, lookup_never_creates, undecodable_name_with_fence_fails_startup, marker_in_invalid_directory_fails_startup}`; legacy behaviour kept: `destroy_on_error_without_fences_is_unchanged`, `undecodable_row_without_fences_is_skipped_as_before`; `meta_store::fence_tests::corrupt_fence_row_fails_closed`; restore provenance: `namespace::meta_store::fence_tests::provenance::{bottomless_restore_is_reported (a real bottomless restore against a local S3 endpoint: a metastore opened on an empty directory from a backup reports the restore and its generation and holds the backed-up namespace; one opened with nothing to restore does not), generation_only_after_a_recovery, not_restored_until_recorded_and_first_record_wins}`, `http::admin::fence::tests::restore_provenance_is_reported` (fence view and capability `metastore` object), `tests::fence::admin::capabilities` (not restored: capability endpoint, fence views and the gauge report it) | | 10 | Wrong owner, stale revision, invalid role/state, replay, command-id reuse; replay before revision check | `fence::transition::tests::*` (exhaustive over states × commands) | | 11 | Target creation raced with SQL, dump, replication, lifecycle never observable as writable or readable | landed: `namespace::fence::target::tests::create_race_never_observable` (parked after the rows commit and before the config is published and the namespace loaded: SQL connections, stats, replication `hello` (never `UNAVAILABLE`), create, delete and fork of the name are denied or find nothing; afterwards the target is loaded behind the quarantine gate, SQL reads and WAL writes are refused, and lifecycle and replication are refused with `MIGRATION_TARGET_QUARANTINED`), `creating_gate_refuses_before_commit` (parked before the metastore transaction: the same attempts are refused and no database file is created), `create_replay_completes_interrupted_creation` (marker only, after a restart), `create_completes_when_the_caller_goes_away`, `indeterminate_create_is_completed_by_replay`, `create_rejects_existing_name` (a loaded or cold existing name, whose gate never moves, and another operation's target), `abort_keeps_traffic_denied` (also across a restart) | @@ -925,7 +925,7 @@ Planned commits, each leaving the crate building with its tests passing: 11. `libsql-server: deny lifecycle operations on fenced namespaces` 12. `libsql-replication: add stable error code and replicated fence to protocols` 13. `libsql-server: typed fence outcomes across HTTP, Hrana, RPC, dump and replication` -14. `libsql-server: legacy-binary protection and restart/eviction tests` +14. `libsql-server: test legacy fence mirror and read/target crash boundaries`; `libsql-server: fence restart and eviction integration tests` 15. `libsql-server: namespace fence adoption` 16. `libsql-server: namespace fence metrics and audit log` diff --git a/libsql-server/tests/fence/lifecycle.rs b/libsql-server/tests/fence/lifecycle.rs index 2ec5f77e54..5d57915753 100644 --- a/libsql-server/tests/fence/lifecycle.rs +++ b/libsql-server/tests/fence/lifecycle.rs @@ -1,15 +1,18 @@ //! Lifecycle and configuration operations on fenced namespaces, over the admin API //! (`docs/NAMESPACE_FENCE.md` section 3.3, the lifecycle column; section 17 row 17). +use std::sync::Arc; + use hyper::StatusCode; use libsql::Value as SqlValue; use serde_json::{json, Value}; use tempfile::tempdir; +use tokio::sync::Notify; use uuid::Uuid; use super::{ - acquire_body, command_body, connect, load_and_log_id, make_primary, sim, state_of, Admin, - Primary, ADMIN_KEY, + acquire_body, command_body, connect, load_and_log_id, make_primary, make_restartable_primary, + sim, state_of, user_execute, Admin, Primary, ADMIN_KEY, }; fn uuid(n: u128) -> Uuid { @@ -200,3 +203,382 @@ fn lifecycle_rejected_while_fenced() { }); sim.run().unwrap(); } + +#[track_caller] +fn assert_fence(body: &Value, state: &str, revision: u64, write: &str, read: &str) { + assert_eq!(state_of(body), (state, revision), "{body}"); + assert_eq!(body["fence"]["admission"]["write"], write, "{body}"); + assert_eq!(body["fence"]["admission"]["read"], read, "{body}"); +} + +async fn assert_user_locked(ns: &str, sql: &str, code: &str) -> anyhow::Result<()> { + let (status, body) = user_execute(ns, sql).await?; + assert_eq!(status, StatusCode::LOCKED, "{ns}: {body}"); + assert_eq!(body["code"], code, "{ns}: {body}"); + Ok(()) +} + +async fn assert_user_ok(ns: &str, sql: &str) -> anyhow::Result<()> { + let (status, body) = user_execute(ns, sql).await?; + assert_eq!(status, StatusCode::OK, "{ns}: {body}"); + Ok(()) +} + +/// Create a target and move it through validation into `TARGET_WRITE_FENCED`. Returns the exact +/// publish request and response so a restart test can replay the command and compare its receipt. +async fn target_write_fenced( + admin: &Admin, + ns: &str, + op: Uuid, + command_base: u128, +) -> anyhow::Result<(Value, Value)> { + let (status, created) = admin + .command( + ns, + "target/create-quarantined", + command_body(op, uuid(command_base), "ABSENT", 0, json!({})), + ) + .await?; + assert_eq!(status, StatusCode::OK, "{created}"); + let (_, revision) = state_of(&created); + + let (status, sealed) = admin + .command( + ns, + "target/seal-import", + command_body( + op, + uuid(command_base + 1), + "TARGET_QUARANTINED", + revision, + json!({}), + ), + ) + .await?; + assert_eq!(status, StatusCode::OK, "{sealed}"); + let (_, revision) = state_of(&sealed); + + let (status, validated) = admin + .command( + ns, + "target/validation-receipt", + command_body( + op, + uuid(command_base + 2), + "TARGET_VALIDATING", + revision, + json!({ "result": "ok", "summary": "restart integration" }), + ), + ) + .await?; + assert_eq!(status, StatusCode::OK, "{validated}"); + let (_, revision) = state_of(&validated); + + let publish = command_body( + op, + uuid(command_base + 3), + "TARGET_VALIDATING", + revision, + json!({}), + ); + let (status, published) = admin + .command(ns, "target/publish-readable", publish.clone()) + .await?; + assert_eq!(status, StatusCode::OK, "{published}"); + assert_eq!(state_of(&published).0, "TARGET_WRITE_FENCED", "{published}"); + Ok((publish, published)) +} + +/// Capacity eviction removes the live namespace but not its controller. A reload through the user +/// protocol therefore has the same revision/generation and still rejects writes. +#[test] +fn evicted_namespace_reloads_same_gate() { + let mut sim = sim(); + let tmp = tempdir().unwrap(); + make_primary( + &mut sim, + tmp.path().to_path_buf(), + Primary { + max_active_namespaces: 1, + ..Primary::default() + }, + ); + sim.client("client", async { + let admin = Admin::new(Some(ADMIN_KEY)); + admin.create_namespace("evicted").await?; + let log_id = load_and_log_id(&admin, "evicted").await?; + let operation = uuid(0xac00); + let (status, fenced) = admin + .command( + "evicted", + "source/acquire-write-fence", + acquire_body(operation, uuid(0xac01), &log_id), + ) + .await?; + assert_eq!(status, StatusCode::OK, "{fenced}"); + let (state, revision) = state_of(&fenced); + let generation = fenced["fence"]["admission"]["generation"].clone(); + assert_fence(&fenced, "SOURCE_WRITE_FENCED", revision, "closed", "open"); + assert_eq!(state, "SOURCE_WRITE_FENCED"); + + // With capacity one, creating each distinct namespace loads it and forces the prior live + // namespace out. More than one successor also lets the asynchronous eviction listener + // finish before `evicted` is accessed again. + for i in 0..4 { + admin.create_namespace(&format!("pressure-{i}")).await?; + } + + // This read reloads `evicted`; the controller is outside the capacity-limited cache. + assert_user_ok("evicted", "select count(*) from t").await?; + assert_user_locked( + "evicted", + "insert into t values (2)", + "MIGRATION_WRITE_FENCED", + ) + .await?; + let (status, reloaded) = admin.inspect("evicted").await?; + assert_eq!(status, StatusCode::OK, "{reloaded}"); + assert_fence(&reloaded, "SOURCE_WRITE_FENCED", revision, "closed", "open"); + assert_eq!( + reloaded["fence"]["admission"]["generation"], generation, + "{reloaded}" + ); + Ok(()) + }); + sim.run().unwrap(); +} + +/// Every active source/target gate is reconstructed before traffic after a clean server restart, +/// and the exact command which produced it still replays its durable receipt. +#[test] +fn restart_keeps_fence() { + let mut sim = sim(); + let tmp = tempdir().unwrap(); + let restart = Arc::new(Notify::new()); + let restarted = Arc::new(Notify::new()); + make_restartable_primary( + &mut sim, + tmp.path().to_path_buf(), + Primary::default(), + restart.clone(), + restarted.clone(), + ); + sim.client("client", async move { + let admin = Admin::new(Some(ADMIN_KEY)); + + admin.create_namespace("source-write").await?; + let log_id = load_and_log_id(&admin, "source-write").await?; + let source_write_op = uuid(0xac10); + let source_write_request = acquire_body(source_write_op, uuid(0xac11), &log_id); + let (status, source_write) = admin + .command( + "source-write", + "source/acquire-write-fence", + source_write_request.clone(), + ) + .await?; + assert_eq!(status, StatusCode::OK, "{source_write}"); + + admin.create_namespace("source-read").await?; + let log_id = load_and_log_id(&admin, "source-read").await?; + let source_read_op = uuid(0xac20); + let (status, acquired) = admin + .command( + "source-read", + "source/acquire-write-fence", + acquire_body(source_read_op, uuid(0xac21), &log_id), + ) + .await?; + assert_eq!(status, StatusCode::OK, "{acquired}"); + let source_read_request = command_body( + source_read_op, + uuid(0xac22), + "SOURCE_WRITE_FENCED", + state_of(&acquired).1, + json!({ "drain_policy": { "deadline_ms": 5000 } }), + ); + let (status, source_read) = admin + .command( + "source-read", + "source/set-read-fence", + source_read_request.clone(), + ) + .await?; + assert_eq!(status, StatusCode::OK, "{source_read}"); + + let target_quarantined_op = uuid(0xac30); + let target_quarantined_request = + command_body(target_quarantined_op, uuid(0xac31), "ABSENT", 0, json!({})); + let (status, target_quarantined) = admin + .command( + "target-quarantined", + "target/create-quarantined", + target_quarantined_request.clone(), + ) + .await?; + assert_eq!(status, StatusCode::OK, "{target_quarantined}"); + + let (target_write_request, target_write) = + target_write_fenced(&admin, "target-write-fenced", uuid(0xac40), 0xac41).await?; + + let expected = [ + ( + "source-write", + "SOURCE_WRITE_FENCED", + state_of(&source_write).1, + "closed", + "open", + ), + ( + "source-read", + "SOURCE_READ_FENCED", + state_of(&source_read).1, + "closed", + "closed", + ), + ( + "target-quarantined", + "TARGET_QUARANTINED", + state_of(&target_quarantined).1, + "closed", + "closed", + ), + ( + "target-write-fenced", + "TARGET_WRITE_FENCED", + state_of(&target_write).1, + "closed", + "open", + ), + ]; + for (ns, state, revision, write, read) in expected { + let (status, body) = admin.inspect(ns).await?; + assert_eq!(status, StatusCode::OK, "{body}"); + assert_fence(&body, state, revision, write, read); + } + + restart.notify_waiters(); + restarted.notified().await; + // A response proves the restarted admin server is serving from the rebuilt store. + let admin = Admin::new(Some(ADMIN_KEY)); + assert_eq!(admin.get("/v1/fence/capabilities").await?.0, StatusCode::OK); + + for (ns, state, revision, write, read) in expected { + let (status, body) = admin.inspect(ns).await?; + assert_eq!(status, StatusCode::OK, "{body}"); + assert_fence(&body, state, revision, write, read); + } + + for (ns, route, request, original) in [ + ( + "source-write", + "source/acquire-write-fence", + source_write_request, + source_write, + ), + ( + "source-read", + "source/set-read-fence", + source_read_request, + source_read, + ), + ( + "target-quarantined", + "target/create-quarantined", + target_quarantined_request, + target_quarantined, + ), + ( + "target-write-fenced", + "target/publish-readable", + target_write_request, + target_write, + ), + ] { + let (status, replay) = admin.command(ns, route, request).await?; + assert_eq!(status, StatusCode::OK, "{ns}: {replay}"); + assert_eq!(replay["replayed"], true, "{ns}: {replay}"); + assert_eq!(replay["receipt"], original["receipt"], "{ns}: {replay}"); + } + + assert_user_ok("source-write", "select count(*) from t").await?; + assert_user_locked( + "source-write", + "insert into t values (2)", + "MIGRATION_WRITE_FENCED", + ) + .await?; + assert_user_locked("source-read", "select * from t", "MIGRATION_READ_FENCED").await?; + assert_user_locked( + "target-quarantined", + "select 1", + "MIGRATION_TARGET_QUARANTINED", + ) + .await?; + assert_user_ok("target-write-fenced", "select 1").await?; + assert_user_locked( + "target-write-fenced", + "create table denied (x)", + "MIGRATION_WRITE_FENCED", + ) + .await?; + Ok(()) + }); + sim.run().unwrap(); +} + +/// `TARGET_WRITABLE` is durable too: a restart cannot put the target back in quarantine or close +/// writes, and a lost enable-writes response remains inspectable through exact replay. +#[test] +fn restart_after_enable_writes_stays_writable() { + let mut sim = sim(); + let tmp = tempdir().unwrap(); + let restart = Arc::new(Notify::new()); + let restarted = Arc::new(Notify::new()); + make_restartable_primary( + &mut sim, + tmp.path().to_path_buf(), + Primary::default(), + restart.clone(), + restarted.clone(), + ); + sim.client("client", async move { + let admin = Admin::new(Some(ADMIN_KEY)); + let operation = uuid(0xac50); + let (_, published) = + target_write_fenced(&admin, "target-writable", operation, 0xac51).await?; + let enable = command_body( + operation, + uuid(0xac55), + "TARGET_WRITE_FENCED", + state_of(&published).1, + json!({}), + ); + let (status, enabled) = admin + .command("target-writable", "target/enable-writes", enable.clone()) + .await?; + assert_eq!(status, StatusCode::OK, "{enabled}"); + let revision = state_of(&enabled).1; + assert_fence(&enabled, "TARGET_WRITABLE", revision, "open", "open"); + assert_user_ok("target-writable", "create table before_restart (x)").await?; + + restart.notify_waiters(); + restarted.notified().await; + let admin = Admin::new(Some(ADMIN_KEY)); + let (status, body) = admin.inspect("target-writable").await?; + assert_eq!(status, StatusCode::OK, "{body}"); + assert_fence(&body, "TARGET_WRITABLE", revision, "open", "open"); + assert_user_ok("target-writable", "insert into before_restart values (1)").await?; + assert_user_ok("target-writable", "create table after_restart (x)").await?; + + let (status, replay) = admin + .command("target-writable", "target/enable-writes", enable) + .await?; + assert_eq!(status, StatusCode::OK, "{replay}"); + assert_eq!(replay["replayed"], true, "{replay}"); + assert_eq!(replay["receipt"], enabled["receipt"], "{replay}"); + assert_fence(&replay, "TARGET_WRITABLE", revision, "open", "open"); + Ok(()) + }); + sim.run().unwrap(); +} diff --git a/libsql-server/tests/fence/mod.rs b/libsql-server/tests/fence/mod.rs index 5e97ed7ec1..b7a64e2b42 100644 --- a/libsql-server/tests/fence/mod.rs +++ b/libsql-server/tests/fence/mod.rs @@ -7,6 +7,7 @@ mod lifecycle; mod protocol; use std::path::PathBuf; +use std::sync::Arc; use std::time::Duration; use hyper::StatusCode; @@ -18,6 +19,7 @@ use libsql_server::config::{ }; use s3s::header::AUTHORIZATION; use serde_json::{json, Value}; +use tokio::sync::Notify; use turmoil::{Builder, Sim}; use uuid::Uuid; @@ -28,6 +30,7 @@ use crate::common::net::{ pub const ADMIN_KEY: &str = "fence-admin-key"; +#[derive(Clone, Copy)] pub struct Primary { /// `None` starts the admin API without an auth key. pub admin_key: Option<&'static str>, @@ -36,6 +39,8 @@ pub struct Primary { pub user_credential: Option<&'static str>, /// The fence adoption key; `None` leaves adoption disabled. pub adoption_key: Option<&'static str>, + /// Capacity of the live namespace cache. Fence controllers live outside this cache. + pub max_active_namespaces: usize, } impl Default for Primary { @@ -45,6 +50,7 @@ impl Default for Primary { fence_enabled: true, user_credential: None, adoption_key: None, + max_active_namespaces: 100, } } } @@ -55,46 +61,79 @@ pub fn sim() -> Sim<'static> { .build() } -/// A primary on host `primary`: user API on 8080, admin API on 9090. -pub fn make_primary(sim: &mut Sim, path: PathBuf, primary: Primary) { - init_tracing(); +async fn primary_server(path: PathBuf, primary: Primary) -> anyhow::Result { let Primary { admin_key, fence_enabled, user_credential, adoption_key, + max_active_namespaces, } = primary; + Ok(TestServer { + path: path.into(), + user_api_config: UserApiConfig { + auth_strategy: match user_credential { + Some(credential) => Auth::new(HttpBasic::new(credential.into())), + None => UserApiConfig::::default().auth_strategy, + }, + ..Default::default() + }, + admin_api_config: Some(AdminApiConfig { + acceptor: TurmoilAcceptor::bind(([0, 0, 0, 0], 9090)).await?, + connector: TurmoilConnector, + disable_metrics: true, + auth_key: admin_key.map(Into::into), + }), + rpc_server_config: Some(RpcServerConfig { + acceptor: TurmoilAcceptor::bind(([0, 0, 0, 0], 4567)).await?, + tls_config: None, + }), + meta_store_config: MetaStoreConfig { + namespace_fence: fence_enabled, + namespace_fence_adoption_key: adoption_key.and_then(FenceAdoptionKey::new), + ..Default::default() + }, + disable_namespaces: false, + disable_default_namespace: true, + max_active_namespaces, + ..Default::default() + }) +} + +/// A primary on host `primary`: user API on 8080, admin API on 9090. +pub fn make_primary(sim: &mut Sim, path: PathBuf, primary: Primary) { + init_tracing(); sim.host("primary", move || { let path = path.clone(); async move { - let server = TestServer { - path: path.into(), - user_api_config: UserApiConfig { - auth_strategy: match user_credential { - Some(credential) => Auth::new(HttpBasic::new(credential.into())), - None => UserApiConfig::::default().auth_strategy, - }, - ..Default::default() - }, - admin_api_config: Some(AdminApiConfig { - acceptor: TurmoilAcceptor::bind(([0, 0, 0, 0], 9090)).await?, - connector: TurmoilConnector, - disable_metrics: true, - auth_key: admin_key.map(Into::into), - }), - rpc_server_config: Some(RpcServerConfig { - acceptor: TurmoilAcceptor::bind(([0, 0, 0, 0], 4567)).await?, - tls_config: None, - }), - meta_store_config: MetaStoreConfig { - namespace_fence: fence_enabled, - namespace_fence_adoption_key: adoption_key.and_then(FenceAdoptionKey::new), - ..Default::default() - }, - disable_namespaces: false, - disable_default_namespace: true, - ..Default::default() - }; + primary_server(path, primary).await?.start_sim(8080).await?; + Ok(()) + } + }); +} + +/// A primary that shuts down when `restart` is notified, then starts again on the same path. +/// `restarted` is notified after the second server has rebound the admin and RPC listeners; an +/// admin request made after that notification is the readiness barrier for the restarted server. +pub fn make_restartable_primary( + sim: &mut Sim, + path: PathBuf, + primary: Primary, + restart: Arc, + restarted: Arc, +) { + init_tracing(); + sim.host("primary", move || { + let path = path.clone(); + let restart = restart.clone(); + let restarted = restarted.clone(); + async move { + let mut server = primary_server(path.clone(), primary).await?; + server.shutdown = restart; + server.start_sim(8080).await?; + + let server = primary_server(path, primary).await?; + restarted.notify_one(); server.start_sim(8080).await?; Ok(()) } @@ -266,6 +305,18 @@ pub fn connect(ns: &str) -> anyhow::Result { Ok(db.connect()?) } +/// Execute one statement through the Hrana v1 user endpoint, returning its typed body. +pub async fn user_execute(ns: &str, sql: &str) -> anyhow::Result<(StatusCode, Value)> { + let response = Client::new() + .post( + &format!("http://{ns}.primary:8080/v1/execute"), + json!({ "stmt": { "sql": sql } }), + ) + .await?; + let status = response.status(); + Ok((status, response.json_value().await?)) +} + /// Load `ns` on the server with one write, and return the replication log id the server /// reports for it. pub async fn load_and_log_id(admin: &Admin, ns: &str) -> anyhow::Result { From 034d33e31b0ac391d9ac782c853dcf31e4e89d94 Mon Sep 17 00:00:00 2001 From: River Date: Wed, 30 Sep 2026 12:22:10 +0000 Subject: [PATCH 29/33] libsql-server: namespace fence metrics and audit log Every fence command the server answers (committed, replayed or refused) now emits one structured event under the libsql_server::fence::audit tracing target, with the namespace, operation and command id, outcome, revisions, state before and after, drain kind and duration, forced actions, replay/conflict and server instance; a committed adoption keeps its approvers, incident reference and reason. Metrics, all with bounded labels (docs/NAMESPACE_FENCE.md section 15): transitions by command and outcome, drain duration by kind, forced rollbacks and cancellations by kind, replays and conflicts, denials by code and surface (http, hrana, rpc, proxy, dump, replication, admin_shell, lifecycle), adoptions, and two gauges computed from the fence registry when /metrics is read: namespaces by role and state, and the age of the oldest active fence. Co-authored-by: Tomasz Szymczyszyn --- docs/NAMESPACE_FENCE.md | 14 +- libsql-server/src/admin_shell.rs | 6 +- libsql-server/src/error.rs | 17 +- libsql-server/src/hrana/batch.rs | 8 +- libsql-server/src/hrana/stmt.rs | 8 +- libsql-server/src/http/admin/mod.rs | 6 +- libsql-server/src/http/user/dump.rs | 20 +- libsql-server/src/namespace/fence/audit.rs | 893 ++++++++++++++++-- .../src/namespace/fence/controller.rs | 19 +- libsql-server/src/namespace/fence/drain.rs | 21 +- libsql-server/src/namespace/fence/import.rs | 20 +- libsql-server/src/namespace/fence/read.rs | 19 +- libsql-server/src/namespace/fence/registry.rs | 19 +- libsql-server/src/namespace/meta_store.rs | 15 +- libsql-server/src/namespace/store.rs | 33 +- libsql-server/src/rpc/proxy.rs | 26 +- .../src/rpc/replication/replication_log.rs | 7 +- libsql-server/tests/fence/mod.rs | 1 + libsql-server/tests/fence/observability.rs | 275 ++++++ 19 files changed, 1300 insertions(+), 127 deletions(-) create mode 100644 libsql-server/tests/fence/observability.rs diff --git a/docs/NAMESPACE_FENCE.md b/docs/NAMESPACE_FENCE.md index a757c61e46..68d611ca9f 100644 --- a/docs/NAMESPACE_FENCE.md +++ b/docs/NAMESPACE_FENCE.md @@ -723,7 +723,7 @@ Implementation (`http/admin/fence.rs`, `NamespaceStore::execute_fence_command_au - **What changes.** The owner, the revision, `last_command_id`, `written_by` and the appended adoption entry; the receipt carries the same entry. The state, and so every admission, is unchanged, as are the identity, the frozen boundary, the validation and the saved legacy values; the legacy mirror in the config row names the new owner. Because the owner changed, the write generation moves (section 7.2: nothing can write in a state adoption can act on, so this refuses nothing that was admitted) and every capability issued to the old owner is revoked: the old owner's import or validation session stops working and the new owner opens its own. `RELEASED`, `TARGET_WRITABLE` and `TARGET_ABORTED` are refused with `INVALID_FENCE_TRANSITION` / `operation_finished`. - **After a metastore rollback.** When the marker is ahead of the metastore (`metastore_behind_marker`: the fence row is missing or older), the adoption names the marker's state, revision and owner (the `marker_record` that `InspectFence` reports), and commits the marker's record with the new owner at the marker's revision + 1, as an update of the older row or an insert when the row is gone. The name leaves the unavailable set, its gate is the re-established record's, and its in-memory config gets the namespace's own `block_*` values back (the saved values in the record), as startup does for every established record. No other unavailable reason can be adopted: a corrupt record, an unsupported format version, an interrupted target creation (which only its own replay completes) and an indeterminate commit (which only its own replay reconciles) keep refusing it. - **A name the metastore holds no configuration for.** A metastore restored from a backup older than the namespace itself has neither the fence row nor the config row. Adoption is then refused with `FENCE_PRECONDITION_FAILED` / `namespace_config_missing`, after the authorisation checks, and writes nothing; the name stays `UNKNOWN_UNAVAILABLE`. The marker holds the fence record, not the namespace's configuration (its JWT key, size limit, durability mode, backup id, attach and shared-schema settings), and re-creating the config row from defaults would silently change who can read the namespace and how it is stored once it is released. Recovering such a namespace is an operator decision outside the fence: put its config row back (for example from a newer metastore backup), after which the adoption applies; or discard the namespace directory, marker included. -- **Audit.** Every committed adoption (not a replay, not a refusal) emits one `info` event with target `libsql_server::fence::audit` and `event = "namespace_fence_adopted"`, carrying the namespace, command, outcome, state, previous and new operation id, command id, approvers, incident reference, reason, revisions before and after and the server instance. Section 15's events for the other transitions use the same target. +- **Audit.** Every committed adoption (not a replay, not a refusal) emits one `info` event with target `libsql_server::fence::audit` and `event = "namespace_fence_adopted"`, carrying the namespace, command, outcome, state, previous and new operation id, command id, approvers, incident reference, reason, revisions before and after and the server instance. Every other command answer is an event under the same target (section 15); a replay of an adoption is an ordinary command event. ## 13. Deployment and compatibility (Contract) @@ -828,6 +828,16 @@ Metrics (labels are bounded; namespace, operation id, command id, revision and c Every transition emits one structured log event (target `libsql_server::fence::audit`) with namespace, operation id, command id, command, outcome, revisions, state before and after, drain duration and forced actions, replay/conflict, server instance, and for adoption the approvers and incident reference. +Implementation (`namespace/fence/audit.rs`): + +- **One event per command answered.** `FenceController::execute` (every command but `CreateTargetQuarantined`) and `NamespaceStore::create_target_quarantined` run the command on their own task under the transition lock and, before releasing it, emit one `info` event and count it in `libsql_server_fence_transitions_total{command, outcome}`: `event = "namespace_fence_command"` for an answer (`APPLIED`, `ALREADY_APPLIED`, `DRAINING`, with `replay = none | replay | resume`, where `resume` is a replay of a `DRAINING` command that picks its drain up again), `"namespace fence command refused"` for a refusal (the outcome code and `detail`, the expected revision, the error text, `replay = conflict` for `FENCE_COMMAND_CONFLICT`); a committed adoption is `event = "namespace_fence_adopted"` with the fields of section 12. `state_before` is the published state when the command took the lock; `drain`/`drain_ms` and `forced` are what the command's drain did. A command refused before it reaches the controller (a namespace that cannot be loaded for an acquisition, an admin request that fails validation) is not a transition and emits nothing. A non-fence failure is counted with `outcome = "ERROR"`. +- **Drain duration** is observed when a drain is proven, from the start of the wait (after the `*_DRAINING` commit, or the replay that resumes it) to the proof; a drain that answers `DRAINING` is not observed. +- **Forced actions** are counted when the drain acts: `rollback` per connection manager holding a writer when `force_rollback` fires (write and import drains); `sql_cancel`, `dump_cancel` and `stream_termination` per read lease the read drain asks to stop at its deadline. +- **Replays**: `result = replay` for an answer that was a replay or a resumption, `result = conflict` for a command-id reuse. +- **Denials** are counted at the surface that refuses: `http` (every fence error answered as an HTTP error body: legacy `/`, the admin API's lifecycle refusals, schema errors), `hrana` (step and request errors on `/v1`, `/v2`, `/v3`, cursors and WebSocket), `rpc` (the primary's proxy service: step errors carrying `stable_code` and typed statuses), `proxy` (a replica mapping a denial its primary returned), `dump` (refused or cancelled `/dump`), `replication` (the replication service), `admin_shell` (a query the admin shell's read admission refuses; a write the WAL gate refuses inside the admin shell is not counted separately) and `lifecycle` (the lifecycle checks of section 13.5 and the metastore's config-write and delete refusals). A denial is counted once at each surface it crosses: a lifecycle refusal over the admin API is `lifecycle` and `http`; a write through a replica is `rpc` on the primary and `proxy` plus the user protocol on the replica. +- **Gauges** are computed from the fence registry when `/metrics` is read: `libsql_server_fence_namespaces{role = source | target | unknown, state}` for every state a record or an unavailable namespace can be in (each pair is written, so an emptied state reads 0; `unknown` is `UNKNOWN_UNAVAILABLE`), and `libsql_server_fence_oldest_active_age_seconds` from the oldest active record's creation time (0 when there is none). +- `libsql_server_fence_adoptions_total` counts committed adoptions; `libsql_server_metastore_restored_from_backup` is set at startup (section 13.3). + ## 16. Test strategy (Design) - **No new dependency.** The crate has no failpoint library. Race tests use `#[cfg(test)]` hooks: `FenceTestHooks` holds named points (`AfterInstallingGate`, `AfterClosingReads`, `BeforeReadLeaseCancel`, `BeforeMetastoreCommit`, `AfterMetastoreCommit`, `BeforeGatePublish`, `InBeginWriteTxnAfterCheck`, `AfterManagerRelease`, `BeforeBoundaryCapture`, `AfterTargetRowsCommitted`, `BeforeResponse`), each able to park the task on a pair of `Notify`s (`pause_at` returns handles to wait until the task arrives and to release it) or to inject an error or an indeterminate commit. An armed point fires once. `BeforeMetastoreCommit` is reached immediately before the metastore transaction is started (an injected error there is a failure before commit); an injected indeterminate outcome at `AfterMetastoreCommit` is a commit that happened but was not acknowledged. Hooks compile only in the library's own test build, so they cost nothing in release builds; integration tests under `tests/` cover protocol behaviour and do not rely on hooks. @@ -863,7 +873,7 @@ Planned test names; the table is updated as tests land. | 17 | Delete, reset, fork, restore, config, schema mutation rejected | landed: `tests::fence::lifecycle::lifecycle_rejected_while_fenced` (over the admin API, for a write-fenced source and a quarantined target: delete, fork as source and as destination, create with a `dump_url` whose file does not exist, create over the record, linking to a shared schema at creation and config `POST` are all `423` with the fence code; state, revision and data are unchanged and no copy exists; a target cannot be created with a shared schema; after release the source takes writes and config, fork and delete work again); `namespace::store::fence_tests::{reset_refused_while_fenced (called directly and as the replicator's reset callback; after release reset works and wipes the data), lifecycle_refused_while_fenced (fork either side, create over, delete, config and shared-schema link in the metastore transaction)}`; `schema::scheduler::test::fence::{acquire_rejects_shared_schema, migration_not_registered_while_linked_namespace_fenced}` | | 18 | Codes through HTTP, Hrana, RPC, dump, replication, replica write proxy; distinguishable from auth/timeout/not-found; old peers compatible; no retry loops | user protocols landed: `tests::fence::protocol::{http_codes (legacy `/`, `/v1/execute`, `/v1/batch` for a write-fenced write, a read-fenced read, a quarantined target and a namespace whose marker cannot be decoded: `423` with `code`, and `detail` where there is one), hrana_http_codes (`/v2`, `/v3` pipelines and `/v3/cursor`: step and whole-request errors carry the code and the baton stays usable), hrana_ws_codes (the same over a WebSocket, whose stream reads again after the read fence is cleared), dump_codes, auth_and_not_found_distinct (`401` without or with a wrong credential and `404` for a missing namespace, with no fence code, on the same fenced server)}`, `error::fence_tests::fence_errors_carry_code` (the body through every wrapper; `block_*`'s `Blocked` keeps its mapping); RPC and replica write proxy landed: `rpc::proxy::fence_tests::rpc_codes` (on the primary's proxy service: a write-fenced write is a step error with `SQL_ERROR` + `stable_code`, reads are served, a read-fenced read is a program error with the code when streamed and the typed `FAILED_PRECONDITION` status when unary, and a namespace whose fence state is unknown is refused with the typed status before any connection), `rpc::proxy::fence_tests::replica_maps_proxied_denials` (step, program and connection-status denials become the fence error on the replica; an older primary's error without `stable_code`, an unknown code and other errors keep their mapping), `namespace::fence::outcome::tests::peer_denials_round_trip`, `tests::fence::protocol::{replica_proxy_preserves_code (writes through a replica to a write-fenced primary: `423` + `MIGRATION_WRITE_FENCED` on legacy `/` and `/v1/execute`, the step error on `/v1/batch`, the Hrana error code on `/v2` and `/v3`; reads on the replica served; writes through the replica work after release), denial_not_retried (each refused write is delegated exactly once and answered well within the write proxy's first retry backoff)}`; capability: `tests::fence::admin::capabilities` asserts `proxy_stable_code: true`; replication and replica servers landed: `tests::fence::protocol::{replication_codes (a raw peer of the primary's internal replication service: `hello` carries the write fence's state and revision, an open `log_entries` stream ends with `FAILED_PRECONDITION` + `x-libsql-fence-code` `MIGRATION_READ_FENCED` under the read fence, `hello`, `log_entries` and `snapshot` are then refused with it, `hello` is answered again after the clear and carries no fence after release, and a quarantined target refuses `hello` with `MIGRATION_TARGET_QUARANTINED`), replica_reads_denied_while_source_read_fenced (the replica refuses local reads with `423` + `MIGRATION_READ_FENCED` on legacy `/` and `/v1/execute` and the Hrana code on `/v2` within 500 ms of simulated time after the read fence is acknowledged, and still 30 s later), replica_backs_off_on_fence_code (4 to 9 refused, counted attempts over 60 s of simulated time; a fixed 1 s retry makes 57), replica_resumes_after_clear_read_fence (reads served again within 16 s of the clear, and a write after release is replicated), replica_lazy_creation_refused_by_fence (a read of a quarantined target through a replica that has never loaded it answers `423` + `MIGRATION_TARGET_QUARANTINED` within 500 ms of simulated time, twice, leaving no `dbs/` directory; a directory that was already there is kept; after publication the replica creates and serves the name, and after enable-writes a write through it succeeds)}`, `namespace::meta_store::fence_tests::forget_unstored_only_unused_unstored_entries`, `namespace::fence::registry::tests::forget_idle_only_unreferenced_plain_controllers`, `namespace::fence::replica::tests::{refusal_from_typed_status_only, hello_fence_denies_only_read_denying_states, backoff_doubles_to_its_cap, observed_denial_refuses_local_reads_and_cancels_leases}`; landed: `libsql-replication` `rpc::test::{proxy_error_stable_code_is_additive, replicated_fence_is_additive}` (each new field is skipped by a peer that does not know it, absent from an older peer's message, and absent fields encode exactly as before), `namespace::fence::stream::tests::hello_carries_replicated_fence` (no fence before acquisition and after release; state and revision while write-fenced; the stored configuration never carries it) | | 19 | Corrupt or unknown durable fence state fails closed | `fence::store::tests::corrupt_payload_fails_closed`, `unknown_format_version_fails_closed` | -| 20 | Metrics and audit logs | `tests::fence::observability::metrics_and_labels`; `fence::audit::tests::audit_event_fields` | +| 20 | Metrics and audit logs | landed: `namespace::fence::audit::tests::{audit_event_fields (one event per answer: a committed drain with state before and after, revisions, drain kind and duration and forced actions; a replay; a command-id conflict with its code; a non-fence failure), adoption_event_fields, metrics_and_bounded_labels (every metric of section 15 with its labels, denials at all eight surfaces, the gauges per role and state including zeroed states and the oldest active age; no label value is a namespace, operation or command id)}`; over a running server: `tests::fence::observability::metrics_and_labels` (a write drain with a forced rollback, a replay, a conflict, a Hrana write denial and a lifecycle refusal over the admin API: transitions, replays, forced, drain histogram, denials under `hrana`, `lifecycle` and `http`, the namespace gauge and oldest-age gauge before and after release, and only bounded label keys and values) | | 21 | Capability discovery and mixed-version protection | capability discovery landed: `tests::fence::admin::{capabilities, capabilities_when_disabled}`; legacy mirror and foreign-key guard landed: `namespace::fence::tests::legacy_mirror::legacy_mirror_and_fk_guard` (a source and two targets walked through every stored state: the config row's `block_*` fields hold the mirror of section 13.2 and nothing else in the row changes, a config write is refused and changes nothing, an older binary's delete fails on the foreign key; release and write enable restore the namespace's own values, later config writes are stored as written, and after a restart the rows are unchanged and the in-memory config holds the namespace's own values, including a config written after the release); an older binary is not run (bounded, see section 18) | | 22 | Adoption is two-person/audited, keeps admission closed, cannot reverse publication | landed: `namespace::fence::tests::adoption::{adopt_requires_key_and_two_approvers (no key, one approver, duplicate or blank approvers, three approvers, blank incident or reason: `adoption_not_authorised`, nothing changes), adopt_keeps_gates_closed (write-fenced source: owner and revision move, state, admissions, boundary and saved values do not, writes still refused, replay with or without the key returns the receipt, old owner `FENCE_OWNED_BY_ANOTHER_OPERATION`, new owner releases), adopt_quarantined_target (the import capability moves to the new owner; SQL still `MIGRATION_TARGET_QUARANTINED`), adopt_cannot_touch_writable (`TARGET_WRITABLE`, `TARGET_ABORTED`), adopt_cannot_touch_released, adopt_recovers_metastore_rollback (fence row gone, and fence row at an older revision: re-established from the marker, served again behind the same gate, own `block_*` values back in memory, new owner finishes), adopt_recovered_name_without_config_row (`namespace_config_missing`, nothing written, still unavailable), adoption_key_matching}`, `namespace::fence::audit::tests::adoption_event_fields`; pure transition: `namespace::fence::transition::tests::{adopt_requires_key_and_two_approvers, adopt_keeps_gates_closed}`; over HTTP: `tests::fence::admin::{adopt_over_http (no key, wrong key, no admin credential, bad approvers, unknown field, success, replay, old and new owner, finished operation), adopt_disabled_without_key}` | | — | Import API usable by bulk import | landed: `namespace::fence::import::tests::import_session_loads_dump_into_quarantined_target` (a dump exported by the server from a source with tables, keys, a foreign key, an index, an autoincrement table, a trigger, a view and an FTS5 table loads through `ImportSession::load_dump`; after the seal the target's schema, rows, view and full-text results equal the source's) | diff --git a/libsql-server/src/admin_shell.rs b/libsql-server/src/admin_shell.rs index 84f11e7fe8..a463944427 100644 --- a/libsql-server/src/admin_shell.rs +++ b/libsql-server/src/admin_shell.rs @@ -87,11 +87,15 @@ fn run_admitted( }) { Ok(lease) => lease, Err(e) => { + crate::namespace::fence::audit::denied( + &e, + crate::namespace::fence::audit::DenialSurface::AdminShell, + ); return Ok(rpc::Response { resp: Some(Resp::Error(rpc::Error { error: e.to_string(), })), - }) + }); } }; let res = run_one(conn, q); diff --git a/libsql-server/src/error.rs b/libsql-server/src/error.rs index b6903c4a82..55627a1d64 100644 --- a/libsql-server/src/error.rs +++ b/libsql-server/src/error.rs @@ -174,7 +174,13 @@ impl Error { crate::namespace::fence::outcome::FenceError::from_proxy_stable_code(code, &e.message) }); match fence { - Some(fence) => Error::NamespaceFence(fence), + Some(fence) => { + crate::namespace::fence::audit::denied( + &fence, + crate::namespace::fence::audit::DenialSurface::Proxy, + ); + Error::NamespaceFence(fence) + } None => Error::RpcQueryError(e), } } @@ -184,7 +190,13 @@ impl Error { /// anything else unchanged. pub(crate) fn from_proxy_status(status: tonic::Status) -> Self { match crate::namespace::fence::outcome::FenceError::from_grpc_status(&status) { - Some(fence) => Error::NamespaceFence(fence), + Some(fence) => { + crate::namespace::fence::audit::denied( + &fence, + crate::namespace::fence::audit::DenialSurface::Proxy, + ); + Error::NamespaceFence(fence) + } None => Error::RpcQueryExecutionError(status), } } @@ -195,6 +207,7 @@ impl Error { pub(crate) fn fence_error_response( e: &crate::namespace::fence::outcome::FenceError, ) -> axum::response::Response { + crate::namespace::fence::audit::denied(e, crate::namespace::fence::audit::DenialSurface::Http); let status = e.http_status(); tracing::debug!("HTTP API: {status}, {e}"); (status, axum::Json(e.http_error_body())).into_response() diff --git a/libsql-server/src/hrana/batch.rs b/libsql-server/src/hrana/batch.rs index 4292660088..baaf96184f 100644 --- a/libsql-server/src/hrana/batch.rs +++ b/libsql-server/src/hrana/batch.rs @@ -187,7 +187,13 @@ pub fn batch_error_from_sqld_error(sqld_error: SqldError) -> Result { BatchError::ResponseTooLarge } - SqldError::NamespaceFence(e) => BatchError::Fence(e), + SqldError::NamespaceFence(e) => { + crate::namespace::fence::audit::denied( + &e, + crate::namespace::fence::audit::DenialSurface::Hrana, + ); + BatchError::Fence(e) + } sqld_error => return Err(sqld_error), }) } diff --git a/libsql-server/src/hrana/stmt.rs b/libsql-server/src/hrana/stmt.rs index 49bb65d2f7..88824ced5c 100644 --- a/libsql-server/src/hrana/stmt.rs +++ b/libsql-server/src/hrana/stmt.rs @@ -220,7 +220,13 @@ pub fn stmt_error_from_sqld_error(sqld_error: SqldError) -> Result Ok(StmtError::Blocked { reason }), SqldError::RpcQueryError(e) => Ok(StmtError::Proxy(e.message)), - SqldError::NamespaceFence(e) => Ok(StmtError::Fence(e)), + SqldError::NamespaceFence(e) => { + crate::namespace::fence::audit::denied( + &e, + crate::namespace::fence::audit::DenialSurface::Hrana, + ); + Ok(StmtError::Fence(e)) + } SqldError::RusqliteError(rusqlite_error) | SqldError::RusqliteErrorExtended(rusqlite_error, _) => match rusqlite_error { rusqlite::Error::SqliteFailure(sqlite_error, Some(message)) => { diff --git a/libsql-server/src/http/admin/mod.rs b/libsql-server/src/http/admin/mod.rs index c54461184f..584e2e814a 100644 --- a/libsql-server/src/http/admin/mod.rs +++ b/libsql-server/src/http/admin/mod.rs @@ -244,8 +244,10 @@ async fn handle_get_index() -> &'static str { "Welcome to the sqld admin API" } -async fn handle_metrics(State(metrics): State) -> String { - metrics.render() +async fn handle_metrics(State(app_state): State>>) -> String { + // The fence gauges are computed from the registry when they are read. + app_state.namespaces.update_fence_gauges(); + app_state.metrics.render() } async fn handle_get_config( diff --git a/libsql-server/src/http/user/dump.rs b/libsql-server/src/http/user/dump.rs index ae0260482e..f278fb9a48 100644 --- a/libsql-server/src/http/user/dump.rs +++ b/libsql-server/src/http/user/dump.rs @@ -131,8 +131,13 @@ pub(crate) async fn dump_stream( conn_maker: Arc>, preserve_row_ids: bool, ) -> crate::Result>> { - let (lease, cancel) = - acquire_stream_lease(fence, LeaseKind::Dump).map_err(Error::NamespaceFence)?; + let (lease, cancel) = acquire_stream_lease(fence, LeaseKind::Dump).map_err(|e| { + crate::namespace::fence::audit::denied( + &e, + crate::namespace::fence::audit::DenialSurface::Dump, + ); + Error::NamespaceFence(e) + })?; let conn = conn_maker.create().await?; @@ -151,9 +156,14 @@ pub(crate) async fn dump_stream( }); match result { Ok(()) => Ok(()), - Err(_) if cancel.is_cancelled() => Err(Error::NamespaceFence(cancelled_by_read_fence( - LeaseKind::Dump, - ))), + Err(_) if cancel.is_cancelled() => { + let e = cancelled_by_read_fence(LeaseKind::Dump); + crate::namespace::fence::audit::denied( + &e, + crate::namespace::fence::audit::DenialSurface::Dump, + ); + Err(Error::NamespaceFence(e)) + } Err(e) => Err(e.into()), } }); diff --git a/libsql-server/src/namespace/fence/audit.rs b/libsql-server/src/namespace/fence/audit.rs index e727a0fea7..063d2484cb 100644 --- a/libsql-server/src/namespace/fence/audit.rs +++ b/libsql-server/src/namespace/fence/audit.rs @@ -1,54 +1,431 @@ -//! The fence audit log (`docs/NAMESPACE_FENCE.md` sections 12 and 15): structured events under -//! the tracing target [`AUDIT_TARGET`], so that a log pipeline can route them apart from the -//! server's operational logs. +//! Fence observability (`docs/NAMESPACE_FENCE.md` sections 12 and 15): the fence metrics and +//! the audit log. +//! +//! Every fence command a server answers emits one structured event under the tracing target +//! [`AUDIT_TARGET`], so that a log pipeline can route them apart from the server's operational +//! logs, and is counted in the metrics below. Metric labels are bounded: a namespace, an +//! operation id, a command id, a revision or a caller never appears as a label value; those +//! are in the audit event. -use crate::namespace::meta_store::FenceCommit; +use std::time::Duration; + +use uuid::Uuid; + +use super::command::CommandKind; +use super::outcome::FenceError; +use super::registry::FenceRegistry; +use super::state::FenceState; +use crate::namespace::meta_store::{FenceCommit, FenceCommitKind}; use crate::namespace::NamespaceName; /// The tracing target of every fence audit event. pub const AUDIT_TARGET: &str = "libsql_server::fence::audit"; -/// One audit event for a committed `AdoptFence`: who adopted what from whom, the two recorded -/// approvers, the incident reference and the reason, and the revisions. The server cannot -/// verify the approvers (section 12); the event is the record of what the request claimed. -pub fn adoption(namespace: &NamespaceName, commit: &FenceCommit) { - let receipt = &commit.receipt; - let Some(adoption) = &receipt.adoption else { - return; - }; - tracing::info!( - target: AUDIT_TARGET, - event = "namespace_fence_adopted", - namespace = %namespace, - command = receipt.command.as_str(), - outcome = receipt.outcome.as_str(), - state = receipt.state_after.as_str(), - previous_operation_id = %adoption.previous_operation_id, - operation_id = %adoption.new_operation_id, - command_id = %adoption.command_id, - approvers = ?adoption.approvers, - incident_ref = %adoption.incident_ref, - reason = %adoption.reason, - revision_before = receipt.revision_before, - revision_after = receipt.revision_after, - server_instance = %receipt.instance_id, - "namespace fence adopted" +pub const TRANSITIONS_TOTAL: &str = "libsql_server_fence_transitions_total"; +pub const DRAIN_DURATION_SECONDS: &str = "libsql_server_fence_drain_duration_seconds"; +pub const FORCED_TOTAL: &str = "libsql_server_fence_forced_total"; +pub const REPLAYS_TOTAL: &str = "libsql_server_fence_replays_total"; +pub const DENIALS_TOTAL: &str = "libsql_server_fence_denials_total"; +pub const NAMESPACES: &str = "libsql_server_fence_namespaces"; +pub const OLDEST_ACTIVE_AGE_SECONDS: &str = "libsql_server_fence_oldest_active_age_seconds"; +pub const ADOPTIONS_TOTAL: &str = "libsql_server_fence_adoptions_total"; + +/// The label value of a command answered with an error that is not a fence outcome (an I/O or +/// metastore failure), in place of an outcome code. +pub const OTHER_ERROR: &str = "ERROR"; + +/// Describe the fence metrics to the recorder, once, so that `/metrics` carries their help +/// text. Called when the namespace store starts. +pub fn describe_metrics() { + static ONCE: std::sync::Once = std::sync::Once::new(); + ONCE.call_once(|| { + metrics::describe_counter!( + TRANSITIONS_TOTAL, + "fence commands answered, by command and outcome code (replays and refusals included)" + ); + metrics::describe_histogram!( + DRAIN_DURATION_SECONDS, + metrics::Unit::Seconds, + "time from the start of a fence drain to its proof, by kind (write, read, import)" + ); + metrics::describe_counter!( + FORCED_TOTAL, + "work ended by a fence drain at its deadline, by kind (rollback, sql_cancel, \ + dump_cancel, stream_termination)" + ); + metrics::describe_counter!( + REPLAYS_TOTAL, + "fence commands that were replays of a recorded command, or reused its command id \ + for a different request (conflict)" + ); + metrics::describe_counter!( + DENIALS_TOTAL, + "requests refused by a namespace fence, by outcome code and surface" + ); + metrics::describe_gauge!( + NAMESPACES, + "namespaces with fence state on this server, by role and state" + ); + metrics::describe_gauge!( + OLDEST_ACTIVE_AGE_SECONDS, + metrics::Unit::Seconds, + "age of the oldest active fence record on this server, 0 when there is none" + ); + metrics::describe_counter!(ADOPTIONS_TOTAL, "committed fence adoptions"); + }); +} + +/// The drain a command waited on. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum DrainKind { + Write, + Read, + Import, +} + +impl DrainKind { + pub const fn as_str(self) -> &'static str { + match self { + DrainKind::Write => "write", + DrainKind::Read => "read", + DrainKind::Import => "import", + } + } +} + +/// Work a drain ended at its deadline instead of waiting for it. +#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord)] +pub enum ForcedKind { + /// A write or import transaction rolled back (`on_deadline: force_rollback`). + Rollback, + /// A running SQL program cancelled by the read drain. + SqlCancel, + /// A dump cancelled by the read drain. + DumpCancel, + /// A replication stream terminated by the read drain. + StreamTermination, +} + +impl ForcedKind { + pub const fn as_str(self) -> &'static str { + match self { + ForcedKind::Rollback => "rollback", + ForcedKind::SqlCancel => "sql_cancel", + ForcedKind::DumpCancel => "dump_cancel", + ForcedKind::StreamTermination => "stream_termination", + } + } +} + +/// Where a request refused by a fence was refused. +/// +/// A denial is counted once at each surface it crosses: a write sent to a replica and refused +/// by the primary is counted under `rpc` on the primary, and under `proxy` and the replica's +/// user protocol on the replica; a lifecycle operation refused over the admin API is counted +/// under `lifecycle` and `http`. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum DenialSurface { + /// An HTTP error response: legacy `/`, the admin API's lifecycle routes, schema errors. + Http, + /// A Hrana error: `/v1`, `/v2`, `/v3`, cursors and WebSocket streams. + Hrana, + /// The primary's proxy service answering a replica. + Rpc, + /// A replica mapping a denial its primary returned through the write proxy. + Proxy, + /// `/dump`. + Dump, + /// The replication service (`hello`, `log_entries`, `batch_log_entries`, `snapshot`). + Replication, + /// The admin shell. + AdminShell, + /// Delete, reset, fork, restore, config and schema operations. + Lifecycle, +} + +impl DenialSurface { + pub const ALL: [DenialSurface; 8] = [ + DenialSurface::Http, + DenialSurface::Hrana, + DenialSurface::Rpc, + DenialSurface::Proxy, + DenialSurface::Dump, + DenialSurface::Replication, + DenialSurface::AdminShell, + DenialSurface::Lifecycle, + ]; + + pub const fn as_str(self) -> &'static str { + match self { + DenialSurface::Http => "http", + DenialSurface::Hrana => "hrana", + DenialSurface::Rpc => "rpc", + DenialSurface::Proxy => "proxy", + DenialSurface::Dump => "dump", + DenialSurface::Replication => "replication", + DenialSurface::AdminShell => "admin_shell", + DenialSurface::Lifecycle => "lifecycle", + } + } +} + +/// Count a request refused by a fence at `surface`. +pub fn denied(error: &FenceError, surface: DenialSurface) { + metrics::increment_counter!( + DENIALS_TOTAL, + "code" => error.outcome().as_str(), + "surface" => surface.as_str(), ); } +/// Count `count` pieces of work of `kind` ended by a drain. +pub fn forced(kind: ForcedKind, count: usize) { + if count > 0 { + metrics::counter!(FORCED_TOTAL, count as u64, "kind" => kind.as_str()); + } +} + +/// What a command's drain did, filled in by the drain while it runs under the transition and +/// reported with the command's audit event. +#[derive(Debug, Clone, Default)] +pub struct CommandReport { + drain: Option<(DrainKind, Duration)>, + forced: Vec, +} + +impl CommandReport { + /// The drain of `kind` was proven after `duration`. Observed in the drain histogram. + pub fn drained(&mut self, kind: DrainKind, duration: Duration) { + metrics::histogram!( + DRAIN_DURATION_SECONDS, + duration.as_secs_f64(), + "kind" => kind.as_str(), + ); + self.drain = Some((kind, duration)); + } + + /// The drain ended `count` pieces of work of `kind` at its deadline. Counted. + pub fn forced(&mut self, kind: ForcedKind, count: usize) { + if count == 0 { + return; + } + forced(kind, count); + if !self.forced.contains(&kind) { + self.forced.push(kind); + self.forced.sort(); + } + } + + pub fn drain(&self) -> Option<(DrainKind, Duration)> { + self.drain + } + + pub fn forced_kinds(&self) -> &[ForcedKind] { + &self.forced + } + + fn drain_ms(&self) -> Option { + self.drain.map(|(_, d)| d.as_millis()) + } + + fn forced_list(&self) -> String { + self.forced + .iter() + .map(|k| k.as_str()) + .collect::>() + .join(",") + } +} + +/// The request fields of a command, taken before the command runs, for its audit event. +#[derive(Debug, Clone)] +pub struct CommandAudit { + pub namespace: NamespaceName, + pub operation_id: Uuid, + pub command_id: Uuid, + pub command: CommandKind, + pub expected_revision: u64, + /// The published state when the command took the transition lock. + pub state_before: FenceState, +} + +impl CommandAudit { + pub fn new(request: &super::command::FenceRequest, state_before: FenceState) -> Self { + Self { + namespace: request.namespace.clone(), + operation_id: request.operation_id, + command_id: request.command_id, + command: request.command.kind(), + expected_revision: request.expected_revision, + state_before, + } + } +} + +/// Count a command's answer and emit its audit event: one per command answered, whether it +/// committed, replayed a recorded answer or was refused. +pub fn command_finished( + audit: &CommandAudit, + result: &crate::Result, + report: &CommandReport, +) { + let command = audit.command.as_str(); + match result { + Ok(commit) => { + let receipt = &commit.receipt; + let replay = match commit.kind { + FenceCommitKind::Committed => None, + FenceCommitKind::Replayed => Some("replay"), + FenceCommitKind::Resumed => Some("resume"), + }; + metrics::increment_counter!( + TRANSITIONS_TOTAL, + "command" => command, + "outcome" => receipt.outcome.as_str(), + ); + if replay.is_some() { + metrics::increment_counter!(REPLAYS_TOTAL, "result" => "replay"); + } + let adoption = receipt + .adoption + .as_ref() + .filter(|_| commit.kind == FenceCommitKind::Committed); + if let Some(adoption) = adoption { + metrics::increment_counter!(ADOPTIONS_TOTAL); + tracing::info!( + target: AUDIT_TARGET, + event = "namespace_fence_adopted", + namespace = %audit.namespace, + command, + outcome = receipt.outcome.as_str(), + state_before = audit.state_before.as_str(), + state = receipt.state_after.as_str(), + previous_operation_id = %adoption.previous_operation_id, + operation_id = %adoption.new_operation_id, + command_id = %adoption.command_id, + approvers = ?adoption.approvers, + incident_ref = %adoption.incident_ref, + reason = %adoption.reason, + revision_before = receipt.revision_before, + revision_after = receipt.revision_after, + server_instance = %receipt.instance_id, + "namespace fence adopted" + ); + return; + } + tracing::info!( + target: AUDIT_TARGET, + event = "namespace_fence_command", + namespace = %audit.namespace, + operation_id = %audit.operation_id, + command_id = %audit.command_id, + command, + outcome = receipt.outcome.as_str(), + replay = replay.unwrap_or("none"), + revision_before = receipt.revision_before, + revision_after = receipt.revision_after, + state_before = audit.state_before.as_str(), + state_after = receipt.state_after.as_str(), + drain = report.drain.map(|(k, _)| k.as_str()).unwrap_or("none"), + drain_ms = ?report.drain_ms(), + forced = %report.forced_list(), + server_instance = %receipt.instance_id, + "namespace fence command answered" + ); + } + Err(error) => { + let fence = error.fence_error(); + let outcome = fence.map(|e| e.outcome().as_str()).unwrap_or(OTHER_ERROR); + metrics::increment_counter!( + TRANSITIONS_TOTAL, + "command" => command, + "outcome" => outcome, + ); + let conflict = fence + .is_some_and(|e| e.outcome() == super::outcome::FenceOutcome::FenceCommandConflict); + if conflict { + metrics::increment_counter!(REPLAYS_TOTAL, "result" => "conflict"); + } + tracing::info!( + target: AUDIT_TARGET, + event = "namespace_fence_command", + namespace = %audit.namespace, + operation_id = %audit.operation_id, + command_id = %audit.command_id, + command, + outcome, + detail = fence.and_then(|e| e.detail()).map(|d| d.as_str()).unwrap_or("none"), + replay = if conflict { "conflict" } else { "none" }, + expected_revision = audit.expected_revision, + state_before = audit.state_before.as_str(), + drain = report.drain.map(|(k, _)| k.as_str()).unwrap_or("none"), + drain_ms = ?report.drain_ms(), + forced = %report.forced_list(), + server_instance = %super::server_identity().instance_id, + error = %error, + "namespace fence command refused" + ); + } + } +} + +/// Set the fence gauges from the registry: how many namespaces are in each state, by role, and +/// the age of the oldest active record. Every (role, state) pair is written, so a state that +/// emptied reads 0. Called before `/metrics` is rendered. +pub fn update_gauges(registry: &FenceRegistry, now_ms: i64) { + let census = registry.census(); + for state in FenceState::ALL { + let Some(role) = gauge_role(state) else { + continue; + }; + let count = census.iter().filter(|(s, _)| *s == state).count(); + metrics::gauge!( + NAMESPACES, + count as f64, + "role" => role, + "state" => state.as_str(), + ); + } + let oldest = census + .iter() + .filter_map(|(state, created_at_ms)| created_at_ms.filter(|_| state.is_active())) + .min(); + let age = oldest + .map(|created| (now_ms.saturating_sub(created)).max(0) as f64 / 1000.0) + .unwrap_or(0.0); + metrics::gauge!(OLDEST_ACTIVE_AGE_SECONDS, age); +} + +/// The `role` label of a state counted in [`NAMESPACES`]: the record's role, `unknown` for a +/// namespace whose fence state cannot be established, none for an ordinary namespace. +fn gauge_role(state: FenceState) -> Option<&'static str> { + match state.role() { + Some(role) => Some(match role { + super::state::Role::Source => "source", + super::state::Role::Target => "target", + }), + None if state == FenceState::UnknownUnavailable => Some("unknown"), + None => None, + } +} + #[cfg(test)] mod tests { + use std::collections::HashMap; use std::io::Write; use std::sync::{Arc, Mutex}; + use metrics_util::debugging::{DebugValue, DebuggingRecorder, Snapshotter}; + use metrics_util::MetricKind; use uuid::Uuid; use super::*; use crate::namespace::fence::command::{AdoptArgs, FenceCommand, FenceRequest}; use crate::namespace::fence::outcome::FenceOutcome; + use crate::namespace::fence::record::tests::sample_record; use crate::namespace::fence::record::{Adoption, CommandReceipt}; use crate::namespace::fence::state::FenceState; - use crate::namespace::meta_store::FenceCommitKind; + use crate::namespace::fence::store::StoredFence; #[derive(Clone, Default)] struct Captured(Arc>>); @@ -64,7 +441,7 @@ mod tests { } } - fn capture(f: impl FnOnce()) -> String { + pub(crate) fn capture(f: impl FnOnce()) -> String { let captured = Captured::default(); let writer = captured.clone(); let subscriber = tracing_subscriber::fmt() @@ -77,32 +454,54 @@ mod tests { String::from_utf8(bytes).unwrap() } - fn commit(adoption: Option) -> FenceCommit { - let args = AdoptArgs { - current_operation_id: Uuid::from_u128(0xa), - approvers: vec!["alice".into(), "bob".into()], - incident_ref: "INC-1".into(), - reason: "control record lost".into(), - }; - let request = FenceRequest { + fn adopt_request() -> FenceRequest { + FenceRequest { namespace: "ns".into(), operation_id: Uuid::from_u128(0xb), command_id: Uuid::from_u128(3), expected_state: FenceState::SourceWriteFenced, expected_revision: 2, - command: FenceCommand::AdoptFence(args), - }; + command: FenceCommand::AdoptFence(AdoptArgs { + current_operation_id: Uuid::from_u128(0xa), + approvers: vec!["alice".into(), "bob".into()], + incident_ref: "INC-1".into(), + reason: "control record lost".into(), + }), + } + } + + fn acquire_request() -> FenceRequest { + FenceRequest { + namespace: "ns".into(), + operation_id: Uuid::from_u128(0xc), + command_id: Uuid::from_u128(4), + expected_state: FenceState::Unfenced, + expected_revision: 0, + command: FenceCommand::AcquireSourceWriteFence { + expected_log_id: Uuid::from_u128(0x10), + drain_policy: None, + }, + } + } + + fn commit( + request: &FenceRequest, + kind: FenceCommitKind, + outcome: FenceOutcome, + (revision_before, revision_after): (u64, u64), + adoption: Option, + ) -> FenceCommit { FenceCommit { - kind: FenceCommitKind::Committed, + kind, receipt: CommandReceipt { namespace: "ns".into(), operation_id: request.operation_id, command_id: request.command_id, command: request.command.kind(), fingerprint: request.fingerprint(), - outcome: FenceOutcome::Applied, - revision_before: 2, - revision_after: 3, + outcome, + revision_before, + revision_after, state_after: FenceState::SourceWriteFenced, applied_at_ms: 1_000, instance_id: Uuid::from_u128(0x99), @@ -113,11 +512,8 @@ mod tests { } } - /// A committed adoption is one event under the audit target with every field section 12 - /// asks for; a receipt without an adoption emits nothing. - #[test] - fn adoption_event_fields() { - let entry = Adoption { + fn adoption_entry() -> Adoption { + Adoption { previous_operation_id: Uuid::from_u128(0xa), new_operation_id: Uuid::from_u128(0xb), command_id: Uuid::from_u128(3), @@ -126,31 +522,382 @@ mod tests { reason: "control record lost".into(), at_ms: 1_000, revision: 3, - }; - let out = capture(|| super::adoption(&"ns".into(), &commit(Some(entry.clone())))); - assert_eq!(out.lines().count(), 1, "{out}"); - for expected in [ - AUDIT_TARGET, - "namespace fence adopted", - "event=\"namespace_fence_adopted\"", - "namespace=ns", - "command=\"AdoptFence\"", - "outcome=\"APPLIED\"", - "state=\"SOURCE_WRITE_FENCED\"", - &format!("previous_operation_id={}", Uuid::from_u128(0xa)), - &format!("operation_id={}", Uuid::from_u128(0xb)), - &format!("command_id={}", Uuid::from_u128(3)), - "approvers=[\"alice\", \"bob\"]", - "incident_ref=INC-1", - "reason=control record lost", - "revision_before=2", - "revision_after=3", - &format!("server_instance={}", Uuid::from_u128(0x99)), - ] { + } + } + + #[track_caller] + fn assert_contains_all(out: &str, expected: &[&str]) { + for expected in expected { assert!(out.contains(expected), "`{expected}` missing from {out}"); } + } + + /// A committed adoption is one event under the audit target with every field section 12 + /// asks for; a replay of it is an ordinary command event without them. + #[test] + fn adoption_event_fields() { + let request = adopt_request(); + let audit = CommandAudit::new(&request, FenceState::SourceWriteFenced); + let committed = commit( + &request, + FenceCommitKind::Committed, + FenceOutcome::Applied, + (2, 3), + Some(adoption_entry()), + ); + let out = capture(|| command_finished(&audit, &Ok(committed), &CommandReport::default())); + assert_eq!(out.lines().count(), 1, "{out}"); + assert_contains_all( + &out, + &[ + AUDIT_TARGET, + "namespace fence adopted", + "event=\"namespace_fence_adopted\"", + "namespace=ns", + "command=\"AdoptFence\"", + "outcome=\"APPLIED\"", + "state_before=\"SOURCE_WRITE_FENCED\"", + "state=\"SOURCE_WRITE_FENCED\"", + &format!("previous_operation_id={}", Uuid::from_u128(0xa)), + &format!("operation_id={}", Uuid::from_u128(0xb)), + &format!("command_id={}", Uuid::from_u128(3)), + "approvers=[\"alice\", \"bob\"]", + "incident_ref=INC-1", + "reason=control record lost", + "revision_before=2", + "revision_after=3", + &format!("server_instance={}", Uuid::from_u128(0x99)), + ], + ); + + let replayed = commit( + &request, + FenceCommitKind::Replayed, + FenceOutcome::Applied, + (2, 3), + Some(adoption_entry()), + ); + let out = capture(|| command_finished(&audit, &Ok(replayed), &CommandReport::default())); + assert_eq!(out.lines().count(), 1, "{out}"); + assert!(!out.contains("namespace_fence_adopted"), "{out}"); + assert_contains_all( + &out, + &["event=\"namespace_fence_command\"", "replay=\"replay\""], + ); + } + + /// Every answer is one event under the audit target: a committed drain with its duration + /// and forced actions, a replay, and a refusal (a command-id conflict) with its code. + #[test] + fn audit_event_fields() { + let request = acquire_request(); + let audit = CommandAudit::new(&request, FenceState::Unfenced); + let mut report = CommandReport::default(); + report.forced(ForcedKind::Rollback, 1); + report.forced(ForcedKind::Rollback, 1); + report.forced(ForcedKind::SqlCancel, 0); + report.drained(DrainKind::Write, Duration::from_millis(1_500)); + assert_eq!(report.forced_kinds(), &[ForcedKind::Rollback]); + + let committed = commit( + &request, + FenceCommitKind::Committed, + FenceOutcome::Applied, + (0, 2), + None, + ); + let out = capture(|| command_finished(&audit, &Ok(committed), &report)); + assert_eq!(out.lines().count(), 1, "{out}"); + assert_contains_all( + &out, + &[ + AUDIT_TARGET, + "namespace fence command answered", + "event=\"namespace_fence_command\"", + "namespace=ns", + &format!("operation_id={}", Uuid::from_u128(0xc)), + &format!("command_id={}", Uuid::from_u128(4)), + "command=\"AcquireSourceWriteFence\"", + "outcome=\"APPLIED\"", + "replay=\"none\"", + "revision_before=0", + "revision_after=2", + "state_before=\"UNFENCED\"", + "state_after=\"SOURCE_WRITE_FENCED\"", + "drain=\"write\"", + "drain_ms=Some(1500)", + "forced=rollback", + &format!("server_instance={}", Uuid::from_u128(0x99)), + ], + ); + + let out = capture(|| { + let replayed = commit( + &request, + FenceCommitKind::Replayed, + FenceOutcome::Applied, + (0, 2), + None, + ); + command_finished(&audit, &Ok(replayed), &CommandReport::default()) + }); + assert_contains_all(&out, &["replay=\"replay\"", "drain=\"none\"", "forced= "]); + + let conflict = FenceError::new(FenceOutcome::FenceCommandConflict, "reused command id"); + let out = capture(|| { + command_finished( + &audit, + &Err(crate::Error::NamespaceFence(conflict)), + &CommandReport::default(), + ) + }); + assert_eq!(out.lines().count(), 1, "{out}"); + assert_contains_all( + &out, + &[ + AUDIT_TARGET, + "namespace fence command refused", + "outcome=\"FENCE_COMMAND_CONFLICT\"", + "replay=\"conflict\"", + "expected_revision=0", + "state_before=\"UNFENCED\"", + &format!( + "server_instance={}", + super::super::server_identity().instance_id + ), + ], + ); + + let out = capture(|| { + command_finished( + &audit, + &Err(crate::Error::NamespaceStoreShutdown), + &CommandReport::default(), + ) + }); + assert_contains_all(&out, &["outcome=\"ERROR\"", "detail=\"none\""]); + } + + type Snapshot = HashMap<(MetricKind, String, Vec<(String, String)>), DebugValue>; + + fn snapshot() -> Snapshot { + Snapshotter::current_thread_snapshot() + .expect("per-thread recorder installed") + .into_vec() + .into_iter() + .map(|(key, _, _, value)| { + let (kind, key) = key.into_parts(); + let mut labels: Vec<_> = key + .labels() + .map(|l| (l.key().to_string(), l.value().to_string())) + .collect(); + labels.sort(); + ((kind, key.name().to_string(), labels), value) + }) + .collect() + } + + fn labels(pairs: &[(&str, &str)]) -> Vec<(String, String)> { + let mut labels: Vec<_> = pairs + .iter() + .map(|(k, v)| (k.to_string(), v.to_string())) + .collect(); + labels.sort(); + labels + } + + #[track_caller] + fn counter(s: &Snapshot, name: &str, pairs: &[(&str, &str)]) -> u64 { + match s.get(&(MetricKind::Counter, name.to_string(), labels(pairs))) { + Some(DebugValue::Counter(v)) => *v, + other => panic!("counter {name} {pairs:?}: {other:?} in {s:?}"), + } + } + + #[track_caller] + fn gauge(s: &Snapshot, name: &str, pairs: &[(&str, &str)]) -> f64 { + match s.get(&(MetricKind::Gauge, name.to_string(), labels(pairs))) { + Some(DebugValue::Gauge(v)) => v.0, + other => panic!("gauge {name} {pairs:?}: {other:?} in {s:?}"), + } + } + + /// Each metric of section 15 is recorded with its bounded labels, and no label value is a + /// namespace, an operation id or a command id. + #[test] + fn metrics_and_bounded_labels() { + let _ = DebuggingRecorder::per_thread().install(); + let acquire = acquire_request(); + let audit = CommandAudit::new(&acquire, FenceState::Unfenced); + let mut report = CommandReport::default(); + report.forced(ForcedKind::Rollback, 2); + report.forced(ForcedKind::StreamTermination, 1); + report.drained(DrainKind::Write, Duration::from_millis(20)); + let applied = commit( + &acquire, + FenceCommitKind::Committed, + FenceOutcome::Applied, + (0, 2), + None, + ); + let replayed = FenceCommit { + kind: FenceCommitKind::Replayed, + ..applied.clone() + }; + let conflict = FenceError::new(FenceOutcome::FenceCommandConflict, "reused command id"); + let _ = capture(|| { + command_finished(&audit, &Ok(applied), &report); + command_finished(&audit, &Ok(replayed), &CommandReport::default()); + command_finished( + &audit, + &Err(crate::Error::NamespaceFence(conflict)), + &CommandReport::default(), + ); + let adopt = adopt_request(); + let adopted = commit( + &adopt, + FenceCommitKind::Committed, + FenceOutcome::Applied, + (2, 3), + Some(adoption_entry()), + ); + command_finished( + &CommandAudit::new(&adopt, FenceState::SourceWriteFenced), + &Ok(adopted), + &CommandReport::default(), + ); + }); + let fenced = FenceError::new(FenceOutcome::MigrationWriteFenced, "fenced"); + for surface in DenialSurface::ALL { + denied(&fenced, surface); + } + + let mut quarantined = sample_record(); + quarantined.namespace = "t1".into(); + quarantined.role = super::super::state::Role::Target; + quarantined.state = FenceState::TargetQuarantined; + quarantined.created_at_ms = 10_000; + let mut released = sample_record(); + released.namespace = "s2".into(); + released.state = FenceState::Released; + released.created_at_ms = 1_000; + let registry = FenceRegistry::seeded([ + ("s1".into(), StoredFence::Record(sample_record())), + ("s2".into(), StoredFence::Record(released)), + ("t1".into(), StoredFence::Record(quarantined)), + ( + "u1".into(), + StoredFence::Unavailable { + detail: super::super::outcome::FenceDetail::CorruptRecord, + reason: "test".into(), + marker: None, + }, + ), + ( + "plain".into(), + StoredFence::None { + namespace_exists: true, + }, + ), + ]); + // s1 (active) was created at 100 ms, s2 (released, not active) at 1 s. + update_gauges(®istry, 60_100); + + let s = snapshot(); + let cmd = [("command", "AcquireSourceWriteFence")]; + assert_eq!( + counter(&s, TRANSITIONS_TOTAL, &[cmd[0], ("outcome", "APPLIED")]), + 2 + ); + assert_eq!( + counter( + &s, + TRANSITIONS_TOTAL, + &[cmd[0], ("outcome", "FENCE_COMMAND_CONFLICT")] + ), + 1 + ); + assert_eq!(counter(&s, REPLAYS_TOTAL, &[("result", "replay")]), 1); + assert_eq!(counter(&s, REPLAYS_TOTAL, &[("result", "conflict")]), 1); + assert_eq!(counter(&s, FORCED_TOTAL, &[("kind", "rollback")]), 2); + assert_eq!( + counter(&s, FORCED_TOTAL, &[("kind", "stream_termination")]), + 1 + ); + assert_eq!(counter(&s, ADOPTIONS_TOTAL, &[]), 1); + match s.get(&( + MetricKind::Histogram, + DRAIN_DURATION_SECONDS.to_string(), + labels(&[("kind", "write")]), + )) { + Some(DebugValue::Histogram(v)) => assert_eq!(v.len(), 1), + other => panic!("drain histogram: {other:?}"), + } + for surface in DenialSurface::ALL { + assert_eq!( + counter( + &s, + DENIALS_TOTAL, + &[ + ("code", "MIGRATION_WRITE_FENCED"), + ("surface", surface.as_str()) + ] + ), + 1 + ); + } + let ns = |role, state| gauge(&s, NAMESPACES, &[("role", role), ("state", state)]); + assert_eq!(ns("source", "SOURCE_WRITE_FENCED"), 1.0); + assert_eq!(ns("source", "RELEASED"), 1.0); + assert_eq!(ns("target", "TARGET_QUARANTINED"), 1.0); + assert_eq!(ns("unknown", "UNKNOWN_UNAVAILABLE"), 1.0); + assert_eq!(ns("source", "SOURCE_DRAINING"), 0.0); + assert_eq!(ns("target", "TARGET_WRITABLE"), 0.0); + assert_eq!(gauge(&s, OLDEST_ACTIVE_AGE_SECONDS, &[]), 60.0); + + let forbidden = [ + "ns".to_string(), + "s1".to_string(), + "t1".to_string(), + Uuid::from_u128(0xb).to_string(), + Uuid::from_u128(0xc).to_string(), + Uuid::from_u128(3).to_string(), + Uuid::from_u128(4).to_string(), + ]; + for ((_, name, labels), _) in &s { + if !name.starts_with("libsql_server_fence_") { + continue; + } + for (key, value) in labels { + assert!( + matches!( + key.as_str(), + "command" + | "outcome" + | "kind" + | "result" + | "code" + | "surface" + | "role" + | "state" + ), + "{name}: unexpected label {key}" + ); + assert!(!forbidden.contains(value), "{name}: label {key}={value}"); + } + } - let out = capture(|| super::adoption(&"ns".into(), &commit(None))); - assert!(out.is_empty(), "{out}"); + // An emptied registry sets every gauge back to zero. + update_gauges(&FenceRegistry::seeded([]), 60_100); + let s = snapshot(); + assert_eq!( + gauge( + &s, + NAMESPACES, + &[("role", "source"), ("state", "SOURCE_WRITE_FENCED")] + ), + 0.0 + ); + assert_eq!(gauge(&s, OLDEST_ACTIVE_AGE_SECONDS, &[]), 0.0); } } diff --git a/libsql-server/src/namespace/fence/controller.rs b/libsql-server/src/namespace/fence/controller.rs index bc358818a4..11c0bfec25 100644 --- a/libsql-server/src/namespace/fence/controller.rs +++ b/libsql-server/src/namespace/fence/controller.rs @@ -501,18 +501,28 @@ impl FenceController { /// Cancel every read lease held now (the read drain's deadline). Each lease's work is asked /// to stop once; the leases stay counted until they are actually released. Returns how many /// were asked. - pub(crate) fn cancel_read_leases(&self) -> usize { + /// [`cancel_read_leases`](Self::cancel_read_leases), counted by the kind of work asked to + /// stop. + pub(crate) fn cancel_read_leases_by_kind(&self) -> ReadLeaseCounts { let leases = self.read_leases.lock(); - let mut asked = 0; + let mut asked = ReadLeaseCounts::default(); for entry in leases.live.values() { if !entry.cancelled.swap(true, Ordering::AcqRel) { (entry.cancel)(); - asked += 1; + match entry.kind { + LeaseKind::Sql => asked.sql += 1, + LeaseKind::Dump => asked.dump += 1, + LeaseKind::Replication => asked.replication += 1, + } } } asked } + pub(crate) fn cancel_read_leases(&self) -> usize { + self.cancel_read_leases_by_kind().total() + } + /// On a replica server: publish what the replicator learned of the primary's fence /// (section 6.2). `Some` denies normal reads and streams of the local copy with that error /// and asks every read lease held now to stop, so that work admitted before the replica @@ -757,6 +767,7 @@ impl FenceController { let guard = self.transition_lock.clone().lock_owned().await; Transition { controller: self.clone(), + report: Default::default(), _guard: guard, } } @@ -879,6 +890,8 @@ impl FenceController { /// dropped. pub struct Transition { controller: Arc, + /// What the command's drain did, for its audit event (section 15). + pub(crate) report: super::audit::CommandReport, _guard: OwnedMutexGuard<()>, } diff --git a/libsql-server/src/namespace/fence/drain.rs b/libsql-server/src/namespace/fence/drain.rs index f05924a92c..edd0a22795 100644 --- a/libsql-server/src/namespace/fence/drain.rs +++ b/libsql-server/src/namespace/fence/drain.rs @@ -16,6 +16,7 @@ use tokio::time::Instant; use crate::namespace::meta_store::{FenceCommit, FenceContext, MetaStore}; +use super::audit::{self, CommandAudit, CommandReport, DrainKind, ForcedKind}; use super::command::{DrainPolicy, FenceCommand, FenceRequest, OnDeadline}; use super::controller::{FenceController, LiveWriteDrain, Transition}; use super::hooks::HookPoint; @@ -52,7 +53,8 @@ impl FenceController { let meta = meta.clone(); tokio::spawn(async move { let mut transition = this.begin_transition().await; - match request.command { + let audit = CommandAudit::new(&request, this.gate().state()); + let result = match request.command { FenceCommand::AcquireSourceWriteFence { .. } => { acquire_source_write_fence(&mut transition, &meta, request, ctx).await } @@ -63,7 +65,9 @@ impl FenceController { super::import::seal_target_import(&mut transition, &meta, request, ctx).await } _ => transition.apply(&meta, request, ctx).await, - } + }; + audit::command_finished(&audit, &result, &transition.report); + result }) .await? } @@ -121,10 +125,14 @@ pub async fn acquire_source_write_fence( let drain_key = (commit.receipt.operation_id, commit.receipt.command_id); // Steps 5 and 6. - let boundary = match drain_writers(&controller, policy).await { + let started = Instant::now(); + let boundary = match drain_writers(&controller, policy, &mut transition.report).await { Some(boundary) => boundary, None => return Ok(commit), }; + transition + .report + .drained(DrainKind::Write, started.elapsed()); // Step 7. let acquired_on = commit.record.as_ref().and_then(|r| r.identity.log_id); @@ -156,6 +164,7 @@ pub async fn acquire_source_write_fence( async fn drain_writers( controller: &FenceController, policy: DrainPolicy, + report: &mut CommandReport, ) -> Option { let namespace = controller.namespace().clone(); let deadline_after = Duration::from_millis(policy.deadline_ms); @@ -178,6 +187,10 @@ async fn drain_writers( match policy.on_deadline { OnDeadline::ForceRollback if !forced => { forced = true; + report.forced( + ForcedKind::Rollback, + sources.iter().filter(|s| s.manager.has_writer()).count(), + ); for source in &sources { let manager = source.manager.clone(); // The rollback takes the connection's lock, which a running program @@ -283,7 +296,7 @@ fn capture_boundary( Ok(FrozenBoundary { log_id, frame_no }) } -pub(super) fn now_ms() -> i64 { +pub(crate) fn now_ms() -> i64 { std::time::SystemTime::now() .duration_since(std::time::UNIX_EPOCH) .map_or(0, |d| i64::try_from(d.as_millis()).unwrap_or(i64::MAX)) diff --git a/libsql-server/src/namespace/fence/import.rs b/libsql-server/src/namespace/fence/import.rs index d4d36451c3..206d6c1561 100644 --- a/libsql-server/src/namespace/fence/import.rs +++ b/libsql-server/src/namespace/fence/import.rs @@ -28,6 +28,7 @@ use crate::namespace::configurator::{load_dump_sql, read_dump}; use crate::namespace::meta_store::{FenceCommit, FenceContext, MetaStore}; use crate::namespace::replication_wal::ReplicationWalWrapper; +use super::audit::{CommandReport, DrainKind, ForcedKind}; use super::capability::MigrationCapability; use super::command::{DrainPolicy, FenceCommand, FenceRequest, OnDeadline}; use super::controller::{FenceController, LiveWriteDrain, Transition}; @@ -172,10 +173,14 @@ pub async fn seal_target_import( } let drain_key = (commit.receipt.operation_id, commit.receipt.command_id); - if !drain_import_writers(&controller, policy).await { + let started = Instant::now(); + if !drain_import_writers(&controller, policy, &mut transition.report).await { return Ok(commit); } + transition + .report + .drained(DrainKind::Import, started.elapsed()); ctx.now_ms = now_ms(); transition .complete_drain(meta, drain_key, DrainCompletion::TargetImport, ctx) @@ -187,7 +192,11 @@ pub async fn seal_target_import( /// still holding the slot (an idle session's open transaction; a running call ends its own) /// and waits again for the same deadline, but at least [`FORCED_ROLLBACK_GRACE`]. `false` when /// the drain could not be proven within the policy. -async fn drain_import_writers(controller: &FenceController, policy: DrainPolicy) -> bool { +async fn drain_import_writers( + controller: &FenceController, + policy: DrainPolicy, + report: &mut CommandReport, +) -> bool { let namespace = controller.namespace().clone(); let deadline_after = Duration::from_millis(policy.deadline_ms); let mut deadline = Instant::now() + deadline_after; @@ -206,7 +215,12 @@ async fn drain_import_writers(controller: &FenceController, policy: DrainPolicy) match policy.on_deadline { OnDeadline::ForceRollback if !forced => { forced = true; - for source in controller.live_write_drains() { + let sources = controller.live_write_drains(); + report.forced( + ForcedKind::Rollback, + sources.iter().filter(|s| s.manager.has_writer()).count(), + ); + for source in sources { let manager = source.manager.clone(); // The rollback takes the connection's lock, which a running import call // holds; the release it causes is what the drain keeps waiting for. diff --git a/libsql-server/src/namespace/fence/read.rs b/libsql-server/src/namespace/fence/read.rs index 50feb0b3b1..643e006d01 100644 --- a/libsql-server/src/namespace/fence/read.rs +++ b/libsql-server/src/namespace/fence/read.rs @@ -15,6 +15,7 @@ use tokio::time::Instant; use crate::namespace::meta_store::{FenceCommit, FenceContext, MetaStore}; +use super::audit::{CommandReport, DrainKind, ForcedKind}; use super::command::{DrainPolicy, FenceCommand, FenceRequest}; use super::controller::{FenceController, Transition}; use super::drain::{now_ms, FORCED_ROLLBACK_GRACE}; @@ -73,10 +74,14 @@ pub async fn set_source_read_fence( let drain_key = (commit.receipt.operation_id, commit.receipt.command_id); // Step 4. - if !drain_readers(&controller, policy).await { + let started = Instant::now(); + if !drain_readers(&controller, policy, &mut transition.report).await { return Ok(commit); } + transition + .report + .drained(DrainKind::Read, started.elapsed()); // Step 5. ctx.now_ms = now_ms(); transition @@ -89,7 +94,11 @@ pub async fn set_source_read_fence( /// through their own), and the drain keeps waiting for the actual releases for one more /// deadline, but at least [`FORCED_ROLLBACK_GRACE`]. `false` when leases are still held then: /// the answer is `DRAINING`, never a guess. -async fn drain_readers(controller: &FenceController, policy: DrainPolicy) -> bool { +async fn drain_readers( + controller: &FenceController, + policy: DrainPolicy, + report: &mut CommandReport, +) -> bool { let namespace = controller.namespace().clone(); let deadline_after = Duration::from_millis(policy.deadline_ms); let deadline = Instant::now() + deadline_after; @@ -98,7 +107,11 @@ async fn drain_readers(controller: &FenceController, policy: DrainPolicy) -> boo } let _ = controller.hook(HookPoint::BeforeReadLeaseCancel).await; let held = controller.read_lease_counts(); - let asked = controller.cancel_read_leases(); + let asked = controller.cancel_read_leases_by_kind(); + report.forced(ForcedKind::SqlCancel, asked.sql); + report.forced(ForcedKind::DumpCancel, asked.dump); + report.forced(ForcedKind::StreamTermination, asked.replication); + let asked = asked.total(); tracing::info!( %namespace, deadline_ms = policy.deadline_ms, diff --git a/libsql-server/src/namespace/fence/registry.rs b/libsql-server/src/namespace/fence/registry.rs index 7dd52425ba..2c2e508652 100644 --- a/libsql-server/src/namespace/fence/registry.rs +++ b/libsql-server/src/namespace/fence/registry.rs @@ -110,10 +110,14 @@ impl FenceRegistry { /// installed, a target being created, an indeterminate commit or an unavailable state. A /// name without a controller has no fence state and is not refused here. pub fn check_lifecycle(&self, namespace: &NamespaceName) -> Result<(), FenceError> { - match self.get(namespace) { + let refused = match self.get(namespace) { Some(controller) => controller.gate().permits(OperationClass::Lifecycle), None => Ok(()), + }; + if let Err(e) = &refused { + super::audit::denied(e, super::audit::DenialSurface::Lifecycle); } + refused } /// How many namespaces have an active fence (`docs/NAMESPACE_FENCE.md` section 4.4, @@ -132,6 +136,19 @@ impl FenceRegistry { .count() } + /// The published state of every namespace the registry holds, with the creation time of + /// its record when it has one (for the fence gauges, section 15). + pub fn census(&self) -> Vec<(super::state::FenceState, Option)> { + let controllers: Vec<_> = self.controllers.lock().values().cloned().collect(); + controllers + .iter() + .map(|controller| { + let gate = controller.gate(); + (gate.state(), gate.fence.record().map(|r| r.created_at_ms)) + }) + .collect() + } + pub fn len(&self) -> usize { self.controllers.lock().len() } diff --git a/libsql-server/src/namespace/meta_store.rs b/libsql-server/src/namespace/meta_store.rs index ff8c9f3ec1..6dc5de14d2 100644 --- a/libsql-server/src/namespace/meta_store.rs +++ b/libsql-server/src/namespace/meta_store.rs @@ -653,6 +653,17 @@ fn unavailable_error(stored: &StoredFence) -> FenceError { }) } +/// A config write or a delete under the stored fence: refused (and counted as a lifecycle +/// denial) unless its state permits lifecycle operations. +fn lifecycle_permitted(stored: &StoredFence) -> std::result::Result<(), FenceError> { + stored.permits(OperationClass::Lifecycle).inspect_err(|e| { + crate::namespace::fence::audit::denied( + e, + crate::namespace::fence::audit::DenialSurface::Lifecycle, + ) + }) +} + /// Why a name that has no config must not be created: its directory holds a marker, so it is a /// target being created or a namespace the metastore lost (section 13.3). fn marker_denial(dbs_path: &Path, namespace: &NamespaceName) -> Result> { @@ -745,7 +756,7 @@ fn try_process( if inner.fence.tables { let (stored, _) = fence_store::read_fence(&tx, &inner.dbs_path, namespace).map_err(fence_store_error)?; - stored.permits(OperationClass::Lifecycle)?; + lifecycle_permitted(&stored)?; } if let Some(schema) = config.shared_schema_name.as_ref() { if inner.db_kind.is_primary() { @@ -1344,7 +1355,7 @@ impl MetaStore { if self.inner.fence.tables { let (stored, _) = fence_store::read_fence(&tx, &self.inner.dbs_path, &namespace) .map_err(fence_store_error)?; - stored.permits(OperationClass::Lifecycle)?; + lifecycle_permitted(&stored)?; if !matches!(stored, StoredFence::None { .. }) { // The marker goes before the commit: a crash in between leaves a record // without a marker, which is repaired on load, rather than a marker diff --git a/libsql-server/src/namespace/store.rs b/libsql-server/src/namespace/store.rs index 0ac3c42465..a4c09790fe 100644 --- a/libsql-server/src/namespace/store.rs +++ b/libsql-server/src/namespace/store.rs @@ -21,6 +21,7 @@ use crate::stats::Stats; use super::broadcasters::{BroadcasterHandle, BroadcasterRegistry}; use super::configurator::{DynConfigurator, NamespaceConfigurators}; +use super::fence::audit::CommandAudit; use super::fence::capability::CapabilityPurpose; use super::fence::command::{FenceCommand, FenceRequest}; use super::fence::controller::{FenceController, Transition}; @@ -32,9 +33,7 @@ use super::fence::registry::FenceRegistry; use super::fence::state::{FenceState, Role}; use super::fence::store::StoredFence; use super::fence::target::{self, CreateTargetRequest, ValidationSession}; -use super::meta_store::{ - FenceCommit, FenceCommitKind, FenceContext, FenceInspection, MetaStore, MetaStoreHandle, -}; +use super::meta_store::{FenceCommit, FenceContext, FenceInspection, MetaStore, MetaStoreHandle}; use super::schema_lock::SchemaLocksRegistry; use super::{Namespace, ResetCb, ResetOp, ResolveNamespacePathFn, RestoreOption}; @@ -102,6 +101,7 @@ impl NamespaceStore { // Every namespace with fence state gets its controller before anything is served // (section 8.5). + super::fence::audit::describe_metrics(); let fences = FenceRegistry::seeded(metadata.load_fences().await?); if fences.len() > 0 { tracing::info!("loaded {} namespace fence controllers", fences.len()); @@ -641,8 +641,9 @@ impl NamespaceStore { /// [`execute_fence_command`](Self::execute_fence_command) for a request that may carry /// the adoption key: `adoption_authorised` says whether it did (section 12). Only - /// `AdoptFence` looks at it. A committed adoption is written to the audit log (target - /// `libsql_server::fence::audit`) with its approvers, incident reference and reason. + /// `AdoptFence` looks at it. Like every command, a committed adoption is written to the + /// audit log (target `libsql_server::fence::audit`), with its approvers, incident reference + /// and reason. pub(crate) async fn execute_fence_command_authorised( &self, request: FenceRequest, @@ -654,13 +655,7 @@ impl NamespaceStore { let controller = self.inner.fences.controller(&namespace); let mut ctx = FenceContext::now(server, None); ctx.adoption_authorised = adoption_authorised; - let commit = controller - .execute(&self.inner.metadata, request, ctx) - .await?; - if commit.kind == FenceCommitKind::Committed { - super::fence::audit::adoption(&namespace, &commit); - } - return Ok(commit); + return controller.execute(&self.inner.metadata, request, ctx).await; } let controller = match request.command { FenceCommand::CreateTargetQuarantined { .. } => { @@ -755,9 +750,13 @@ impl NamespaceStore { let this = self.clone(); tokio::spawn(async move { let mut transition = controller.begin_transition().await; + let audit = CommandAudit::new(&request, controller.gate().state()); let ctx = FenceContext::now(server, None); - this.create_target_under(&mut transition, request, ctx) - .await + let result = this + .create_target_under(&mut transition, request, ctx) + .await; + super::fence::audit::command_finished(&audit, &result, &transition.report); + result }) .await? } @@ -1010,6 +1009,12 @@ impl NamespaceStore { self.inner.fences.active_count() } + /// Set the fence gauges (`docs/NAMESPACE_FENCE.md` section 15) from the registry, before + /// `/metrics` is rendered. + pub(crate) fn update_fence_gauges(&self) { + super::fence::audit::update_gauges(&self.inner.fences, super::fence::drain::now_ms()); + } + pub(crate) fn schema_locks(&self) -> &SchemaLocksRegistry { &self.inner.schema_locks } diff --git a/libsql-server/src/rpc/proxy.rs b/libsql-server/src/rpc/proxy.rs index b189b1143e..51b46c0dec 100644 --- a/libsql-server/src/rpc/proxy.rs +++ b/libsql-server/src/rpc/proxy.rs @@ -46,6 +46,12 @@ pub mod rpc { let stable_code = other .fence_error() .and_then(|e| e.outcome().proxy_stable_code()); + if let Some(fence) = other.fence_error().filter(|_| stable_code.is_some()) { + crate::namespace::fence::audit::denied( + fence, + crate::namespace::fence::audit::DenialSurface::Rpc, + ); + } let code = match other { _ if stable_code.is_some() => ErrorCode::SqlError, SqldError::LibSqlInvalidQueryParams(_) => ErrorCode::SqlError, @@ -326,11 +332,12 @@ impl ProxyService { crate::error::Error::NamespaceDoesntExist(_) => None, // A namespace the fence refuses is refused with the typed status, never // retried by the write proxy (`docs/NAMESPACE_FENCE.md` section 6.1). - e if fence_status(e).is_some() => Err(fence_status(e).unwrap())?, - _ => Err(tonic::Status::internal(format!( - "Error fetching jwt key for a namespace: {}", - e - )))?, + e => Err(fence_status(e).unwrap_or_else(|| { + tonic::Status::internal(format!( + "Error fetching jwt key for a namespace: {}", + e + )) + }))?, }, Ok(Err(e)) => Err(tonic::Status::internal(format!( "Error fetching jwt key for a namespace: {}", @@ -582,8 +589,15 @@ pub async fn garbage_collect(clients: &mut HashMap> /// The typed status of a fence denial on the proxy service (`docs/NAMESPACE_FENCE.md` section /// 6.1): `FAILED_PRECONDITION` with the stable code, never `UNAVAILABLE`, which the write /// proxy retries without bound. +/// Counted as a denial on the `rpc` surface when it is one. fn fence_status(e: &crate::error::Error) -> Option { - e.fence_error()?.to_grpc_status() + let fence = e.fence_error()?; + let status = fence.to_grpc_status()?; + crate::namespace::fence::audit::denied( + fence, + crate::namespace::fence::audit::DenialSurface::Rpc, + ); + Some(status) } /// The status for an error looking up the namespace a proxy request names. diff --git a/libsql-server/src/rpc/replication/replication_log.rs b/libsql-server/src/rpc/replication/replication_log.rs index 7b6c5d51d7..1502205f7c 100644 --- a/libsql-server/src/rpc/replication/replication_log.rs +++ b/libsql-server/src/rpc/replication/replication_log.rs @@ -87,10 +87,9 @@ impl ReplicationLogService { /// The status of a replication call (or stream) that the namespace fence refuses. Counted /// every time, and logged at most once per namespace per [`FENCE_DENIAL_LOG_INTERVAL`]. fn fence_denied(&self, namespace: &NamespaceName, call: &str, error: FenceError) -> Status { - metrics::increment_counter!( - "libsql_server_fence_denials_total", - "code" => error.outcome().as_str(), - "surface" => "replication", + crate::namespace::fence::audit::denied( + &error, + crate::namespace::fence::audit::DenialSurface::Replication, ); let now = Instant::now(); let log = { diff --git a/libsql-server/tests/fence/mod.rs b/libsql-server/tests/fence/mod.rs index b7a64e2b42..e7dddf1785 100644 --- a/libsql-server/tests/fence/mod.rs +++ b/libsql-server/tests/fence/mod.rs @@ -4,6 +4,7 @@ mod admin; mod lifecycle; +mod observability; mod protocol; use std::path::PathBuf; diff --git a/libsql-server/tests/fence/observability.rs b/libsql-server/tests/fence/observability.rs new file mode 100644 index 0000000000..8b8b3c28db --- /dev/null +++ b/libsql-server/tests/fence/observability.rs @@ -0,0 +1,275 @@ +//! Fence metrics over a running server (`docs/NAMESPACE_FENCE.md` section 15). + +use hyper::StatusCode; +use metrics_util::debugging::DebugValue; +use metrics_util::MetricKind; +use serde_json::json; +use tempfile::tempdir; +use uuid::Uuid; + +use super::{ + acquire_body, command_body, connect, load_and_log_id, make_primary, sim, state_of, + user_execute, Admin, Primary, ADMIN_KEY, +}; + +fn uuid(n: u128) -> Uuid { + Uuid::from_u128(n) +} + +/// The value of the metric `name` of `kind` whose labels are exactly `labels`. +fn metric(kind: MetricKind, name: &str, labels: &[(&str, &str)]) -> Option { + let snapshot = crate::common::snapshot_metrics(); + let mut wanted: Vec<(String, String)> = labels + .iter() + .map(|(k, v)| (k.to_string(), v.to_string())) + .collect(); + wanted.sort(); + snapshot + .snapshot() + .iter() + .find(|(key, _)| { + let mut have: Vec<(String, String)> = key + .key() + .labels() + .map(|l| (l.key().to_string(), l.value().to_string())) + .collect(); + have.sort(); + key.kind() == kind && key.key().name() == name && have == wanted + }) + .map(|(_, (_, _, value))| match value { + DebugValue::Counter(v) => DebugValue::Counter(*v), + DebugValue::Gauge(v) => DebugValue::Gauge(*v), + DebugValue::Histogram(v) => DebugValue::Histogram(v.clone()), + }) +} + +#[track_caller] +fn counter(name: &str, labels: &[(&str, &str)]) -> u64 { + match metric(MetricKind::Counter, name, labels) { + Some(DebugValue::Counter(v)) => v, + other => panic!("counter {name} {labels:?}: {other:?}"), + } +} + +#[track_caller] +fn gauge(name: &str, labels: &[(&str, &str)]) -> f64 { + match metric(MetricKind::Gauge, name, labels) { + Some(DebugValue::Gauge(v)) => v.0, + other => panic!("gauge {name} {labels:?}: {other:?}"), + } +} + +/// After a source walk with a forced rollback, a replay, a command-id conflict and denials on +/// the user protocol and a lifecycle route, every fence metric of section 15 is present with +/// its bounded labels, and no fence metric has a label naming the namespace, the operation or +/// a command. +#[test] +fn metrics_and_labels() { + let mut sim = sim(); + let tmp = tempdir().unwrap(); + make_primary(&mut sim, tmp.path().to_path_buf(), Primary::default()); + sim.client("client", async { + let admin = Admin::new(Some(ADMIN_KEY)); + admin.create_namespace("src").await?; + let log_id = load_and_log_id(&admin, "src").await?; + let op = uuid(0xa); + + // A transaction admitted before the fence is still open at the deadline: the drain + // rolls it back and is then proven. + let conn = connect("src")?; + let tx = conn.transaction().await?; + tx.execute("insert into t values (2)", ()).await?; + let acquire = command_body( + op, + uuid(1), + "UNFENCED", + 0, + json!({ + "expected_namespace_identity": { "log_id": log_id }, + "drain_policy": { "deadline_ms": 100, "on_deadline": "force_rollback" }, + }), + ); + let (status, acquired) = admin + .command("src", "source/acquire-write-fence", acquire.clone()) + .await?; + assert_eq!(status, StatusCode::OK, "{acquired}"); + assert_eq!(state_of(&acquired).0, "SOURCE_WRITE_FENCED", "{acquired}"); + assert!(tx.commit().await.is_err()); + + // A replay, and a conflict on the same command id. + let (status, replay) = admin + .command("src", "source/acquire-write-fence", acquire) + .await?; + assert_eq!(status, StatusCode::OK, "{replay}"); + assert_eq!(replay["replayed"], true); + let (status, body) = admin + .command( + "src", + "source/acquire-write-fence", + acquire_body(op, uuid(1), &log_id), + ) + .await?; + assert_eq!(status, StatusCode::CONFLICT, "{body}"); + + // Denials: a write over Hrana, and a config change (a lifecycle operation) over the + // admin API. (A delete is refused on a blocking thread, which this test's per-thread + // metrics recorder does not see.) + let (status, body) = user_execute("src", "insert into t values (3)").await?; + assert_eq!(status, StatusCode::LOCKED, "{body}"); + assert_eq!(body["code"], "MIGRATION_WRITE_FENCED", "{body}"); + let (status, body) = admin + .post( + "/v1/namespaces/src/config", + json!({ "block_reads": false, "block_writes": true, "block_reason": null }), + ) + .await?; + assert_eq!(status, StatusCode::LOCKED, "{body}"); + + // The gauges are computed when `/metrics` is read. + let (status, _) = admin.get("/metrics").await?; + assert_eq!(status, StatusCode::OK); + + // First: a snapshot drains the recorded histogram values. + match metric( + MetricKind::Histogram, + "libsql_server_fence_drain_duration_seconds", + &[("kind", "write")], + ) { + Some(DebugValue::Histogram(v)) => assert_eq!(v.len(), 1), + other => panic!("drain histogram: {other:?}"), + } + let acquire = ("command", "AcquireSourceWriteFence"); + assert_eq!( + counter( + "libsql_server_fence_transitions_total", + &[acquire, ("outcome", "APPLIED")] + ), + 2 + ); + assert_eq!( + counter( + "libsql_server_fence_transitions_total", + &[acquire, ("outcome", "FENCE_COMMAND_CONFLICT")] + ), + 1 + ); + assert_eq!( + counter("libsql_server_fence_replays_total", &[("result", "replay")]), + 1 + ); + assert_eq!( + counter( + "libsql_server_fence_replays_total", + &[("result", "conflict")] + ), + 1 + ); + assert_eq!( + counter("libsql_server_fence_forced_total", &[("kind", "rollback")]), + 1 + ); + for surface in ["hrana", "lifecycle", "http"] { + assert!( + counter( + "libsql_server_fence_denials_total", + &[("code", "MIGRATION_WRITE_FENCED"), ("surface", surface)] + ) >= 1, + "{surface}" + ); + } + assert_eq!( + gauge( + "libsql_server_fence_namespaces", + &[("role", "source"), ("state", "SOURCE_WRITE_FENCED")] + ), + 1.0 + ); + assert_eq!( + gauge( + "libsql_server_fence_namespaces", + &[("role", "target"), ("state", "TARGET_QUARANTINED")] + ), + 0.0 + ); + let age = gauge("libsql_server_fence_oldest_active_age_seconds", &[]); + assert!(age >= 0.0, "{age}"); + + // No fence metric carries a namespace, operation or command id as a label value. + let forbidden = [ + "src".to_string(), + op.to_string(), + uuid(1).to_string(), + log_id.clone(), + ]; + let snapshot = crate::common::snapshot_metrics(); + let mut seen = 0; + for (key, _) in snapshot.snapshot() { + let name = key.key().name(); + if !name.starts_with("libsql_server_fence_") { + continue; + } + seen += 1; + for label in key.key().labels() { + assert!( + matches!( + label.key(), + "command" + | "outcome" + | "kind" + | "result" + | "code" + | "surface" + | "role" + | "state" + ), + "{name}: unexpected label {}", + label.key() + ); + assert!( + !forbidden.iter().any(|f| f == label.value()), + "{name}: {}={}", + label.key(), + label.value() + ); + } + } + assert!(seen > 10, "{seen}"); + + // Releasing moves the source out of the write-fenced count. + let (status, body) = admin + .command( + "src", + "source/release-write-fence", + command_body( + op, + uuid(2), + "SOURCE_WRITE_FENCED", + state_of(&acquired).1, + json!({}), + ), + ) + .await?; + assert_eq!(status, StatusCode::OK, "{body}"); + admin.get("/metrics").await?; + assert_eq!( + gauge( + "libsql_server_fence_namespaces", + &[("role", "source"), ("state", "SOURCE_WRITE_FENCED")] + ), + 0.0 + ); + assert_eq!( + gauge( + "libsql_server_fence_namespaces", + &[("role", "source"), ("state", "RELEASED")] + ), + 1.0 + ); + assert_eq!( + gauge("libsql_server_fence_oldest_active_age_seconds", &[]), + 0.0 + ); + Ok(()) + }); + sim.run().unwrap(); +} From dd4c476e65709f16847dcd7475ec03f9e265bac4 Mon Sep 17 00:00:00 2001 From: River Date: Wed, 30 Sep 2026 12:34:59 +0000 Subject: [PATCH 30/33] libsql-server: complete namespace fence acceptance tests Add a test that the admin shell can neither read nor write a quarantined migration target, through its own entry point and with a raw write that skips its read admission, while the operation's import session keeps working. Map every acceptance requirement in docs/NAMESPACE_FENCE.md section 17 to the tests that cover it, correct the names of the corrupt-state tests, list the transition and CAS tests that cover ownership, revision and replay outcomes, and note the replica connection path in the code-path coverage table. Co-authored-by: Tomasz Szymczyszyn --- docs/NAMESPACE_FENCE.md | 54 +++++++++---------- libsql-server/src/admin_shell.rs | 91 ++++++++++++++++++++++++++++++++ 2 files changed, 118 insertions(+), 27 deletions(-) diff --git a/docs/NAMESPACE_FENCE.md b/docs/NAMESPACE_FENCE.md index 68d611ca9f..e6b772a2b9 100644 --- a/docs/NAMESPACE_FENCE.md +++ b/docs/NAMESPACE_FENCE.md @@ -795,7 +795,7 @@ How each path that can reach namespace data or lifecycle is covered. File refere | `connection/connection_manager.rs` | Authoritative gate in `begin_write_txn` before `acquire()`, generation check, non-`BUSY` refusal; classed queue entries; queue wake on generation change; active-writer query, release notification, `abort_active()`. | | `connection/legacy.rs` | `FenceConnState` wired into every `LegacyConnection`; the controller is passed to `MakeLegacyConnection::new` before the first connection. `with_raw` users are covered by the WAL gate. | | HTTP `/`, `/v1/execute`, `/v1/batch`, Hrana `/v2`, `/v3`, cursors, WebSocket, dev route | Core checks and WAL gate; typed status and `code` field; Hrana codes, for step and whole-request denials alike, with the stream left usable (section 6.0). | -| `rpc/proxy.rs`, `rpc/streaming_exec.rs`, `rpc/replica_proxy.rs`, `connection/write_proxy.rs` | `stable_code` set on the primary for step and program denials (streamed and unary); namespace-lookup, JWT-key-lookup, connection-creation and unary whole-program denials are `FAILED_PRECONDITION` with the code, never `UNAVAILABLE`; the replica maps both back to `Error::NamespaceFence` and never retries them; the replica proxy forwards them unchanged (section 6.1). | +| `rpc/proxy.rs`, `rpc/streaming_exec.rs`, `rpc/replica_proxy.rs`, `connection/write_proxy.rs`, `database/replica.rs` | `stable_code` set on the primary for step and program denials (streamed and unary); namespace-lookup, JWT-key-lookup, connection-creation and unary whole-program denials are `FAILED_PRECONDITION` with the code, never `UNAVAILABLE`; the replica maps both back to `Error::NamespaceFence` and never retries them; the replica proxy forwards them unchanged (section 6.1). A replica server's connections (`database/replica.rs`) run local reads on a `LegacyConnection` that carries the replica's controller and delegate every write to the primary; their WAL is a passthrough, so write authority is always the primary's gate. | | `namespace/meta_store.rs` `handle()`, `restore()`, `maybe_recover_from_fs`, `destroy_on_error`, `process`/`try_process`, `remove`, bottomless metastore restore | Non-creating lookups; fail-closed decoding; marker-aware recovery; rename-aside; publish only after commit; fence check inside the config and remove transactions; the bottomless restore's outcome is kept (`metastore_connection_maker_with_provenance`) and reported in the admin API, a gauge and a startup warning (section 13.3). | | `namespace/store.rs` `with`, `load_namespace`, `make_namespace`, eviction | Registry check before setup; controller passed into setup; eviction keeps the registry entry. | | `store.rs` `create`, `destroy`, `reset`, `fork`, `checkpoint`, restore options | `check_lifecycle` (section 13.5): create refuses a name whose fence denies lifecycle work; `CreateTargetQuarantined` is the atomic quarantined create; destroy is refused in the metastore transaction; reset and fork as source are checked under the namespace's transition lock; fork's destination is checked before anything is stored and again under its entry lock; every restore (create with a dump, reset, fork to a point in time) is one of these paths and is refused there; checkpoint uses a non-creating lookup and skips vacuum. | @@ -850,40 +850,40 @@ Implementation (`namespace/fence/audit.rs`): ## 17. Acceptance tests map -Planned test names; the table is updated as tests land. +Every row of the issue's acceptance list, with the tests that cover it. Every named test is in this change and passes (`cargo nextest run -p libsql-server` and `-p libsql_replication`); where a requirement can only be partly proven in the test suite, the row says so and section 18 states the limit. Race tests use the `FenceTestHooks` points of section 16 or simulated time, never sleeps, for admission against persistence, drain against commit, restart and response loss. Tests named `tests::fence::*` are integration tests of `libsql-server/tests/fence/` against a running server; the rest are unit tests of the library. -| # | Requirement | Planned tests | +| # | Requirement | Tests | |---|---|---| -| 1 | Concurrent acquisition by two operations: one owner, typed conflict for the loser | landed: `namespace::fence::drain::tests::acquire_race_single_owner` (the first acquisition is parked after closing admission while the second waits on the transition lock); over the admin API: `tests::fence::admin::concurrent_acquire_one_owner` (two operations acquire at once over HTTP: one `200 APPLIED`, the other `409 FENCE_OWNED_BY_ANOTHER_OPERATION` with the winner in its fence view); admin walks: `tests::fence::admin::{source_walk_over_http, target_walk_over_http, inspect_reports_drain_counters, mutating_routes_require_admin_key}` | -| 2 | Active writer commits or is rolled back before freeze acknowledgement; nothing commits after | landed: `namespace::fence::drain::tests::{active_writer_commits_before_ack, forced_rollback_before_ack, no_commit_after_ack}` (the boundary equals the last committed replication frame and no frame follows it; autocommit, `BEGIN IMMEDIATE`, DDL and a pre-fence read transaction upgrading are refused), `installing_gate_closes_writes_before_persisting`, `refused_acquire_reopens_writes`, `release_reopens_with_new_generation` | -| 3 | Autocommit, explicit transactions, queued writers, batches, DDL, schema jobs, old WebSockets, read-to-write upgrades cannot bypass | landed: `connection::connection_manager::fence_tests::{fence_rejects_read_to_write_upgrade, fence_rejects_ddl_and_pragma, fence_rejects_raw_with_raw_write}` (autocommit, explicit transactions, DDL, header-writing pragma, `BEGIN IMMEDIATE`, `VACUUM`, `with_raw` users); `connection::connection_manager::fence_tests::fence_rejects_queued_writer` (a writer parked in the queue behind an open transaction leaves it with `MIGRATION_WRITE_FENCED` when the fence changes, and the holder keeps the slot); maintenance and vacuum under a fence: `queued_checkpoint_survives_fence_wake`, `checkpoint_allowed_while_fenced`, `vacuum_skipped_while_fenced`; drain primitives: `abort_active_tolerates_closed_connection`, `release_notifies_drain_waiters`, `fence::controller::tests::write_queues_are_woken_on_every_generation_change`; schema jobs: `schema::scheduler::test::fence::{acquire_rejects_shared_schema (a shared schema and a linked namespace cannot be fenced; a fenced namespace cannot be linked by create or config), migration_not_registered_while_linked_namespace_fenced}`; old WebSockets and batches over the protocols: `tests::fence::protocol::{old_ws_session_cannot_write (a WebSocket session whose transaction began before the fence cannot write while fenced nor after the release; after a rollback the same session writes), batch_denied_mid_batch (the read before the write runs, the write step gets the code, a step conditional on it is skipped and one conditional on its failure runs)}` | -| 4 | Program that captured config before the fence is rejected at the WAL | landed: `connection::connection_manager::fence_tests::wal_gate_rejects_program_admitted_before_fence` (a SQL function parks the program between admission and its write while the fence is acquired and released) | -| 5 | Pre-fence transactions cannot write after release or publication | landed: `connection::connection_manager::fence_tests::stale_generation_cannot_write_after_release`, `namespace::fence::target::tests::stale_generation_cannot_write_after_enable_writes`; unfenced behaviour unchanged: `unfenced_namespace_is_unchanged` | -| 6 | Acquisition timeout returns `DRAINING`, admission stays closed | landed: `namespace::fence::drain::tests::{deadline_returns_draining_and_stays_closed, replay_of_draining_resumes_and_completes}` | -| 7 | Restart at every persistence boundary; indeterminate persistence keeps the gate closed until same-command reconciliation | landed: `namespace::fence::tests::restart_at_each_boundary` (a crash at each of 17 boundaries of `AcquireSourceWriteFence` and `ReleaseSourceWriteFence`: after the `INSTALLING` gate, before, at and after each metastore commit, with and without a lagging marker, before publication, before boundary capture and before the response; the restart recovers the prior or the committed state, admits writes only if an opening transition committed, repairs the marker, and a replay finishes the command), `restart_in_draining_waits_for_the_same_command` (a writer active at the crash; nothing advances until the same command is replayed, which completes at once), `indeterminate_commit_keeps_gate_closed` (through the drain path, both when the commit happened and when it did not), `acquire_response_loss_resolved_by_replay_and_inspect`; `fence::controller::tests::{indeterminate_commit_keeps_writes_closed_until_replayed, indeterminate_commit_that_did_not_apply_is_retried_by_replay, failed_before_commit_leaves_gate_unchanged, publication_happens_before_the_response, committed_command_is_published_when_the_caller_goes_away}`, `namespace::store::fence_tests::restart_installs_the_durable_gate_before_serving`; the read fence, the seal and write enable: `namespace::fence::tests::read_and_target_boundaries::restart_at_each_read_and_target_boundary` (a crash at each of 29 boundaries of `SetSourceReadFence`, `SealTargetImport` and `EnableTargetWrites`: after the read-closing or `INSTALLING` gate, before, at and after each metastore commit, with and without a lagging marker, before publication, before the response, and while the drain waits for a reader or an import call that was running when it started; the restart recovers the prior or the committed state with no in-memory gate, reader or import call surviving, keeps reads closed once `SOURCE_READ_DRAINING` committed and import closed once `TARGET_IMPORT_DRAINING` committed, admits target writes only if `TARGET_WRITABLE` committed, repairs the marker, and a replay completes an interrupted drain at once or returns the stored result); the rebuilt log after a restart: `namespace::fence::tests::current_log_id_is_the_rebuilt_log_after_restart` (the stored record keeps the acquisition log and its boundary; the controller's `current_log_id`, which the admin view reports, names the rebuilt log); integration restarts: `tests::fence::lifecycle::{restart_keeps_fence (a graceful same-path TestServer restart reconstructs SOURCE_WRITE_FENCED, SOURCE_READ_FENCED, TARGET_QUARANTINED and TARGET_WRITE_FENCED admission before traffic, exact command replays return their stored receipts, and the user protocol observes the same denials), restart_after_enable_writes_stays_writable (TARGET_WRITABLE stays readable/writable after restart and exact EnableTargetWrites replay returns the stored receipt)}` | -| 8 | Evict and lazily reload a fenced namespace; identical admission | landed: `tests::fence::lifecycle::evicted_namespace_reloads_same_gate` (a one-entry namespace cache is put under capacity pressure by four other loaded namespaces; a user-protocol reload of the fenced namespace keeps the exact revision and generation, serves reads and returns `423 MIGRATION_WRITE_FENCED` for writes); unit-level identity: `namespace::store::fence_tests::evicted_namespace_reloads_with_the_same_controller`, `fence::registry::tests::seeded_from_load_fences_including_recovered_names` | -| 9 | Filesystem recovery, `destroy_on_error`, undecodable records, missing target quarantine, metastore backup rollback fail closed with provenance | `meta_store::fence_tests::recovery::{fs_recovery_with_marker_unavailable, destroy_on_error_keeps_fenced_unavailable, undecodable_row_unavailable, incomplete_target_unavailable, metastore_rollback_detected_by_marker, lookup_never_creates, undecodable_name_with_fence_fails_startup, marker_in_invalid_directory_fails_startup}`; legacy behaviour kept: `destroy_on_error_without_fences_is_unchanged`, `undecodable_row_without_fences_is_skipped_as_before`; `meta_store::fence_tests::corrupt_fence_row_fails_closed`; restore provenance: `namespace::meta_store::fence_tests::provenance::{bottomless_restore_is_reported (a real bottomless restore against a local S3 endpoint: a metastore opened on an empty directory from a backup reports the restore and its generation and holds the backed-up namespace; one opened with nothing to restore does not), generation_only_after_a_recovery, not_restored_until_recorded_and_first_record_wins}`, `http::admin::fence::tests::restore_provenance_is_reported` (fence view and capability `metastore` object), `tests::fence::admin::capabilities` (not restored: capability endpoint, fence views and the gauge report it) | -| 10 | Wrong owner, stale revision, invalid role/state, replay, command-id reuse; replay before revision check | `fence::transition::tests::*` (exhaustive over states × commands) | -| 11 | Target creation raced with SQL, dump, replication, lifecycle never observable as writable or readable | landed: `namespace::fence::target::tests::create_race_never_observable` (parked after the rows commit and before the config is published and the namespace loaded: SQL connections, stats, replication `hello` (never `UNAVAILABLE`), create, delete and fork of the name are denied or find nothing; afterwards the target is loaded behind the quarantine gate, SQL reads and WAL writes are refused, and lifecycle and replication are refused with `MIGRATION_TARGET_QUARANTINED`), `creating_gate_refuses_before_commit` (parked before the metastore transaction: the same attempts are refused and no database file is created), `create_replay_completes_interrupted_creation` (marker only, after a restart), `create_completes_when_the_caller_goes_away`, `indeterminate_create_is_completed_by_replay`, `create_rejects_existing_name` (a loaded or cold existing name, whose gate never moves, and another operation's target), `abort_keeps_traffic_denied` (also across a restart) | -| 12 | Only the matching import capability writes a quarantined target; admin credentials and admin shell cannot | landed: `namespace::fence::import::tests::import_requires_matching_capability` (plain connections with and without raw access, as the admin shell uses; issuing to another operation or at another revision; at the WAL, capabilities the server never issued, of another operation, at another revision, for validation, and revoked); `namespace::fence::target::tests::{create_race_never_observable, abort_keeps_traffic_denied}` (raw DDL refused); planned: `tests::fence::admin::admin_shell_cannot_write_quarantined` | -| 13 | Seal enters `TARGET_IMPORT_DRAINING`, waits, reaches `TARGET_VALIDATING`, cannot resume import; only a durable validation receipt permits idempotent publication | seal landed: `namespace::fence::import::tests::{seal_waits_for_import_writers (an import call parked inside its write transaction: import is closed at once, the transaction commits, and the seal reaches its completion commit only afterwards), seal_deadline_leaves_import_draining_until_replayed (an idle session's open transaction; `DRAINING`, also across a restart; another operation refused, the owner's new seal joins; the replay completes), seal_force_rollback_ends_open_import_transaction, sealed_target_rejects_import}`; validation/publication landed: `namespace::fence::target::tests::{validation_session_is_read_only (query-only plus WAL defence, live capability checks, server snapshot and a concurrent exact replay while the receipt is committed but not published), publish_requires_validation_receipt, publish_is_idempotent}` | -| 14 | Enable writes idempotent, survives restart and response loss, irreversible | landed: `namespace::fence::target::tests::{enable_writes_idempotent_and_irreversible, enable_writes_survives_restart}` | -| 15 | Lost `EnableTargetWrites` response resolved from receipt/state | landed: `namespace::fence::target::tests::enable_writes_response_loss_resolved` | -| 16 | Read fence drains SQL, dump, `log_entries`, `snapshot`, including dead peers and forced termination | SQL landed: `namespace::fence::read::tests::{read_fence_waits_for_running_program, program_after_closing_gate_is_refused (parked after the read-closing gate, before the CAS), read_fence_cancels_at_deadline, unreleased_lease_answers_draining_and_replay_completes, idle_txn_fails_on_next_program, clear_read_fence_reopens_reads_not_writes, refused_read_fence_reopens_reads, attach_of_read_fenced_namespace_denied}`, `admin_shell::fence_tests::admin_shell_read_denied`; dump and replication landed: `namespace::fence::stream::tests::{dump_lease_released_on_cancel (a dump blocked mid-row on a peer that stopped reading is cancelled at the deadline, its lease released without the peer, the fence acknowledged, and the body ends with the fence error and no `COMMIT;`), read_fence_waits_for_dump, dump_refused_while_read_fenced, log_entries_stream_ends_typed, stream_lease_released_without_peer_read (a dead peer), snapshot_stream_ends_typed, replication_calls_denied_while_read_fenced, read_fence_forced_termination}`; the response `/dump` returns: `namespace::fence::stream::tests::dump_response_aborted_on_cancel` (a dump cancelled by the read drain fails its response body, with the fence error and no `COMMIT;`); over HTTP: `tests::fence::protocol::dump_codes` (complete under a write fence; `423` + code under a read fence and on a quarantined target) | -| 17 | Delete, reset, fork, restore, config, schema mutation rejected | landed: `tests::fence::lifecycle::lifecycle_rejected_while_fenced` (over the admin API, for a write-fenced source and a quarantined target: delete, fork as source and as destination, create with a `dump_url` whose file does not exist, create over the record, linking to a shared schema at creation and config `POST` are all `423` with the fence code; state, revision and data are unchanged and no copy exists; a target cannot be created with a shared schema; after release the source takes writes and config, fork and delete work again); `namespace::store::fence_tests::{reset_refused_while_fenced (called directly and as the replicator's reset callback; after release reset works and wipes the data), lifecycle_refused_while_fenced (fork either side, create over, delete, config and shared-schema link in the metastore transaction)}`; `schema::scheduler::test::fence::{acquire_rejects_shared_schema, migration_not_registered_while_linked_namespace_fenced}` | -| 18 | Codes through HTTP, Hrana, RPC, dump, replication, replica write proxy; distinguishable from auth/timeout/not-found; old peers compatible; no retry loops | user protocols landed: `tests::fence::protocol::{http_codes (legacy `/`, `/v1/execute`, `/v1/batch` for a write-fenced write, a read-fenced read, a quarantined target and a namespace whose marker cannot be decoded: `423` with `code`, and `detail` where there is one), hrana_http_codes (`/v2`, `/v3` pipelines and `/v3/cursor`: step and whole-request errors carry the code and the baton stays usable), hrana_ws_codes (the same over a WebSocket, whose stream reads again after the read fence is cleared), dump_codes, auth_and_not_found_distinct (`401` without or with a wrong credential and `404` for a missing namespace, with no fence code, on the same fenced server)}`, `error::fence_tests::fence_errors_carry_code` (the body through every wrapper; `block_*`'s `Blocked` keeps its mapping); RPC and replica write proxy landed: `rpc::proxy::fence_tests::rpc_codes` (on the primary's proxy service: a write-fenced write is a step error with `SQL_ERROR` + `stable_code`, reads are served, a read-fenced read is a program error with the code when streamed and the typed `FAILED_PRECONDITION` status when unary, and a namespace whose fence state is unknown is refused with the typed status before any connection), `rpc::proxy::fence_tests::replica_maps_proxied_denials` (step, program and connection-status denials become the fence error on the replica; an older primary's error without `stable_code`, an unknown code and other errors keep their mapping), `namespace::fence::outcome::tests::peer_denials_round_trip`, `tests::fence::protocol::{replica_proxy_preserves_code (writes through a replica to a write-fenced primary: `423` + `MIGRATION_WRITE_FENCED` on legacy `/` and `/v1/execute`, the step error on `/v1/batch`, the Hrana error code on `/v2` and `/v3`; reads on the replica served; writes through the replica work after release), denial_not_retried (each refused write is delegated exactly once and answered well within the write proxy's first retry backoff)}`; capability: `tests::fence::admin::capabilities` asserts `proxy_stable_code: true`; replication and replica servers landed: `tests::fence::protocol::{replication_codes (a raw peer of the primary's internal replication service: `hello` carries the write fence's state and revision, an open `log_entries` stream ends with `FAILED_PRECONDITION` + `x-libsql-fence-code` `MIGRATION_READ_FENCED` under the read fence, `hello`, `log_entries` and `snapshot` are then refused with it, `hello` is answered again after the clear and carries no fence after release, and a quarantined target refuses `hello` with `MIGRATION_TARGET_QUARANTINED`), replica_reads_denied_while_source_read_fenced (the replica refuses local reads with `423` + `MIGRATION_READ_FENCED` on legacy `/` and `/v1/execute` and the Hrana code on `/v2` within 500 ms of simulated time after the read fence is acknowledged, and still 30 s later), replica_backs_off_on_fence_code (4 to 9 refused, counted attempts over 60 s of simulated time; a fixed 1 s retry makes 57), replica_resumes_after_clear_read_fence (reads served again within 16 s of the clear, and a write after release is replicated), replica_lazy_creation_refused_by_fence (a read of a quarantined target through a replica that has never loaded it answers `423` + `MIGRATION_TARGET_QUARANTINED` within 500 ms of simulated time, twice, leaving no `dbs/` directory; a directory that was already there is kept; after publication the replica creates and serves the name, and after enable-writes a write through it succeeds)}`, `namespace::meta_store::fence_tests::forget_unstored_only_unused_unstored_entries`, `namespace::fence::registry::tests::forget_idle_only_unreferenced_plain_controllers`, `namespace::fence::replica::tests::{refusal_from_typed_status_only, hello_fence_denies_only_read_denying_states, backoff_doubles_to_its_cap, observed_denial_refuses_local_reads_and_cancels_leases}`; landed: `libsql-replication` `rpc::test::{proxy_error_stable_code_is_additive, replicated_fence_is_additive}` (each new field is skipped by a peer that does not know it, absent from an older peer's message, and absent fields encode exactly as before), `namespace::fence::stream::tests::hello_carries_replicated_fence` (no fence before acquisition and after release; state and revision while write-fenced; the stored configuration never carries it) | -| 19 | Corrupt or unknown durable fence state fails closed | `fence::store::tests::corrupt_payload_fails_closed`, `unknown_format_version_fails_closed` | -| 20 | Metrics and audit logs | landed: `namespace::fence::audit::tests::{audit_event_fields (one event per answer: a committed drain with state before and after, revisions, drain kind and duration and forced actions; a replay; a command-id conflict with its code; a non-fence failure), adoption_event_fields, metrics_and_bounded_labels (every metric of section 15 with its labels, denials at all eight surfaces, the gauges per role and state including zeroed states and the oldest active age; no label value is a namespace, operation or command id)}`; over a running server: `tests::fence::observability::metrics_and_labels` (a write drain with a forced rollback, a replay, a conflict, a Hrana write denial and a lifecycle refusal over the admin API: transitions, replays, forced, drain histogram, denials under `hrana`, `lifecycle` and `http`, the namespace gauge and oldest-age gauge before and after release, and only bounded label keys and values) | -| 21 | Capability discovery and mixed-version protection | capability discovery landed: `tests::fence::admin::{capabilities, capabilities_when_disabled}`; legacy mirror and foreign-key guard landed: `namespace::fence::tests::legacy_mirror::legacy_mirror_and_fk_guard` (a source and two targets walked through every stored state: the config row's `block_*` fields hold the mirror of section 13.2 and nothing else in the row changes, a config write is refused and changes nothing, an older binary's delete fails on the foreign key; release and write enable restore the namespace's own values, later config writes are stored as written, and after a restart the rows are unchanged and the in-memory config holds the namespace's own values, including a config written after the release); an older binary is not run (bounded, see section 18) | -| 22 | Adoption is two-person/audited, keeps admission closed, cannot reverse publication | landed: `namespace::fence::tests::adoption::{adopt_requires_key_and_two_approvers (no key, one approver, duplicate or blank approvers, three approvers, blank incident or reason: `adoption_not_authorised`, nothing changes), adopt_keeps_gates_closed (write-fenced source: owner and revision move, state, admissions, boundary and saved values do not, writes still refused, replay with or without the key returns the receipt, old owner `FENCE_OWNED_BY_ANOTHER_OPERATION`, new owner releases), adopt_quarantined_target (the import capability moves to the new owner; SQL still `MIGRATION_TARGET_QUARANTINED`), adopt_cannot_touch_writable (`TARGET_WRITABLE`, `TARGET_ABORTED`), adopt_cannot_touch_released, adopt_recovers_metastore_rollback (fence row gone, and fence row at an older revision: re-established from the marker, served again behind the same gate, own `block_*` values back in memory, new owner finishes), adopt_recovered_name_without_config_row (`namespace_config_missing`, nothing written, still unavailable), adoption_key_matching}`, `namespace::fence::audit::tests::adoption_event_fields`; pure transition: `namespace::fence::transition::tests::{adopt_requires_key_and_two_approvers, adopt_keeps_gates_closed}`; over HTTP: `tests::fence::admin::{adopt_over_http (no key, wrong key, no admin credential, bad approvers, unknown field, success, replay, old and new owner, finished operation), adopt_disabled_without_key}` | -| — | Import API usable by bulk import | landed: `namespace::fence::import::tests::import_session_loads_dump_into_quarantined_target` (a dump exported by the server from a source with tables, keys, a foreign key, an index, an autoincrement table, a trigger, a view and an FTS5 table loads through `ImportSession::load_dump`; after the seal the target's schema, rows, view and full-text results equal the source's) | +| 1 | Concurrent acquisition by two operations: one owner, typed conflict for the loser | `namespace::fence::drain::tests::acquire_race_single_owner` (the first acquisition is parked after closing admission while the second waits on the transition lock); over the admin API: `tests::fence::admin::concurrent_acquire_one_owner` (two operations acquire at once over HTTP: one `200 APPLIED`, the other `409 FENCE_OWNED_BY_ANOTHER_OPERATION` with the winner in its fence view); admin walks: `tests::fence::admin::{source_walk_over_http, target_walk_over_http, inspect_reports_drain_counters, mutating_routes_require_admin_key}` | +| 2 | Active writer commits or is rolled back before freeze acknowledgement; nothing commits after | `namespace::fence::drain::tests::{active_writer_commits_before_ack, forced_rollback_before_ack, no_commit_after_ack}` (the boundary equals the last committed replication frame and no frame follows it; autocommit, `BEGIN IMMEDIATE`, DDL and a pre-fence read transaction upgrading are refused), `installing_gate_closes_writes_before_persisting`, `refused_acquire_reopens_writes`, `release_reopens_with_new_generation` | +| 3 | Autocommit, explicit transactions, queued writers, batches, DDL, schema jobs, old WebSockets, read-to-write upgrades cannot bypass | `connection::connection_manager::fence_tests::{fence_rejects_read_to_write_upgrade, fence_rejects_ddl_and_pragma, fence_rejects_raw_with_raw_write}` (autocommit, explicit transactions, DDL, header-writing pragma, `BEGIN IMMEDIATE`, `VACUUM`, `with_raw` users); `connection::connection_manager::fence_tests::fence_rejects_queued_writer` (a writer parked in the queue behind an open transaction leaves it with `MIGRATION_WRITE_FENCED` when the fence changes, and the holder keeps the slot); maintenance and vacuum under a fence: `queued_checkpoint_survives_fence_wake`, `checkpoint_allowed_while_fenced`, `vacuum_skipped_while_fenced`; drain primitives: `abort_active_tolerates_closed_connection`, `release_notifies_drain_waiters`, `fence::controller::tests::write_queues_are_woken_on_every_generation_change`; schema jobs: `schema::scheduler::test::fence::{acquire_rejects_shared_schema (a shared schema and a linked namespace cannot be fenced; a fenced namespace cannot be linked by create or config), migration_not_registered_while_linked_namespace_fenced}`; old WebSockets and batches over the protocols: `tests::fence::protocol::{old_ws_session_cannot_write (a WebSocket session whose transaction began before the fence cannot write while fenced nor after the release; after a rollback the same session writes), batch_denied_mid_batch (the read before the write runs, the write step gets the code, a step conditional on it is skipped and one conditional on its failure runs)}` | +| 4 | Program that captured config before the fence is rejected at the WAL | `connection::connection_manager::fence_tests::wal_gate_rejects_program_admitted_before_fence` (a SQL function parks the program between admission and its write while the fence is acquired and released) | +| 5 | Pre-fence transactions cannot write after release or publication | `connection::connection_manager::fence_tests::stale_generation_cannot_write_after_release`, `namespace::fence::target::tests::stale_generation_cannot_write_after_enable_writes`; unfenced behaviour unchanged: `unfenced_namespace_is_unchanged` | +| 6 | Acquisition timeout returns `DRAINING`, admission stays closed | `namespace::fence::drain::tests::{deadline_returns_draining_and_stays_closed, replay_of_draining_resumes_and_completes}` | +| 7 | Restart at every persistence boundary; indeterminate persistence keeps the gate closed until same-command reconciliation | `namespace::fence::tests::restart_at_each_boundary` (a crash at each of 17 boundaries of `AcquireSourceWriteFence` and `ReleaseSourceWriteFence`: after the `INSTALLING` gate, before, at and after each metastore commit, with and without a lagging marker, before publication, before boundary capture and before the response; the restart recovers the prior or the committed state, admits writes only if an opening transition committed, repairs the marker, and a replay finishes the command), `restart_in_draining_waits_for_the_same_command` (a writer active at the crash; nothing advances until the same command is replayed, which completes at once), `indeterminate_commit_keeps_gate_closed` (through the drain path, both when the commit happened and when it did not), `acquire_response_loss_resolved_by_replay_and_inspect`; `fence::controller::tests::{indeterminate_commit_keeps_writes_closed_until_replayed, indeterminate_commit_that_did_not_apply_is_retried_by_replay, failed_before_commit_leaves_gate_unchanged, publication_happens_before_the_response, committed_command_is_published_when_the_caller_goes_away}`, `namespace::store::fence_tests::restart_installs_the_durable_gate_before_serving`; the read fence, the seal and write enable: `namespace::fence::tests::read_and_target_boundaries::restart_at_each_read_and_target_boundary` (a crash at each of 29 boundaries of `SetSourceReadFence`, `SealTargetImport` and `EnableTargetWrites`: after the read-closing or `INSTALLING` gate, before, at and after each metastore commit, with and without a lagging marker, before publication, before the response, and while the drain waits for a reader or an import call that was running when it started; the restart recovers the prior or the committed state with no in-memory gate, reader or import call surviving, keeps reads closed once `SOURCE_READ_DRAINING` committed and import closed once `TARGET_IMPORT_DRAINING` committed, admits target writes only if `TARGET_WRITABLE` committed, repairs the marker, and a replay completes an interrupted drain at once or returns the stored result); the rebuilt log after a restart: `namespace::fence::tests::current_log_id_is_the_rebuilt_log_after_restart` (the stored record keeps the acquisition log and its boundary; the controller's `current_log_id`, which the admin view reports, names the rebuilt log); integration restarts: `tests::fence::lifecycle::{restart_keeps_fence (a graceful same-path TestServer restart reconstructs SOURCE_WRITE_FENCED, SOURCE_READ_FENCED, TARGET_QUARANTINED and TARGET_WRITE_FENCED admission before traffic, exact command replays return their stored receipts, and the user protocol observes the same denials), restart_after_enable_writes_stays_writable (TARGET_WRITABLE stays readable/writable after restart and exact EnableTargetWrites replay returns the stored receipt)}` | +| 8 | Evict and lazily reload a fenced namespace; identical admission | `tests::fence::lifecycle::evicted_namespace_reloads_same_gate` (a one-entry namespace cache is put under capacity pressure by four other loaded namespaces; a user-protocol reload of the fenced namespace keeps the exact revision and generation, serves reads and returns `423 MIGRATION_WRITE_FENCED` for writes); unit-level identity: `namespace::store::fence_tests::evicted_namespace_reloads_with_the_same_controller`, `fence::registry::tests::seeded_from_load_fences_including_recovered_names` | +| 9 | Filesystem recovery, `destroy_on_error`, undecodable records, missing target quarantine, metastore backup rollback fail closed with provenance | `namespace::meta_store::fence_tests::recovery::{fs_recovery_with_marker_unavailable, destroy_on_error_keeps_fenced_unavailable, undecodable_row_unavailable, incomplete_target_unavailable, metastore_rollback_detected_by_marker, lookup_never_creates, undecodable_name_with_fence_fails_startup, marker_in_invalid_directory_fails_startup}`; legacy behaviour kept: `namespace::meta_store::fence_tests::recovery::{destroy_on_error_without_fences_is_unchanged, undecodable_row_without_fences_is_skipped_as_before}`; `namespace::meta_store::fence_tests::corrupt_fence_row_fails_closed`; restore provenance: `namespace::meta_store::fence_tests::provenance::{bottomless_restore_is_reported (a real bottomless restore against a local S3 endpoint: a metastore opened on an empty directory from a backup reports the restore and its generation and holds the backed-up namespace; one opened with nothing to restore does not), generation_only_after_a_recovery, not_restored_until_recorded_and_first_record_wins}`, `http::admin::fence::tests::restore_provenance_is_reported` (fence view and capability `metastore` object), `tests::fence::admin::capabilities` (not restored: capability endpoint, fence views and the gauge report it) | +| 10 | Wrong owner, stale revision, invalid role/state, replay, command-id reuse; replay before revision check | `namespace::fence::transition::tests::{exhaustive_owner_commands (every state × command pair of the owning operation against the full expected table: legal transitions, `ALREADY_APPLIED` goal states, drain joins and a typed refusal for everything else), wrong_owner_is_refused, stale_revision_is_refused, role_mismatch, exact_replay_after_revision_advanced (a replay is answered from its receipt before the revision is checked), command_id_reuse_with_different_fingerprint_conflicts, unavailable_refuses_everything_else, target_writable_is_irreversible, revision_increases_by_one_per_applied_transition, release_from_draining_is_a_precommit_rollback, owner_joins_its_own_drain_with_a_new_command}`; the durable CAS: `namespace::meta_store::fence_tests::{concurrent_cas_has_exactly_one_winner, fence_cas_persists_across_restart, fence_cas_and_config_writes_serialise, target_creation_is_atomic_and_replayable}`, `namespace::fence::store::tests::record_round_trips_and_cas_checks_revision` | +| 11 | Target creation raced with SQL, dump, replication, lifecycle never observable as writable or readable | `namespace::fence::target::tests::create_race_never_observable` (parked after the rows commit and before the config is published and the namespace loaded: SQL connections, stats, replication `hello` (never `UNAVAILABLE`), create, delete and fork of the name are denied or find nothing; afterwards the target is loaded behind the quarantine gate, SQL reads and WAL writes are refused, and lifecycle and replication are refused with `MIGRATION_TARGET_QUARANTINED`), `creating_gate_refuses_before_commit` (parked before the metastore transaction: the same attempts are refused and no database file is created), `create_replay_completes_interrupted_creation` (marker only, after a restart), `create_completes_when_the_caller_goes_away`, `indeterminate_create_is_completed_by_replay`, `create_rejects_existing_name` (a loaded or cold existing name, whose gate never moves, and another operation's target), `abort_keeps_traffic_denied` (also across a restart) | +| 12 | Only the matching import capability writes a quarantined target; admin credentials and admin shell cannot | `namespace::fence::import::tests::import_requires_matching_capability` (plain connections with and without raw access, as the admin shell uses; issuing to another operation or at another revision; at the WAL, capabilities the server never issued, of another operation, at another revision, for validation, and revoked); `namespace::fence::target::tests::{create_race_never_observable, abort_keeps_traffic_denied}` (raw DDL refused); the admin shell: `admin_shell::fence_tests::admin_shell_cannot_write_quarantined` (through the shell's own entry point, every query, read or write, DDL, pragma or `BEGIN IMMEDIATE`, is refused with `MIGRATION_TARGET_QUARANTINED`; a raw write that skipped the shell's read admission is still refused at the WAL; the import session keeps writing and the shell changed nothing); ordinary credentials over HTTP: `tests::fence::admin::target_walk_over_http` (full-access user SQL is refused while the target is quarantined; the admin API has no SQL route besides the admin shell and the capability-scoped `validation-query`) | +| 13 | Seal enters `TARGET_IMPORT_DRAINING`, waits, reaches `TARGET_VALIDATING`, cannot resume import; only a durable validation receipt permits idempotent publication | seal: `namespace::fence::import::tests::{seal_waits_for_import_writers (an import call parked inside its write transaction: import is closed at once, the transaction commits, and the seal reaches its completion commit only afterwards), seal_deadline_leaves_import_draining_until_replayed (an idle session's open transaction; `DRAINING`, also across a restart; another operation refused, the owner's new seal joins; the replay completes), seal_force_rollback_ends_open_import_transaction, sealed_target_rejects_import}`; validation/publication: `namespace::fence::target::tests::{validation_session_is_read_only (query-only plus WAL defence, live capability checks, server snapshot and a concurrent exact replay while the receipt is committed but not published), publish_requires_validation_receipt, publish_is_idempotent}` | +| 14 | Enable writes idempotent, survives restart and response loss, irreversible | `namespace::fence::target::tests::{enable_writes_idempotent_and_irreversible, enable_writes_survives_restart}` | +| 15 | Lost `EnableTargetWrites` response resolved from receipt/state | `namespace::fence::target::tests::enable_writes_response_loss_resolved` (the response is dropped after the commit; an exact replay returns the stored receipt and `InspectFence` shows `TARGET_WRITABLE`); over HTTP the replay of an answered command returns its receipt: `tests::fence::admin::target_walk_over_http`, `tests::fence::lifecycle::restart_after_enable_writes_stays_writable`. Classifying a response that neither replay nor inspection can answer as `COMMIT_UNKNOWN` is the caller's (section 18) | +| 16 | Read fence drains SQL, dump, `log_entries`, `snapshot`, including dead peers and forced termination | SQL: `namespace::fence::read::tests::{read_fence_waits_for_running_program, program_after_closing_gate_is_refused (parked after the read-closing gate, before the CAS), read_fence_cancels_at_deadline, unreleased_lease_answers_draining_and_replay_completes, idle_txn_fails_on_next_program, clear_read_fence_reopens_reads_not_writes, refused_read_fence_reopens_reads, attach_of_read_fenced_namespace_denied}`, `admin_shell::fence_tests::admin_shell_read_denied`; dump and replication: `namespace::fence::stream::tests::{dump_lease_released_on_cancel (a dump blocked mid-row on a peer that stopped reading is cancelled at the deadline, its lease released without the peer, the fence acknowledged, and the body ends with the fence error and no `COMMIT;`), read_fence_waits_for_dump, dump_refused_while_read_fenced, log_entries_stream_ends_typed, stream_lease_released_without_peer_read (a dead peer), snapshot_stream_ends_typed, replication_calls_denied_while_read_fenced, read_fence_forced_termination}`; the response `/dump` returns: `namespace::fence::stream::tests::dump_response_aborted_on_cancel` (a dump cancelled by the read drain fails its response body, with the fence error and no `COMMIT;`); over HTTP: `tests::fence::protocol::dump_codes` (complete under a write fence; `423` + code under a read fence and on a quarantined target) | +| 17 | Delete, reset, fork, restore, config, schema mutation rejected | `tests::fence::lifecycle::lifecycle_rejected_while_fenced` (over the admin API, for a write-fenced source and a quarantined target: delete, fork as source and as destination, create with a `dump_url` whose file does not exist, create over the record, linking to a shared schema at creation and config `POST` are all `423` with the fence code; state, revision and data are unchanged and no copy exists; a target cannot be created with a shared schema; after release the source takes writes and config, fork and delete work again); `namespace::store::fence_tests::{reset_refused_while_fenced (called directly and as the replicator's reset callback; after release reset works and wipes the data), lifecycle_refused_while_fenced (fork either side, create over, delete, config and shared-schema link in the metastore transaction)}`; `schema::scheduler::test::fence::{acquire_rejects_shared_schema, migration_not_registered_while_linked_namespace_fenced}` | +| 18 | Codes through HTTP, Hrana, RPC, dump, replication, replica write proxy; distinguishable from auth/timeout/not-found; old peers compatible; no retry loops | user protocols: `tests::fence::protocol::{http_codes (legacy `/`, `/v1/execute`, `/v1/batch` for a write-fenced write, a read-fenced read, a quarantined target and a namespace whose marker cannot be decoded: `423` with `code`, and `detail` where there is one), hrana_http_codes (`/v2`, `/v3` pipelines and `/v3/cursor`: step and whole-request errors carry the code and the baton stays usable), hrana_ws_codes (the same over a WebSocket, whose stream reads again after the read fence is cleared), dump_codes, auth_and_not_found_distinct (`401` without or with a wrong credential and `404` for a missing namespace, with no fence code, on the same fenced server)}`, `error::fence_tests::fence_errors_carry_code` (the body through every wrapper; `block_*`'s `Blocked` keeps its mapping); RPC and replica write proxy: `rpc::proxy::fence_tests::rpc_codes` (on the primary's proxy service: a write-fenced write is a step error with `SQL_ERROR` + `stable_code`, reads are served, a read-fenced read is a program error with the code when streamed and the typed `FAILED_PRECONDITION` status when unary, and a namespace whose fence state is unknown is refused with the typed status before any connection), `rpc::proxy::fence_tests::replica_maps_proxied_denials` (step, program and connection-status denials become the fence error on the replica; an older primary's error without `stable_code`, an unknown code and other errors keep their mapping), `namespace::fence::outcome::tests::peer_denials_round_trip`, `tests::fence::protocol::{replica_proxy_preserves_code (writes through a replica to a write-fenced primary: `423` + `MIGRATION_WRITE_FENCED` on legacy `/` and `/v1/execute`, the step error on `/v1/batch`, the Hrana error code on `/v2` and `/v3`; reads on the replica served; writes through the replica work after release), denial_not_retried (each refused write is delegated exactly once and answered well within the write proxy's first retry backoff)}`; capability: `tests::fence::admin::capabilities` asserts `proxy_stable_code: true`; replication and replica servers: `tests::fence::protocol::{replication_codes (a raw peer of the primary's internal replication service: `hello` carries the write fence's state and revision, an open `log_entries` stream ends with `FAILED_PRECONDITION` + `x-libsql-fence-code` `MIGRATION_READ_FENCED` under the read fence, `hello`, `log_entries` and `snapshot` are then refused with it, `hello` is answered again after the clear and carries no fence after release, and a quarantined target refuses `hello` with `MIGRATION_TARGET_QUARANTINED`), replica_reads_denied_while_source_read_fenced (the replica refuses local reads with `423` + `MIGRATION_READ_FENCED` on legacy `/` and `/v1/execute` and the Hrana code on `/v2` within 500 ms of simulated time after the read fence is acknowledged, and still 30 s later), replica_backs_off_on_fence_code (4 to 9 refused, counted attempts over 60 s of simulated time; a fixed 1 s retry makes 57), replica_resumes_after_clear_read_fence (reads served again within 16 s of the clear, and a write after release is replicated), replica_lazy_creation_refused_by_fence (a read of a quarantined target through a replica that has never loaded it answers `423` + `MIGRATION_TARGET_QUARANTINED` within 500 ms of simulated time, twice, leaving no `dbs/` directory; a directory that was already there is kept; after publication the replica creates and serves the name, and after enable-writes a write through it succeeds)}`, `namespace::meta_store::fence_tests::forget_unstored_only_unused_unstored_entries`, `namespace::fence::registry::tests::forget_idle_only_unreferenced_plain_controllers`, `namespace::fence::replica::tests::{refusal_from_typed_status_only, hello_fence_denies_only_read_denying_states, backoff_doubles_to_its_cap, observed_denial_refuses_local_reads_and_cancels_leases}`; `libsql-replication` `rpc::test::{proxy_error_stable_code_is_additive, replicated_fence_is_additive}` (each new field is skipped by a peer that does not know it, absent from an older peer's message, and absent fields encode exactly as before), `namespace::fence::stream::tests::hello_carries_replicated_fence` (no fence before acquisition and after release; state and revision while write-fenced; the stored configuration never carries it) | +| 19 | Corrupt or unknown durable fence state fails closed | `namespace::fence::store::tests::corrupt_rows_are_unavailable` (an unknown format version, a revision column that disagrees with the record, an undecodable payload and a negative revision each read as `FENCE_STATE_UNAVAILABLE` with its detail, and deny every class but maintenance and observability), `namespace::fence::record::tests::{unknown_format_version_is_rejected, garbage_is_rejected, revision_column_must_match, invalid_records_are_rejected, invalid_receipts_are_rejected}`, `namespace::meta_store::fence_tests::corrupt_fence_row_fails_closed` (startup keeps the namespace unavailable rather than guessing: not served, not recreatable, config not writable), `namespace::meta_store::fence_tests::recovery::{undecodable_row_unavailable, undecodable_name_with_fence_fails_startup}`; over HTTP: `tests::fence::protocol::http_codes` (a namespace whose marker cannot be decoded) | +| 20 | Metrics and audit logs | `namespace::fence::audit::tests::{audit_event_fields (one event per answer: a committed drain with state before and after, revisions, drain kind and duration and forced actions; a replay; a command-id conflict with its code; a non-fence failure), adoption_event_fields, metrics_and_bounded_labels (every metric of section 15 with its labels, denials at all eight surfaces, the gauges per role and state including zeroed states and the oldest active age; no label value is a namespace, operation or command id)}`; over a running server: `tests::fence::observability::metrics_and_labels` (a write drain with a forced rollback, a replay, a conflict, a Hrana write denial and a lifecycle refusal over the admin API: transitions, replays, forced, drain histogram, denials under `hrana`, `lifecycle` and `http`, the namespace gauge and oldest-age gauge before and after release, and only bounded label keys and values) | +| 21 | Capability discovery and mixed-version protection | capability discovery: `tests::fence::admin::{capabilities, capabilities_when_disabled}`; legacy mirror and foreign-key guard: `namespace::fence::tests::legacy_mirror::legacy_mirror_and_fk_guard` (a source and two targets walked through every stored state: the config row's `block_*` fields hold the mirror of section 13.2 and nothing else in the row changes, a config write is refused and changes nothing, an older binary's delete fails on the foreign key; release and write enable restore the namespace's own values, later config writes are stored as written, and after a restart the rows are unchanged and the in-memory config holds the namespace's own values, including a config written after the release); an older binary is not run (bounded, see section 18) | +| 22 | Adoption is two-person/audited, keeps admission closed, cannot reverse publication | `namespace::fence::tests::adoption::{adopt_requires_key_and_two_approvers (no key, one approver, duplicate or blank approvers, three approvers, blank incident or reason: `adoption_not_authorised`, nothing changes), adopt_keeps_gates_closed (write-fenced source: owner and revision move, state, admissions, boundary and saved values do not, writes still refused, replay with or without the key returns the receipt, old owner `FENCE_OWNED_BY_ANOTHER_OPERATION`, new owner releases), adopt_quarantined_target (the import capability moves to the new owner; SQL still `MIGRATION_TARGET_QUARANTINED`), adopt_cannot_touch_writable (`TARGET_WRITABLE`, `TARGET_ABORTED`), adopt_cannot_touch_released, adopt_recovers_metastore_rollback (fence row gone, and fence row at an older revision: re-established from the marker, served again behind the same gate, own `block_*` values back in memory, new owner finishes), adopt_recovered_name_without_config_row (`namespace_config_missing`, nothing written, still unavailable), adoption_key_matching}`, `namespace::fence::audit::tests::adoption_event_fields`; pure transition: `namespace::fence::transition::tests::{adopt_requires_key_and_two_approvers, adopt_keeps_gates_closed}`; over HTTP: `tests::fence::admin::{adopt_over_http (no key, wrong key, no admin credential, bad approvers, unknown field, success, replay, old and new owner, finished operation), adopt_disabled_without_key}` | +| — | Import API usable by bulk import | `namespace::fence::import::tests::import_session_loads_dump_into_quarantined_target` (a dump exported by the server from a source with tables, keys, a foreign key, an index, an autoincrement table, a trigger, a view and an FTS5 table loads through `ImportSession::load_dump`; after the seal the target's schema, rows, view and full-text results equal the source's) | ## 18. Limits What this design and its tests do not prove: - **Mixed-version behaviour** is tested at the data and wire level (legacy mirror, foreign-key guard, proto unknown-field handling, capability endpoint), not by running an older binary against the same metastore. -- **Representative data** is synthetic. Real production schemas are not part of the test suite. +- **Representative data** is synthetic: multi-table schemas with keys, indexes, triggers, views and an FTS5 table (section 16), under the server's default settings plus the fence tests' small caches and short drain deadlines. Real production schemas and their settings are not part of the test suite. - **Client retry policy** of SDKs was not audited; the server guarantees only that fence denials use codes that are not conventionally retried. - **"Commit unknown"** is the caller's classification when neither replay nor inspection answers; the server's part is that replay and inspection always answer when the server is reachable. - **Two-person adoption** is a separate secret plus a recorded two-approver request, not verified identities. diff --git a/libsql-server/src/admin_shell.rs b/libsql-server/src/admin_shell.rs index a463944427..9f072993ae 100644 --- a/libsql-server/src/admin_shell.rs +++ b/libsql-server/src/admin_shell.rs @@ -347,4 +347,95 @@ mod fence_tests { ); assert_eq!(s.fence.read_lease_counts().total(), 0); } + + /// Section 17 row 12: a quarantined target is written only through its operation's import + /// capability. The admin shell, which reaches the namespace with admin authority and runs raw + /// SQL, can neither read nor write it: every query is refused by the read admission with the + /// quarantine code, and a raw write that skipped the admission is still refused at the WAL. + /// The import session keeps working, and nothing the shell sent changed the data. + #[tokio::test(flavor = "multi_thread")] + async fn admin_shell_cannot_write_quarantined() { + use crate::namespace::fence::target::tests::{create, create_request, OP as TARGET_OP}; + use crate::namespace::open_test_store as open_store; + + let dir = tempfile::tempdir().unwrap(); + let store = open_store(dir.path()).await; + create(&store, create_request("tgt", 1)) + .await + .unwrap() + .unwrap(); + let mut session = store + .open_import_session("tgt".into(), TARGET_OP, 1) + .await + .unwrap(); + session + .with_raw(|c| c.execute_batch("create table t (x); insert into t values (1)")) + .await + .unwrap() + .unwrap(); + + let shell = AdminShell::new(store.clone()); + let sql = [ + "insert into t values (2)", + "delete from t", + "create table u (y)", + "pragma user_version = 7", + "begin immediate", + "select count(*) from t", + ]; + let queries = tokio_stream::iter(sql.map(|q| Ok(rpc::Query { query: q.into() }))); + let responses: Vec<_> = shell + .with_namespace(Bytes::from_static(b"tgt"), queries) + .await + .unwrap() + .collect() + .await; + assert_eq!(responses.len(), sql.len()); + for (q, resp) in sql.iter().zip(&responses) { + let resp = resp.as_ref().unwrap(); + assert!( + error(resp).starts_with("MIGRATION_TARGET_QUARANTINED"), + "{q}: {}", + error(resp) + ); + } + + // Without the shell's read admission, the raw write is refused by the WAL gate: the + // connection holds no capability. + let (fence, maker) = store + .with("tgt".into(), |ns| { + (ns.fence().clone(), ns.db.connection_maker()) + }) + .await + .unwrap(); + let conn = maker.create().await.unwrap(); + for q in ["insert into t values (3)", "create table v (z)"] { + let resp = conn.with_raw(|c| run_one(c, q.into())).unwrap(); + assert!(error(&resp).contains("authoriz"), "{q}: {}", error(&resp)); + } + assert_eq!(fence.read_lease_counts().total(), 0); + + // The capability still writes, and it sees only its own rows. + let rows: i64 = session + .with_raw(|c| { + c.execute("insert into t values (4)", ())?; + c.query_row("select count(*) from t", (), |r| r.get(0)) + }) + .await + .unwrap() + .unwrap(); + assert_eq!(rows, 2); + let tables: i64 = session + .with_raw(|c| { + c.query_row( + "select count(*) from sqlite_schema where type = 'table'", + (), + |r| r.get(0), + ) + }) + .await + .unwrap() + .unwrap(); + assert_eq!(tables, 1); + } } From 5b63b7dbffd0684a2336fd81cf338049877eb062 Mon Sep 17 00:00:00 2001 From: River Date: Wed, 30 Sep 2026 12:47:00 +0000 Subject: [PATCH 31/33] libsql-server: drop transitional dead-code allowances in the fence module The fence module carried a module-wide `allow(dead_code)` while its consumers landed across the series. Everything is wired now, so remove it and deal with the five items it was hiding: - delete the `InBeginWriteTxnAfterCheck` and `AfterManagerRelease` hook points, which no code path reaches and no test arms; - delete the unused `AuditReport::drain` accessor and the `FenceController::cancel_read_leases` wrapper (its one test now uses `cancel_read_leases_by_kind`); - compile `DenialSurface::ALL`, `AuditReport::forced_kinds` and the `Fail`/`Indeterminate` hook outcomes only in the library's test build, which is the only place they are used or produced. Co-authored-by: Tomasz Szymczyszyn --- docs/NAMESPACE_FENCE.md | 2 +- libsql-server/src/namespace/fence/audit.rs | 6 ++---- libsql-server/src/namespace/fence/controller.rs | 11 ++++------- libsql-server/src/namespace/fence/hooks.rs | 7 +++---- libsql-server/src/namespace/fence/mod.rs | 11 +++-------- libsql-server/src/namespace/fence/stream.rs | 2 +- 6 files changed, 14 insertions(+), 25 deletions(-) diff --git a/docs/NAMESPACE_FENCE.md b/docs/NAMESPACE_FENCE.md index e6b772a2b9..489922a01b 100644 --- a/docs/NAMESPACE_FENCE.md +++ b/docs/NAMESPACE_FENCE.md @@ -840,7 +840,7 @@ Implementation (`namespace/fence/audit.rs`): ## 16. Test strategy (Design) -- **No new dependency.** The crate has no failpoint library. Race tests use `#[cfg(test)]` hooks: `FenceTestHooks` holds named points (`AfterInstallingGate`, `AfterClosingReads`, `BeforeReadLeaseCancel`, `BeforeMetastoreCommit`, `AfterMetastoreCommit`, `BeforeGatePublish`, `InBeginWriteTxnAfterCheck`, `AfterManagerRelease`, `BeforeBoundaryCapture`, `AfterTargetRowsCommitted`, `BeforeResponse`), each able to park the task on a pair of `Notify`s (`pause_at` returns handles to wait until the task arrives and to release it) or to inject an error or an indeterminate commit. An armed point fires once. `BeforeMetastoreCommit` is reached immediately before the metastore transaction is started (an injected error there is a failure before commit); an injected indeterminate outcome at `AfterMetastoreCommit` is a commit that happened but was not acknowledged. Hooks compile only in the library's own test build, so they cost nothing in release builds; integration tests under `tests/` cover protocol behaviour and do not rely on hooks. +- **No new dependency.** The crate has no failpoint library. Race tests use `#[cfg(test)]` hooks: `FenceTestHooks` holds named points (`AfterInstallingGate`, `AfterClosingReads`, `BeforeReadLeaseCancel`, `BeforeMetastoreCommit`, `AfterMetastoreCommit`, `BeforeGatePublish`, `BeforeBoundaryCapture`, `AfterTargetRowsCommitted`, `BeforeResponse`), each able to park the task on a pair of `Notify`s (`pause_at` returns handles to wait until the task arrives and to release it) or to inject an error or an indeterminate commit. An armed point fires once. `BeforeMetastoreCommit` is reached immediately before the metastore transaction is started (an injected error there is a failure before commit); an injected indeterminate outcome at `AfterMetastoreCommit` is a commit that happened but was not acknowledged. Hooks compile only in the library's own test build, so they cost nothing in release builds; integration tests under `tests/` cover protocol behaviour and do not rely on hooks. - **Restart** at a boundary: reopen `MetaStore` and rebuild the registry on the same temporary directory, the way existing metastore tests do; integration tests stop and start a `TestServer` on the same path. - **Restart** at a boundary, as a crash: `namespace::fence::tests` runs each server lifetime on a runtime of its own and ends it by shutting the runtime down without any shutdown code and leaking the `NamespaceStore`, with the command parked at the hook point; the next lifetime opens a new `NamespaceStore` (metastore and registry) on the same directory. The namespace's `.sentinel` stays behind, so the restart takes the real dirty-recovery path. - **Response loss**: the test drops the command future after `AfterMetastoreCommit` and then replays or inspects. diff --git a/libsql-server/src/namespace/fence/audit.rs b/libsql-server/src/namespace/fence/audit.rs index 063d2484cb..009dee94f5 100644 --- a/libsql-server/src/namespace/fence/audit.rs +++ b/libsql-server/src/namespace/fence/audit.rs @@ -144,6 +144,7 @@ pub enum DenialSurface { } impl DenialSurface { + #[cfg(test)] pub const ALL: [DenialSurface; 8] = [ DenialSurface::Http, DenialSurface::Hrana, @@ -216,10 +217,7 @@ impl CommandReport { } } - pub fn drain(&self) -> Option<(DrainKind, Duration)> { - self.drain - } - + #[cfg(test)] pub fn forced_kinds(&self) -> &[ForcedKind] { &self.forced } diff --git a/libsql-server/src/namespace/fence/controller.rs b/libsql-server/src/namespace/fence/controller.rs index 11c0bfec25..4b05a249db 100644 --- a/libsql-server/src/namespace/fence/controller.rs +++ b/libsql-server/src/namespace/fence/controller.rs @@ -500,9 +500,7 @@ impl FenceController { /// Cancel every read lease held now (the read drain's deadline). Each lease's work is asked /// to stop once; the leases stay counted until they are actually released. Returns how many - /// were asked. - /// [`cancel_read_leases`](Self::cancel_read_leases), counted by the kind of work asked to - /// stop. + /// were asked, counted by the kind of work asked to stop. pub(crate) fn cancel_read_leases_by_kind(&self) -> ReadLeaseCounts { let leases = self.read_leases.lock(); let mut asked = ReadLeaseCounts::default(); @@ -519,10 +517,6 @@ impl FenceController { asked } - pub(crate) fn cancel_read_leases(&self) -> usize { - self.cancel_read_leases_by_kind().total() - } - /// On a replica server: publish what the replicator learned of the primary's fence /// (section 6.2). `Some` denies normal reads and streams of the local copy with that error /// and asks every read lease held now to stop, so that work admitted before the replica @@ -1032,9 +1026,11 @@ impl Transition { let result = match controller.hook(HookPoint::BeforeMetastoreCommit).await { HookOutcome::Continue => run.await, + #[cfg(test)] HookOutcome::Fail(e) => return Err(e.into()), // A commit that failed without applying, but whose outcome the controller cannot // know (test hook). + #[cfg(test)] HookOutcome::Indeterminate => { Err(indeterminate(key, "the commit was not acknowledged (test hook)").into()) } @@ -1043,6 +1039,7 @@ impl Transition { let result = match result { Ok(commit) => match controller.hook(HookPoint::AfterMetastoreCommit).await { HookOutcome::Continue => Ok(commit), + #[cfg(test)] HookOutcome::Indeterminate | HookOutcome::Fail(_) => Err(indeterminate( key, "the commit was not acknowledged (test hook)", diff --git a/libsql-server/src/namespace/fence/hooks.rs b/libsql-server/src/namespace/fence/hooks.rs index 75631e122c..23f05d3336 100644 --- a/libsql-server/src/namespace/fence/hooks.rs +++ b/libsql-server/src/namespace/fence/hooks.rs @@ -18,6 +18,7 @@ use parking_lot::Mutex; #[cfg(test)] use tokio::sync::Notify; +#[cfg(test)] use super::outcome::FenceError; /// A named point on a fence transition or a gated path. @@ -35,10 +36,6 @@ pub enum HookPoint { AfterMetastoreCommit, /// The committed result is about to be published to the gate. BeforeGatePublish, - /// In `begin_write_txn`, after the gate check admitted the transaction. - InBeginWriteTxnAfterCheck, - /// The connection manager released the write slot. - AfterManagerRelease, /// A drain is about to read the frozen boundary. BeforeBoundaryCapture, /// The rows of a quarantined target were committed; the target is not published yet. @@ -70,7 +67,9 @@ pub enum HookAction { #[derive(Debug, Clone, PartialEq, Eq)] pub enum HookOutcome { Continue, + #[cfg(test)] Fail(FenceError), + #[cfg(test)] Indeterminate, } diff --git a/libsql-server/src/namespace/fence/mod.rs b/libsql-server/src/namespace/fence/mod.rs index 46ed08eb5e..a4f5ba9130 100644 --- a/libsql-server/src/namespace/fence/mod.rs +++ b/libsql-server/src/namespace/fence/mod.rs @@ -9,17 +9,12 @@ //! ([`transition`]), the metastore tables, compare-and-swap and marker file that persist them //! ([`store`], driven by `MetaStore::apply_fence_command`), and the in-memory authority built //! on them: the per-namespace [`controller`] with its gate and read leases, the positive write -//! [`drain`], the source [`read`] fence and its -//! [`stream`] leases for dump and replication, quarantined migration [`target`]s with their -//! [`capability`]-scoped [`import`] sessions and seal drain, the [`registry`] that holds the controllers outside the namespace cache, the +//! [`drain`], the source [`read`] fence and its [`stream`] leases for dump and replication, +//! quarantined migration [`target`]s with their [`capability`]-scoped [`import`] sessions and +//! seal drain, the [`registry`] that holds the controllers outside the namespace cache, the //! [`replica`]-server view of a primary's fence, the [`audit`] log, and the test [`hooks`] on //! their paths. -// The persistence, controller and protocol layers that consume these types land in the -// following commits of this series; until then most of the module is unused by the rest of -// the crate. This attribute is removed once they are wired. -#![allow(dead_code)] - pub mod audit; pub mod capability; pub mod command; diff --git a/libsql-server/src/namespace/fence/stream.rs b/libsql-server/src/namespace/fence/stream.rs index 2711105a2a..dd763abacf 100644 --- a/libsql-server/src/namespace/fence/stream.rs +++ b/libsql-server/src/namespace/fence/stream.rs @@ -730,7 +730,7 @@ mod tests { fence_status as fn(FenceError) -> tonic::Status, ) .unwrap(); - assert_eq!(s.fence.cancel_read_leases(), 1); + assert_eq!(s.fence.cancel_read_leases_by_kind().total(), 1); until_released(&s.fence).await; let status = tokio::time::timeout(PROMPT, stream.next()) .await From 3da5eac5a3357373a8fdc610df42cffb9db7ef46 Mon Sep 17 00:00:00 2001 From: River Date: Wed, 30 Sep 2026 13:17:11 +0000 Subject: [PATCH 32/33] libsql-server: fix fence replay and retry edge cases Keep stale DRAINING receipts historical once their exact drain state and revision have been superseded, and preserve resumed attribution when a live drain completes. This prevents old commands from cancelling or completing work in newer states while keeping replay responses and audit classification accurate. Count fence errors delivered as terminal replication-stream statuses in the consecutive-refusal backoff, so reconnect pacing doubles from the first refusal as documented. Co-authored-by: Tomasz Szymczyszyn --- docs/NAMESPACE_FENCE.md | 10 ++- libsql-server/src/http/admin/fence.rs | 2 +- libsql-server/src/namespace/fence/drain.rs | 58 +++++++++++-- libsql-server/src/namespace/fence/import.rs | 15 +++- libsql-server/src/namespace/fence/read.rs | 15 +++- .../src/namespace/fence/transition.rs | 84 +++++++++++++++++-- libsql-server/src/namespace/meta_store.rs | 17 ++-- .../src/replication/replicator_client.rs | 24 ++++-- 8 files changed, 187 insertions(+), 38 deletions(-) diff --git a/docs/NAMESPACE_FENCE.md b/docs/NAMESPACE_FENCE.md index 489922a01b..8cea5311be 100644 --- a/docs/NAMESPACE_FENCE.md +++ b/docs/NAMESPACE_FENCE.md @@ -228,7 +228,7 @@ GET /v1/namespaces/:namespace/fence POST /v1/namespaces/:namespace/fence/source/acquire-write-fence ``` -`AcquireSourceWriteFence`. Extra body fields: `expected_namespace_identity: { "log_id": "" }` (the replication log id the caller observed) and `drain_policy: { "deadline_ms": , "on_deadline": "fail" | "force_rollback" }`. Returns `APPLIED` with `SOURCE_WRITE_FENCED` and the frozen boundary, or `DRAINING` with `SOURCE_DRAINING` if the deadline passed under `fail`, or if the request was cut short. Replaying the same command resumes the same drain. +`AcquireSourceWriteFence`. Extra body fields: `expected_namespace_identity: { "log_id": "" }` (the replication log id the caller observed) and `drain_policy: { "deadline_ms": , "on_deadline": "fail" | "force_rollback" }`. Returns `APPLIED` with `SOURCE_WRITE_FENCED` and the frozen boundary, or `DRAINING` with `SOURCE_DRAINING` if the deadline passed under `fail`, or if the request was cut short. Replaying the same command resumes the same drain while its state and revision are still current. ```HTTP POST /v1/namespaces/:namespace/fence/source/set-read-fence @@ -321,7 +321,9 @@ For every mutating command, under the per-namespace transition lock: 2. Compute the fingerprint: SHA-256 over the deterministic protobuf encoding of the command *including* namespace, `operation_id`, command kind, `expected_state`, `expected_revision` and every argument, *excluding* `command_id`. 3. Look up `(namespace, operation_id, command_id)`: - found, same fingerprint, final outcome: return the stored result with `"replayed": true`. This holds even though the revision has since advanced. - - found, same fingerprint, in-progress (`DRAINING`, or an indeterminate commit being reconciled): resume that same command (section 8.4). + - found, same fingerprint, `DRAINING` while the record is still in that command's draining state at the receipt's revision: resume that same command (section 8.4). + - found, same fingerprint, `DRAINING` after `ReleaseSourceWriteFence`, `ClearSourceReadFence` or `AbortQuarantinedTarget` superseded the drain: return the stored historical result with `"replayed": true`; do not run the old drain against the newer state. + - found, same fingerprint, an indeterminate commit being reconciled: resume that same command (section 8.4). - found, different fingerprint: `FENCE_COMMAND_CONFLICT`. Nothing changes. 4. Only a non-replay proceeds. A namespace whose control state cannot be established refuses everything with `FENCE_STATE_UNAVAILABLE`, except the two commands that reconcile it: a replay of the `CreateTargetQuarantined` that left the marker (section 10.1), and `AdoptFence` after a metastore rollback (section 12). 5. Owner check (`FENCE_OWNED_BY_ANOTHER_OPERATION`) against an unfinished record. @@ -528,7 +530,7 @@ This makes the WAL gate independent of statement classification: DDL, misclassif 4. CAS `SOURCE_DRAINING` in the metastore with receipt outcome `DRAINING`. - Committed: the commit publishes `SOURCE_DRAINING` in place of the `INSTALLING` gate (write admission never reopens in between). Continue. - A replay of a finished acquisition, or `ALREADY_APPLIED`: publish the durable state, respond with the stored result. - - A replay of the `DRAINING` receipt, or a new command of the owner joining the drain the record is in: nothing is written; continue at step 5. + - A replay of a `DRAINING` receipt while the record is still in that command's draining state at the receipt's revision, or a new command of the owner joining that state: nothing is written; continue at step 5. A drain superseded by release is returned as a replay without continuing. - Proven not committed (the transition function refused it, or the transaction failed before `COMMIT`): remove the `INSTALLING` gate, which moves the generation again, and return the error. A transaction opened while it was up therefore cannot write afterwards. - Unknown (error on `COMMIT`, the task died): the gate closes as indeterminate (section 7.2) and the response is `FENCE_COMMIT_INDETERMINATE`. It is never treated as not applied. 5. For each live source, wait until its connection manager has no connection holding the write slot for a write (a checkpoint, `Maintenance`, may hold it): enable the manager's release `Notify`, check `has_writer()`, and wait for the notification. Elapsed time and `txn_timeout` are never evidence; with admission closed nobody queues behind the holder, so its slot is not stolen by the timeout either. Because admission is closed, a manager seen without a writer stays without one, so the managers are waited for in turn. @@ -540,7 +542,7 @@ This makes the WAL gate independent of statement classification: DDL, misclassif ### 8.4 Reconciliation and resumption -- Replay of a command whose receipt says `DRAINING` resumes at step 5, with the request's own deadline counted from the replay. After a restart there is no pre-cutoff writer (SQLite recovery discards uncommitted work), so it completes at once. +- Replay of a command whose receipt says `DRAINING` resumes at step 5 only while the durable record is still in that command's draining state at the same revision, with the request's own deadline counted from the replay. The revision check distinguishes a later read-fence cycle under the same operation and state. If that replay proves the drain, its response still has `"replayed": true` and the audit event classifies it as `resume`, even though completing the drain wrote the final receipt. After a restart there is no pre-cutoff writer (SQLite recovery discards uncommitted work), so it completes at once. If an explicit release, read-fence clear or target abort has superseded the drain, replay returns the stored historical `DRAINING` result without restarting it. - Replay of a command held `Indeterminate` re-reads the metastore: if the row shows the command applied, it continues from that durable point; if it shows it did not, it retries the same CAS. Other commands receive `FENCE_COMMIT_INDETERMINATE` until then. After a restart the gate reflects whatever is durable, which by definition was never acknowledged as open. - Opening transitions (`ReleaseSourceWriteFence`, `EnableTargetWrites`) follow **commit → publish the exact revision to the gate → respond `APPLIED`**. A crash after commit and before publication sends no success, and startup recovers the committed gate before exposing the namespace. diff --git a/libsql-server/src/http/admin/fence.rs b/libsql-server/src/http/admin/fence.rs index a4b94b86bb..45622fe782 100644 --- a/libsql-server/src/http/admin/fence.rs +++ b/libsql-server/src/http/admin/fence.rs @@ -482,7 +482,7 @@ fn success_reply( let outcome = commit.receipt.outcome; let body = json!({ "outcome": outcome.as_str(), - "replayed": commit.kind == crate::namespace::meta_store::FenceCommitKind::Replayed, + "replayed": commit.kind != crate::namespace::meta_store::FenceCommitKind::Committed, "fence": fence, "receipt": receipt_json(&commit.receipt), "drain": drain_json(controller.as_deref()), diff --git a/libsql-server/src/namespace/fence/drain.rs b/libsql-server/src/namespace/fence/drain.rs index edd0a22795..a0aef2f2cb 100644 --- a/libsql-server/src/namespace/fence/drain.rs +++ b/libsql-server/src/namespace/fence/drain.rs @@ -14,7 +14,7 @@ use std::time::Duration; use tokio::time::Instant; -use crate::namespace::meta_store::{FenceCommit, FenceContext, MetaStore}; +use crate::namespace::meta_store::{FenceCommit, FenceCommitKind, FenceContext, MetaStore}; use super::audit::{self, CommandAudit, CommandReport, DrainKind, ForcedKind}; use super::command::{DrainPolicy, FenceCommand, FenceRequest, OnDeadline}; @@ -118,10 +118,13 @@ pub async fn acquire_source_write_fence( return Err(e); } }; - if commit.receipt.outcome != FenceOutcome::Draining { - // A replay of a finished acquisition, or ALREADY_APPLIED. + if commit.kind == FenceCommitKind::Replayed || commit.receipt.outcome != FenceOutcome::Draining + { + // A replay whose drain was superseded, a replay of a finished acquisition, or + // ALREADY_APPLIED. return Ok(commit); } + let resumed = commit.kind == FenceCommitKind::Resumed; let drain_key = (commit.receipt.operation_id, commit.receipt.command_id); // Steps 5 and 6. @@ -146,14 +149,18 @@ pub async fn acquire_source_write_fence( ); } ctx.now_ms = now_ms(); - transition + let mut completed = transition .complete_drain( meta, drain_key, DrainCompletion::SourceWrites { boundary }, ctx, ) - .await + .await?; + if resumed { + completed.kind = FenceCommitKind::Resumed; + } + Ok(completed) } /// Wait until no connection manager of the namespace has a writer holding its write slot, and @@ -698,6 +705,7 @@ pub(crate) mod tests { raw(&holder, "commit").await.unwrap(); let committed = s.frame_no(); let done = s.execute(s.acquire(OP, 1, policy)).await.unwrap().unwrap(); + assert_eq!(done.kind, FenceCommitKind::Resumed); assert_eq!(done.receipt.outcome, FenceOutcome::Applied); assert_eq!(done.receipt.command_id, Uuid::from_u128(1)); assert_eq!( @@ -716,6 +724,46 @@ pub(crate) mod tests { assert_eq!(boundary(&again), boundary(&done)); } + /// A drain that was explicitly rolled back by `ReleaseSourceWriteFence` is historical: an + /// exact replay returns its stored `DRAINING` receipt without trying to complete that drain + /// against the newer `RELEASED` record. + #[tokio::test(flavor = "multi_thread")] + async fn replay_of_released_drain_does_not_resume_it() { + let s = Source::new().await; + let holder = s.conn().await; + raw(&holder, "begin immediate; insert into t values (1);") + .await + .unwrap(); + let policy = DrainPolicy { + deadline_ms: 0, + on_deadline: OnDeadline::Fail, + }; + let acquire = s.acquire(OP, 1, policy); + let first = s.execute(acquire.clone()).await.unwrap().unwrap(); + assert_eq!(first.receipt.outcome, FenceOutcome::Draining); + + let release = FenceRequest { + namespace: "ns".into(), + operation_id: OP, + command_id: Uuid::from_u128(2), + expected_state: FenceState::SourceDraining, + expected_revision: s.fence.gate().revision(), + command: FenceCommand::ReleaseSourceWriteFence, + }; + let released = s.execute(release).await.unwrap().unwrap(); + assert_eq!(released.receipt.outcome, FenceOutcome::Applied); + assert_eq!(s.fence.gate().state(), FenceState::Released); + raw(&holder, "commit").await.unwrap(); + + let replay = s.execute(acquire).await.unwrap().unwrap(); + assert_eq!(replay.kind, FenceCommitKind::Replayed); + assert_eq!(replay.receipt.outcome, FenceOutcome::Draining); + assert_eq!(s.fence.gate().state(), FenceState::Released); + raw(&s.conn().await, "insert into t values (2)") + .await + .unwrap(); + } + /// Releasing the write fence commits, publishes a new write generation and only then /// answers: new programs write again, and a transaction that began under the fence cannot. #[tokio::test(flavor = "multi_thread")] diff --git a/libsql-server/src/namespace/fence/import.rs b/libsql-server/src/namespace/fence/import.rs index 206d6c1561..a8a20a3a31 100644 --- a/libsql-server/src/namespace/fence/import.rs +++ b/libsql-server/src/namespace/fence/import.rs @@ -25,7 +25,7 @@ use crate::connection::legacy::LegacyConnection; use crate::connection::Connection as _; use crate::error::Error; use crate::namespace::configurator::{load_dump_sql, read_dump}; -use crate::namespace::meta_store::{FenceCommit, FenceContext, MetaStore}; +use crate::namespace::meta_store::{FenceCommit, FenceCommitKind, FenceContext, MetaStore}; use crate::namespace::replication_wal::ReplicationWalWrapper; use super::audit::{CommandReport, DrainKind, ForcedKind}; @@ -168,9 +168,11 @@ pub async fn seal_target_import( return Err(e); } }; - if commit.receipt.outcome != FenceOutcome::Draining { + if commit.kind == FenceCommitKind::Replayed || commit.receipt.outcome != FenceOutcome::Draining + { return Ok(commit); } + let resumed = commit.kind == FenceCommitKind::Resumed; let drain_key = (commit.receipt.operation_id, commit.receipt.command_id); let started = Instant::now(); @@ -182,9 +184,13 @@ pub async fn seal_target_import( .report .drained(DrainKind::Import, started.elapsed()); ctx.now_ms = now_ms(); - transition + let mut completed = transition .complete_drain(meta, drain_key, DrainCompletion::TargetImport, ctx) - .await + .await?; + if resumed { + completed.kind = FenceCommitKind::Resumed; + } + Ok(completed) } /// Wait until no import call is running and no connection manager of the target has a writer @@ -624,6 +630,7 @@ mod tests { .await .unwrap() .unwrap(); + assert_eq!(replay.kind, FenceCommitKind::Resumed); assert_eq!(replay.receipt.outcome, FenceOutcome::Applied); assert_eq!( (fence.gate().state(), fence.gate().revision()), diff --git a/libsql-server/src/namespace/fence/read.rs b/libsql-server/src/namespace/fence/read.rs index 643e006d01..0603e962b8 100644 --- a/libsql-server/src/namespace/fence/read.rs +++ b/libsql-server/src/namespace/fence/read.rs @@ -13,7 +13,7 @@ use std::time::Duration; use tokio::time::Instant; -use crate::namespace::meta_store::{FenceCommit, FenceContext, MetaStore}; +use crate::namespace::meta_store::{FenceCommit, FenceCommitKind, FenceContext, MetaStore}; use super::audit::{CommandReport, DrainKind, ForcedKind}; use super::command::{DrainPolicy, FenceCommand, FenceRequest}; @@ -68,9 +68,11 @@ pub async fn set_source_read_fence( return Err(e); } }; - if commit.receipt.outcome != FenceOutcome::Draining { + if commit.kind == FenceCommitKind::Replayed || commit.receipt.outcome != FenceOutcome::Draining + { return Ok(commit); } + let resumed = commit.kind == FenceCommitKind::Resumed; let drain_key = (commit.receipt.operation_id, commit.receipt.command_id); // Step 4. @@ -84,9 +86,13 @@ pub async fn set_source_read_fence( .drained(DrainKind::Read, started.elapsed()); // Step 5. ctx.now_ms = now_ms(); - transition + let mut completed = transition .complete_drain(meta, drain_key, DrainCompletion::SourceReads, ctx) - .await + .await?; + if resumed { + completed.kind = FenceCommitKind::Resumed; + } + Ok(completed) } /// Wait until every read lease of the namespace is released. At the deadline the leases still @@ -433,6 +439,7 @@ pub(crate) mod tests { // The program was cancelled by the fence; it reports the fence, not its rows. read_fenced(&running.await.unwrap().unwrap_err()); let replayed = s.execute(request).await.unwrap(); + assert_eq!(replayed.as_ref().unwrap().kind, FenceCommitKind::Resumed); assert_eq!(fence_outcome(&replayed), FenceOutcome::Applied); assert_eq!(s.fence.gate().state(), FenceState::SourceReadFenced); } diff --git a/libsql-server/src/namespace/fence/transition.rs b/libsql-server/src/namespace/fence/transition.rs index 9c664b320c..1a4b5fceb7 100644 --- a/libsql-server/src/namespace/fence/transition.rs +++ b/libsql-server/src/namespace/fence/transition.rs @@ -10,8 +10,9 @@ //! Checks run in this order, and the order is part of the contract: //! //! 1. **Replay.** A stored receipt with the same fingerprint is answered from the receipt -//! (`Replay`, or `Resume` for a drain still in progress), whatever has happened to the -//! record since. A stored receipt with a different fingerprint is `FENCE_COMMAND_CONFLICT`. +//! (`Resume` only while the record is still in that command's draining state and revision, +//! otherwise `Replay`), whatever has happened to the record since. A stored receipt with a +//! different fingerprint is `FENCE_COMMAND_CONFLICT`. //! 2. **Unavailable state.** A record the server cannot establish refuses everything with //! `FENCE_STATE_UNAVAILABLE`, except the two commands that can reconcile it: a replay of the //! `CreateTargetQuarantined` that left the marker, and an adoption after a metastore @@ -260,10 +261,22 @@ pub fn apply( ), )); } - return Ok(if existing.is_final() { - Decision::Replay(existing.clone()) - } else { + // A DRAINING receipt resumes only while its exact drain is still the durable state. + // Release, clear-read and abort may supersede one, and a source may begin another read + // drain later under the same operation and state; the record revision distinguishes that + // later cycle. Replaying an older command must return its stored answer without running + // it against the newer record (and potentially cancelling newly admitted work). + let still_draining = matches!( + current, + CurrentFence::Record(record) + if record.operation_id == existing.operation_id + && drain_state(existing.command) == Some(record.state) + && existing.revision_after == record.revision + ); + return Ok(if !existing.is_final() && still_draining { Decision::Resume(existing.clone()) + } else { + Decision::Replay(existing.clone()) }); } @@ -646,9 +659,12 @@ pub fn complete_drain( completion: DrainCompletion, env: &ApplyEnv, ) -> Result<(NamespaceFenceRecord, CommandReceipt), FenceError> { - if receipt.outcome != FenceOutcome::Draining || receipt.operation_id != record.operation_id { + if receipt.outcome != FenceOutcome::Draining + || receipt.operation_id != record.operation_id + || receipt.revision_after != record.revision + { return Err(invalid( - "there is no drain of the owning operation to complete", + "there is no matching drain of the owning operation to complete", )); } @@ -1274,6 +1290,57 @@ mod tests { assert_eq!(receipt.revision_after, 2); } + #[test] + fn superseded_draining_receipt_is_replayed_without_resuming() { + let mut source = Harness::source(); + let acquire = source.request(OP, command(CommandKind::AcquireSourceWriteFence)); + source.run_request(&acquire, &env()).unwrap(); + source + .run(OP, CommandKind::ReleaseSourceWriteFence) + .unwrap(); + let replay = source.decide(&acquire, &env()).unwrap(); + assert!(matches!( + replay, + Decision::Replay(ref receipt) if receipt.outcome == O::Draining + )); + assert_eq!(source.state(), S::Released); + + let mut source = Harness::in_state(S::SourceWriteFenced); + let read_fence = source.request(OP, command(CommandKind::SetSourceReadFence)); + source.run_request(&read_fence, &env()).unwrap(); + source.run(OP, CommandKind::ClearSourceReadFence).unwrap(); + let replay = source.decide(&read_fence, &env()).unwrap(); + assert!(matches!( + replay, + Decision::Replay(ref receipt) if receipt.outcome == O::Draining + )); + assert_eq!(source.state(), S::SourceWriteFenced); + + // Starting another read drain under the same operation and state does not make the + // earlier cycle live again: only receipts at the current drain revision may resume it. + let later_read_fence = source.request(OP, command(CommandKind::SetSourceReadFence)); + source.run_request(&later_read_fence, &env()).unwrap(); + assert!(matches!( + source.decide(&read_fence, &env()).unwrap(), + Decision::Replay(ref receipt) if receipt.outcome == O::Draining + )); + assert!(matches!( + source.decide(&later_read_fence, &env()).unwrap(), + Decision::Resume(ref receipt) if receipt.outcome == O::Draining + )); + + let mut target = Harness::in_state(S::TargetQuarantined); + let seal = target.request(OP, command(CommandKind::SealTargetImport)); + target.run_request(&seal, &env()).unwrap(); + target.run(OP, CommandKind::AbortQuarantinedTarget).unwrap(); + let replay = target.decide(&seal, &env()).unwrap(); + assert!(matches!( + replay, + Decision::Replay(ref receipt) if receipt.outcome == O::Draining + )); + assert_eq!(target.state(), S::TargetAborted); + } + #[test] fn command_id_reuse_with_different_fingerprint_conflicts() { let mut h = Harness::source(); @@ -1515,6 +1582,9 @@ mod tests { let mut other = receipt.clone(); other.operation_id = OTHER_OP; assert!(complete_drain(&record, &other, boundary, &env()).is_err()); + let mut stale = receipt.clone(); + stale.revision_after -= 1; + assert!(complete_drain(&record, &stale, boundary, &env()).is_err()); // Not draining any more. let h = Harness::in_state(S::SourceWriteFenced); diff --git a/libsql-server/src/namespace/meta_store.rs b/libsql-server/src/namespace/meta_store.rs index 6dc5de14d2..4a63018ac3 100644 --- a/libsql-server/src/namespace/meta_store.rs +++ b/libsql-server/src/namespace/meta_store.rs @@ -951,14 +951,17 @@ fn apply_fence_command( let decision = transition::apply(stored.as_current(), existing.as_ref(), request, &env)?; let (record, receipt) = match decision { - Decision::Replay(receipt) | Decision::Resume(receipt) => { - let kind = if receipt.is_final() { - FenceCommitKind::Replayed - } else { - FenceCommitKind::Resumed - }; + Decision::Replay(receipt) => { + return Ok(FenceCommit { + kind: FenceCommitKind::Replayed, + receipt, + record: stored.record().cloned(), + created_config: None, + }); + } + Decision::Resume(receipt) => { return Ok(FenceCommit { - kind, + kind: FenceCommitKind::Resumed, receipt, record: stored.record().cloned(), created_config: None, diff --git a/libsql-server/src/replication/replicator_client.rs b/libsql-server/src/replication/replicator_client.rs index 9d9a6df78f..541f3b68e6 100644 --- a/libsql-server/src/replication/replicator_client.rs +++ b/libsql-server/src/replication/replicator_client.rs @@ -1,5 +1,6 @@ use std::path::Path; use std::pin::Pin; +use std::sync::atomic::{AtomicU32, Ordering}; use std::sync::Arc; use bytes::Bytes; @@ -114,8 +115,9 @@ pub struct Client { /// The namespace's fence controller on this replica server, on which the primary's fence /// is published as a local read denial (`docs/NAMESPACE_FENCE.md` section 6.2). fence: Arc, - /// Replication calls the primary's fence refused since the last `hello` it answered. - fence_refusals: u32, + /// Replication calls the primary's fence refused since the last `hello` it answered. Shared + /// with active frame streams so a refusal delivered as their terminal status is counted too. + fence_refusals: Arc, } impl Client { @@ -137,14 +139,14 @@ impl Client { wal_impl: wal_flavor, first_sync_since_handshake: true, fence, - fence_refusals: 0, + fence_refusals: Arc::new(AtomicU32::new(0)), }) } /// Replication calls the primary's fence refused in a row, since the last `hello` it /// answered. The replica's replication loop paces its reconnects by it. pub(crate) fn fence_refusals(&self) -> u32 { - self.fence_refusals + self.fence_refusals.load(Ordering::Relaxed) } /// Publish what the primary said of its fence as this replica's local read denial, logging @@ -171,7 +173,11 @@ impl Client { fn status_error(&mut self, status: Status) -> Error { let error = replica::replicator_error(status); if let Some(refusal) = PrimaryFenceRefusal::of(&error) { - self.fence_refusals = self.fence_refusals.saturating_add(1); + self.fence_refusals + .fetch_update(Ordering::Relaxed, Ordering::Relaxed, |count| { + Some(count.saturating_add(1)) + }) + .ok(); metrics::increment_counter!( "libsql_server_replica_fence_refusals_total", "code" => refusal.0.outcome().as_str(), @@ -189,10 +195,16 @@ impl Client { stream: tonic::Streaming, ) -> impl Stream> + Send + 'static { let fence = self.fence.clone(); + let fence_refusals = self.fence_refusals.clone(); let namespace = self.namespace.clone(); stream.map_err(move |status| { let error = replica::replicator_error(status); if let Some(refusal) = PrimaryFenceRefusal::of(&error) { + fence_refusals + .fetch_update(Ordering::Relaxed, Ordering::Relaxed, |count| { + Some(count.saturating_add(1)) + }) + .ok(); metrics::increment_counter!( "libsql_server_replica_fence_refusals_total", "code" => refusal.0.outcome().as_str(), @@ -253,7 +265,7 @@ impl ReplicatorClient for Client { let hello = resp.into_inner(); verify_session_token(&hello.session_token).map_err(Error::Client)?; // The primary answers `hello` only where its fence admits replication. - self.fence_refusals = 0; + self.fence_refusals.store(0, Ordering::Relaxed); self.observe_primary_fence(replica::denial_from_hello( hello.config.as_ref().and_then(|c| c.fence.as_ref()), )); From 6bdbff868db7ee27740b5ccdc048b91a7b129f04 Mon Sep 17 00:00:00 2001 From: River Date: Wed, 30 Sep 2026 13:29:48 +0000 Subject: [PATCH 33/33] libsql-server: expect resumed drain attribution after restart Co-authored-by: Tomasz Szymczyszyn --- libsql-server/src/namespace/fence/tests.rs | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/libsql-server/src/namespace/fence/tests.rs b/libsql-server/src/namespace/fence/tests.rs index 8791ee2e54..5f2c5fec85 100644 --- a/libsql-server/src/namespace/fence/tests.rs +++ b/libsql-server/src/namespace/fence/tests.rs @@ -610,7 +610,7 @@ fn restart_at(case: &Boundary) { // The drain that was requested resumes and completes at once: recovery // discarded any uncommitted work. The boundary is on the live, rebuilt log. let commit = replay.unwrap(); - assert_eq!(commit.kind, FenceCommitKind::Committed, "{name}"); + assert_eq!(commit.kind, FenceCommitKind::Resumed, "{name}"); assert_eq!(commit.receipt.outcome, FenceOutcome::Applied, "{name}"); assert_eq!(commit.receipt.command_id, request.command_id); assert_eq!(commit.receipt.revision_after, 2, "{name}"); @@ -1569,10 +1569,10 @@ mod read_and_target_boundaries { } let replay = server.execute(request.clone()).await.unwrap().unwrap(); - let kind = if case.durable == Durable::Final { - FenceCommitKind::Replayed - } else { - FenceCommitKind::Committed + let kind = match case.durable { + Durable::Nothing => FenceCommitKind::Committed, + Durable::Draining => FenceCommitKind::Resumed, + Durable::Final => FenceCommitKind::Replayed, }; assert_eq!(replay.kind, kind, "{name}"); assert_eq!(replay.receipt.outcome, FenceOutcome::Applied, "{name}");